summaryrefslogtreecommitdiff
path: root/include/linux
diff options
context:
space:
mode:
Diffstat (limited to 'include/linux')
-rw-r--r--include/linux/acpi.h30
-rw-r--r--include/linux/amd-iommu.h13
-rw-r--r--include/linux/apple-gmux.h2
-rw-r--r--include/linux/arm-rsi-cmds.h239
-rw-r--r--include/linux/arm-smccc-bus.h48
-rw-r--r--include/linux/arm-smccc-rsi.h193
-rw-r--r--include/linux/audit.h6
-rw-r--r--include/linux/binfmts.h3
-rw-r--r--include/linux/bio-integrity.h3
-rw-r--r--include/linux/bio.h17
-rw-r--r--include/linux/bitmap.h14
-rw-r--r--include/linux/blkdev.h19
-rw-r--r--include/linux/bnge/hsi.h136
-rw-r--r--include/linux/bootconfig.h7
-rw-r--r--include/linux/bpf-cgroup-defs.h1
-rw-r--r--include/linux/bpf-cgroup.h28
-rw-r--r--include/linux/bpf.h222
-rw-r--r--include/linux/bpf_crypto.h24
-rw-r--r--include/linux/bpf_lsm.h11
-rw-r--r--include/linux/bpf_verifier.h356
-rw-r--r--include/linux/btf.h24
-rw-r--r--include/linux/buffer_head.h94
-rw-r--r--include/linux/cache.h1
-rw-r--r--include/linux/cacheinfo.h12
-rw-r--r--include/linux/capability.h13
-rw-r--r--include/linux/ccp.h58
-rw-r--r--include/linux/cgroup-defs.h5
-rw-r--r--include/linux/cgroup.h3
-rw-r--r--include/linux/cleanup.h7
-rw-r--r--include/linux/clk-provider.h25
-rw-r--r--include/linux/clk/tegra.h2
-rw-r--r--include/linux/clk/ti.h2
-rw-r--r--include/linux/compaction.h11
-rw-r--r--include/linux/compiler.h22
-rw-r--r--include/linux/configfs.h73
-rw-r--r--include/linux/console.h2
-rw-r--r--include/linux/coredump.h37
-rw-r--r--include/linux/cpufreq.h3
-rw-r--r--include/linux/cpumask.h42
-rw-r--r--include/linux/cred.h17
-rw-r--r--include/linux/damon.h102
-rw-r--r--include/linux/dax.h12
-rw-r--r--include/linux/dcache.h38
-rw-r--r--include/linux/device-id/ap.h2
-rw-r--r--include/linux/device-id/apr.h21
-rw-r--r--include/linux/device-id/arm_smccc.h19
-rw-r--r--include/linux/device-id/scmi.h17
-rw-r--r--include/linux/device/bus.h1
-rw-r--r--include/linux/device/driver.h31
-rw-r--r--include/linux/dibs.h6
-rw-r--r--include/linux/dma-buf.h1
-rw-r--r--include/linux/dma-fence-array.h1
-rw-r--r--include/linux/dma-fence-chain.h9
-rw-r--r--include/linux/dma-fence.h8
-rw-r--r--include/linux/dma-map-ops.h4
-rw-r--r--include/linux/dma-mapping.h4
-rw-r--r--include/linux/dmaengine.h9
-rw-r--r--include/linux/edac.h3
-rw-r--r--include/linux/entry-common.h9
-rw-r--r--include/linux/ethtool.h12
-rw-r--r--include/linux/f2fs_fs.h140
-rw-r--r--include/linux/fault-inject.h10
-rw-r--r--include/linux/fdtable.h15
-rw-r--r--include/linux/file.h130
-rw-r--r--include/linux/fileattr.h2
-rw-r--r--include/linux/filter.h46
-rw-r--r--include/linux/firewire.h50
-rw-r--r--include/linux/firmware/cirrus/cs_dsp.h2
-rw-r--r--include/linux/firmware/imx/se_api.h14
-rw-r--r--include/linux/firmware/qcom/qcom_scm.h29
-rw-r--r--include/linux/firmware/xlnx-zynqmp-ufs.h8
-rw-r--r--include/linux/firmware/xlnx-zynqmp.h5
-rw-r--r--include/linux/fs.h127
-rw-r--r--include/linux/fs_context.h4
-rw-r--r--include/linux/fscache-cache.h2
-rw-r--r--include/linux/fscache.h53
-rw-r--r--include/linux/ftrace.h5
-rw-r--r--include/linux/generic_pt/common.h6
-rw-r--r--include/linux/generic_pt/iommu.h12
-rw-r--r--include/linux/gfp.h2
-rw-r--r--include/linux/gfp_types.h2
-rw-r--r--include/linux/gpu_buddy.h101
-rw-r--r--include/linux/hazptr.h325
-rw-r--r--include/linux/hdmi.h12
-rw-r--r--include/linux/hiddev.h2
-rw-r--r--include/linux/hrtimer.h10
-rw-r--r--include/linux/hrtimer_rearm.h2
-rw-r--r--include/linux/huge_mm.h9
-rw-r--r--include/linux/hugetlb.h50
-rw-r--r--include/linux/hugetlb_inline.h28
-rw-r--r--include/linux/hwmon.h12
-rw-r--r--include/linux/i2c.h36
-rw-r--r--include/linux/i3c/device.h4
-rw-r--r--include/linux/i3c/hub.h92
-rw-r--r--include/linux/i3c/master.h20
-rw-r--r--include/linux/idr.h5
-rw-r--r--include/linux/ieee80211-eht.h18
-rw-r--r--include/linux/ieee80211-mesh.h4
-rw-r--r--include/linux/ieee80211-uhr.h101
-rw-r--r--include/linux/ieee80211.h25
-rw-r--r--include/linux/if_vlan.h3
-rw-r--r--include/linux/igmp.h10
-rw-r--r--include/linux/input/samsung-keypad.h39
-rw-r--r--include/linux/intel_vsec.h14
-rw-r--r--include/linux/interrupt.h6
-rw-r--r--include/linux/interrupt_rc.h22
-rw-r--r--include/linux/io_uring.h32
-rw-r--r--include/linux/io_uring/cmd.h34
-rw-r--r--include/linux/io_uring_types.h2
-rw-r--r--include/linux/iomap.h38
-rw-r--r--include/linux/iommu-dma.h2
-rw-r--r--include/linux/irq-entry-common.h17
-rw-r--r--include/linux/irqchip/arm-gic-v5.h256
-rw-r--r--include/linux/irqchip/arm-vgic-info.h5
-rw-r--r--include/linux/irqdomain.h18
-rw-r--r--include/linux/kdev_t.h2
-rw-r--r--include/linux/kernel-page-flags.h1
-rw-r--r--include/linux/kernel.h20
-rw-r--r--include/linux/kernel_stat.h11
-rw-r--r--include/linux/kexec.h17
-rw-r--r--include/linux/kexec_handover.h16
-rw-r--r--include/linux/key.h2
-rw-r--r--include/linux/kho/abi/pci.h66
-rw-r--r--include/linux/kmod.h12
-rw-r--r--include/linux/kprobes.h1
-rw-r--r--include/linux/kvm_host.h102
-rw-r--r--include/linux/libata.h5
-rw-r--r--include/linux/list.h6
-rw-r--r--include/linux/liveupdate.h22
-rw-r--r--include/linux/llc.h5
-rw-r--r--include/linux/lsm_audit.h7
-rw-r--r--include/linux/lsm_hook_defs.h53
-rw-r--r--include/linux/lsm_hooks.h1
-rw-r--r--include/linux/mailbox/riscv-rpmi-message.h11
-rw-r--r--include/linux/maple_tree.h4
-rw-r--r--include/linux/mdio-mux.h2
-rw-r--r--include/linux/memblock.h31
-rw-r--r--include/linux/memcontrol.h452
-rw-r--r--include/linux/mempool.h7
-rw-r--r--include/linux/mfd/88pm886.h7
-rw-r--r--include/linux/mfd/cs42l43-regs.h1
-rw-r--r--include/linux/mfd/da9150/core.h1
-rw-r--r--include/linux/mfd/khadas-mcu.h19
-rw-r--r--include/linux/mfd/lm3533.h75
-rw-r--r--include/linux/mfd/motorola-cpcap.h7
-rw-r--r--include/linux/mfd/rohm-bd73800.h306
-rw-r--r--include/linux/mfd/rohm-generic.h1
-rw-r--r--include/linux/micrel_phy.h5
-rw-r--r--include/linux/minmax.h6
-rw-r--r--include/linux/mlx5/eswitch.h3
-rw-r--r--include/linux/mm.h400
-rw-r--r--include/linux/mm_inline.h115
-rw-r--r--include/linux/mm_types.h81
-rw-r--r--include/linux/mmap_lock.h108
-rw-r--r--include/linux/mmc/host.h6
-rw-r--r--include/linux/mmzone.h222
-rw-r--r--include/linux/mnt_idmapping.h24
-rw-r--r--include/linux/mod_devicetable.h2
-rw-r--r--include/linux/module_symbol.h7
-rw-r--r--include/linux/mount.h4
-rw-r--r--include/linux/mtd/bbm.h2
-rw-r--r--include/linux/mtd/lpc32xx_mlc.h17
-rw-r--r--include/linux/mtd/lpc32xx_slc.h17
-rw-r--r--include/linux/mtd/nand-qpic-common.h1
-rw-r--r--include/linux/mtd/nand.h2
-rw-r--r--include/linux/mtd/rawnand.h2
-rw-r--r--include/linux/mtd/sh_flctl.h4
-rw-r--r--include/linux/mtd/spi-nor.h10
-rw-r--r--include/linux/mtd/spinand.h3
-rw-r--r--include/linux/namei.h19
-rw-r--r--include/linux/net.h34
-rw-r--r--include/linux/netdevice.h26
-rw-r--r--include/linux/netfilter_arp/arp_tables.h19
-rw-r--r--include/linux/netfilter_netdev.h2
-rw-r--r--include/linux/netfs.h119
-rw-r--r--include/linux/netpoll.h5
-rw-r--r--include/linux/nfs.h55
-rw-r--r--include/linux/nfs3.h43
-rw-r--r--include/linux/nfs4.h6
-rw-r--r--include/linux/nfs_fh.h63
-rw-r--r--include/linux/nfs_fs.h6
-rw-r--r--include/linux/nfs_fs_sb.h8
-rw-r--r--include/linux/nfs_page.h8
-rw-r--r--include/linux/nfs_ssc.h69
-rw-r--r--include/linux/nfs_xdr.h2
-rw-r--r--include/linux/nfsd_ssc.h38
-rw-r--r--include/linux/nfslocalio.h11
-rw-r--r--include/linux/ns/ns_common_types.h9
-rw-r--r--include/linux/nvme-tcp.h18
-rw-r--r--include/linux/once_lite.h8
-rw-r--r--include/linux/page-flags.h64
-rw-r--r--include/linux/page_counter.h3
-rw-r--r--include/linux/pagemap.h68
-rw-r--r--include/linux/panic.h5
-rw-r--r--include/linux/pci-epf.h3
-rw-r--r--include/linux/pci.h14
-rw-r--r--include/linux/pci_liveupdate.h72
-rw-r--r--include/linux/perf/riscv_pmu.h2
-rw-r--r--include/linux/perf_event.h40
-rw-r--r--include/linux/perf_regs.h36
-rw-r--r--include/linux/pgtable.h24
-rw-r--r--include/linux/phy.h28
-rw-r--r--include/linux/phy/phy.h13
-rw-r--r--include/linux/phylink.h2
-rw-r--r--include/linux/platform_data/microchip-ksz.h2
-rw-r--r--include/linux/platform_data/mipi-i3c-hci.h4
-rw-r--r--include/linux/platform_data/x86/asus-wmi.h4
-rw-r--r--include/linux/platform_data/x86/int3472.h2
-rw-r--r--include/linux/platform_device.h30
-rw-r--r--include/linux/pm.h93
-rw-r--r--include/linux/pm_domain.h9
-rw-r--r--include/linux/pm_qos.h9
-rw-r--r--include/linux/pm_runtime.h409
-rw-r--r--include/linux/posix-timers.h39
-rw-r--r--include/linux/posix_acl.h24
-rw-r--r--include/linux/power/bq27xxx_battery.h1
-rw-r--r--include/linux/power_supply.h24
-rw-r--r--include/linux/preempt.h24
-rw-r--r--include/linux/property.h5
-rw-r--r--include/linux/psp-sev.h86
-rw-r--r--include/linux/ptdump.h4
-rw-r--r--include/linux/quotaops.h6
-rw-r--r--include/linux/rbtree_augmented.h35
-rw-r--r--include/linux/rcupdate.h10
-rw-r--r--include/linux/rcuref.h2
-rw-r--r--include/linux/regulator/driver.h2
-rw-r--r--include/linux/regulator/pca9450.h4
-rw-r--r--include/linux/resctrl.h19
-rw-r--r--include/linux/rhashtable-types.h74
-rw-r--r--include/linux/rhashtable.h19
-rw-r--r--include/linux/ring_buffer.h24
-rw-r--r--include/linux/rmap.h2
-rw-r--r--include/linux/rolling_buffer.h6
-rw-r--r--include/linux/rtnetlink.h7
-rw-r--r--include/linux/sched.h68
-rw-r--r--include/linux/sched/cputime.h6
-rw-r--r--include/linux/sched/ext.h38
-rw-r--r--include/linux/sched/rt.h2
-rw-r--r--include/linux/sched/signal.h38
-rw-r--r--include/linux/sched/task.h1
-rw-r--r--include/linux/sched/topology.h4
-rw-r--r--include/linux/sched/user.h3
-rw-r--r--include/linux/scmi_protocol.h20
-rw-r--r--include/linux/security.h143
-rw-r--r--include/linux/set_memory.h120
-rw-r--r--include/linux/shmem_fs.h12
-rw-r--r--include/linux/skbuff.h38
-rw-r--r--include/linux/soc/qcom/apr.h4
-rw-r--r--include/linux/soc/qcom/geni-se.h15
-rw-r--r--include/linux/soc/qcom/llcc-qcom.h6
-rw-r--r--include/linux/soc/qcom/tc9563.h19
-rw-r--r--include/linux/soc/qcom/ubwc.h3
-rw-r--r--include/linux/soc/ti/k3-ringacc.h31
-rw-r--r--include/linux/spi/spi.h9
-rw-r--r--include/linux/splice.h4
-rw-r--r--include/linux/srcu.h83
-rw-r--r--include/linux/srcutiny.h18
-rw-r--r--include/linux/srcutree.h19
-rw-r--r--include/linux/stmmac.h2
-rw-r--r--include/linux/string.h13
-rw-r--r--include/linux/string_choices.h6
-rw-r--r--include/linux/sunrpc/svc_xprt.h5
-rw-r--r--include/linux/swap.h59
-rw-r--r--include/linux/swapops.h9
-rw-r--r--include/linux/thermal.h19
-rw-r--r--include/linux/thunderbolt.h3
-rw-r--r--include/linux/torture.h6
-rw-r--r--include/linux/trace_remote_event.h2
-rw-r--r--include/linux/tty.h2
-rw-r--r--include/linux/tty_driver.h6
-rw-r--r--include/linux/tty_port.h22
-rw-r--r--include/linux/uidgid.h12
-rw-r--r--include/linux/usb/serial.h1
-rw-r--r--include/linux/user_namespace.h11
-rw-r--r--include/linux/userfaultfd_k.h1
-rw-r--r--include/linux/vga_switcheroo.h29
-rw-r--r--include/linux/vgaarb.h8
-rw-r--r--include/linux/vmalloc.h4
-rw-r--r--include/linux/vmemmap-optimization.h115
-rw-r--r--include/linux/wait.h2
-rw-r--r--include/linux/wait_bit.h26
-rw-r--r--include/linux/xattr.h20
-rw-r--r--include/linux/zsmalloc.h4
-rw-r--r--include/linux/zswap.h8
284 files changed, 6831 insertions, 2655 deletions
diff --git a/include/linux/acpi.h b/include/linux/acpi.h
index ddacac812094..aef0a01f4cfa 100644
--- a/include/linux/acpi.h
+++ b/include/linux/acpi.h
@@ -25,17 +25,19 @@ struct irq_domain_ops;
#define _LINUX
#endif
#include <acpi/acpi.h>
+#include <acpi/acpi_bus.h>
#include <acpi/acpi_numa.h>
#ifdef CONFIG_ACPI
+DEFINE_FREE(acpi_object_free, union acpi_object *, if (_T) ACPI_FREE(_T));
+
#include <linux/list.h>
#include <linux/dynamic_debug.h>
#include <linux/module.h>
#include <linux/mutex.h>
#include <linux/fw_table.h>
-#include <acpi/acpi_bus.h>
#include <acpi/acpi_drivers.h>
#include <acpi/acpi_io.h>
#include <asm/acpi.h>
@@ -434,7 +436,7 @@ extern acpi_handle ec_get_handle(void);
extern bool acpi_is_pnp_device(struct acpi_device *);
-#if defined(CONFIG_ACPI_WMI) || defined(CONFIG_ACPI_WMI_MODULE)
+#if IS_ENABLED(CONFIG_ACPI_WMI)
typedef void (*wmi_notify_handler) (union acpi_object *data, void *context);
@@ -1305,34 +1307,34 @@ void __acpi_handle_debug(struct _ddebug *descriptor, acpi_handle handle, const c
#endif
/*
- * acpi_handle_<level>: Print message with ACPI prefix and object path
+ * acpi_handle_<level> - Print a message with ACPI prefix and object path
*
- * These interfaces acquire the global namespace mutex to obtain an object
- * path. In interrupt context, it shows the object path as <n/a>.
+ * In thread context, the global namespace mutex is acquired to obtain the
+ * object path. In interrupt context, the object path is shown as <n/a>.
*/
#define acpi_handle_emerg(handle, fmt, ...) \
- acpi_handle_printk(KERN_EMERG, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_EMERG, handle, dev_fmt(fmt), ##__VA_ARGS__)
#define acpi_handle_alert(handle, fmt, ...) \
- acpi_handle_printk(KERN_ALERT, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_ALERT, handle, dev_fmt(fmt), ##__VA_ARGS__)
#define acpi_handle_crit(handle, fmt, ...) \
- acpi_handle_printk(KERN_CRIT, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_CRIT, handle, dev_fmt(fmt), ##__VA_ARGS__)
#define acpi_handle_err(handle, fmt, ...) \
- acpi_handle_printk(KERN_ERR, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_ERR, handle, dev_fmt(fmt), ##__VA_ARGS__)
#define acpi_handle_warn(handle, fmt, ...) \
- acpi_handle_printk(KERN_WARNING, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_WARNING, handle, dev_fmt(fmt), ##__VA_ARGS__)
#define acpi_handle_notice(handle, fmt, ...) \
- acpi_handle_printk(KERN_NOTICE, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_NOTICE, handle, dev_fmt(fmt), ##__VA_ARGS__)
#define acpi_handle_info(handle, fmt, ...) \
- acpi_handle_printk(KERN_INFO, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_INFO, handle, dev_fmt(fmt), ##__VA_ARGS__)
#if defined(DEBUG)
#define acpi_handle_debug(handle, fmt, ...) \
- acpi_handle_printk(KERN_DEBUG, handle, fmt, ##__VA_ARGS__)
+ acpi_handle_printk(KERN_DEBUG, handle, dev_fmt(fmt), ##__VA_ARGS__)
#else
#if defined(CONFIG_DYNAMIC_DEBUG)
#define acpi_handle_debug(handle, fmt, ...) \
_dynamic_func_call(fmt, __acpi_handle_debug, \
- handle, pr_fmt(fmt), ##__VA_ARGS__)
+ handle, dev_fmt(fmt), ##__VA_ARGS__)
#else
#define acpi_handle_debug(handle, fmt, ...) \
({ \
diff --git a/include/linux/amd-iommu.h b/include/linux/amd-iommu.h
index edcee9f5335a..104817e8d897 100644
--- a/include/linux/amd-iommu.h
+++ b/include/linux/amd-iommu.h
@@ -11,11 +11,11 @@
#include <linux/types.h>
struct amd_iommu;
+struct pci_dev;
#ifdef CONFIG_AMD_IOMMU
struct task_struct;
-struct pci_dev;
extern void amd_iommu_detect(void);
@@ -76,4 +76,15 @@ static inline int amd_iommu_snp_disable(void) { return 0; }
static inline bool amd_iommu_sev_tio_supported(void) { return false; }
#endif
+#ifdef CONFIG_AMD_IOMMU
+int amd_iommu_enable_perfopt(struct pci_dev *pdev);
+void amd_iommu_disable_perfopt(struct pci_dev *pdev);
+#else
+static inline int amd_iommu_enable_perfopt(struct pci_dev *pdev)
+{
+ return 0;
+}
+static inline void amd_iommu_disable_perfopt(struct pci_dev *pdev) { }
+#endif
+
#endif /* _ASM_X86_AMD_IOMMU_H */
diff --git a/include/linux/apple-gmux.h b/include/linux/apple-gmux.h
index 206d97ffda79..d7939c3a08fd 100644
--- a/include/linux/apple-gmux.h
+++ b/include/linux/apple-gmux.h
@@ -109,7 +109,7 @@ static inline bool apple_gmux_detect(struct pnp_dev *pnp_dev, enum apple_gmux_ty
if (!adev)
return false;
- dev = get_device(acpi_get_first_physical_node(adev));
+ dev = acpi_bus_get_primary_device(adev);
acpi_dev_put(adev);
if (!dev)
return false;
diff --git a/include/linux/arm-rsi-cmds.h b/include/linux/arm-rsi-cmds.h
new file mode 100644
index 000000000000..3f7a6a833993
--- /dev/null
+++ b/include/linux/arm-rsi-cmds.h
@@ -0,0 +1,239 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Copyright (C) 2023 ARM Ltd.
+ */
+
+#ifndef __LINUX_ARM_RSI_CMDS_H_
+#define __LINUX_ARM_RSI_CMDS_H_
+
+#include <linux/arm-smccc-rsi.h>
+#include <linux/jump_label.h>
+#include <linux/string.h>
+#include <asm/memory.h>
+
+#ifdef CONFIG_ARM_RMM_RSI
+DECLARE_STATIC_KEY_FALSE(rsi_present);
+
+void __init arm64_rsi_init(void);
+
+bool arm64_rsi_is_protected(phys_addr_t base, size_t size);
+
+static inline bool is_realm_world(void)
+{
+ return static_branch_unlikely(&rsi_present);
+}
+#else
+static inline void arm64_rsi_init(void) { }
+
+static inline bool arm64_rsi_is_protected(phys_addr_t base, size_t size)
+{
+ return false;
+}
+
+static inline bool is_realm_world(void) { return false; }
+#endif
+
+#define RSI_GRANULE_SHIFT 12
+#define RSI_GRANULE_SIZE (_AC(1, UL) << RSI_GRANULE_SHIFT)
+
+enum ripas {
+ RSI_RIPAS_EMPTY = 0,
+ RSI_RIPAS_RAM = 1,
+ RSI_RIPAS_DESTROYED = 2,
+ RSI_RIPAS_DEV = 3,
+};
+
+static inline unsigned long rsi_request_version(unsigned long req,
+ unsigned long *out_lower,
+ unsigned long *out_higher)
+{
+ struct arm_smccc_res res;
+
+ arm_smccc_smc(SMC_RSI_ABI_VERSION, req, 0, 0, 0, 0, 0, 0, &res);
+
+ if (out_lower)
+ *out_lower = res.a1;
+ if (out_higher)
+ *out_higher = res.a2;
+
+ return res.a0;
+}
+
+static inline unsigned long rsi_get_realm_config(struct realm_config *cfg)
+{
+ struct arm_smccc_res res;
+
+ arm_smccc_smc(SMC_RSI_REALM_CONFIG, virt_to_phys(cfg),
+ 0, 0, 0, 0, 0, 0, &res);
+ return res.a0;
+}
+
+static inline unsigned long rsi_ipa_state_get(phys_addr_t start,
+ phys_addr_t end,
+ enum ripas *state,
+ phys_addr_t *top)
+{
+ struct arm_smccc_res res;
+
+ arm_smccc_smc(SMC_RSI_IPA_STATE_GET,
+ start, end, 0, 0, 0, 0, 0,
+ &res);
+
+ if (res.a0 == RSI_SUCCESS) {
+ if (top)
+ *top = res.a1;
+ if (state)
+ *state = res.a2;
+ }
+
+ return res.a0;
+}
+
+static inline long rsi_set_addr_range_state(phys_addr_t start,
+ phys_addr_t end,
+ enum ripas state,
+ unsigned long flags,
+ phys_addr_t *top)
+{
+ struct arm_smccc_res res;
+
+ arm_smccc_smc(SMC_RSI_IPA_STATE_SET, start, end, state,
+ flags, 0, 0, 0, &res);
+
+ if (top)
+ *top = res.a1;
+
+ if (res.a2 != RSI_ACCEPT)
+ return -EPERM;
+
+ return res.a0;
+}
+
+static inline int rsi_set_memory_range(phys_addr_t start, phys_addr_t end,
+ enum ripas state, unsigned long flags)
+{
+ unsigned long ret;
+ phys_addr_t top;
+
+ while (start != end) {
+ ret = rsi_set_addr_range_state(start, end, state, flags, &top);
+ if (ret || top < start || top > end)
+ return -EINVAL;
+ start = top;
+ }
+
+ return 0;
+}
+
+/*
+ * Convert the specified range to RAM. Do not use this if you rely on the
+ * contents of a page that may already be in RAM state.
+ */
+static inline int rsi_set_memory_range_protected(phys_addr_t start,
+ phys_addr_t end)
+{
+ return rsi_set_memory_range(start, end, RSI_RIPAS_RAM,
+ RSI_CHANGE_DESTROYED);
+}
+
+/*
+ * Convert the specified range to RAM. Do not convert any pages that may have
+ * been DESTROYED, without our permission.
+ */
+static inline int rsi_set_memory_range_protected_safe(phys_addr_t start,
+ phys_addr_t end)
+{
+ return rsi_set_memory_range(start, end, RSI_RIPAS_RAM,
+ RSI_NO_CHANGE_DESTROYED);
+}
+
+static inline int rsi_set_memory_range_shared(phys_addr_t start,
+ phys_addr_t end)
+{
+ return rsi_set_memory_range(start, end, RSI_RIPAS_EMPTY,
+ RSI_CHANGE_DESTROYED);
+}
+
+#define RSI_ATTEST_CHALLENGE_MIN_SIZE 32
+#define RSI_ATTEST_CHALLENGE_MAX_SIZE 64
+
+struct rsi_attestation_token_init_args {
+ unsigned long fid;
+ u8 challenge[RSI_ATTEST_CHALLENGE_MAX_SIZE];
+};
+
+/**
+ * rsi_attestation_token_init - Initialise the operation to retrieve an
+ * attestation token.
+ *
+ * @challenge: The challenge data to be used in the attestation token
+ * generation.
+ * @size: Size of the challenge data in bytes.
+ *
+ * Initialises the attestation token generation and returns an upper bound
+ * on the attestation token size that can be used to allocate an adequate
+ * buffer. The caller is expected to subsequently call
+ * rsi_attestation_token_continue() to retrieve the attestation token data on
+ * the same CPU.
+ *
+ * Returns:
+ * On success, returns the upper limit of the attestation report size.
+ * Otherwise, -EINVAL
+ */
+static inline long
+rsi_attestation_token_init(const u8 *challenge, unsigned long size)
+{
+ union {
+ struct arm_smccc_1_2_regs regs;
+ struct rsi_attestation_token_init_args init;
+ } args = { 0 };
+
+ if (!challenge || size < RSI_ATTEST_CHALLENGE_MIN_SIZE ||
+ size > RSI_ATTEST_CHALLENGE_MAX_SIZE)
+ return -EINVAL;
+
+ args.init.fid = SMC_RSI_ATTESTATION_TOKEN_INIT;
+ memcpy(args.init.challenge, challenge, size);
+ arm_smccc_1_2_smc(&args.regs, &args.regs);
+
+ if (args.regs.a0 == RSI_SUCCESS)
+ return args.regs.a1;
+
+ return -EINVAL;
+}
+
+/**
+ * rsi_attestation_token_continue - Continue the operation to retrieve an
+ * attestation token.
+ *
+ * @granule: {I}PA of the Granule to which the token will be written.
+ * @offset: Offset within Granule to start of buffer in bytes.
+ * @size: The size of the buffer.
+ * @len: The number of bytes written to the buffer.
+ *
+ * Retrieves up to a RSI_GRANULE_SIZE worth of token data per call. The caller
+ * is expected to call rsi_attestation_token_init() before calling this
+ * function to retrieve the attestation token.
+ *
+ * Return:
+ * * %RSI_SUCCESS - Attestation token retrieved successfully.
+ * * %RSI_INCOMPLETE - Token generation is not complete.
+ * * %RSI_ERROR_INPUT - A parameter was not valid.
+ * * %RSI_ERROR_STATE - Attestation not in progress.
+ */
+static inline unsigned long rsi_attestation_token_continue(phys_addr_t granule,
+ unsigned long offset,
+ unsigned long size,
+ unsigned long *len)
+{
+ struct arm_smccc_res res;
+
+ arm_smccc_1_1_invoke(SMC_RSI_ATTESTATION_TOKEN_CONTINUE,
+ granule, offset, size, 0, &res);
+
+ if (len)
+ *len = res.a1;
+ return res.a0;
+}
+
+#endif /* __LINUX_ARM_RSI_CMDS_H_ */
diff --git a/include/linux/arm-smccc-bus.h b/include/linux/arm-smccc-bus.h
new file mode 100644
index 000000000000..05c0c850510e
--- /dev/null
+++ b/include/linux/arm-smccc-bus.h
@@ -0,0 +1,48 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Copyright (C) 2026 Arm Limited
+ */
+#ifndef __LINUX_ARM_SMCCC_BUS_H
+#define __LINUX_ARM_SMCCC_BUS_H
+
+#include <linux/device.h>
+#include <linux/device-id/arm_smccc.h>
+#include <linux/module.h>
+
+struct arm_smccc_device {
+ struct device dev;
+ u32 func_id;
+};
+
+#define to_arm_smccc_device(d) container_of_const(d, struct arm_smccc_device, dev)
+
+struct arm_smccc_driver {
+ struct device_driver driver;
+
+ const char *name;
+ int (*probe)(struct arm_smccc_device *sdev);
+ void (*remove)(struct arm_smccc_device *sdev);
+ const struct arm_smccc_device_id *id_table;
+};
+
+#define to_arm_smccc_driver(d) \
+ container_of_const(d, struct arm_smccc_driver, driver)
+
+int arm_smccc_driver_register(struct arm_smccc_driver *driver,
+ struct module *owner, const char *mod_name);
+void arm_smccc_driver_unregister(struct arm_smccc_driver *driver);
+struct arm_smccc_device *arm_smccc_device_register(const char *name, u32 func_id);
+void arm_smccc_device_unregister(struct arm_smccc_device *smcc_dev);
+
+#define arm_smccc_register(driver) \
+ arm_smccc_driver_register(driver, THIS_MODULE, KBUILD_MODNAME)
+#define arm_smccc_unregister(driver) \
+ arm_smccc_driver_unregister(driver)
+
+#define module_arm_smccc_driver(__arm_smccc_driver) \
+ module_driver(__arm_smccc_driver, arm_smccc_register, \
+ arm_smccc_unregister)
+
+extern const struct bus_type arm_smccc_bus_type;
+
+#endif /* __LINUX_ARM_SMCCC_BUS_H */
diff --git a/include/linux/arm-smccc-rsi.h b/include/linux/arm-smccc-rsi.h
new file mode 100644
index 000000000000..fddb77986f70
--- /dev/null
+++ b/include/linux/arm-smccc-rsi.h
@@ -0,0 +1,193 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Copyright (C) 2023 ARM Ltd.
+ */
+
+#ifndef __LINUX_ARM_SMCCC_RSI_H_
+#define __LINUX_ARM_SMCCC_RSI_H_
+
+#include <linux/arm-smccc.h>
+
+/*
+ * This file describes the Realm Services Interface (RSI) Application Binary
+ * Interface (ABI) for SMC calls made from within the Realm to the RMM and
+ * serviced by the RMM.
+ */
+
+/*
+ * The major version number of the RSI implementation. This is increased when
+ * the binary format or semantics of the SMC calls change.
+ */
+#define RSI_ABI_VERSION_MAJOR UL(1)
+
+/*
+ * The minor version number of the RSI implementation. This is increased when
+ * a bug is fixed, or a feature is added without breaking binary compatibility.
+ */
+#define RSI_ABI_VERSION_MINOR UL(0)
+
+#define RSI_ABI_VERSION ((RSI_ABI_VERSION_MAJOR << 16) | \
+ RSI_ABI_VERSION_MINOR)
+
+#define RSI_ABI_VERSION_GET_MAJOR(_version) ((_version) >> 16)
+#define RSI_ABI_VERSION_GET_MINOR(_version) ((_version) & 0xFFFF)
+
+#define RSI_SUCCESS UL(0)
+#define RSI_ERROR_INPUT UL(1)
+#define RSI_ERROR_STATE UL(2)
+#define RSI_INCOMPLETE UL(3)
+#define RSI_ERROR_UNKNOWN UL(4)
+
+#define SMC_RSI_FID(n) ARM_SMCCC_CALL_VAL(ARM_SMCCC_FAST_CALL, \
+ ARM_SMCCC_SMC_64, \
+ ARM_SMCCC_OWNER_STANDARD, \
+ n)
+
+/*
+ * Returns RSI version.
+ *
+ * arg1 == Requested interface revision
+ * ret0 == Status / error
+ * ret1 == Lower implemented interface revision
+ * ret2 == Higher implemented interface revision
+ */
+#define SMC_RSI_ABI_VERSION SMC_RSI_FID(0x190)
+
+/*
+ * Read feature register.
+ *
+ * arg1 == Feature register index
+ * ret0 == Status / error
+ * ret1 == Feature register value
+ */
+#define SMC_RSI_FEATURES SMC_RSI_FID(0x191)
+
+/*
+ * Read measurement for the current Realm.
+ *
+ * arg1 == Index, which measurements slot to read
+ * ret0 == Status / error
+ * ret1 == Measurement value, bytes: 0 - 7
+ * ret2 == Measurement value, bytes: 8 - 15
+ * ret3 == Measurement value, bytes: 16 - 23
+ * ret4 == Measurement value, bytes: 24 - 31
+ * ret5 == Measurement value, bytes: 32 - 39
+ * ret6 == Measurement value, bytes: 40 - 47
+ * ret7 == Measurement value, bytes: 48 - 55
+ * ret8 == Measurement value, bytes: 56 - 63
+ */
+#define SMC_RSI_MEASUREMENT_READ SMC_RSI_FID(0x192)
+
+/*
+ * Extend Realm Extensible Measurement (REM) value.
+ *
+ * arg1 == Index, which measurements slot to extend
+ * arg2 == Size of realm measurement in bytes, max 64 bytes
+ * arg3 == Measurement value, bytes: 0 - 7
+ * arg4 == Measurement value, bytes: 8 - 15
+ * arg5 == Measurement value, bytes: 16 - 23
+ * arg6 == Measurement value, bytes: 24 - 31
+ * arg7 == Measurement value, bytes: 32 - 39
+ * arg8 == Measurement value, bytes: 40 - 47
+ * arg9 == Measurement value, bytes: 48 - 55
+ * arg10 == Measurement value, bytes: 56 - 63
+ * ret0 == Status / error
+ */
+#define SMC_RSI_MEASUREMENT_EXTEND SMC_RSI_FID(0x193)
+
+/*
+ * Initialize the operation to retrieve an attestation token.
+ *
+ * arg1 == Challenge value, bytes: 0 - 7
+ * arg2 == Challenge value, bytes: 8 - 15
+ * arg3 == Challenge value, bytes: 16 - 23
+ * arg4 == Challenge value, bytes: 24 - 31
+ * arg5 == Challenge value, bytes: 32 - 39
+ * arg6 == Challenge value, bytes: 40 - 47
+ * arg7 == Challenge value, bytes: 48 - 55
+ * arg8 == Challenge value, bytes: 56 - 63
+ * ret0 == Status / error
+ * ret1 == Upper bound of token size in bytes
+ */
+#define SMC_RSI_ATTESTATION_TOKEN_INIT SMC_RSI_FID(0x194)
+
+/*
+ * Continue the operation to retrieve an attestation token.
+ *
+ * arg1 == The IPA of token buffer
+ * arg2 == Offset within the granule of the token buffer
+ * arg3 == Size of the granule buffer
+ * ret0 == Status / error
+ * ret1 == Length of token bytes copied to the granule buffer
+ */
+#define SMC_RSI_ATTESTATION_TOKEN_CONTINUE SMC_RSI_FID(0x195)
+
+#ifndef __ASSEMBLER__
+
+struct realm_config {
+ union {
+ struct {
+ unsigned long ipa_bits; /* Width of IPA in bits */
+ unsigned long hash_algo; /* Hash algorithm */
+ };
+ u8 pad[0x200];
+ };
+ union {
+ u8 rpv[64]; /* Realm Personalization Value */
+ u8 pad2[0xe00];
+ };
+ /*
+ * The RMM requires the configuration structure to be aligned to a 4k
+ * boundary, ensure this happens by aligning this structure.
+ */
+} __aligned(0x1000);
+
+#endif /* __ASSEMBLER__ */
+
+/*
+ * Read configuration for the current Realm.
+ *
+ * arg1 == struct realm_config addr
+ * ret0 == Status / error
+ */
+#define SMC_RSI_REALM_CONFIG SMC_RSI_FID(0x196)
+
+/*
+ * Request RIPAS of a target IPA range to be changed to a specified value.
+ *
+ * arg1 == Base IPA address of target region
+ * arg2 == Top of the region
+ * arg3 == RIPAS value
+ * arg4 == flags
+ * ret0 == Status / error
+ * ret1 == Top of modified IPA range
+ * ret2 == Whether the Host accepted or rejected the request
+ */
+#define SMC_RSI_IPA_STATE_SET SMC_RSI_FID(0x197)
+
+#define RSI_NO_CHANGE_DESTROYED UL(0)
+#define RSI_CHANGE_DESTROYED UL(1)
+
+#define RSI_ACCEPT UL(0)
+#define RSI_REJECT UL(1)
+
+/*
+ * Get RIPAS of a target IPA range.
+ *
+ * arg1 == Base IPA of target region
+ * arg2 == End of target IPA region
+ * ret0 == Status / error
+ * ret1 == Top of IPA region which has the reported RIPAS value
+ * ret2 == RIPAS value
+ */
+#define SMC_RSI_IPA_STATE_GET SMC_RSI_FID(0x198)
+
+/*
+ * Make a Host call.
+ *
+ * arg1 == IPA of host call structure
+ * ret0 == Status / error
+ */
+#define SMC_RSI_HOST_CALL SMC_RSI_FID(0x199)
+
+#endif /* __LINUX_ARM_SMCCC_RSI_H_ */
diff --git a/include/linux/audit.h b/include/linux/audit.h
index 45abb3722d30..18b44a1c5397 100644
--- a/include/linux/audit.h
+++ b/include/linux/audit.h
@@ -547,7 +547,7 @@ static inline void audit_openat2_how(struct open_how *how)
static inline void audit_log_kern_module(const char *name)
{
- if (!audit_dummy_context())
+ if (unlikely(!audit_dummy_context()))
__audit_log_kern_module(name);
}
@@ -563,7 +563,7 @@ static inline void audit_tk_injoffset(struct timespec64 offset)
if (offset.tv_sec == 0 && offset.tv_nsec == 0)
return;
- if (!audit_dummy_context())
+ if (unlikely(!audit_dummy_context()))
__audit_tk_injoffset(offset);
}
@@ -586,7 +586,7 @@ static inline void audit_ntp_set_new(struct audit_ntp_data *ad,
static inline void audit_ntp_log(const struct audit_ntp_data *ad)
{
- if (!audit_dummy_context())
+ if (unlikely(!audit_dummy_context()))
__audit_ntp_log(ad);
}
diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h
index f686a37f7a0a..2e87faf9a8c2 100644
--- a/include/linux/binfmts.h
+++ b/include/linux/binfmts.h
@@ -128,7 +128,8 @@ struct linux_binfmt {
struct module *module;
int (*load_binary)(struct linux_binprm *);
#ifdef CONFIG_COREDUMP
- int (*core_dump)(struct coredump_params *cprm);
+ /* Returns true if the whole coredump was written. */
+ bool (*core_dump)(struct coredump_params *cprm);
unsigned long min_coredump; /* minimal dump size */
#endif
} __randomize_layout;
diff --git a/include/linux/bio-integrity.h b/include/linux/bio-integrity.h
index 0ea2a8bf7efb..a954c97be0b3 100644
--- a/include/linux/bio-integrity.h
+++ b/include/linux/bio-integrity.h
@@ -151,7 +151,6 @@ void bio_integrity_setup_default(struct bio *bio);
unsigned int fs_bio_integrity_alloc(struct bio *bio);
void fs_bio_integrity_free(struct bio *bio);
void fs_bio_integrity_generate(struct bio *bio);
-int fs_bio_integrity_verify(struct bio *bio, sector_t sector,
- unsigned int size);
+int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter);
#endif /* _LINUX_BIO_INTEGRITY_H */
diff --git a/include/linux/bio.h b/include/linux/bio.h
index bb3235497e67..75f8086351c0 100644
--- a/include/linux/bio.h
+++ b/include/linux/bio.h
@@ -256,6 +256,12 @@ static inline struct folio *bio_first_folio_all(struct bio *bio)
return page_folio(bio_first_page_all(bio));
}
+static inline struct bio_vec *bio_last_bvec_all(struct bio *bio)
+{
+ WARN_ON_ONCE(bio_flagged(bio, BIO_CLONED));
+ return &bio->bi_io_vec[bio->bi_vcnt - 1];
+}
+
/**
* struct folio_iter - State for iterating all folios in a bio.
* @folio: The current folio we're iterating. NULL after the last folio.
@@ -479,6 +485,7 @@ static inline void bio_init_inline(struct bio *bio, struct block_device *bdev,
extern void bio_uninit(struct bio *);
void bio_reset(struct bio *bio, struct block_device *bdev, blk_opf_t opf);
void bio_reuse(struct bio *bio, blk_opf_t opf);
+void bio_prepare_reissue(struct bio *bio, struct block_device *bdev);
void bio_chain(struct bio *, struct bio *);
void bio_await(struct bio *bio, void *priv,
void (*submit)(struct bio *bio, void *priv));
@@ -516,16 +523,18 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data,
size_t len, enum req_op op);
int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter,
- unsigned mem_align_mask, unsigned len_align_mask);
+ unsigned maxlen, unsigned mem_align_mask,
+ unsigned len_align_mask);
bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter);
void __bio_release_pages(struct bio *bio, bool mark_dirty);
extern void bio_set_pages_dirty(struct bio *bio);
extern void bio_check_pages_dirty(struct bio *bio);
-int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen,
- size_t minsize);
-void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty);
+int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize);
+void bio_free_folios(struct bio *bio);
+int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter,
+ size_t maxlen, size_t minsize);
extern void bio_copy_data(struct bio *dst, struct bio *src);
extern void bio_free_pages(struct bio *bio);
diff --git a/include/linux/bitmap.h b/include/linux/bitmap.h
index 7df1573a409c..adafbcf2016b 100644
--- a/include/linux/bitmap.h
+++ b/include/linux/bitmap.h
@@ -52,6 +52,7 @@ struct device;
* bitmap_complement(dst, src, nbits) *dst = ~(*src)
* bitmap_equal(src1, src2, nbits) Are *src1 and *src2 equal?
* bitmap_intersects(src1, src2, nbits) Do *src1 and *src2 overlap?
+ * bitmap_intersects_and(src1, src2, src3, nbits) Do *src1, *src2 and *src3 overlap?
* bitmap_subset(src1, src2, nbits) Is *src1 a subset of *src2?
* bitmap_empty(src, nbits) Are all bits zero in *src?
* bitmap_full(src, nbits) Are all bits set in *src?
@@ -181,6 +182,9 @@ void __bitmap_replace(unsigned long *dst,
const unsigned long *mask, unsigned int nbits);
bool __bitmap_intersects(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int nbits);
+bool __bitmap_intersects_and(const unsigned long *bitmap1,
+ const unsigned long *bitmap2,
+ const unsigned long *bitmap3, unsigned int nbits);
bool __bitmap_subset(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int nbits);
unsigned int __bitmap_weight(const unsigned long *bitmap, unsigned int nbits);
@@ -446,6 +450,16 @@ bool bitmap_intersects(const unsigned long *src1, const unsigned long *src2, uns
}
static __always_inline
+bool bitmap_intersects_and(const unsigned long *src1, const unsigned long *src2,
+ const unsigned long *src3, unsigned int nbits)
+{
+ if (small_const_nbits(nbits))
+ return ((*src1 & *src2 & *src3) & BITMAP_LAST_WORD_MASK(nbits)) != 0;
+ else
+ return __bitmap_intersects_and(src1, src2, src3, nbits);
+}
+
+static __always_inline
bool bitmap_subset(const unsigned long *src1, const unsigned long *src2, unsigned int nbits)
{
if (small_const_nbits(nbits))
diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h
index 4f7905c3412b..d003a9d2d1f6 100644
--- a/include/linux/blkdev.h
+++ b/include/linux/blkdev.h
@@ -191,14 +191,17 @@ struct gendisk {
#ifdef CONFIG_BLK_DEV_ZONED
/*
* Zoned block device information. Reads of this information must be
- * protected with blk_queue_enter() / blk_queue_exit(). Modifying this
- * information is only allowed while no requests are being processed.
- * See also blk_mq_freeze_queue() and blk_mq_unfreeze_queue().
+ * protected with blk_queue_enter() / blk_queue_exit() or by holding a
+ * lock on zone_revalidate_mutex. blk_revalidate_disk_zones() may modify
+ * this information while no requests are being processed (disk queue
+ * frozen with blk_mq_freeze_queue()) and while holding a lock on
+ * zone_revalidate_mutex.
*/
+ struct mutex zone_revalidate_mutex;
unsigned int nr_zones;
unsigned int zone_capacity;
unsigned int last_zone_capacity;
- u8 __rcu *zones_cond;
+ u8 __rcu *zones_state;
unsigned int zone_wplugs_hash_bits;
atomic_t nr_zone_wplugs;
spinlock_t zone_wplugs_hash_lock;
@@ -1816,9 +1819,11 @@ static inline int bio_split_rw_at(struct bio *bio,
*/
static inline unsigned int max_integrity_io_size(struct queue_limits *lim)
{
- return min_t(unsigned int, lim->max_segment_size,
- (BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size) <<
- lim->integrity.interval_exp);
+ u64 max_intervals;
+
+ max_intervals = BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size;
+ return min_t(u64, lim->max_segment_size,
+ max_intervals << lim->integrity.interval_exp);
}
#define DEFINE_IO_COMP_BATCH(name) struct io_comp_batch name = { }
diff --git a/include/linux/bnge/hsi.h b/include/linux/bnge/hsi.h
index 1f7bd96415a5..d4fe662d60b1 100644
--- a/include/linux/bnge/hsi.h
+++ b/include/linux/bnge/hsi.h
@@ -8607,6 +8607,7 @@ struct hwrm_cfa_l2_filter_alloc_input {
#define CFA_L2_FILTER_ALLOC_REQ_ENABLES_MIRROR_VNIC_ID 0x10000UL
#define CFA_L2_FILTER_ALLOC_REQ_ENABLES_NUM_VLANS 0x20000UL
#define CFA_L2_FILTER_ALLOC_REQ_ENABLES_T_NUM_VLANS 0x40000UL
+ #define CFA_L2_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX 0x80000UL
u8 l2_addr[6];
u8 num_vlans;
u8 t_num_vlans;
@@ -8663,7 +8664,8 @@ struct hwrm_cfa_l2_filter_alloc_input {
#define CFA_L2_FILTER_ALLOC_REQ_PRI_HINT_MIN 0x4UL
#define CFA_L2_FILTER_ALLOC_REQ_PRI_HINT_LAST CFA_L2_FILTER_ALLOC_REQ_PRI_HINT_MIN
u8 unused_5;
- __le32 unused_6;
+ __le16 rfs_ring_tbl_idx;
+ __le16 unused_6;
__le64 l2_filter_id_hint;
};
@@ -8900,6 +8902,95 @@ struct hwrm_cfa_tunnel_filter_free_output {
u8 valid;
};
+/* hwrm_cfa_ntuple_filter_alloc_input (size:1024b/128B) */
+struct hwrm_cfa_ntuple_filter_alloc_input {
+ __le16 req_type;
+ __le16 cmpl_ring;
+ __le16 seq_id;
+ __le16 target_id;
+ __le64 resp_addr;
+ __le32 flags;
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_LOOPBACK 0x1UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DROP 0x2UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_METER 0x4UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DEST_FID 0x8UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_ARP_REPLY 0x10UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DEST_RFS_RING_IDX 0x20UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_NO_L2_CONTEXT 0x40UL
+ __le32 enables;
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_L2_FILTER_ID 0x1UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_ETHERTYPE 0x2UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_TUNNEL_TYPE 0x4UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_MACADDR 0x8UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_IPADDR_TYPE 0x10UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_IPADDR 0x20UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_IPADDR_MASK 0x40UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_IPADDR 0x80UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_IPADDR_MASK 0x100UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_IP_PROTOCOL 0x200UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_PORT 0x400UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_PORT_MASK 0x800UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_PORT 0x1000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_PORT_MASK 0x2000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_PRI_HINT 0x4000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_NTUPLE_FILTER_ID 0x8000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_ID 0x10000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_MIRROR_VNIC_ID 0x20000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_MACADDR 0x40000UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX 0x80000UL
+ __le64 l2_filter_id;
+ u8 src_macaddr[6];
+ __be16 ethertype;
+ u8 ip_addr_type;
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_UNKNOWN 0x0UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV4 0x4UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV6 0x6UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV6
+ u8 ip_protocol;
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_UNKNOWN 0x0UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_TCP 0x6UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_UDP 0x11UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_ICMP 0x1UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_ICMPV6 0x3aUL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_RSVD 0xffUL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_RSVD
+ __le16 dst_id;
+ __le16 rfs_ring_tbl_idx;
+ u8 tunnel_type;
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_NONTUNNEL 0x0UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN 0x1UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_NVGRE 0x2UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_L2GRE 0x3UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_IPIP 0x4UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_GENEVE 0x5UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_MPLS 0x6UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_STT 0x7UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_IPGRE 0x8UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN_V4 0x9UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_IPGRE_V1 0xaUL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_L2_ETYPE 0xbUL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN_GPE_V6 0xcUL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN_GPE 0x10UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_ANYTUNNEL 0xffUL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_ANYTUNNEL
+ u8 pri_hint;
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_NO_PREFER 0x0UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_ABOVE 0x1UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_BELOW 0x2UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_HIGHEST 0x3UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_LOWEST 0x4UL
+ #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_LOWEST
+ __be32 src_ipaddr[4];
+ __be32 src_ipaddr_mask[4];
+ __be32 dst_ipaddr[4];
+ __be32 dst_ipaddr_mask[4];
+ __be16 src_port;
+ __be16 src_port_mask;
+ __be16 dst_port;
+ __be16 dst_port_mask;
+ __le64 ntuple_filter_id_hint;
+};
+
/* hwrm_cfa_ntuple_filter_alloc_output (size:192b/24B) */
struct hwrm_cfa_ntuple_filter_alloc_output {
__le16 error_code;
@@ -9014,6 +9105,49 @@ struct hwrm_cfa_ntuple_filter_cfg_output {
u8 valid;
};
+/* hwrm_cfa_adv_flow_mgnt_qcaps_input (size:256b/32B) */
+struct hwrm_cfa_adv_flow_mgnt_qcaps_input {
+ __le16 req_type;
+ __le16 cmpl_ring;
+ __le16 seq_id;
+ __le16 target_id;
+ __le64 resp_addr;
+ __le32 unused_0[4];
+};
+
+/* hwrm_cfa_adv_flow_mgnt_qcaps_output (size:128b/16B) */
+struct hwrm_cfa_adv_flow_mgnt_qcaps_output {
+ __le16 error_code;
+ __le16 req_type;
+ __le16 seq_id;
+ __le16 resp_len;
+ __le32 flags;
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_HND_16BIT_SUPPORTED 0x1UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_HND_64BIT_SUPPORTED 0x2UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_BATCH_DELETE_SUPPORTED 0x4UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_RESET_ALL_SUPPORTED 0x8UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_DEST_FUNC_SUPPORTED 0x10UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_TX_EEM_FLOW_SUPPORTED 0x20UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RX_EEM_FLOW_SUPPORTED 0x40UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_COUNTER_ALLOC_SUPPORTED 0x80UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RFS_RING_TBL_IDX_SUPPORTED 0x100UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_UNTAGGED_VLAN_SUPPORTED 0x200UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_XDP_SUPPORTED 0x400UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_L2_HEADER_SOURCE_FIELDS_SUPPORTED 0x800UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_RX_ARP_SUPPORTED 0x1000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RFS_RING_TBL_IDX_V2_SUPPORTED 0x2000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_RX_ETHERTYPE_IP_SUPPORTED 0x4000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_TRUFLOW_CAPABLE 0x8000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_L2_FILTER_TRAFFIC_TYPE_L2_ROCE_SUPPORTED 0x10000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_LAG_SUPPORTED 0x20000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_NO_L2CTX_SUPPORTED 0x40000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NIC_FLOW_STATS_SUPPORTED 0x80000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_RX_EXT_IP_PROTO_SUPPORTED 0x100000UL
+ #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RFS_RING_TBL_IDX_V3_SUPPORTED 0x200000UL
+ u8 unused_0[3];
+ u8 valid;
+};
+
/* hwrm_tunnel_dst_port_alloc_input (size:192b/24B) */
struct hwrm_tunnel_dst_port_alloc_input {
__le16 req_type;
diff --git a/include/linux/bootconfig.h b/include/linux/bootconfig.h
index deda507500da..fdc15b6f4d1b 100644
--- a/include/linux/bootconfig.h
+++ b/include/linux/bootconfig.h
@@ -291,12 +291,7 @@ int __init xbc_init(const char *buf, size_t size, const char **emsg, int *epos);
int __init xbc_get_info(int *node_size, size_t *data_size);
/* XBC cleanup data structures */
-void __init _xbc_exit(bool early);
-
-static __always_inline void xbc_exit(void)
-{
- _xbc_exit(false);
-}
+void __init xbc_exit(void);
/* XBC embedded bootconfig data in kernel */
#ifdef CONFIG_BOOT_CONFIG_EMBED
diff --git a/include/linux/bpf-cgroup-defs.h b/include/linux/bpf-cgroup-defs.h
index c9e6b26abab6..0147b8bec973 100644
--- a/include/linux/bpf-cgroup-defs.h
+++ b/include/linux/bpf-cgroup-defs.h
@@ -47,6 +47,7 @@ enum cgroup_bpf_attach_type {
CGROUP_INET6_GETSOCKNAME,
CGROUP_UNIX_GETSOCKNAME,
CGROUP_INET_SOCK_RELEASE,
+ CGROUP_TCP_SOCK_OPS,
CGROUP_LSM_START,
CGROUP_LSM_END = CGROUP_LSM_START + CGROUP_LSM_NUM - 1,
MAX_CGROUP_BPF_ATTACH_TYPE
diff --git a/include/linux/bpf-cgroup.h b/include/linux/bpf-cgroup.h
index 4d0cc65976a1..8a75a6cd7309 100644
--- a/include/linux/bpf-cgroup.h
+++ b/include/linux/bpf-cgroup.h
@@ -100,6 +100,8 @@ struct bpf_cgroup_storage {
struct bpf_cgroup_link {
struct bpf_link link;
struct cgroup *cgroup;
+ struct bpf_map *map;
+ wait_queue_head_t wait_hup;
};
struct bpf_prog_list {
@@ -110,6 +112,18 @@ struct bpf_prog_list {
u32 flags;
};
+#define bpf_cgroup_struct_ops_foreach(var, item, cgrp, atype) \
+ for (item = rcu_dereference((cgrp)->bpf.effective[atype])->items;\
+ ((var) = READ_ONCE(item->kdata)); \
+ item++)
+
+static inline bool cgroup_bpf_is_struct_ops_atype(enum cgroup_bpf_attach_type atype)
+{
+ return atype == CGROUP_TCP_SOCK_OPS;
+}
+void cgroup_bpf_struct_ops_register(int atype, u32 type_id, void *cfi_stubs, bool mult_trace);
+int cgroup_bpf_struct_ops_attach(struct bpf_map *map, const union bpf_attr *attr);
+
void __init cgroup_bpf_lifetime_notifier_init(void);
int __cgroup_bpf_run_filter_skb(struct sock *sk,
@@ -479,6 +493,20 @@ static inline int bpf_percpu_cgroup_storage_update(struct bpf_map *map,
return 0;
}
+static inline bool cgroup_bpf_is_struct_ops_atype(int atype)
+{
+ return false;
+}
+static inline void cgroup_bpf_struct_ops_register(int atype, u32 type_id, void *cfi_stubs,
+ bool mult_trace)
+{
+}
+static inline int cgroup_bpf_struct_ops_attach(struct bpf_map *map,
+ const union bpf_attr *attr)
+{
+ return -EOPNOTSUPP;
+}
+
#define cgroup_bpf_enabled(atype) (0)
#define BPF_CGROUP_RUN_SA_PROG_LOCK(sk, uaddr, uaddrlen, atype, t_ctx) ({ 0; })
#define BPF_CGROUP_RUN_SA_PROG(sk, uaddr, uaddrlen, atype) ({ 0; })
diff --git a/include/linux/bpf.h b/include/linux/bpf.h
index ffa5626411ac..d23a188a6f40 100644
--- a/include/linux/bpf.h
+++ b/include/linux/bpf.h
@@ -17,6 +17,7 @@
#include <linux/numa.h>
#include <linux/mm_types.h>
#include <linux/wait.h>
+#include <linux/irq_work_types.h>
#include <linux/refcount.h>
#include <linux/mutex.h>
#include <linux/module.h>
@@ -90,6 +91,7 @@ struct bpf_map_ops {
struct bpf_map *(*map_alloc)(union bpf_attr *attr);
void (*map_release)(struct bpf_map *map, struct file *map_file);
void (*map_free)(struct bpf_map *map);
+ void (*map_free_pre_rcu)(struct bpf_map *map);
int (*map_get_next_key)(struct bpf_map *map, void *key, void *next_key);
void (*map_release_uref)(struct bpf_map *map);
void *(*map_lookup_elem_sys_only)(struct bpf_map *map, void *key);
@@ -214,6 +216,7 @@ enum btf_field_type {
BPF_UPTR = (1 << 11),
BPF_RES_SPIN_LOCK = (1 << 12),
BPF_TASK_WORK = (1 << 13),
+ BPF_RCU_HEAD = (1 << 14),
};
enum bpf_cgroup_storage_type {
@@ -268,6 +271,7 @@ struct btf_record {
int wq_off;
int refcount_off;
int task_work_off;
+ int rcu_head_off;
struct btf_field fields[];
};
@@ -341,8 +345,18 @@ struct bpf_map {
s64 __percpu *elem_count;
u64 cookie; /* write-once */
char *excl_prog_sha;
+ /*
+ * Which programs use the map, see bpf_map_claim(): 0 - none so far,
+ * aux of the program - only that one, the same with BPF_MAP_USER_PATCHED
+ * set - only that one and it stored the addresses of its functions into
+ * the map, BPF_MAP_USER_MANY - more than one.
+ */
+ unsigned long user;
};
+#define BPF_MAP_USER_MANY 1UL
+#define BPF_MAP_USER_PATCHED 1UL
+
static inline const char *btf_field_type_name(enum btf_field_type type)
{
switch (type) {
@@ -373,6 +387,8 @@ static inline const char *btf_field_type_name(enum btf_field_type type)
return "bpf_refcount";
case BPF_TASK_WORK:
return "bpf_task_work";
+ case BPF_RCU_HEAD:
+ return "bpf_rcu_head";
default:
WARN_ON_ONCE(1);
return "unknown";
@@ -413,6 +429,8 @@ static inline u32 btf_field_type_size(enum btf_field_type type)
return sizeof(struct bpf_refcount);
case BPF_TASK_WORK:
return sizeof(struct bpf_task_work);
+ case BPF_RCU_HEAD:
+ return sizeof(struct bpf_rcu_head);
default:
WARN_ON_ONCE(1);
return 0;
@@ -447,6 +465,8 @@ static inline u32 btf_field_type_align(enum btf_field_type type)
return __alignof__(struct bpf_refcount);
case BPF_TASK_WORK:
return __alignof__(struct bpf_task_work);
+ case BPF_RCU_HEAD:
+ return __alignof__(struct bpf_rcu_head);
default:
WARN_ON_ONCE(1);
return 0;
@@ -479,6 +499,7 @@ static inline void bpf_obj_init_field(const struct btf_field *field, void *addr)
case BPF_KPTR_PERCPU:
case BPF_UPTR:
case BPF_TASK_WORK:
+ case BPF_RCU_HEAD:
break;
default:
WARN_ON_ONCE(1);
@@ -572,7 +593,7 @@ static inline void bpf_obj_memcpy(struct btf_record *rec,
if (long_memcpy)
bpf_long_memcpy(dst, src, size);
else
- memcpy(dst, src, size);
+ data_race(memcpy(dst, src, size));
return;
}
@@ -580,10 +601,10 @@ static inline void bpf_obj_memcpy(struct btf_record *rec,
u32 next_off = rec->fields[i].offset;
u32 sz = next_off - curr_off;
- memcpy(dst + curr_off, src + curr_off, sz);
+ data_race(memcpy(dst + curr_off, src + curr_off, sz));
curr_off += rec->fields[i].size + sz;
}
- memcpy(dst + curr_off, src + curr_off, size - curr_off);
+ data_race(memcpy(dst + curr_off, src + curr_off, size - curr_off));
}
static inline void copy_map_value(struct bpf_map *map, void *dst, void *src)
@@ -780,7 +801,10 @@ enum bpf_type_flag {
*/
PTR_UNTRUSTED = BIT(6 + BPF_BASE_TYPE_BITS),
- /* MEM can be uninitialized. */
+ /*
+ * MEM can be uninitialized. Generic memory outputs need not be fully
+ * initialized by the callee.
+ */
MEM_UNINIT = BIT(7 + BPF_BASE_TYPE_BITS),
/* DYNPTR points to memory local to the bpf program. */
@@ -874,7 +898,7 @@ enum bpf_type_flag {
/* function argument constraints */
enum bpf_arg_type {
- ARG_DONTCARE = 0, /* unused argument in helper function */
+ ARG_UNUSED = 0, /* unused argument; terminates argument iteration */
/* the following constraints used to prototype
* bpf_map_lookup/update/delete_elem() functions
@@ -894,6 +918,7 @@ enum bpf_arg_type {
ARG_PTR_TO_CTX, /* pointer to context */
ARG_ANYTHING, /* any (initialized) argument is ok */
+ ARG_SCALAR, /* scalar argument */
ARG_PTR_TO_SPIN_LOCK, /* pointer to bpf_spin_lock */
ARG_PTR_TO_SOCK_COMMON, /* pointer to sock_common */
ARG_PTR_TO_SOCKET, /* pointer to bpf_sock (fullsock) */
@@ -908,6 +933,24 @@ enum bpf_arg_type {
ARG_PTR_TO_TIMER, /* pointer to bpf_timer */
ARG_KPTR_XCHG_DEST, /* pointer to destination that kptrs are bpf_kptr_xchg'd into */
ARG_PTR_TO_DYNPTR, /* pointer to bpf_dynptr. See bpf_type_flag for dynptr type */
+
+ ARG_CONST_SCALAR, /* scalar known at verification time */
+ ARG_CONST_MEM_SIZE, /* ARG_MEM_SIZE that must be constant */
+ ARG_PTR_TO_ALLOC_BTF_ID, /* pointer to an allocated object */
+ ARG_PTR_TO_REFCOUNTED_KPTR, /* pointer to a refcounted local kptr */
+ ARG_PTR_TO_ITER, /* pointer to an iterator */
+ ARG_PTR_TO_LIST_HEAD, /* pointer to bpf_list_head */
+ ARG_PTR_TO_LIST_NODE, /* pointer to bpf_list_node */
+ ARG_PTR_TO_RB_ROOT, /* pointer to bpf_rb_root */
+ ARG_PTR_TO_RB_NODE, /* pointer to bpf_rb_node */
+ ARG_PTR_TO_WORKQUEUE, /* pointer to bpf_wq */
+ ARG_PTR_TO_TASK_WORK, /* pointer to bpf_task_work */
+ ARG_PTR_TO_RCU_HEAD, /* pointer to bpf_rcu_head */
+ ARG_PTR_TO_IRQ_FLAG, /* pointer to saved IRQ flags on the stack */
+ ARG_PTR_TO_RES_SPIN_LOCK, /* pointer to bpf_res_spin_lock */
+ ARG_PTR_TO_CTX_OUT, /* hook output argument passed through from ctx */
+ ARG_PTR_TO_PROG_AUX, /* pointer to the caller's bpf_prog_aux */
+ ARG_IGNORE, /* argument the verifier does not check at all */
__BPF_ARG_TYPE_MAX,
/* Extended arg_types. */
@@ -976,6 +1019,13 @@ static_assert(__BPF_RET_TYPE_MAX <= BPF_BASE_TYPE_LIMIT);
*/
#define MAX_BPF_FUNC_REG_ARGS 5
+/* A by-value argument takes two eightbytes at most, so the maximum number of
+ * argument slots of any function is 2 * MAX_BPF_FUNC_ARGS. A local array may
+ * need that size for processing, although eventually the maximum slots will
+ * be capped at MAX_BPF_FUNC_ARGS.
+ */
+#define MAX_BPF_FUNC_ARG_SLOTS (2 * MAX_BPF_FUNC_ARGS)
+
/* eBPF function prototype used by verifier to allow BPF_CALLs from eBPF programs
* to in-kernel helper functions and for adjusting imm32 field in BPF_CALL
* instructions after verifying
@@ -1004,13 +1054,13 @@ struct bpf_func_proto {
};
union {
struct {
- u32 *arg1_btf_id;
- u32 *arg2_btf_id;
- u32 *arg3_btf_id;
- u32 *arg4_btf_id;
- u32 *arg5_btf_id;
+ const u32 *arg1_btf_id;
+ const u32 *arg2_btf_id;
+ const u32 *arg3_btf_id;
+ const u32 *arg4_btf_id;
+ const u32 *arg5_btf_id;
};
- u32 *arg_btf_id[MAX_BPF_FUNC_ARGS];
+ const u32 *arg_btf_id[MAX_BPF_FUNC_ARGS];
struct {
size_t arg1_size;
size_t arg2_size;
@@ -1113,6 +1163,7 @@ struct bpf_insn_access_aux {
u32 ref_id;
};
};
+ u32 mem_size;
struct bpf_verifier_log *log; /* for verbose logs */
bool is_retval; /* is accessing function return value ? */
};
@@ -1193,6 +1244,9 @@ struct bpf_prog_offload {
u32 jited_len;
};
+/* The argument is aligned to 16 bytes. */
+#define BTF_FMODEL_ALIGN16_ARG BIT(0)
+
/* The argument is signed. */
#define BTF_FMODEL_SIGNED_ARG BIT(1)
@@ -1210,6 +1264,11 @@ struct btf_func_model {
u8 arg_flags[MAX_BPF_FUNC_ARGS];
};
+static inline u32 btf_func_model_arg_slots(const struct btf_func_model *m, u32 arg)
+{
+ return (m->arg_size[arg] + sizeof(u64) - 1) / sizeof(u64);
+}
+
/* Restore arguments before returning from trampoline to let original function
* continue executing. This flag is used for fentry progs when there are no
* fexit progs.
@@ -1257,11 +1316,15 @@ struct btf_func_model {
#define BPF_TRAMP_F_INDIRECT BIT(8)
/* Each call __bpf_prog_enter + call bpf_func + call __bpf_prog_exit is ~50
- * bytes on x86.
+ * bytes on x86. The trampoline image has to fit in PAGE_SIZE.
*/
enum {
-#if defined(__s390x__)
+#if defined(__s390x__) || defined(__powerpc64__)
BPF_MAX_TRAMP_LINKS = 27,
+#elif defined(__x86_64__)
+ BPF_MAX_TRAMP_LINKS = 36,
+#elif defined(__aarch64__)
+ BPF_MAX_TRAMP_LINKS = 37,
#else
BPF_MAX_TRAMP_LINKS = 38,
#endif
@@ -1314,6 +1377,7 @@ int arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *image, void *i
void *arch_alloc_bpf_trampoline(unsigned int size);
void arch_free_bpf_trampoline(void *image, unsigned int size);
int __must_check arch_protect_bpf_trampoline(void *image, unsigned int size);
+int arch_bpf_trampoline_skip(void *nop, void *target);
int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *func_addr);
@@ -1362,19 +1426,48 @@ enum bpf_tramp_prog_type {
BPF_TRAMP_FSESSION,
};
+/*
+ * Each prog call in a trampoline image is preceded by a nop. When the prog is
+ * detached, the nop is patched to a jump to target, right after the call, so
+ * that tasks still running in the image skip the prog.
+ */
+struct bpf_tramp_skip {
+ struct bpf_prog *prog;
+ void *nop;
+ void *target;
+};
+
struct bpf_tramp_image {
void *image;
int size;
struct bpf_ksym ksym;
struct percpu_ref pcref;
- void *ip_after_call;
- void *ip_epilogue;
+ bool call_orig;
+ /* entry in tr->images, the image holds a reference on tr */
+ struct bpf_trampoline *tr;
+ struct list_head list;
+ int nr_skips;
+ struct bpf_tramp_skip *skips;
union {
struct rcu_head rcu;
struct work_struct work;
};
};
+static inline void bpf_tramp_image_add_skip(struct bpf_tramp_image *im, struct bpf_prog *prog,
+ void *nop, void *target)
+{
+ struct bpf_tramp_skip *skip;
+
+ /* struct_ops trampolines and arch_bpf_trampoline_size() have no image */
+ if (!im || !im->skips)
+ return;
+ skip = &im->skips[im->nr_skips++];
+ skip->prog = prog;
+ skip->nop = nop;
+ skip->target = target;
+}
+
struct bpf_trampoline {
/* hlist for trampoline_key_table */
struct hlist_node hlist_key;
@@ -1401,6 +1494,8 @@ struct bpf_trampoline {
int progs_cnt[BPF_TRAMP_MAX];
/* Executable image of trampoline */
struct bpf_tramp_image *cur_image;
+ /* Images not freed yet, cur_image and older ones still in use */
+ struct list_head images;
/* Used as temporary old image storage for multi_attach */
struct {
struct bpf_tramp_image *old_image;
@@ -1650,8 +1745,9 @@ static inline void bpf_trampoline_set_flags(struct bpf_trampoline *tr, u32 flags
struct bpf_func_info_aux {
u16 linkage;
bool unreliable;
- bool called : 1;
- bool verified : 1;
+ /* Indexed by in_sleepable. */
+ bool called[2];
+ bool verified[2];
};
enum bpf_jit_poke_reason {
@@ -1683,6 +1779,7 @@ struct bpf_ctx_arg_aux {
struct btf *btf;
u32 btf_id;
u32 ref_id;
+ u32 mem_size;
bool refcounted;
};
@@ -1711,12 +1808,18 @@ enum {
};
struct bpf_stream {
- atomic_t capacity;
+ refcount_t refcnt;
+ atomic_t capacity; /* bytes reserved against the stream limit */
+ atomic_t readable; /* published bytes available to readers */
struct llist_head log; /* list of in-flight stream elements in LIFO order */
struct mutex lock; /* lock protecting backlog_{head,tail} */
struct llist_node *backlog_head; /* list of in-flight stream elements in FIFO order */
struct llist_node *backlog_tail; /* tail of the list above */
+ wait_queue_head_t waitq;
+ struct irq_work notify_work;
+ bool notify_used; /* notify_work was queued at least once */
+ bool dead;
};
struct bpf_stream_stage {
@@ -1735,6 +1838,7 @@ enum bpf_sig_keyring {
BPF_SIG_KEYRING_SECONDARY,
BPF_SIG_KEYRING_PLATFORM,
BPF_SIG_KEYRING_USER,
+ BPF_SIG_KEYRING_BPF,
};
struct bpf_prog_aux {
@@ -1768,12 +1872,12 @@ struct bpf_prog_aux {
bool offload_requested; /* Program is bound and offloaded to the netdev. */
bool attach_btf_trace; /* true if attaching to BTF-enabled raw tp */
bool attach_tracing_prog; /* true if tracing another tracing program */
+ bool tramp_linked; /* true if it was ever linked to a trampoline */
bool func_proto_unreliable;
bool tail_call_reachable;
bool xdp_has_frags;
bool exception_cb;
bool exception_boundary;
- bool is_extended; /* true if extended by freplace program */
bool jits_use_priv_stack;
bool priv_stack_requested;
bool changes_pkt_data;
@@ -1785,7 +1889,7 @@ struct bpf_prog_aux {
u8 verdict;
} sig;
u64 prog_array_member_cnt; /* counts how many times as member of prog_array */
- struct mutex ext_mutex; /* mutex for is_extended and prog_array_member_cnt */
+ struct mutex ext_mutex; /* mutex for freplace_link_cnt and prog_array_member_cnt */
struct bpf_arena *arena;
void (*recursion_detected)(struct bpf_prog *prog); /* callback if recursion is detected */
/* BTF_KIND_FUNC_PROTO for valid attach_btf_id */
@@ -1817,6 +1921,7 @@ struct bpf_prog_aux {
char name[BPF_OBJ_NAME_LEN];
u64 (*bpf_exception_cb)(u64 cookie, u64 sp, u64 bp, u64, u64);
u16 stack_arg_sp_adjust;
+ u16 freplace_link_cnt; /* counts freplace links extending this prog */
#ifdef CONFIG_SECURITY
void *security;
#endif
@@ -1854,7 +1959,7 @@ struct bpf_prog_aux {
struct work_struct work;
struct rcu_head rcu;
};
- struct bpf_stream stream[2];
+ struct bpf_stream *stream[2];
struct mutex st_ops_assoc_mutex;
struct bpf_map __rcu *st_ops_assoc;
};
@@ -2098,6 +2203,18 @@ struct btf_member;
* unloaded while in use.
* @name: The name of the struct bpf_struct_ops object.
* @func_models: Func models
+ * @cgroup_atype: A value in enum cgroup_bpf_attach_type for cgroup attachment.
+ * 0 means the struct_ops type does not support cgroup attachment.
+ * If cgroup_atype is non-zero, the @reg and @unreg must be NULL
+ * because the attachment/detachment will be handled by the bpf core.
+ * @free_after_tasks_rcu_gp: Set to true if it needs the bpf core to wait for
+ * a tasks_rcu gp before freeing the struct_ops map
+ * and its progs. It is unnecessary if the @unreg
+ * has waited for the correct rcu gp or the @unreg
+ * has ensured all struct_ops prog has finished running.
+ * @free_after_mult_rcu_gp: Same as @free_after_tasks_rcu_gp but waiting for
+ * both tasks_trace_rcu and regular rcu grace period.
+ * It is usually needed if the struct_ops has sleepable prog.
*/
struct bpf_struct_ops {
const struct bpf_verifier_ops *verifier_ops;
@@ -2116,6 +2233,9 @@ struct bpf_struct_ops {
struct module *owner;
const char *name;
struct btf_func_model func_models[BPF_STRUCT_OPS_MAX_NR_MEMBERS];
+ int cgroup_atype;
+ bool free_after_tasks_rcu_gp;
+ bool free_after_mult_rcu_gp;
};
/* Every member of a struct_ops type has an instance even a member is not
@@ -2250,6 +2370,12 @@ u32 bpf_struct_ops_id(const void *kdata);
int bpf_struct_ops_for_each_prog(const void *kdata,
int (*cb)(struct bpf_prog *prog, void *data),
void *data);
+void *bpf_struct_ops_map_kdata(struct bpf_map *map);
+void *bpf_struct_ops_map_cfi_stubs(struct bpf_map *map);
+bool bpf_struct_ops_valid_to_reg(struct bpf_map *map);
+int bpf_struct_ops_link_update_check(struct bpf_map *new_map, struct bpf_map *old_map,
+ struct bpf_map *expected_old_map);
+int bpf_struct_ops_map_cgroup_atype(struct bpf_map *map);
#ifdef CONFIG_NET
/* Define it here to avoid the use of forward declaration */
@@ -2314,6 +2440,32 @@ static inline void bpf_map_struct_ops_info_fill(struct bpf_map_info *info, struc
static inline void bpf_struct_ops_desc_release(struct bpf_struct_ops_desc *st_ops_desc)
{
}
+static inline u32 bpf_struct_ops_id(void *kdata)
+{
+ return 0;
+}
+static inline void *bpf_struct_ops_map_kdata(struct bpf_map *map)
+{
+ return NULL;
+}
+static inline int bpf_struct_ops_map_cgroup_atype(struct bpf_map *map)
+{
+ return 0;
+}
+static inline void *bpf_struct_ops_map_cfi_stubs(struct bpf_map *map)
+{
+ return NULL;
+}
+static inline bool bpf_struct_ops_valid_to_reg(struct bpf_map *map)
+{
+ return false;
+}
+static inline int bpf_struct_ops_link_update_check(struct bpf_map *new_map,
+ struct bpf_map *old_map,
+ struct bpf_map *expected_old_map)
+{
+ return -EOPNOTSUPP;
+}
#endif
@@ -2489,7 +2641,10 @@ u64 bpf_event_output(struct bpf_map *map, u64 flags, void *meta, u64 meta_size,
* since other cpus are walking the array of pointers in parallel.
*/
struct bpf_prog_array_item {
- struct bpf_prog *prog;
+ union {
+ struct bpf_prog *prog;
+ void *kdata;
+ };
union {
struct bpf_cgroup_storage *cgroup_storage[MAX_BPF_CGROUP_STORAGE_TYPE];
u64 bpf_cookie;
@@ -2531,6 +2686,7 @@ int bpf_prog_array_copy(struct bpf_prog_array *old_array,
struct bpf_prog *include_prog,
u64 bpf_cookie,
struct bpf_prog_array **new_array);
+struct bpf_prog *bpf_prog_dummy(void);
struct bpf_run_ctx {};
@@ -2549,6 +2705,7 @@ struct bpf_trace_run_ctx {
struct bpf_tramp_run_ctx {
struct bpf_run_ctx run_ctx;
u64 bpf_cookie;
+ int retval;
struct bpf_run_ctx *saved_run_ctx;
};
@@ -3819,6 +3976,8 @@ struct bpf_key {
#if defined(CONFIG_KEYS) && defined(CONFIG_BPF_SYSCALL)
struct bpf_key *bpf_lookup_user_key(s32 serial, u64 flags);
struct bpf_key *bpf_lookup_system_key(u64 id);
+struct bpf_key *bpf_lookup_keyring(void);
+bool bpf_keyring_enforced(void);
void bpf_key_put(struct bpf_key *bkey);
int bpf_verify_pkcs7_signature(const struct bpf_dynptr *data_p,
const struct bpf_dynptr *sig_p,
@@ -3839,6 +3998,16 @@ static inline struct bpf_key *bpf_lookup_system_key(u64 id)
return NULL;
}
+static inline struct bpf_key *bpf_lookup_keyring(void)
+{
+ return NULL;
+}
+
+static inline bool bpf_keyring_enforced(void)
+{
+ return false;
+}
+
static inline void bpf_key_put(struct bpf_key *bkey)
{
}
@@ -4099,9 +4268,10 @@ void bpf_bprintf_cleanup(struct bpf_bprintf_data *data);
int bpf_try_get_buffers(struct bpf_bprintf_buffers **bufs);
void bpf_put_buffers(void);
-void bpf_prog_stream_init(struct bpf_prog *prog);
+int bpf_prog_stream_init(struct bpf_prog *prog, gfp_t gfp_extra_flags);
void bpf_prog_stream_free(struct bpf_prog *prog);
-int bpf_prog_stream_read(struct bpf_prog *prog, enum bpf_stream_id stream_id, void __user *buf, int len);
+int bpf_prog_stream_read(struct bpf_prog *prog, enum bpf_stream_id stream_id, void __user *buf, u32 len);
+int bpf_prog_stream_new_fd(struct bpf_prog *prog, enum bpf_stream_id stream_id, u32 flags);
void bpf_stream_stage_init(struct bpf_stream_stage *ss);
void bpf_stream_stage_free(struct bpf_stream_stage *ss);
__printf(2, 3)
@@ -4164,7 +4334,7 @@ struct bpf_prog *bpf_prog_find_from_stack(void);
int bpf_insn_array_init(struct bpf_map *map, const struct bpf_prog *prog);
int bpf_insn_array_ready(struct bpf_map *map);
void bpf_insn_array_release(struct bpf_map *map);
-void bpf_insn_array_adjust(struct bpf_map *map, u32 off, u32 len);
+void bpf_insn_array_adjust(struct bpf_map *map, u32 first, u32 len);
void bpf_insn_array_adjust_after_remove(struct bpf_map *map, u32 off, u32 len);
#ifdef CONFIG_BPF_SYSCALL
@@ -4209,7 +4379,7 @@ static inline int bpf_map_check_op_flags(struct bpf_map *map, u64 flags, u64 all
return -EINVAL;
cpu = flags >> 32;
- if ((flags & BPF_F_CPU) && cpu >= num_possible_cpus())
+ if ((flags & BPF_F_CPU) && (cpu >= nr_cpu_ids || !cpu_possible(cpu)))
return -ERANGE;
}
diff --git a/include/linux/bpf_crypto.h b/include/linux/bpf_crypto.h
deleted file mode 100644
index a41e71d4e2d9..000000000000
--- a/include/linux/bpf_crypto.h
+++ /dev/null
@@ -1,24 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0-only */
-/* Copyright (c) 2024 Meta Platforms, Inc. and affiliates. */
-#ifndef _BPF_CRYPTO_H
-#define _BPF_CRYPTO_H
-
-struct bpf_crypto_type {
- void *(*alloc_tfm)(const char *algo);
- void (*free_tfm)(void *tfm);
- int (*has_algo)(const char *algo);
- int (*setkey)(void *tfm, const u8 *key, unsigned int keylen);
- int (*setauthsize)(void *tfm, unsigned int authsize);
- int (*encrypt)(void *tfm, const u8 *src, u8 *dst, unsigned int len, u8 *iv);
- int (*decrypt)(void *tfm, const u8 *src, u8 *dst, unsigned int len, u8 *iv);
- unsigned int (*ivsize)(void *tfm);
- unsigned int (*statesize)(void *tfm);
- u32 (*get_flags)(void *tfm);
- struct module *owner;
- char name[14];
-};
-
-int bpf_crypto_register_type(const struct bpf_crypto_type *type);
-int bpf_crypto_unregister_type(const struct bpf_crypto_type *type);
-
-#endif /* _BPF_CRYPTO_H */
diff --git a/include/linux/bpf_lsm.h b/include/linux/bpf_lsm.h
index dda272d78f01..eb4a38eef87e 100644
--- a/include/linux/bpf_lsm.h
+++ b/include/linux/bpf_lsm.h
@@ -21,6 +21,13 @@ extern bool bpf_lsm_initialized __ro_after_init;
#include <linux/lsm_hook_defs.h>
#undef LSM_HOOK
+/*
+ * Number of xattr slots the BPF LSM reserves in the array handed to
+ * security_inode_init_security(), i.e. the maximum number of labels
+ * a policy may attach to an inode while it is being created.
+ */
+#define BPF_LSM_INODE_INIT_XATTRS 2
+
struct bpf_storage_blob {
struct bpf_local_storage __rcu *storage;
};
@@ -28,7 +35,7 @@ struct bpf_storage_blob {
extern struct lsm_blob_sizes bpf_lsm_blob_sizes;
int bpf_lsm_verify_prog(struct bpf_verifier_log *vlog,
- const struct bpf_prog *prog);
+ struct bpf_prog *prog);
bool bpf_lsm_is_sleepable_hook(u32 btf_id);
bool bpf_lsm_is_trusted(const struct bpf_prog *prog);
@@ -71,7 +78,7 @@ static inline bool bpf_lsm_is_trusted(const struct bpf_prog *prog)
}
static inline int bpf_lsm_verify_prog(struct bpf_verifier_log *vlog,
- const struct bpf_prog *prog)
+ struct bpf_prog *prog)
{
return -EOPNOTSUPP;
}
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 5fad59fdab0d..811342e3c041 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -19,11 +19,12 @@
* that converting umax_value to int cannot overflow.
*/
#define BPF_MAX_VAR_SIZ (1 << 29)
-/* size of tmp_str_buf in bpf_verifier.
- * we need at least 306 bytes to fit full stack mask representation
- * (in the "-8,-16,...,-512" form)
+/*
+ * size of tmp_str_buf in bpf_verifier.
+ * we need at least 1399 bytes to fit full stack mask representation
+ * (in the "-8,-16,...,-2048" form)
*/
-#define TMP_STR_BUF_LEN 320
+#define TMP_STR_BUF_LEN 1408
/* Patch buffer size */
#define INSN_BUF_SIZE 32
@@ -45,18 +46,20 @@ struct bpf_reg_state {
union {
/* valid when type == PTR_TO_PACKET */
int range;
+ /*
+ * Valid when type == PTR_TO_STACK. Inside the callee two registers
+ * can be both PTR_TO_STACK like R1=fp-8 and R2=fp-8, but one of them
+ * points to this function stack while another to the caller's stack.
+ * To differentiate them 'frameno' is used which is an index in
+ * bpf_verifier_state->frame[] array pointing to bpf_func_state.
+ */
+ u8 frameno;
- /* valid when type == CONST_PTR_TO_MAP | PTR_TO_MAP_VALUE |
- * PTR_TO_MAP_VALUE_OR_NULL
+ /*
+ * For CONST_PTR_TO_MAP, PTR_TO_MAP_KEY, PTR_TO_MAP_VALUE and
+ * PTR_TO_INSN.
*/
- struct {
- struct bpf_map *map_ptr;
- /* To distinguish map lookups from outer map
- * the map_uid is non-zero for registers
- * pointing to inner maps.
- */
- u32 map_uid;
- };
+ struct bpf_map *map_ptr;
/* for PTR_TO_BTF_ID */
struct {
@@ -155,13 +158,12 @@ struct bpf_reg_state {
* gets parent_id set to the dynptr's id.
*/
u32 parent_id;
- /* Inside the callee two registers can be both PTR_TO_STACK like
- * R1=fp-8 and R2=fp-8, but one of them points to this function stack
- * while another to the caller's stack. To differentiate them 'frameno'
- * is used which is an index in bpf_verifier_state->frame[] array
- * pointing to bpf_func_state.
+ /*
+ * Distinguishes inner-map lookups and their keys and values. Zero for
+ * other registers. Kept outside the metadata union for ID remapping
+ * during state comparisons.
*/
- u32 frameno;
+ u32 map_uid;
/* if (!precise && SCALAR_VALUE) min/max/tnum don't affect safety */
bool precise;
};
@@ -242,55 +244,18 @@ enum bpf_stack_slot_type {
#define BPF_REG_SIZE 8 /* size of eBPF register in bytes */
+/*
+ * Largest number of BPF_REG_SIZE stack slots a single frame can have, sized
+ * for the largest stack budget any JIT supports. A frame may use any part of
+ * its program's budget; check_max_stack_depth() enforces the budget on the
+ * combined depth of frames sharing the kernel stack and on each frame using
+ * a private stack.
+ */
+#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK_JIT / BPF_REG_SIZE)
+
/* 4-byte stack slot granularity for liveness analysis */
#define BPF_HALF_REG_SIZE 4
#define STACK_SLOT_SZ 4
-#define STACK_SLOTS (MAX_BPF_STACK / BPF_HALF_REG_SIZE) /* 128 */
-
-typedef struct {
- u64 v[2];
-} spis_t;
-
-#define SPIS_ZERO ((spis_t){})
-#define SPIS_ALL ((spis_t){{ U64_MAX, U64_MAX }})
-
-static inline bool spis_is_zero(spis_t s)
-{
- return s.v[0] == 0 && s.v[1] == 0;
-}
-
-static inline bool spis_equal(spis_t a, spis_t b)
-{
- return a.v[0] == b.v[0] && a.v[1] == b.v[1];
-}
-
-static inline spis_t spis_or(spis_t a, spis_t b)
-{
- return (spis_t){{ a.v[0] | b.v[0], a.v[1] | b.v[1] }};
-}
-
-static inline spis_t spis_and(spis_t a, spis_t b)
-{
- return (spis_t){{ a.v[0] & b.v[0], a.v[1] & b.v[1] }};
-}
-
-static inline spis_t spis_not(spis_t s)
-{
- return (spis_t){{ ~s.v[0], ~s.v[1] }};
-}
-
-static inline bool spis_test_bit(spis_t s, u32 slot)
-{
- return s.v[slot / 64] & BIT_ULL(slot % 64);
-}
-
-static inline void spis_or_range(spis_t *mask, u32 lo, u32 hi)
-{
- u32 w;
-
- for (w = lo; w <= hi && w < STACK_SLOTS; w++)
- mask->v[w / 64] |= BIT_ULL(w % 64);
-}
#define BPF_REGMASK_ARGS ((1 << BPF_REG_1) | (1 << BPF_REG_2) | \
(1 << BPF_REG_3) | (1 << BPF_REG_4) | \
@@ -420,30 +385,36 @@ enum {
INSN_F_STACK_ARG_ACCESS = BIT(3),
};
+/* Registers linked to one jump condition that a history entry can record */
+#define BPF_LINKED_REGS_MAX 5
+
struct bpf_jmp_history_entry {
/* insn idx can't be bigger than 1 million */
u32 idx : 20;
u32 frame : 4; /* stack access frame number */
- u32 spi : 6; /* stack slot index (0..63) */
- u32 : 2;
- u32 prev_idx : 20;
/* special INSN_F_xxx flags */
u32 flags : 4;
- u32 : 8;
+ u32 : 4;
+ u32 prev_idx : 20;
+ u32 spi : 12; /* stack slot index */
/*
- * additional registers that need precision tracking when this
- * jump is backtracked, vector of five 11-bit records
+ * Scalar registers and spilled scalars linked to the condition of
+ * this jump, which need precision tracking together when the jump is
+ * backtracked. Each is packed as 4 bits of frame number, one bit
+ * telling a register from a stack slot and 11 bits of register or
+ * slot index, see linked_regs_pack().
*/
- u64 linked_regs;
+ u16 linked_regs[BPF_LINKED_REGS_MAX];
+ u8 linked_regs_cnt;
};
static_assert(MAX_CALL_FRAMES <= (1 << 4));
-static_assert(MAX_BPF_STACK / 8 <= (1 << 6));
+static_assert(MAX_BPF_STACK_SLOTS <= (1 << 12));
/* Maximum number of bpf_reg_state objects that can exist at once */
#define MAX_STACK_ARG_SLOTS (MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS)
-#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK / BPF_REG_SIZE + \
- MAX_STACK_ARG_SLOTS) * MAX_CALL_FRAMES)
+#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK_SLOTS + MAX_STACK_ARG_SLOTS) * \
+ MAX_CALL_FRAMES)
struct bpf_verifier_state {
/* call stack tracking */
struct bpf_func_state *frame[MAX_CALL_FRAMES];
@@ -529,12 +500,31 @@ struct bpf_verifier_state {
u32 may_goto_depth;
};
+/* Number of BPF_REG_SIZE stack slots tracked for the frame so far. */
+static inline u32 bpf_stack_nr_slots(const struct bpf_func_state *frame)
+{
+ return frame->allocated_stack / BPF_REG_SIZE;
+}
+
+/*
+ * Stack slot @spi of @frame, covering bytes [fp - (spi + 1) * 8, fp - spi * 8).
+ * The caller must ensure spi < bpf_stack_nr_slots(frame), see grow_stack_state().
+ */
+static inline struct bpf_stack_state *bpf_stack_slot(const struct bpf_func_state *frame, u32 spi)
+{
+ return &frame->stack[spi];
+}
+
static inline struct bpf_reg_state *
bpf_get_spilled_reg(int slot, struct bpf_func_state *frame, u32 mask)
{
- if (slot < frame->allocated_stack / BPF_REG_SIZE &&
- (1 << frame->stack[slot].slot_type[BPF_REG_SIZE - 1]) & mask)
- return &frame->stack[slot].spilled_ptr;
+ struct bpf_stack_state *ss;
+
+ if (slot >= bpf_stack_nr_slots(frame))
+ return NULL;
+ ss = bpf_stack_slot(frame, slot);
+ if ((1 << ss->slot_type[BPF_REG_SIZE - 1]) & mask)
+ return &ss->spilled_ptr;
return NULL;
}
@@ -550,7 +540,7 @@ bpf_get_spilled_stack_arg(int slot, struct bpf_func_state *frame)
/* Iterate over 'frame', setting 'reg' to either NULL or a spilled register. */
#define bpf_for_each_spilled_reg(iter, frame, reg, mask) \
for (iter = 0, reg = bpf_get_spilled_reg(iter, frame, mask); \
- iter < frame->allocated_stack / BPF_REG_SIZE; \
+ iter < bpf_stack_nr_slots(frame); \
iter++, reg = bpf_get_spilled_reg(iter, frame, mask))
/* Iterate over 'frame', setting 'reg' to either NULL or a spilled stack arg. */
@@ -575,7 +565,7 @@ bpf_get_spilled_stack_arg(int slot, struct bpf_func_state *frame)
bpf_for_each_spilled_reg(___j, __state, __reg, __mask) { \
if (!__reg) \
continue; \
- __stack = &__state->stack[___j]; \
+ __stack = bpf_stack_slot(__state, ___j); \
(void)(__expr); \
} \
__stack = NULL; \
@@ -670,7 +660,6 @@ struct bpf_insn_aux_data {
/* remember the offset of node field within type to rewrite */
u64 insert_off;
};
- struct bpf_iarray *jt; /* jump table for gotox or bpf_tailcall call instruction */
struct btf_struct_meta *kptr_struct_meta;
u64 map_key_state; /* constant (32 bit) key tracking for maps */
int ctx_field_size; /* the ctx field size for load insn, maybe 0 */
@@ -679,6 +668,7 @@ struct bpf_insn_aux_data {
bool nospec_result; /* result is unsafe under speculation, nospec must follow */
bool zext_dst; /* this insn zero extends dst reg */
bool needs_zext; /* alu op needs to clear upper bits */
+ bool prevent_zext; /* alu op cannot be zext (already used with 64-bit scalars) */
bool non_sleepable; /* helper/kfunc may be called from non-sleepable context */
bool is_iter_next; /* bpf_iter_<type>_next() kfunc call */
bool call_with_percpu_alloc_ptr; /* {this,per}_cpu_ptr() with prog percpu alloc */
@@ -706,6 +696,9 @@ struct bpf_insn_aux_data {
*/
u32 calls_callback:1;
u32 indirect_target:1; /* if it is an indirect jump target */
+ u32 non_stack_access:1; /* instruction can access non-stack memory */
+ /* true if some jump or call instruction targets this instruction */
+ u32 jump_target:1;
/*
* CFG strongly connected component this instruction belongs to,
* zero if it is a singleton SCC.
@@ -730,7 +723,11 @@ struct bpf_insn_aux_data {
#define MAX_USED_MAPS 64 /* max number of maps accessed by one eBPF program */
#define MAX_USED_BTFS 64 /* max number of BTFs accessed by one BPF program */
-#define BPF_VERIFIER_TMP_LOG_SIZE 1024
+/*
+ * Longest line the verifier log can carry: a full stack mask of
+ * MAX_BPF_STACK_SLOTS slots, see TMP_STR_BUF_LEN, plus its prefix.
+ */
+#define BPF_VERIFIER_TMP_LOG_SIZE 2048
struct bpf_verifier_log {
/* Logical start and end positions of a "log window" of the verifier log.
@@ -783,6 +780,21 @@ int bpf_log_attr_finalize(struct bpf_log_attr *attr, struct bpf_verifier_log *lo
#define BPF_MAX_SUBPROGS 256
+/*
+ * A pointer to a static subprog in the value of a frozen read-only array map:
+ * a 64-bit value that is the offset in bytes of the first instruction of
+ * the subprog in the program.
+ */
+struct bpf_func_ptr {
+ struct bpf_map *map;
+ u32 map_off; /* offset of the pointer in the value of the map */
+ u32 orig_off; /* what the map has: the first instruction of the subprog */
+ u32 xlated_off; /* the same after instructions were patched and removed */
+};
+
+/* the subprog that a bpf_func_ptr pointed to was removed as dead code */
+#define BPF_FUNC_PTR_DELETED ((u32)-1)
+
struct bpf_subprog_arg_info {
enum bpf_arg_type arg_type;
union {
@@ -802,7 +814,12 @@ struct bpf_subprog_info {
u32 start; /* insn idx of function entry point */
u32 linfo_idx; /* The idx to the main_prog->aux->linfo */
u32 postorder_start; /* The idx to the env->cfg.insn_postorder */
- u32 exit_idx; /* Index of one of the BPF_EXIT instructions in this subprogram */
+ /*
+ * Index of one of the BPF_EXIT instructions in this subprogram, or
+ * U32_MAX when it has none.
+ */
+ u32 exit_idx;
+ struct bpf_iarray *jt; /* jump table shared by all gotox of this subprogram */
u16 stack_depth; /* max. stack depth used by this function */
u16 stack_extra;
u32 insns_total;
@@ -819,11 +836,13 @@ struct bpf_subprog_info {
bool is_async_cb: 1;
bool is_exception_cb: 1;
bool args_cached: 1;
+ /* true if the return value is passed in the R0:R2 register pair */
+ bool ret_reg_pair: 1;
/* true if bpf_fastcall stack region is used by functions that can't be inlined */
bool keep_fastcall_stack: 1;
bool changes_pkt_data: 1;
bool might_sleep: 1;
- u8 arg_cnt:4;
+ u8 arg_slot_cnt:4;
enum priv_stack_mode priv_stack_mode;
struct bpf_subprog_arg_info args[MAX_BPF_FUNC_ARGS];
@@ -833,8 +852,8 @@ struct bpf_subprog_info {
static inline u16 bpf_in_stack_arg_cnt(const struct bpf_subprog_info *sub)
{
- if (sub->arg_cnt > MAX_BPF_FUNC_REG_ARGS)
- return sub->arg_cnt - MAX_BPF_FUNC_REG_ARGS;
+ if (sub->arg_slot_cnt > MAX_BPF_FUNC_REG_ARGS)
+ return sub->arg_slot_cnt - MAX_BPF_FUNC_REG_ARGS;
return 0;
}
@@ -845,7 +864,7 @@ struct backtrack_state {
struct bpf_verifier_env *env;
u32 frame;
u32 reg_masks[MAX_CALL_FRAMES];
- u64 stack_masks[MAX_CALL_FRAMES];
+ unsigned long stack_masks[MAX_CALL_FRAMES][BITS_TO_LONGS(MAX_BPF_STACK_SLOTS)];
u8 stack_arg_masks[MAX_CALL_FRAMES];
};
@@ -957,9 +976,24 @@ struct bpf_verifier_env {
const struct bpf_line_info *prev_linfo;
struct bpf_verifier_log log;
struct bpf_diag *diag;
+ struct bpf_func_proto bpf_subprog_scratch;
struct bpf_subprog_info subprog_info[BPF_MAX_SUBPROGS + 2]; /* max + 2 for the fake and exception subprogs */
/* subprog indices sorted in topological order: leaves first, callers last */
int subprog_topo_order[BPF_MAX_SUBPROGS + 2];
+ /*
+ * Pointers to static subprogs found in frozen read-only maps of the
+ * program, see resolve_func_ptrs(). Sorted by map and map_off.
+ */
+ struct bpf_func_ptr *func_ptrs;
+ u32 func_ptr_cnt;
+ bool has_callx;
+ /*
+ * Call graph edges created by callx instructions. A bitmap of
+ * subprog_cnt * subprog_cnt bits, where bit (caller * subprog_cnt + callee)
+ * is set when the main verification pass sees 'caller' calling 'callee'
+ * via callx. Allocated when the first such edge is recorded.
+ */
+ unsigned long *callx_edges;
union {
struct bpf_idmap idmap_scratch;
struct bpf_idset idset_scratch;
@@ -975,11 +1009,13 @@ struct bpf_verifier_env {
int cur_stack;
/* current position in the insn_postorder vector */
int cur_postorder;
+ u32 gotox_edges;
+ bool subprog_jts_ready;
} cfg;
struct backtrack_state bt;
struct bpf_jmp_history_entry *cur_hist_ent;
- /* Per-callsite copy of parent's converged at_stack_in for cross-frame fills. */
- struct arg_track **callsite_at_stack;
+ /* Per-callsite snapshot of the parent's tracked spill slots for cross-frame fills. */
+ struct spill_snapshot **callsite_at_stack;
u32 pass_cnt; /* number of times do_check() was called */
u32 subprog_cnt;
/* number of instructions analyzed by the verifier */
@@ -988,6 +1024,8 @@ struct bpf_verifier_env {
u32 prev_jmps_processed, jmps_processed;
/* maximum combined stack depth */
u32 max_stack_depth;
+ /* stack budget of the program, see bpf_prog_stack_limit() */
+ u32 stack_limit;
/* total verification time */
u64 verification_time;
/* maximum number of verifier states kept in 'branching' instructions */
@@ -1023,7 +1061,7 @@ struct bpf_verifier_env {
*/
u32 scratched_regs;
/* Same as scratched_regs but for stack slots */
- u64 scratched_stack_slots;
+ DECLARE_BITMAP(scratched_stack_slots, MAX_BPF_STACK_SLOTS);
u64 prev_log_pos, prev_insn_print_pos;
/* buffer used to temporary hold constants as scalar registers */
struct bpf_reg_state fake_reg[1];
@@ -1055,8 +1093,13 @@ static inline struct bpf_subprog_info *subprog_info(struct bpf_verifier_env *env
return &env->subprog_info[subprog];
}
+static inline bool bpf_ret_reg_pair(struct bpf_verifier_env *env, int subprog)
+{
+ return subprog_info(env, subprog)->ret_reg_pair;
+}
+
struct bpf_call_summary {
- u8 num_params;
+ u8 arg_slot_cnt;
bool is_void;
bool fastcall;
};
@@ -1079,6 +1122,12 @@ static inline bool bpf_pseudo_kfunc_call(const struct bpf_insn *insn)
insn->src_reg == BPF_PSEUDO_KFUNC_CALL;
}
+/* callx: indirect call of a bpf subprog whose address is in insn->dst_reg */
+static inline bool bpf_is_callx(const struct bpf_insn *insn)
+{
+ return insn->code == (BPF_JMP | BPF_CALL | BPF_X);
+}
+
__printf(2, 0) void bpf_verifier_vlog(struct bpf_verifier_log *log,
const char *fmt, va_list args);
__printf(2, 3) void bpf_verifier_log_write(struct bpf_verifier_env *env,
@@ -1142,6 +1191,16 @@ static inline void mark_jmp_point(struct bpf_verifier_env *env, int idx)
env->insn_aux_data[idx].jmp_point = true;
}
+static inline void mark_jump_target(struct bpf_verifier_env *env, int idx)
+{
+ env->insn_aux_data[idx].jump_target = true;
+}
+
+static inline bool bpf_is_jump_target(struct bpf_verifier_env *env, int insn_idx)
+{
+ return env->insn_aux_data[insn_idx].jump_target;
+}
+
static inline struct bpf_func_state *cur_func(struct bpf_verifier_env *env)
{
struct bpf_verifier_state *cur = env->cur_state;
@@ -1185,6 +1244,8 @@ static inline void bpf_trampoline_unpack_key(u64 key, u32 *obj_id, u32 *btf_id)
int bpf_prepare_btf_info(struct bpf_verifier_env *env,
const union bpf_attr *attr, bpfptr_t uattr);
+int bpf_check_core_relo(struct bpf_verifier_env *env,
+ const union bpf_attr *attr, bpfptr_t uattr);
int bpf_check_btf_info(struct bpf_verifier_env *env,
const union bpf_attr *attr, bpfptr_t uattr);
@@ -1207,7 +1268,8 @@ struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx);
void bpf_free_verifier_state(struct bpf_verifier_state *state, bool free_self);
void bpf_free_backedges(struct bpf_scc_visit *visit);
int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state *cur,
- int insn_flags, int spi, int frame, u64 linked_regs);
+ int insn_flags, int spi, int frame, const u16 *linked_regs,
+ u8 linked_regs_cnt);
void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist);
void bpf_mark_reg_not_init(const struct bpf_verifier_env *env,
struct bpf_reg_state *reg);
@@ -1224,11 +1286,36 @@ static inline int bpf_get_spi(s32 off)
return (-off - 1) / BPF_REG_SIZE;
}
+/*
+ * Stack a program may use in total: combined over the frames of a call
+ * chain on the kernel stack, or per frame on a private stack. Any single
+ * frame may reach that deep. Only a JIT that lays out such frames may go
+ * beyond MAX_BPF_STACK, the interpreter's frame size, and only one whose
+ * tail calls let the target set up its own frame: without subprogram
+ * tail calls, do_misc_fixups() gives every program with tail calls a
+ * MAX_BPF_STACK frame, which a deeper frame would overrun.
+ */
+static inline u32 bpf_prog_stack_limit(const struct bpf_prog *prog)
+{
+ /* an offloaded program never runs on the host JIT, whatever it supports */
+ if (prog->jit_requested && !bpf_prog_is_offloaded(prog->aux) &&
+ bpf_jit_supports_large_stack() && bpf_jit_supports_subprog_tailcalls())
+ return MAX_BPF_STACK_JIT;
+ return MAX_BPF_STACK;
+}
+
+/*
+ * Return the function state a stack pointer register refers to. frameno
+ * shares storage with other pointer metadata, so return NULL for any
+ * other register type instead of indexing frame[] with aliased bytes.
+ */
static inline struct bpf_func_state *bpf_func(struct bpf_verifier_env *env,
const struct bpf_reg_state *reg)
{
struct bpf_verifier_state *cur = env->cur_state;
+ if (reg->type != PTR_TO_STACK)
+ return NULL;
return cur->frame[reg->frameno];
}
@@ -1267,12 +1354,7 @@ static inline void bpf_bt_set_frame_reg(struct backtrack_state *bt, u32 frame, u
static inline void bpf_bt_set_frame_slot(struct backtrack_state *bt, u32 frame, u32 slot)
{
- bt->stack_masks[frame] |= 1ull << slot;
-}
-
-static inline void bpf_bt_set_frame_slot_mask(struct backtrack_state *bt, u32 frame, u64 mask)
-{
- bt->stack_masks[frame] |= mask;
+ __set_bit(slot, bt->stack_masks[frame]);
}
static inline void bt_set_frame_stack_arg_slot(struct backtrack_state *bt, u32 frame, u32 slot)
@@ -1287,10 +1369,17 @@ static inline bool bt_is_frame_reg_set(struct backtrack_state *bt, u32 frame, u3
static inline bool bt_is_frame_slot_set(struct backtrack_state *bt, u32 frame, u32 slot)
{
- return bt->stack_masks[frame] & (1ull << slot);
+ return test_bit(slot, bt->stack_masks[frame]);
}
bool bpf_map_is_rdonly(const struct bpf_map *map);
+struct bpf_func_ptr *bpf_map_func_ptrs(struct bpf_verifier_env *env,
+ const struct bpf_map *map, u32 *cnt);
+struct bpf_func_ptr *bpf_map_range_func_ptrs(struct bpf_verifier_env *env,
+ const struct bpf_map *map,
+ u64 off, u64 size, u32 *cnt);
+void bpf_adjust_func_ptrs(struct bpf_verifier_env *env, u32 off, u32 len);
+void bpf_adjust_func_ptrs_after_remove(struct bpf_verifier_env *env, u32 off, u32 len);
int bpf_map_direct_read(struct bpf_map *map, int off, int size, u64 *val,
bool is_ldsx);
@@ -1369,7 +1458,9 @@ static inline bool bpf_type_has_unsafe_modifiers(u32 type)
static inline bool type_is_ptr_alloc_obj(u32 type)
{
- return base_type(type) == PTR_TO_BTF_ID && type_flag(type) & MEM_ALLOC;
+ return base_type(type) == PTR_TO_BTF_ID &&
+ type_flag(type) & MEM_ALLOC &&
+ !(type_flag(type) & PTR_UNTRUSTED);
}
static inline bool type_is_non_owning_ref(u32 type)
@@ -1416,7 +1507,7 @@ static inline void mark_reg_scratched(struct bpf_verifier_env *env, u32 regno)
static inline void mark_stack_slot_scratched(struct bpf_verifier_env *env, u32 spi)
{
- env->scratched_stack_slots |= 1ULL << spi;
+ __set_bit(spi, env->scratched_stack_slots);
}
static inline bool reg_scratched(const struct bpf_verifier_env *env, u32 regno)
@@ -1424,27 +1515,28 @@ static inline bool reg_scratched(const struct bpf_verifier_env *env, u32 regno)
return (env->scratched_regs >> regno) & 1;
}
-static inline bool stack_slot_scratched(const struct bpf_verifier_env *env, u64 regno)
+static inline bool stack_slot_scratched(const struct bpf_verifier_env *env, u32 spi)
{
- return (env->scratched_stack_slots >> regno) & 1;
+ return test_bit(spi, env->scratched_stack_slots);
}
static inline bool verifier_state_scratched(const struct bpf_verifier_env *env)
{
- return env->scratched_regs || env->scratched_stack_slots;
+ return env->scratched_regs ||
+ !bitmap_empty(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS);
}
static inline void mark_verifier_state_clean(struct bpf_verifier_env *env)
{
env->scratched_regs = 0U;
- env->scratched_stack_slots = 0ULL;
+ bitmap_zero(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS);
}
/* Used for printing the entire verifier state. */
static inline void mark_verifier_state_scratched(struct bpf_verifier_env *env)
{
env->scratched_regs = ~0U;
- env->scratched_stack_slots = ~0ULL;
+ bitmap_fill(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS);
}
static inline bool bpf_stack_narrow_access_ok(int off, int fill_size, int spill_size)
@@ -1479,14 +1571,25 @@ struct bpf_subprog_info *bpf_find_containing_subprog(struct bpf_verifier_env *en
const char *bpf_subprog_name(const struct bpf_verifier_env *env, int subprog);
int bpf_jmp_offset(struct bpf_insn *insn);
struct bpf_iarray *bpf_insn_successors(struct bpf_verifier_env *env, u32 idx);
-void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, u64 stack_mask);
+void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, const unsigned long *stack_mask);
bool bpf_subprog_is_global(const struct bpf_verifier_env *env, int subprog);
+/* Kinds of member a by-value struct or union may be composed of. */
+enum btf_member_kind {
+ BTF_MEMBER_SCALAR = BIT(0), /* an int or an enum */
+ BTF_MEMBER_ARENA_PTR = BIT(1), /* a pointer carrying the "arena" type tag */
+};
+
+bool btf_struct_is_composed_of(struct bpf_verifier_env *env, const struct btf *btf,
+ const struct btf_type *t, u32 member_kinds);
+u32 btf_func_arg_align(const struct btf *btf, const struct btf_type *t);
+
int bpf_find_subprog(struct bpf_verifier_env *env, int off);
bool bpf_is_throw_kfunc(struct bpf_insn *insn);
int bpf_compute_const_regs(struct bpf_verifier_env *env);
int bpf_prune_dead_branches(struct bpf_verifier_env *env);
int bpf_check_cfg(struct bpf_verifier_env *env);
+void bpf_free_subprog_jts(struct bpf_verifier_env *env);
int bpf_compute_postorder(struct bpf_verifier_env *env);
int bpf_compute_scc(struct bpf_verifier_env *env);
@@ -1515,12 +1618,15 @@ struct ref_obj_desc {
};
/*
- * A memory argument a call fills in. The verifier allows the stack to be uninitialized if
- * the range is a known constant. Stack slots are marked as STACK_MISC by check_mem_access().
+ * Generic MEM_UNINIT arguments, indexed by ABI slot. var_size_mask excludes
+ * variable-sized buffers from raw mode without losing the output annotation.
+ * size records constant ranges to mark initialized after checking all arguments,
+ * only when the caller is allowed to read uninitialized stack memory.
*/
struct arg_raw_mem_desc {
- u8 regno;
- int size;
+ u16 mask;
+ u16 var_size_mask;
+ int size[MAX_BPF_FUNC_ARGS];
};
/* Size of PTR_TO_MEM returned, taken from a constant allocation-size argument */
@@ -1540,6 +1646,7 @@ struct bpf_call_arg_meta {
struct btf *btf;
u32 func_id;
const struct bpf_func_proto *fn;
+ const struct btf_type *func_proto;
u8 release_regno;
u32 ret_btf_id;
u32 subprogno;
@@ -1547,11 +1654,14 @@ struct bpf_call_arg_meta {
struct bpf_dynptr_desc dynptr;
struct ref_obj_desc ref_obj;
struct ret_mem_desc ret_mem;
+ struct arg_raw_mem_desc arg_raw_mem;
+
+ /* Only set by subprog */
+ bool subprog_may_change_pkt;
/* Only set by kfunc */
bool r0_rdonly;
u32 kfunc_flags;
- const struct btf_type *func_proto;
const char *func_name;
struct arg_constant_desc arg_constant;
@@ -1560,7 +1670,7 @@ struct bpf_call_arg_meta {
* verification logic
* bpf_obj_drop/bpf_percpu_obj_drop
* Record the local kptr type to be drop'd
- * bpf_refcount_acquire (via KF_ARG_PTR_TO_REFCOUNTED_KPTR arg type)
+ * bpf_refcount_acquire (via ARG_PTR_TO_REFCOUNTED_KPTR arg type)
* Record the local kptr type to be refcount_incr'd and use
* arg_owning_ref to determine whether refcount_acquire should be
* fallible
@@ -1568,7 +1678,6 @@ struct bpf_call_arg_meta {
struct btf *arg_btf;
u32 arg_btf_id;
bool arg_owning_ref;
- bool arg_prog;
struct {
struct btf_field *field;
@@ -1586,7 +1695,6 @@ struct bpf_call_arg_meta {
s64 const_map_key;
struct btf *ret_btf;
struct btf_field *kptr_field;
- struct arg_raw_mem_desc arg_raw_mem;
};
int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id,
@@ -1655,6 +1763,21 @@ static inline u64 bpf_map_key_immediate(const struct bpf_insn_aux_data *aux)
return aux->map_key_state & ~(BPF_MAP_KEY_SEEN | BPF_MAP_KEY_POISON);
}
+static inline bool bpf_is_mem_insn(struct bpf_insn *insn)
+{
+ if (BPF_CLASS(insn->code) != BPF_ST &&
+ BPF_CLASS(insn->code) != BPF_STX &&
+ BPF_CLASS(insn->code) != BPF_LDX)
+ return false;
+
+ if (insn->code == (BPF_ST | BPF_NOSPEC))
+ return false;
+
+ return (BPF_MODE(insn->code) == BPF_MEM ||
+ BPF_MODE(insn->code) == BPF_MEMSX ||
+ BPF_MODE(insn->code) == BPF_ATOMIC);
+}
+
#define MAX_PACKET_OFF 0xffff
#define CALLER_SAVED_REGS 6
@@ -1688,7 +1811,6 @@ struct bpf_kfunc_desc_tab {
};
/* Functions exported from verifier.c, used by fixups.c */
-void bpf_clear_insn_aux_data(struct bpf_verifier_env *env, int start, int len);
void bpf_mark_subprog_exc_cb(struct bpf_verifier_env *env, int subprog);
bool bpf_allow_tail_call_in_subprogs(struct bpf_verifier_env *env);
bool bpf_verifier_inlines_helper_call(struct bpf_verifier_env *env, s32 imm);
diff --git a/include/linux/btf.h b/include/linux/btf.h
index 89d5a5c4f117..4b63bb91550a 100644
--- a/include/linux/btf.h
+++ b/include/linux/btf.h
@@ -80,6 +80,7 @@
#define KF_ARENA_ARG2 (1 << 15) /* kfunc takes an arena pointer as its second argument */
#define KF_IMPLICIT_ARGS (1 << 16) /* kfunc has implicit arguments supplied by the verifier */
#define KF_SPINLOCK_SAFE (1 << 17) /* kfunc is allowed inside bpf_spin_lock-ed region */
+#define KF_PERFMON (1 << 18) /* kfunc requires CAP_PERFMON */
/*
* Tag marking a kernel function as a kfunc. This is meant to minimize the
@@ -235,6 +236,7 @@ struct btf_record *btf_parse_fields(const struct btf *btf, const struct btf_type
u32 field_mask, u32 value_size);
int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec);
bool btf_type_is_void(const struct btf_type *t);
+bool btf_type_is_arena_ptr(const struct btf *btf, const struct btf_type *t);
s32 btf_find_by_name_kind(const struct btf *btf, const char *name, u8 kind);
s32 bpf_find_btf_id(const char *name, u32 kind, struct btf **btf_p);
struct btf *btf_get_module_btf(const struct module *module);
@@ -260,6 +262,11 @@ const char *btf_type_str(const struct btf_type *t);
i < btf_type_vlen(datasec_type); \
i++, member++)
+#define for_each_loc(i, locsec_type, member) \
+ for (i = 0, member = btf_type_loc_secinfo(locsec_type); \
+ i < btf_type_vlen(locsec_type); \
+ i++, member++)
+
static inline bool btf_type_is_ptr(const struct btf_type *t)
{
return BTF_INFO_KIND(t->info) == BTF_KIND_PTR;
@@ -328,6 +335,21 @@ static inline u64 btf_enum64_value(const struct btf_enum64 *e)
return ((u64)e->val_hi32 << 32) | e->val_lo32;
}
+static inline struct btf_loc_param *btf_loc_param(const struct btf_type *t)
+{
+ return (struct btf_loc_param *)(t + 1);
+}
+
+static inline __u32 *btf_loc_proto_params(const struct btf_type *t)
+{
+ return (__u32 *)(t + 1);
+}
+
+static inline struct btf_loc *btf_type_loc_secinfo(const struct btf_type *t)
+{
+ return (struct btf_loc *)(t + 1);
+}
+
static inline bool btf_is_composite(const struct btf_type *t)
{
u16 kind = btf_kind(t);
@@ -558,7 +580,7 @@ struct btf_field_desc {
/* member struct size, or zero, if no members */
int m_sz;
/* repeated per-member offsets */
- int m_off_cnt, m_offs[1];
+ int m_off_cnt, m_offs[2];
};
struct btf_field_iter {
diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h
index fd2c7115c054..f8782b719026 100644
--- a/include/linux/buffer_head.h
+++ b/include/linux/buffer_head.h
@@ -59,10 +59,7 @@ struct address_space;
struct buffer_head {
unsigned long b_state; /* buffer state bitmap (see above) */
struct buffer_head *b_this_page;/* circular list of page's buffers */
- union {
- struct page *b_page; /* the page this bh is mapped to */
- struct folio *b_folio; /* the folio this bh is mapped to */
- };
+ struct folio *b_folio; /* the folio this bh is mapped to */
sector_t b_blocknr; /* start block number */
size_t b_size; /* size of mapping */
@@ -172,15 +169,39 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh)
static inline unsigned long bh_offset(const struct buffer_head *bh)
{
- return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1);
+ return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1);
}
-/* If we *know* page->private refers to buffer_heads */
-#define page_buffers(page) \
- ({ \
- BUG_ON(!PagePrivate(page)); \
- ((struct buffer_head *)page_private(page)); \
- })
+/**
+ * kmap_local_bh - Map the data of a buffer.
+ * @bh: The buffer.
+ *
+ * Buffers usually live in the page cache, but a few are built over memory
+ * which is not. Those carry no folio and b_data is already a kernel address
+ * which is always mapped, so there is nothing to do for them. Pair with
+ * kunmap_local_bh().
+ *
+ * Return: A pointer to the buffer's data.
+ */
+static inline void *kmap_local_bh(const struct buffer_head *bh)
+{
+ if (!bh->b_folio)
+ return bh->b_data;
+ return kmap_local_folio(bh->b_folio, bh_offset(bh));
+}
+
+/**
+ * kunmap_local_bh - Unmap the data of a buffer.
+ * @bh: The buffer.
+ * @addr: The address returned by kmap_local_bh().
+ */
+static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr)
+{
+ if (bh->b_folio)
+ kunmap_local(addr);
+}
+
+/* If we *know* folio->private refers to buffer_heads */
#define folio_buffers(folio) folio_get_private(folio)
void buffer_check_dirty_writeback(struct folio *folio,
@@ -197,7 +218,6 @@ void folio_set_bh(struct buffer_head *bh, struct folio *folio,
unsigned long offset);
struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size,
gfp_t gfp);
-struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size);
struct buffer_head *create_empty_buffers(struct folio *folio,
unsigned long blocksize, unsigned long b_state);
void end_buffer_read_sync(struct buffer_head *bh, int uptodate);
@@ -338,20 +358,58 @@ static inline void bforget(struct buffer_head *bh)
__bforget(bh);
}
-static inline struct buffer_head *
-sb_bread(struct super_block *sb, sector_t block)
+/**
+ * sb_bread - Read a block.
+ * @sb: The superblock to read from.
+ * @block: Block number in units of block size.
+ *
+ * Read a specified block, and return the buffer head that refers
+ * to it. The memory is allocated from the movable area so that it can
+ * be migrated. The returned buffer head has its refcount increased.
+ * The caller should call brelse() when it has finished with the buffer.
+ *
+ * Context: May sleep waiting for I/O.
+ * Return: NULL if the block was unreadable.
+ */
+static inline
+struct buffer_head *sb_bread(struct super_block *sb, sector_t block)
{
return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, __GFP_MOVABLE);
}
-static inline struct buffer_head *
-sb_bread_unmovable(struct super_block *sb, sector_t block)
+/**
+ * sb_bread_unmovable - Read a block.
+ * @sb: The superblock to read from.
+ * @block: Block number in units of block size.
+ *
+ * Read a specified block, and return the buffer head that refers to it.
+ * The memory is allocated from the unmovable area so that pointers into
+ * it remain valid after compaction runs. The returned buffer head has
+ * its refcount increased. The caller should call brelse() when it has
+ * finished with the buffer.
+ *
+ * Context: May sleep waiting for I/O.
+ * Return: NULL if the block was unreadable.
+ */
+static inline
+struct buffer_head *sb_bread_unmovable(struct super_block *sb, sector_t block)
{
return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, 0);
}
-static inline void
-sb_breadahead(struct super_block *sb, sector_t block)
+/**
+ * sb_breadahead - Start readahead.
+ * @sb: Superblock identifying the block device.
+ * @block: The block to read.
+ *
+ * Read this block. The I/O will be flagged as being readahead rather
+ * than immediate read, but (unlike the page cache), surrounding blocks
+ * will not be read.
+ *
+ * Context: May sleep in order to allocate memory.
+ */
+static inline
+void sb_breadahead(struct super_block *sb, sector_t block)
{
__breadahead(sb->s_bdev, block, sb->s_blocksize);
}
diff --git a/include/linux/cache.h b/include/linux/cache.h
index e69768f50d53..b6e857b985bc 100644
--- a/include/linux/cache.h
+++ b/include/linux/cache.h
@@ -89,6 +89,7 @@
*/
#ifndef INTERNODE_CACHE_SHIFT
#define INTERNODE_CACHE_SHIFT L1_CACHE_SHIFT
+#define INTERNODE_CACHE_BYTES (1 << INTERNODE_CACHE_SHIFT)
#endif
#if !defined(____cacheline_internodealigned_in_smp)
diff --git a/include/linux/cacheinfo.h b/include/linux/cacheinfo.h
index fc879ac4cc4f..56fa646df0d1 100644
--- a/include/linux/cacheinfo.h
+++ b/include/linux/cacheinfo.h
@@ -82,6 +82,7 @@ struct cpu_cacheinfo {
};
struct cpu_cacheinfo *get_cpu_cacheinfo(unsigned int cpu);
+int get_cpu_cacheinfo_id(int cpu, int level);
int early_cache_level(unsigned int cpu);
int init_cache_level(unsigned int cpu);
int init_of_cache_level(unsigned int cpu);
@@ -137,17 +138,6 @@ static inline struct cacheinfo *get_cpu_cacheinfo_level(int cpu, int level)
return NULL;
}
-/*
- * Get the id of the cache associated with @cpu at level @level.
- * cpuhp lock must be held.
- */
-static inline int get_cpu_cacheinfo_id(int cpu, int level)
-{
- struct cacheinfo *ci = get_cpu_cacheinfo_level(cpu, level);
-
- return ci ? ci->id : -1;
-}
-
#if defined(CONFIG_ARM64) || defined(CONFIG_ARM)
#define use_arch_cache_info() (true)
#else
diff --git a/include/linux/capability.h b/include/linux/capability.h
index 37db92b3d6f8..622137f66f09 100644
--- a/include/linux/capability.h
+++ b/include/linux/capability.h
@@ -145,6 +145,7 @@ extern bool has_capability_noaudit(struct task_struct *t, int cap);
extern bool has_ns_capability_noaudit(struct task_struct *t,
struct user_namespace *ns, int cap);
extern bool capable(int cap);
+bool capable_noaudit(int cap);
extern bool ns_capable(struct user_namespace *ns, int cap);
extern bool ns_capable_noaudit(struct user_namespace *ns, int cap);
extern bool ns_capable_setid(struct user_namespace *ns, int cap);
@@ -167,6 +168,10 @@ static inline bool capable(int cap)
{
return true;
}
+static inline bool capable_noaudit(int cap)
+{
+ return true;
+}
static inline bool ns_capable(struct user_namespace *ns, int cap)
{
return true;
@@ -181,9 +186,9 @@ static inline bool ns_capable_setid(struct user_namespace *ns, int cap)
}
#endif /* CONFIG_MULTIUSER */
bool privileged_wrt_inode_uidgid(struct user_namespace *ns,
- struct mnt_idmap *idmap,
+ const struct mnt_idmap *idmap,
const struct inode *inode);
-bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap,
+bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap,
const struct inode *inode, int cap);
extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap);
extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns);
@@ -210,11 +215,11 @@ static inline bool checkpoint_restore_ns_capable_noaudit(struct user_namespace *
}
/* audit system wants to get cap info from files as well */
-int get_vfs_caps_from_disk(struct mnt_idmap *idmap,
+int get_vfs_caps_from_disk(const struct mnt_idmap *idmap,
const struct dentry *dentry,
struct cpu_vfs_cap_data *cpu_caps);
-int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry,
+int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry,
const void **ivalue, size_t size);
#endif /* !_LINUX_CAPABILITY_H */
diff --git a/include/linux/ccp.h b/include/linux/ccp.h
index 868924dec5a1..e6c599243f11 100644
--- a/include/linux/ccp.h
+++ b/include/linux/ccp.h
@@ -26,7 +26,7 @@ struct ccp_cmd;
/**
* ccp_present - check if a CCP device is present
*
- * Returns zero if a CCP device is present, -ENODEV otherwise.
+ * Returns: zero if a CCP device is present, -ENODEV otherwise.
*/
int ccp_present(void);
@@ -38,7 +38,7 @@ int ccp_present(void);
/**
* ccp_version - get the version of the CCP
*
- * Returns a positive version number, or zero if no CCP
+ * Returns: a positive version number, or zero if no CCP
*/
unsigned int ccp_version(void);
@@ -61,9 +61,9 @@ unsigned int ccp_version(void);
* will be -EINPROGRESS. Any other "err" value during callback is
* the result of the operation.
*
- * The cmd has been successfully queued if:
- * the return code is -EINPROGRESS or
- * the return code is -EBUSY and CCP_CMD_MAY_BACKLOG flag is set
+ * Returns: The cmd has been successfully queued if:
+ * * the return code is -EINPROGRESS or
+ * * the return code is -EBUSY and CCP_CMD_MAY_BACKLOG flag is set
*/
int ccp_enqueue_cmd(struct ccp_cmd *cmd);
@@ -89,7 +89,7 @@ static inline int ccp_enqueue_cmd(struct ccp_cmd *cmd)
/***** AES engine *****/
/**
- * ccp_aes_type - AES key size
+ * enum ccp_aes_type - AES key size
*
* @CCP_AES_TYPE_128: 128-bit key
* @CCP_AES_TYPE_192: 192-bit key
@@ -99,11 +99,12 @@ enum ccp_aes_type {
CCP_AES_TYPE_128 = 0,
CCP_AES_TYPE_192,
CCP_AES_TYPE_256,
+ /* private: */
CCP_AES_TYPE__LAST,
};
/**
- * ccp_aes_mode - AES operation mode
+ * enum ccp_aes_mode - AES operation mode
*
* @CCP_AES_MODE_ECB: ECB mode
* @CCP_AES_MODE_CBC: CBC mode
@@ -111,6 +112,10 @@ enum ccp_aes_type {
* @CCP_AES_MODE_CFB: CFB mode
* @CCP_AES_MODE_CTR: CTR mode
* @CCP_AES_MODE_CMAC: CMAC mode
+ * @CCP_AES_MODE_GHASH: GHASH mode
+ * @CCP_AES_MODE_GCTR: GCTR mode
+ * @CCP_AES_MODE_GCM: GCM mode
+ * @CCP_AES_MODE_GMAC: GMAC mode
*/
enum ccp_aes_mode {
CCP_AES_MODE_ECB = 0,
@@ -123,11 +128,12 @@ enum ccp_aes_mode {
CCP_AES_MODE_GCTR,
CCP_AES_MODE_GCM,
CCP_AES_MODE_GMAC,
+ /* private: */
CCP_AES_MODE__LAST,
};
/**
- * ccp_aes_mode - AES operation mode
+ * enum ccp_aes_action - AES operation mode
*
* @CCP_AES_ACTION_DECRYPT: AES decrypt operation
* @CCP_AES_ACTION_ENCRYPT: AES encrypt operation
@@ -135,6 +141,7 @@ enum ccp_aes_mode {
enum ccp_aes_action {
CCP_AES_ACTION_DECRYPT = 0,
CCP_AES_ACTION_ENCRYPT,
+ /* private: */
CCP_AES_ACTION__LAST,
};
/* Overloaded field */
@@ -146,6 +153,7 @@ enum ccp_aes_action {
* @type: AES operation key size
* @mode: AES operation mode
* @action: AES operation (decrypt/encrypt)
+ * @authsize: AES block request size
* @key: key to be used for this AES operation
* @key_len: length in bytes of key
* @iv: IV to be used for this AES operation
@@ -156,6 +164,7 @@ enum ccp_aes_action {
* @cmac_final: indicates final operation when running in CMAC mode
* @cmac_key: K1/K2 key used in final CMAC operation
* @cmac_key_len: length in bytes of cmac_key
+ * @aad_len: length in bytes of Additional Authenticated Data
*
* Variables required to be set when calling ccp_enqueue_cmd():
* - type, mode, action, key, key_len, src, dst, src_len
@@ -192,7 +201,7 @@ struct ccp_aes_engine {
/***** XTS-AES engine *****/
/**
- * ccp_xts_aes_unit_size - XTS unit size
+ * enum ccp_xts_aes_unit_size - XTS unit size
*
* @CCP_XTS_AES_UNIT_SIZE_16: Unit size of 16 bytes
* @CCP_XTS_AES_UNIT_SIZE_512: Unit size of 512 bytes
@@ -206,11 +215,13 @@ enum ccp_xts_aes_unit_size {
CCP_XTS_AES_UNIT_SIZE_1024,
CCP_XTS_AES_UNIT_SIZE_2048,
CCP_XTS_AES_UNIT_SIZE_4096,
+ /* private: */
CCP_XTS_AES_UNIT_SIZE__LAST,
};
/**
* struct ccp_xts_aes_engine - CCP XTS AES operation
+ * @type: ccp_aes_type - AES key size
* @action: AES operation (decrypt/encrypt)
* @unit_size: unit size of the XTS operation
* @key: key to be used for this XTS AES operation
@@ -247,11 +258,13 @@ struct ccp_xts_aes_engine {
/***** SHA engine *****/
/**
- * ccp_sha_type - type of SHA operation
+ * enum ccp_sha_type - type of SHA operation
*
* @CCP_SHA_TYPE_1: SHA-1 operation
* @CCP_SHA_TYPE_224: SHA-224 operation
* @CCP_SHA_TYPE_256: SHA-256 operation
+ * @CCP_SHA_TYPE_384: SHA-384 operation
+ * @CCP_SHA_TYPE_512: SHA-512 operation
*/
enum ccp_sha_type {
CCP_SHA_TYPE_1 = 1,
@@ -259,6 +272,7 @@ enum ccp_sha_type {
CCP_SHA_TYPE_256,
CCP_SHA_TYPE_384,
CCP_SHA_TYPE_512,
+ /* private: */
CCP_SHA_TYPE__LAST,
};
@@ -384,7 +398,7 @@ struct ccp_rsa_engine {
/***** Passthru engine *****/
/**
- * ccp_passthru_bitwise - type of bitwise passthru operation
+ * enum ccp_passthru_bitwise - type of bitwise passthru operation
*
* @CCP_PASSTHRU_BITWISE_NOOP: no bitwise operation performed
* @CCP_PASSTHRU_BITWISE_AND: perform bitwise AND of src with mask
@@ -398,11 +412,12 @@ enum ccp_passthru_bitwise {
CCP_PASSTHRU_BITWISE_OR,
CCP_PASSTHRU_BITWISE_XOR,
CCP_PASSTHRU_BITWISE_MASK,
+ /* private: */
CCP_PASSTHRU_BITWISE__LAST,
};
/**
- * ccp_passthru_byteswap - type of byteswap passthru operation
+ * enum ccp_passthru_byteswap - type of byteswap passthru operation
*
* @CCP_PASSTHRU_BYTESWAP_NOOP: no byte swapping performed
* @CCP_PASSTHRU_BYTESWAP_32BIT: swap bytes within 32-bit words
@@ -412,6 +427,7 @@ enum ccp_passthru_byteswap {
CCP_PASSTHRU_BYTESWAP_NOOP = 0,
CCP_PASSTHRU_BYTESWAP_32BIT,
CCP_PASSTHRU_BYTESWAP_256BIT,
+ /* private: */
CCP_PASSTHRU_BYTESWAP__LAST,
};
@@ -450,8 +466,8 @@ struct ccp_passthru_engine {
* @byte_swap: byteswap operation to perform
* @mask: mask to be applied to data
* @mask_len: length in bytes of mask
- * @src: data to be used for this operation
- * @dst: data produced by this operation
+ * @src_dma: data to be used for this operation
+ * @dst_dma: data produced by this operation
* @src_len: length in bytes of data used for this operation
* @final: indicate final pass-through operation
*
@@ -478,7 +494,7 @@ struct ccp_passthru_nomap_engine {
#define CCP_ECC_MAX_OUTPUTS 3
/**
- * ccp_ecc_function - type of ECC function
+ * enum ccp_ecc_function - type of ECC function
*
* @CCP_ECC_FUNCTION_MMUL_384BIT: 384-bit modular multiplication
* @CCP_ECC_FUNCTION_MADD_384BIT: 384-bit modular addition
@@ -564,8 +580,9 @@ struct ccp_ecc_point_math {
* @function: ECC function to perform
* @mod: ECC modulus
* @mod_len: length in bytes of modulus
- * @mm: module math parameters
- * @pm: point math parameters
+ * @u: union for math parameters
+ * @u.mm: module math parameters
+ * @u.pm: point math parameters
* @ecc_result: result of the ECC operation
*
* Variables required to be set when calling ccp_enqueue_cmd():
@@ -589,11 +606,11 @@ struct ccp_ecc_engine {
/**
- * ccp_engine - CCP operation identifiers
+ * enum ccp_engine - CCP operation identifiers
*
* @CCP_ENGINE_AES: AES operation
- * @CCP_ENGINE_XTS_AES: 128-bit XTS AES operation
- * @CCP_ENGINE_RSVD1: unused
+ * @CCP_ENGINE_XTS_AES_128: 128-bit XTS AES operation
+ * @CCP_ENGINE_DES3: 3DES operation
* @CCP_ENGINE_SHA: SHA operation
* @CCP_ENGINE_RSA: RSA operation
* @CCP_ENGINE_PASSTHRU: pass-through operation
@@ -609,6 +626,7 @@ enum ccp_engine {
CCP_ENGINE_PASSTHRU,
CCP_ENGINE_ZLIB_DECOMPRESS,
CCP_ENGINE_ECC,
+ /* private: */
CCP_ENGINE__LAST,
};
diff --git a/include/linux/cgroup-defs.h b/include/linux/cgroup-defs.h
index 7a631a257613..3754d697854b 100644
--- a/include/linux/cgroup-defs.h
+++ b/include/linux/cgroup-defs.h
@@ -527,7 +527,10 @@ struct cgroup {
int nr_threaded_children; /* # of live threaded child cgroups */
- /* sequence number for cgroup.kill, serialized by css_set_lock. */
+ /*
+ * Sequence number for cgroup.kill. Incremented with both cgroup_mutex
+ * and css_set_lock held. Readers hold either one.
+ */
unsigned int kill_seq;
struct kernfs_node *kn; /* cgroup kernfs entry */
diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h
index 5dfa915a630e..2afb4cb2bb4f 100644
--- a/include/linux/cgroup.h
+++ b/include/linux/cgroup.h
@@ -154,6 +154,9 @@ struct cgroup *cgroup_get_from_path(const char *path);
struct cgroup *cgroup_get_from_fd(int fd);
struct cgroup *cgroup_v1v2_get_from_fd(int fd);
+/* Was this controller blocked from v1 hierarchies by cgroup_no_v1= ? */
+bool cgroup1_ssid_disabled(int ssid);
+
int cgroup_attach_task_all(struct task_struct *from, struct task_struct *);
int cgroup_transfer_tasks(struct cgroup *to, struct cgroup *from);
diff --git a/include/linux/cleanup.h b/include/linux/cleanup.h
index b1b5698cbf1b..1fb8058b897d 100644
--- a/include/linux/cleanup.h
+++ b/include/linux/cleanup.h
@@ -261,10 +261,6 @@ const volatile void * __must_check_fn(const volatile void *val)
* CLASS(name, var)(args...):
* declare the variable @var as an instance of the named class
*
- * CLASS_INIT(name, var, init_expr):
- * declare the variable @var as an instance of the named class with
- * custom initialization expression.
- *
* Ex.
*
* DEFINE_CLASS(fdget, struct fd, fdput(_T), fdget(fd), int fd)
@@ -302,9 +298,6 @@ static __always_inline class_##_name##_t class_##_name##ext##_constructor(_init_
class_##_name##_t var __cleanup(class_##_name##_destructor) = \
class_##_name##_constructor
-#define CLASS_INIT(_name, _var, _init_expr) \
- class_##_name##_t _var __cleanup(class_##_name##_destructor) = (_init_expr)
-
#define __scoped_class(_name, var, _label, args...) \
for (CLASS(_name, var)(args); ; ({ goto _label; })) \
if (0) { \
diff --git a/include/linux/clk-provider.h b/include/linux/clk-provider.h
index dc8e161f9c2c..a355d1ee5189 100644
--- a/include/linux/clk-provider.h
+++ b/include/linux/clk-provider.h
@@ -7,6 +7,7 @@
#define __LINUX_CLK_PROVIDER_H
#include <dt-bindings/clock/clock.h>
+#include <linux/bits.h>
#include <linux/of.h>
#include <linux/of_clk.h>
@@ -50,6 +51,7 @@ struct clk;
struct clk_hw;
struct clk_core;
struct dentry;
+struct module;
/**
* struct clk_rate_request - Structure encoding the clk constraints that
@@ -351,8 +353,9 @@ struct clk_init_data {
* @core: pointer to the struct clk_core instance that points back to this
* struct clk_hw instance
*
- * @clk: pointer to the per-user struct clk instance that can be used to call
- * into the clk API
+ * @clk: pointer to the per-user struct clk instance. This will be removed at
+ * some point in the future, Please consider the field obsolete and do not use
+ * it. Use clk_hw_get_clk() to get a per-user struct clk from a struct clk_hw.
*
* @init: pointer to struct clk_init_data that contains the init data shared
* with the common clock framework. This pointer will be set to NULL once
@@ -755,7 +758,7 @@ struct clk_divider {
spinlock_t *lock;
};
-#define clk_div_mask(width) ((1 << (width)) - 1)
+#define clk_div_mask(width) GENMASK((width) - 1, 0)
#define to_clk_divider(_hw) container_of(_hw, struct clk_divider, hw)
#define CLK_DIVIDER_ONE_BASED BIT(0)
@@ -1474,10 +1477,24 @@ static inline struct clk_hw *__clk_get_hw(struct clk *clk)
}
#endif
-struct clk *clk_hw_get_clk(struct clk_hw *hw, const char *con_id);
+struct clk *__clk_hw_get_clk(struct clk_hw *hw, const char *con_id,
+ struct module *owner);
struct clk *devm_clk_hw_get_clk(struct device *dev, struct clk_hw *hw,
const char *con_id);
+/**
+ * clk_hw_get_clk - get clk consumer given a clk_hw
+ * @hw: clk_hw associated with the clk being consumed
+ * @con_id: connection ID string on device
+ *
+ * Return: new clk consumer
+ * This is the function to be used by providers which need
+ * to get a consumer clk and act on the clock element
+ * Calls to this function must be balanced with calls to clk_put()
+ */
+#define clk_hw_get_clk(hw, con_id) \
+ __clk_hw_get_clk((hw), (con_id), THIS_MODULE)
+
unsigned int clk_hw_get_num_parents(const struct clk_hw *hw);
struct clk_hw *clk_hw_get_parent(const struct clk_hw *hw);
struct clk_hw *clk_hw_get_parent_by_index(const struct clk_hw *hw,
diff --git a/include/linux/clk/tegra.h b/include/linux/clk/tegra.h
index 3650e926e93f..f1034e14c1a4 100644
--- a/include/linux/clk/tegra.h
+++ b/include/linux/clk/tegra.h
@@ -170,7 +170,7 @@ struct tegra210_clk_emc_provider {
struct module *owner;
struct device *dev;
- struct tegra210_clk_emc_config *configs;
+ struct tegra210_clk_emc_config *configs __counted_by_ptr(num_configs);
unsigned int num_configs;
int (*set_rate)(struct device *dev,
diff --git a/include/linux/clk/ti.h b/include/linux/clk/ti.h
index 54a3fa370004..61c4abf19995 100644
--- a/include/linux/clk/ti.h
+++ b/include/linux/clk/ti.h
@@ -51,7 +51,7 @@ struct clk_omap_reg {
* @autoidle_mask: mask of the DPLL autoidle mode bitfield in @autoidle_reg
* @freqsel_mask: mask of the DPLL jitter correction bitfield in @control_reg
* @dcc_mask: mask of the DPLL DCC correction bitfield @mult_div1_reg
- * @dcc_rate: rate atleast which DCC @dcc_mask must be set
+ * @dcc_rate: rate at least which DCC @dcc_mask must be set
* @idlest_mask: mask of the DPLL idle status bitfield in @idlest_reg
* @lpmode_mask: mask of the DPLL low-power mode bitfield in @control_reg
* @m4xen_mask: mask of the DPLL M4X multiplier bitfield in @control_reg
diff --git a/include/linux/compaction.h b/include/linux/compaction.h
index 66a2f70e9e01..691c09f0a697 100644
--- a/include/linux/compaction.h
+++ b/include/linux/compaction.h
@@ -81,10 +81,6 @@ static inline unsigned long compact_gap(unsigned int order)
return min(2UL << order, COMPACT_CLUSTER_MAX);
}
-static inline int current_is_kcompactd(void)
-{
- return current->flags & PF_KCOMPACTD;
-}
#ifdef CONFIG_COMPACTION
@@ -103,7 +99,7 @@ extern void compaction_defer_reset(struct zone *zone, int order,
bool compaction_zonelist_suitable(struct alloc_context *ac, int order,
int alloc_flags, gfp_t gfp_mask);
-
+bool current_is_kcompactd(void);
extern void __meminit kcompactd_run(int nid);
extern void __meminit kcompactd_stop(int nid);
extern void wakeup_kcompactd(pg_data_t *pgdat, int order, int highest_zoneidx);
@@ -120,6 +116,11 @@ static inline bool compaction_suitable(struct zone *zone, int order,
return false;
}
+static inline bool current_is_kcompactd(void)
+{
+ return false;
+}
+
static inline void kcompactd_run(int nid)
{
}
diff --git a/include/linux/compiler.h b/include/linux/compiler.h
index cb2f6050bdf7..0ea7150dba7b 100644
--- a/include/linux/compiler.h
+++ b/include/linux/compiler.h
@@ -239,6 +239,23 @@ void ftrace_likely_update(struct ftrace_likely_data *f, int val,
# define TYPEOF_UNQUAL(exp) __typeof__(exp)
#endif
+/*
+ * TYPEOF_NO_ADDRESS_SPACE() - typeof() without the address space qualifiers
+ *
+ * No operator strips only the address space qualifiers: typeof() keeps every
+ * qualifier and TYPEOF_UNQUAL() drops every qualifier, const and volatile
+ * included.
+ *
+ * Approximate one by dropping the qualifiers for sparse only, as it is the
+ * only one that knows about address spaces. The compiler keeps seeing the fully
+ * qualified type, so a missing const or volatile will still throw a warning.
+ */
+#ifdef __CHECKER__
+# define TYPEOF_NO_ADDRESS_SPACE(exp) TYPEOF_UNQUAL(exp)
+#else
+# define TYPEOF_NO_ADDRESS_SPACE(exp) __typeof__(exp)
+#endif
+
#endif /* __KERNEL__ */
#if defined(CONFIG_CFI) && !defined(__DISABLE_EXPORTS) && !defined(BUILD_VDSO)
@@ -275,6 +292,11 @@ static inline void *offset_to_ptr(const int *off)
#define __ADDRESSABLE(sym) \
___ADDRESSABLE(sym, __section(".discard.addressable"))
+/* Enforce static storage duration. */
+#define ASSERT_STATIC_STORAGE(name) \
+ static typeof(name) * const __always_unused \
+ name##_storage_check = &(name)
+
/*
* This returns a constant expression while determining if an argument is
* a constant expression, most importantly without evaluating the argument.
diff --git a/include/linux/configfs.h b/include/linux/configfs.h
index ef65c75beeaa..5bead9173ec1 100644
--- a/include/linux/configfs.h
+++ b/include/linux/configfs.h
@@ -66,8 +66,11 @@ struct config_item_type {
struct module *ct_owner;
const struct configfs_item_operations *ct_item_ops;
const struct configfs_group_operations *ct_group_ops;
- struct configfs_attribute **ct_attrs;
- struct configfs_bin_attribute **ct_bin_attrs;
+ union {
+ struct configfs_attribute **ct_attrs;
+ const struct configfs_attribute *const *ct_attrs_const;
+ };
+ const struct configfs_bin_attribute *const *ct_bin_attrs;
};
/**
@@ -160,41 +163,41 @@ struct configfs_bin_attribute {
ssize_t (*write)(struct config_item *, const void *, size_t);
};
-#define CONFIGFS_BIN_ATTR(_pfx, _name, _priv, _maxsz) \
-static struct configfs_bin_attribute _pfx##attr_##_name = { \
- .cb_attr = { \
- .ca_name = __stringify(_name), \
- .ca_mode = S_IRUGO | S_IWUSR, \
- .ca_owner = THIS_MODULE, \
- }, \
- .cb_private = _priv, \
- .cb_max_size = _maxsz, \
- .read = _pfx##_name##_read, \
- .write = _pfx##_name##_write, \
+#define CONFIGFS_BIN_ATTR(_pfx, _name, _priv, _maxsz) \
+static const struct configfs_bin_attribute _pfx##attr_##_name = { \
+ .cb_attr = { \
+ .ca_name = __stringify(_name), \
+ .ca_mode = S_IRUGO | S_IWUSR, \
+ .ca_owner = THIS_MODULE, \
+ }, \
+ .cb_private = _priv, \
+ .cb_max_size = _maxsz, \
+ .read = _pfx##_name##_read, \
+ .write = _pfx##_name##_write, \
}
-#define CONFIGFS_BIN_ATTR_RO(_pfx, _name, _priv, _maxsz) \
-static struct configfs_bin_attribute _pfx##attr_##_name = { \
- .cb_attr = { \
- .ca_name = __stringify(_name), \
- .ca_mode = S_IRUGO, \
- .ca_owner = THIS_MODULE, \
- }, \
- .cb_private = _priv, \
- .cb_max_size = _maxsz, \
- .read = _pfx##_name##_read, \
+#define CONFIGFS_BIN_ATTR_RO(_pfx, _name, _priv, _maxsz) \
+static const struct configfs_bin_attribute _pfx##attr_##_name = { \
+ .cb_attr = { \
+ .ca_name = __stringify(_name), \
+ .ca_mode = S_IRUGO, \
+ .ca_owner = THIS_MODULE, \
+ }, \
+ .cb_private = _priv, \
+ .cb_max_size = _maxsz, \
+ .read = _pfx##_name##_read, \
}
-#define CONFIGFS_BIN_ATTR_WO(_pfx, _name, _priv, _maxsz) \
-static struct configfs_bin_attribute _pfx##attr_##_name = { \
- .cb_attr = { \
- .ca_name = __stringify(_name), \
- .ca_mode = S_IWUSR, \
- .ca_owner = THIS_MODULE, \
- }, \
- .cb_private = _priv, \
- .cb_max_size = _maxsz, \
- .write = _pfx##_name##_write, \
+#define CONFIGFS_BIN_ATTR_WO(_pfx, _name, _priv, _maxsz) \
+static const struct configfs_bin_attribute _pfx##attr_##_name = { \
+ .cb_attr = { \
+ .ca_name = __stringify(_name), \
+ .ca_mode = S_IWUSR, \
+ .ca_owner = THIS_MODULE, \
+ }, \
+ .cb_private = _priv, \
+ .cb_max_size = _maxsz, \
+ .write = _pfx##_name##_write, \
}
/*
@@ -220,8 +223,8 @@ struct configfs_group_operations {
struct config_group *(*make_group)(struct config_group *group, const char *name);
void (*disconnect_notify)(struct config_group *group, struct config_item *item);
void (*drop_item)(struct config_group *group, struct config_item *item);
- bool (*is_visible)(struct config_item *item, struct configfs_attribute *attr, int n);
- bool (*is_bin_visible)(struct config_item *item, struct configfs_bin_attribute *attr,
+ bool (*is_visible)(struct config_item *item, const struct configfs_attribute *attr, int n);
+ bool (*is_bin_visible)(struct config_item *item, const struct configfs_bin_attribute *attr,
int n);
};
diff --git a/include/linux/console.h b/include/linux/console.h
index d624200cfc17..502d1abe3f50 100644
--- a/include/linux/console.h
+++ b/include/linux/console.h
@@ -173,7 +173,7 @@ static inline void con_debug_leave(void) { }
* @CON_BRL: Indicates a braille device which is exempt from
* receiving the printk spam for obvious reasons.
* @CON_EXTENDED: The console supports the extended output format of
- * /dev/kmesg which requires a larger output buffer.
+ * /dev/kmsg which requires a larger output buffer.
* @CON_SUSPENDED: Indicates if a console is suspended. If true, the
* printing callbacks must not be called.
* @CON_NBCON: Console can operate outside of the legacy style console_lock
diff --git a/include/linux/coredump.h b/include/linux/coredump.h
index 7b38ee2e7913..74af57b9406b 100644
--- a/include/linux/coredump.h
+++ b/include/linux/coredump.h
@@ -6,9 +6,20 @@
#include <linux/mm.h>
#include <linux/fs.h>
#include <linux/sched/coredump.h>
+#include <uapi/linux/coredump.h>
#include <asm/siginfo.h>
#ifdef CONFIG_COREDUMP
+/**
+ * enum coredump_state - what happened while the coredump was written
+ * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump
+ * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it
+ */
+enum coredump_state {
+ COREDUMP_STATE_STARTED = (1U << 0),
+ COREDUMP_STATE_TRUNCATED = (1U << 1),
+};
+
struct core_vma_metadata {
unsigned long start, end;
vm_flags_t flags;
@@ -21,12 +32,20 @@ struct coredump_params {
const kernel_siginfo_t *siginfo;
struct file *file;
unsigned long limit;
- /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */
- unsigned long mm_flags;
+ /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */
+ u64 memory_types;
/* Snapshot of dumpable at dump start. */
enum task_dumpable dumpable;
int cpu;
+ /* COREDUMP_* options negotiated with the coredump server. */
+ u64 mask;
+ /* COREDUMP_STATE_* raised while the coredump is written. */
+ enum coredump_state state;
+ /* Record header scratch, NULL unless the coredump is a record stream. */
+ struct coredump_record_header *record_hdr;
+ /* Bytes handed to the file, record headers included. */
loff_t written;
+ /* Offset in the coredump, record headers excluded. */
loff_t pos;
loff_t to_skip;
int vma_count;
@@ -41,13 +60,13 @@ extern unsigned int core_file_note_size_limit;
* These are the only things you should do on a core-file: use only these
* functions to write out all the necessary info.
*/
-extern void dump_skip_to(struct coredump_params *cprm, unsigned long to);
-extern void dump_skip(struct coredump_params *cprm, size_t nr);
-extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr);
-extern int dump_align(struct coredump_params *cprm, int align);
-int dump_user_range(struct coredump_params *cprm, unsigned long start,
- unsigned long len);
-extern void vfs_coredump(const kernel_siginfo_t *siginfo);
+void dump_skip_to(struct coredump_params *cprm, unsigned long to);
+void dump_skip(struct coredump_params *cprm, size_t nr);
+bool dump_emit(struct coredump_params *cprm, const void *addr, int nr);
+bool dump_align(struct coredump_params *cprm, int align);
+bool dump_user_range(struct coredump_params *cprm, unsigned long start,
+ unsigned long len);
+void vfs_coredump(const kernel_siginfo_t *siginfo);
/*
* Logging for the coredump code, ratelimited.
diff --git a/include/linux/cpufreq.h b/include/linux/cpufreq.h
index 35ce665edfd8..d3d0d9d02aa4 100644
--- a/include/linux/cpufreq.h
+++ b/include/linux/cpufreq.h
@@ -420,6 +420,9 @@ struct cpufreq_driver {
/* Will be called after the driver is fully initialized */
void (*ready)(struct cpufreq_policy *policy);
+ /* Return the capacity reference frequency for policy. */
+ unsigned int (*scale_freq_ref)(struct cpufreq_policy *policy);
+
struct freq_attr **attr;
/* platform specific boost support code */
diff --git a/include/linux/cpumask.h b/include/linux/cpumask.h
index 4c8bb6953107..bf89bb3f30f6 100644
--- a/include/linux/cpumask.h
+++ b/include/linux/cpumask.h
@@ -121,12 +121,20 @@ extern struct cpumask __cpu_enabled_mask;
extern struct cpumask __cpu_present_mask;
extern struct cpumask __cpu_active_mask;
extern struct cpumask __cpu_dying_mask;
+
+#ifdef CONFIG_PREFERRED_CPU
+extern struct cpumask __cpu_preferred_mask;
+#else
+#define __cpu_preferred_mask __cpu_active_mask
+#endif
+
#define cpu_possible_mask ((const struct cpumask *)&__cpu_possible_mask)
#define cpu_online_mask ((const struct cpumask *)&__cpu_online_mask)
#define cpu_enabled_mask ((const struct cpumask *)&__cpu_enabled_mask)
#define cpu_present_mask ((const struct cpumask *)&__cpu_present_mask)
#define cpu_active_mask ((const struct cpumask *)&__cpu_active_mask)
#define cpu_dying_mask ((const struct cpumask *)&__cpu_dying_mask)
+#define cpu_preferred_mask ((const struct cpumask *)&__cpu_preferred_mask)
extern atomic_t __num_online_cpus;
extern unsigned int __num_possible_cpus;
@@ -825,6 +833,24 @@ bool cpumask_intersects(const struct cpumask *src1p, const struct cpumask *src2p
}
/**
+ * cpumask_intersects_and - (*src1p & *src2p & *src3p) != 0
+ * @src1p: the first input
+ * @src2p: the second input
+ * @src3p: the third input
+ *
+ * Return: true if AND of the three cpumasks is non-empty,
+ * otherwise false
+ */
+static __always_inline
+bool cpumask_intersects_and(const struct cpumask *src1p,
+ const struct cpumask *src2p,
+ const struct cpumask *src3p)
+{
+ return bitmap_intersects_and(cpumask_bits(src1p), cpumask_bits(src2p),
+ cpumask_bits(src3p), small_cpumask_bits);
+}
+
+/**
* cpumask_subset - (*src1p & ~*src2p) == 0
* @src1p: the first input
* @src2p: the second input
@@ -1163,6 +1189,12 @@ void init_cpu_possible(const struct cpumask *src);
#define set_cpu_active(cpu, active) assign_cpu((cpu), &__cpu_active_mask, (active))
#define set_cpu_dying(cpu, dying) assign_cpu((cpu), &__cpu_dying_mask, (dying))
+#ifdef CONFIG_PREFERRED_CPU
+#define set_cpu_preferred(cpu, preferred) assign_cpu((cpu), &__cpu_preferred_mask, (preferred))
+#else
+#define set_cpu_preferred(cpu, preferred) do { } while (0)
+#endif
+
void set_cpu_online(unsigned int cpu, bool online);
void set_cpu_possible(unsigned int cpu, bool possible);
@@ -1257,6 +1289,11 @@ static __always_inline bool cpu_dying(unsigned int cpu)
return cpumask_test_cpu(cpu, cpu_dying_mask);
}
+static __always_inline bool cpu_preferred(unsigned int cpu)
+{
+ return cpumask_test_cpu(cpu, cpu_preferred_mask);
+}
+
#else
#define num_online_cpus() 1U
@@ -1295,6 +1332,11 @@ static __always_inline bool cpu_dying(unsigned int cpu)
return false;
}
+static __always_inline bool cpu_preferred(unsigned int cpu)
+{
+ return cpu == 0;
+}
+
#endif /* NR_CPUS > 1 */
#define cpu_is_offline(cpu) unlikely(!cpu_online(cpu))
diff --git a/include/linux/cred.h b/include/linux/cred.h
index 6ef1750c93e2..49c26af37349 100644
--- a/include/linux/cred.h
+++ b/include/linux/cred.h
@@ -180,12 +180,18 @@ static inline bool cap_ambient_invariant_ok(const struct cred *cred)
static inline const struct cred *override_creds(const struct cred *override_cred)
{
- return rcu_replace_pointer(current->cred, override_cred, 1);
+ const struct cred *old = current->cred;
+
+ current->cred = override_cred;
+ return old;
}
static inline const struct cred *revert_creds(const struct cred *revert_cred)
{
- return rcu_replace_pointer(current->cred, revert_cred, 1);
+ const struct cred *override_cred = current->cred;
+
+ current->cred = revert_cred;
+ return override_cred;
}
DEFINE_CLASS(override_creds,
@@ -293,11 +299,10 @@ DEFINE_FREE(put_cred, struct cred *, if (!IS_ERR_OR_NULL(_T)) put_cred(_T))
/**
* current_cred - Access the current task's subjective credentials
*
- * Access the subjective credentials of the current task. RCU-safe,
- * since nobody else can modify it.
+ * Access the subjective credentials of the current task.
+ * Nobody else can modify it.
*/
-#define current_cred() \
- rcu_dereference_protected(current->cred, 1)
+#define current_cred() (current->cred)
/**
* current_real_cred - Access the current task's objective credentials
diff --git a/include/linux/damon.h b/include/linux/damon.h
index 0c8b7ddef9ab..63050eb2206a 100644
--- a/include/linux/damon.h
+++ b/include/linux/damon.h
@@ -117,7 +117,6 @@ struct damon_target {
* @DAMOS_MIGRATE_HOT: Migrate the regions prioritizing warmer regions.
* @DAMOS_MIGRATE_COLD: Migrate the regions prioritizing colder regions.
* @DAMOS_STAT: Do nothing but count the stat.
- * @NR_DAMOS_ACTIONS: Total number of DAMOS actions
*
* The support of each action is up to running &struct damon_operations.
* Refer to 'Operation Action' section of Documentation/mm/damon/design.rst for
@@ -137,7 +136,6 @@ enum damos_action {
DAMOS_MIGRATE_HOT,
DAMOS_MIGRATE_COLD,
DAMOS_STAT, /* Do nothing but only record the stat */
- NR_DAMOS_ACTIONS,
};
/**
@@ -153,9 +151,7 @@ enum damos_action {
* @DAMOS_QUOTA_INACTIVE_MEM_BP: Inactive to total LRU memory ratio.
* @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a
* node in basis points (0-10000).
- * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics.
- *
- * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported.
+ * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio.
*/
enum damos_quota_goal_metric {
DAMOS_QUOTA_USER_INPUT,
@@ -167,12 +163,13 @@ enum damos_quota_goal_metric {
DAMOS_QUOTA_ACTIVE_MEM_BP,
DAMOS_QUOTA_INACTIVE_MEM_BP,
DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP,
- NR_DAMOS_QUOTA_GOAL_METRICS,
+ DAMOS_QUOTA_HUGEPAGE_MEM_BP,
};
/**
* struct damos_quota_goal - DAMOS scheme quota auto-tuning goal.
* @metric: Metric to be used for representing the goal.
+ * @complement: Use the complement of the metric.
* @target_value: Target value of @metric to achieve with the tuning.
* @current_value: Current value of @metric.
* @nid: Node id.
@@ -195,6 +192,7 @@ enum damos_quota_goal_metric {
*/
struct damos_quota_goal {
enum damos_quota_goal_metric metric;
+ bool complement;
unsigned long target_value;
unsigned long current_value;
/* metric-dependent fields */
@@ -311,12 +309,10 @@ struct damos_quota {
*
* @DAMOS_WMARK_NONE: Ignore the watermarks of the given scheme.
* @DAMOS_WMARK_FREE_MEM_RATE: Free memory rate of the system in [0,1000].
- * @NR_DAMOS_WMARK_METRICS: Total number of DAMOS watermark metrics
*/
enum damos_wmark_metric {
DAMOS_WMARK_NONE,
DAMOS_WMARK_FREE_MEM_RATE,
- NR_DAMOS_WMARK_METRICS,
};
/**
@@ -357,7 +353,8 @@ struct damos_watermarks {
* Total bytes that passed ops layer-handled DAMOS filters.
* @qt_exceeds: Total number of times the quota of the scheme has exceeded.
* @nr_snapshots:
- * Total number of DAMON snapshots that the scheme has tried.
+ * Total number of DAMON snapshots that the scheme is completely
+ * tried to be applied.
*
* "Tried an action to a region" in this context means the DAMOS core logic
* determined the region as eligible to apply the action. The access pattern
@@ -396,14 +393,14 @@ struct damos_stat {
* @DAMOS_FILTER_TYPE_UNMAPPED: Unmapped pages.
* @DAMOS_FILTER_TYPE_ADDR: Address range.
* @DAMOS_FILTER_TYPE_TARGET: Data Access Monitoring target.
- * @NR_DAMOS_FILTER_TYPES: Number of filter types.
+ * @DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: probe_hits weighted sum range.
*
- * All types except &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET
- * are handled by the underlying &struct damon_operations as a part of scheme
- * action trying, and therefore accounted as 'tried'. In contrast,
- * &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET filters are handled
- * by the core layer before trying of the action, and therefore not accounted
- * as 'tried'.
+ * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and
+ * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct
+ * damon_operations as a part of scheme action trying, and therefore accounted
+ * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET
+ * and &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM filters are handled by the core layer
+ * before trying of the action, and therefore not accounted as 'tried'.
*
* Support for the operations-handled filters depends on the running
* &struct damon_operations.
@@ -417,7 +414,7 @@ enum damos_filter_type {
DAMOS_FILTER_TYPE_UNMAPPED,
DAMOS_FILTER_TYPE_ADDR,
DAMOS_FILTER_TYPE_TARGET,
- NR_DAMOS_FILTER_TYPES,
+ DAMOS_FILTER_TYPE_PROBE_HITS_WSUM,
};
/**
@@ -431,6 +428,8 @@ enum damos_filter_type {
* &damon_ctx->adaptive_targets if @type is
* DAMOS_FILTER_TYPE_TARGET.
* @sz_range: Size range if @type is DAMOS_FILTER_TYPE_HUGEPAGE_SIZE.
+ * @range_min: Minimum value of range arguments.
+ * @range_max: Maximum value of range arguments.
*
* Before applying the &damos->action to a memory region, DAMOS checks if each
* byte of the region matches to this given condition and avoid applying the
@@ -447,6 +446,10 @@ struct damos_filter {
struct damon_addr_range addr_range;
int target_idx;
struct damon_size_range sz_range;
+ struct {
+ unsigned long range_min;
+ unsigned long range_max;
+ };
};
/* private: */
/* List head for siblings. */
@@ -548,7 +551,7 @@ struct damos_migrate_dests {
*
* After applying the &action to each region, &stat is updated.
*
- * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots be same to or
+ * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots equals or is
* greater than it, the scheme is deactivated.
*/
struct damos {
@@ -627,6 +630,7 @@ enum damon_ops_id {
* @update: Update operations-related data structures.
* @prepare_access_checks: Prepare next access check of target regions.
* @check_accesses: Check the accesses to target regions.
+ * @prep_probes: Prepare applying probes for each region.
* @apply_probes: Apply probes for each region.
* @get_scheme_score: Get the score of a region for a scheme.
* @apply_scheme: Apply a DAMON-based operation scheme.
@@ -654,6 +658,9 @@ enum damon_ops_id {
* last preparation and update the number of observed accesses of each region.
* It should also return max number of observed accesses that made as a result
* of its update. The value will be used for regions adjustment threshold.
+ * @prep_probes should execute required &struct damon_prep for next &struct
+ * damon_probe applications to each region. It should also set
+ * &damon_region->sampling_addr of each region if ``set_samples`` is true.
* @apply_probes should apply the data attribute probes to each region and
* accordingly update the probe hits counter of the region. It should also
* set &damon_region->sampling_addr of each region if ``set_samples`` is true.
@@ -676,6 +683,7 @@ struct damon_operations {
void (*update)(struct damon_ctx *context);
void (*prepare_access_checks)(struct damon_ctx *context);
unsigned int (*check_accesses)(struct damon_ctx *context);
+ void (*prep_probes)(struct damon_ctx *context, bool set_samples);
unsigned int (*apply_probes)(struct damon_ctx *context,
bool set_samples, bool return_max_wsum);
int (*get_scheme_score)(struct damon_ctx *context,
@@ -740,14 +748,41 @@ struct damon_intervals_goal {
};
/**
+ * enum damon_prep_action - DAMON probing preparation action.
+ *
+ * @DAMON_PREP_SET_PGIDLE: Set the probing memory as idle page.
+ */
+enum damon_prep_action {
+ DAMON_PREP_SET_PGIDLE,
+};
+
+/**
+ * struct damon_prep - DAMON probing preparation request.
+ *
+ * @action: Action to do to the probing memory for the preparation.
+ */
+struct damon_prep {
+ enum damon_prep_action action;
+/* private: */
+ /* siblings list. */
+ struct list_head list;
+};
+
+/**
* enum damon_filter_type - Type of &struct damon_filter
*
- * @DAMON_FILTER_TYPE_ANON: Anonymous pages.
- * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages.
+ * @DAMON_FILTER_TYPE_ANON: Anonymous pages.
+ * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages.
+ * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset.
+ * @DAMON_FILTER_TYPE_PGIDLE_SET: Pgidle is set.
+ * @DAMON_FILTER_TYPE_HUGEPAGE_SIZE: Page is part of a hugepage.
*/
enum damon_filter_type {
DAMON_FILTER_TYPE_ANON,
DAMON_FILTER_TYPE_MEMCG,
+ DAMON_FILTER_TYPE_PGIDLE_UNSET,
+ DAMON_FILTER_TYPE_PGIDLE_SET,
+ DAMON_FILTER_TYPE_HUGEPAGE_SIZE,
};
/**
@@ -757,6 +792,8 @@ enum damon_filter_type {
* @matching: Whether this filter is for the type-matching ones.
* @allow: Whether the @type-@matching ones should pass this filter.
* @memcg_id: Memcg id of the question if @type is DAMON_FILTER_MEMCG.
+ * @range_min: Minimum value of range arguments.
+ * @range_max: Maximum value of range arguments.
*/
struct damon_filter {
enum damon_filter_type type;
@@ -764,6 +801,10 @@ struct damon_filter {
bool allow;
union {
u64 memcg_id;
+ struct {
+ unsigned long range_min;
+ unsigned long range_max;
+ };
};
/* private: */
/* Siblings list. */
@@ -778,6 +819,8 @@ struct damon_filter {
struct damon_probe {
unsigned int weight;
/* private: */
+ /* Preparation actions to apply to each probing memory. */
+ struct list_head preps;
/* Filters for assessing if a given region is for this probe. */
struct list_head filters;
/* Siblings list. */
@@ -787,7 +830,8 @@ struct damon_probe {
/**
* struct damon_attrs - Monitoring attributes for accuracy/overhead control.
*
- * @sample_interval: The time between access samplings.
+ * @sample_interval: The time between access samplings. Zero is
+ * accepted.
* @aggr_interval: The time between monitor results aggregations.
* @ops_update_interval: The time between monitoring operations updates.
* @intervals_goal: Intervals auto-tuning goal.
@@ -957,6 +1001,12 @@ static inline unsigned long damon_sz_region(struct damon_region *r)
return r->ar.end - r->ar.start;
}
+#define damon_for_each_prep(p, probe) \
+ list_for_each_entry(p, &(probe)->preps, list)
+
+#define damon_for_each_prep_safe(p, next, probe) \
+ list_for_each_entry_safe(p, next, &(probe)->preps, list)
+
#define damon_for_each_filter(f, p) \
list_for_each_entry(f, &(p)->filters, list)
@@ -1010,6 +1060,9 @@ static inline unsigned long damon_sz_region(struct damon_region *r)
#ifdef CONFIG_DAMON
+struct damon_prep *damon_new_prep(enum damon_prep_action action);
+void damon_add_prep(struct damon_probe *p, struct damon_prep *prep);
+
struct damon_filter *damon_new_filter(enum damon_filter_type type,
bool matching, bool allow);
void damon_add_filter(struct damon_probe *probe, struct damon_filter *f);
@@ -1023,7 +1076,7 @@ unsigned int damon_nr_accesses_mvsum(struct damon_region *r,
struct damon_ctx *ctx);
unsigned char damon_probe_hits_mvsum(int probe_idx, struct damon_region *r,
struct damon_ctx *ctx);
-unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last,
+unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv,
struct damon_ctx *ctx);
int damon_set_regions(struct damon_target *t, struct damon_addr_range *ranges,
@@ -1037,7 +1090,7 @@ bool damos_filter_for_ops(enum damos_filter_type type);
void damos_destroy_filter(struct damos_filter *f);
struct damos_quota_goal *damos_new_quota_goal(
- enum damos_quota_goal_metric metric,
+ enum damos_quota_goal_metric metric, bool complement,
unsigned long target_value);
void damos_add_quota_goal(struct damos_quota *q, struct damos_quota_goal *g);
void damos_destroy_quota_goal(struct damos_quota_goal *goal);
@@ -1054,7 +1107,7 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src);
struct damon_target *damon_new_target(void);
void damon_add_target(struct damon_ctx *ctx, struct damon_target *t);
-bool damon_targets_empty(struct damon_ctx *ctx);
+int damon_set_target_pid(struct damon_target *t, int pid);
void damon_free_target(struct damon_target *t);
void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx);
unsigned int damon_nr_regions(struct damon_target *t);
@@ -1065,7 +1118,6 @@ int damon_set_attrs(struct damon_ctx *ctx, struct damon_attrs *attrs);
void damon_set_schemes(struct damon_ctx *ctx,
struct damos **schemes, ssize_t nr_schemes);
int damon_commit_ctx(struct damon_ctx *old_ctx, struct damon_ctx *new_ctx);
-int damon_nr_running_ctxs(void);
bool damon_is_registered_ops(enum damon_ops_id id);
int damon_register_ops(struct damon_operations *ops);
int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id);
diff --git a/include/linux/dax.h b/include/linux/dax.h
index fe6c3ded1b50..f2d47975d905 100644
--- a/include/linux/dax.h
+++ b/include/linux/dax.h
@@ -155,8 +155,6 @@ int dax_writeback_mapping_range(struct address_space *mapping,
struct dax_device *dax_dev, struct writeback_control *wbc);
int dax_folio_reset_order(struct folio *folio);
-struct page *dax_layout_busy_page(struct address_space *mapping);
-struct page *dax_layout_busy_page_range(struct address_space *mapping, loff_t start, loff_t end);
dax_entry_t dax_lock_folio(struct folio *folio);
void dax_unlock_folio(struct folio *folio, dax_entry_t cookie);
dax_entry_t dax_lock_mapping_entry(struct address_space *mapping,
@@ -173,16 +171,6 @@ static inline int fs_dax_get(struct dax_device *dax_dev, void *holder,
{
return -EOPNOTSUPP;
}
-static inline struct page *dax_layout_busy_page(struct address_space *mapping)
-{
- return NULL;
-}
-
-static inline struct page *dax_layout_busy_page_range(struct address_space *mapping, pgoff_t start, pgoff_t nr_pages)
-{
- return NULL;
-}
-
static inline int dax_writeback_mapping_range(struct address_space *mapping,
struct dax_device *dax_dev, struct writeback_control *wbc)
{
diff --git a/include/linux/dcache.h b/include/linux/dcache.h
index 4b1ff99608e0..adf239f8205f 100644
--- a/include/linux/dcache.h
+++ b/include/linux/dcache.h
@@ -116,6 +116,8 @@ struct dentry {
* possible!
*/
+ /* lockdep tracking of DCACHE_PAR_LOOKUP locks */
+ struct lockdep_map lookup_map;
struct list_head d_lru; /* LRU list */
struct hlist_node d_sib; /* child of parent list */
struct hlist_head d_children; /* our children */
@@ -236,7 +238,9 @@ enum dentry_flags {
DCACHE_PAR_LOOKUP = BIT(24), /* being looked up (with parent locked shared) */
DCACHE_DENTRY_CURSOR = BIT(25),
DCACHE_NORCU = BIT(26), /* No RCU delay for freeing */
- DCACHE_PERSISTENT = BIT(27)
+ DCACHE_PERSISTENT = BIT(27),
+/* 28, 29, 30 free */
+ DCACHE_PRIVATE = BIT(31) /* fs-specific flag */
};
#define DCACHE_MANAGED_DENTRY \
@@ -257,7 +261,9 @@ extern void d_delete(struct dentry *);
extern struct dentry * d_alloc(struct dentry *, const struct qstr *);
extern struct dentry * d_alloc_anon(struct super_block *);
extern struct dentry * d_alloc_parallel(struct dentry *, const struct qstr *);
+extern struct dentry * d_alloc_trylock(struct dentry *, struct qstr *);
extern struct dentry * d_splice_alias(struct inode *, struct dentry *);
+struct dentry *d_duplicate(struct dentry *dentry);
/* weird procfs mess; *NOT* exported */
extern struct dentry * d_splice_alias_ops(struct inode *, struct dentry *,
const struct dentry_operations *);
@@ -553,6 +559,36 @@ static inline int simple_positive(const struct dentry *dentry)
unsigned long vfs_pressure_ratio(unsigned long val);
/**
+ * d_lookup_release - release ownership of DCACHE_PAR_LOOKUP lock
+ * @dentry: dentry that is locked
+ *
+ * If an in-lookup dentry is to be passed to another thread which
+ * will drop the in-lookup lock, then d_lookup_release() must be called
+ * to tell lockdep that this thread no lock holds the lock. The
+ * thread that receives the lock must call d_lookup_acquire() to
+ * acquire the lock.
+ */
+static inline void d_lookup_release(struct dentry *dentry)
+{
+ if (d_in_lookup(dentry))
+ lock_map_release(&dentry->lookup_map);
+}
+
+/**
+ * d_lookup_acquire - acquire ownership of DCACHE_PAR_LOOKUP lock
+ * @dentry: dentry that is locked
+ *
+ * If an in-lookup dentry was passed to this thread, the
+ * d_lookup_acquire() must be called to tell lockdep that this
+ * thread now owns the DCACHE_PAR_LOOKUP lock.
+ */
+static inline void d_lookup_acquire(struct dentry *dentry)
+{
+ if (d_in_lookup(dentry))
+ lock_map_acquire_try(&dentry->lookup_map);
+}
+
+/**
* d_inode - Get the actual inode of this dentry
* @dentry: The dentry to query
*
diff --git a/include/linux/device-id/ap.h b/include/linux/device-id/ap.h
index 0992333a34db..e050abebbf3d 100644
--- a/include/linux/device-id/ap.h
+++ b/include/linux/device-id/ap.h
@@ -4,7 +4,6 @@
#ifdef __KERNEL__
#include <linux/types.h>
-typedef unsigned long kernel_ulong_t;
#endif
#define AP_DEVICE_ID_MATCH_CARD_TYPE 0x01
@@ -14,7 +13,6 @@ typedef unsigned long kernel_ulong_t;
struct ap_device_id {
__u16 match_flags; /* which fields to match against */
__u8 dev_type; /* device type */
- kernel_ulong_t driver_info;
};
#endif /* ifndef LINUX_DEVICE_ID_AP_H */
diff --git a/include/linux/device-id/apr.h b/include/linux/device-id/apr.h
deleted file mode 100644
index f282608ea018..000000000000
--- a/include/linux/device-id/apr.h
+++ /dev/null
@@ -1,21 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0 */
-#ifndef LINUX_DEVICE_ID_APR_H
-#define LINUX_DEVICE_ID_APR_H
-
-#ifdef __KERNEL__
-#include <linux/types.h>
-typedef unsigned long kernel_ulong_t;
-#endif
-
-#define APR_NAME_SIZE 32
-#define APR_MODULE_PREFIX "apr:"
-
-struct apr_device_id {
- char name[APR_NAME_SIZE];
- __u32 domain_id;
- __u32 svc_id;
- __u32 svc_version;
- kernel_ulong_t driver_data; /* Data private to the driver */
-};
-
-#endif /* ifndef LINUX_DEVICE_ID_APR_H */
diff --git a/include/linux/device-id/arm_smccc.h b/include/linux/device-id/arm_smccc.h
new file mode 100644
index 000000000000..a8579832fc1c
--- /dev/null
+++ b/include/linux/device-id/arm_smccc.h
@@ -0,0 +1,19 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __LINUX_DEVICE_ID_ARM_SMCCC_H
+#define __LINUX_DEVICE_ID_ARM_SMCCC_H
+
+#ifdef __KERNEL__
+#include <linux/types.h>
+#endif
+
+#define ARM_SMCCC_MODULE_PREFIX "arm_smccc:"
+
+/**
+ * struct arm_smccc_device_id - Arm SMCCC bus device identifier
+ * @func_id: SMCCC function identifier
+ */
+struct arm_smccc_device_id {
+ __u32 func_id;
+};
+
+#endif /* __LINUX_DEVICE_ID_ARM_SMCCC_H */
diff --git a/include/linux/device-id/scmi.h b/include/linux/device-id/scmi.h
new file mode 100644
index 000000000000..1b4ccfa9dcc5
--- /dev/null
+++ b/include/linux/device-id/scmi.h
@@ -0,0 +1,17 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+#ifndef LINUX_DEVICE_ID_SCMI_H
+#define LINUX_DEVICE_ID_SCMI_H
+
+#ifdef __KERNEL__
+#include <linux/types.h>
+#endif
+
+#define SCMI_NAME_SIZE 32
+#define SCMI_MODULE_PREFIX "scmi:"
+
+struct scmi_device_id {
+ __u8 protocol_id;
+ char name[SCMI_NAME_SIZE];
+};
+
+#endif /* ifndef LINUX_DEVICE_ID_SCMI_H */
diff --git a/include/linux/device/bus.h b/include/linux/device/bus.h
index a38f7229b8f4..89e156b9acb5 100644
--- a/include/linux/device/bus.h
+++ b/include/linux/device/bus.h
@@ -113,6 +113,7 @@ struct bus_type {
bool need_parent_lock;
};
+int __must_check companion_bus_register(const struct bus_type *bus);
int __must_check bus_register(const struct bus_type *bus);
void bus_unregister(const struct bus_type *bus);
diff --git a/include/linux/device/driver.h b/include/linux/device/driver.h
index 768a1334c0a1..29fbc01ef06f 100644
--- a/include/linux/device/driver.h
+++ b/include/linux/device/driver.h
@@ -297,4 +297,35 @@ static int __init __driver##_init(void) \
} \
device_initcall(__driver##_init);
+/**
+ * subsys_driver() - Helper macro for drivers that don't do anything special
+ * in init/exit but have to register earlier, at subsys_initcall level, when
+ * built in. This eliminates a lot of boilerplate. Each driver may only use
+ * this macro once, and calling it replaces the init/exit boilerplate.
+ *
+ * @__driver: driver name
+ * @__register: register function for this driver type
+ * @__unregister: unregister function for this driver type
+ * @...: Additional arguments to be passed to __register and __unregister.
+ *
+ * This is meant to be a direct parallel of module_driver() above, but with
+ * the init call promoted to subsys_initcall() for built-in drivers so they
+ * are available earlier during boot. When built as a module subsys_initcall()
+ * collapses to module_init() and the normal module init/exit paths are used.
+ *
+ * Use this macro to construct bus specific macros for registering drivers,
+ * and do not use it on its own.
+ */
+#define subsys_driver(__driver, __register, __unregister, ...) \
+static int __init __driver##_init(void) \
+{ \
+ return __register(&(__driver), ##__VA_ARGS__); \
+} \
+subsys_initcall(__driver##_init) \
+static void __exit __driver##_exit(void) \
+{ \
+ __unregister(&(__driver), ##__VA_ARGS__); \
+} \
+module_exit(__driver##_exit)
+
#endif /* _DEVICE_DRIVER_H_ */
diff --git a/include/linux/dibs.h b/include/linux/dibs.h
index d3e0777f25ae..0c10c224bcca 100644
--- a/include/linux/dibs.h
+++ b/include/linux/dibs.h
@@ -261,12 +261,12 @@ struct dibs_dev_ops {
* @vlan_id: deprecated, ignored if device does not support vlan
* Upon return in addition the following fields will be valid:
* @dmb_tok: for usage by remote and local devices and clients
- * @cpu_addr: allocated buffer
+ * @cpu_addr: allocated, zerorized buffer
* @idx: dmb index, unique per dibs device
* @dma_addr: to be used by device driver,if applicable
*
- * Allocate a dmb buffer and register it with this device and for this
- * client.
+ * Allocate and zerorize a dmb buffer and register it with this device
+ * and for this client.
* Return: zero on success
*/
int (*register_dmb)(struct dibs_dev *dev, struct dibs_dmb *dmb,
diff --git a/include/linux/dma-buf.h b/include/linux/dma-buf.h
index d1203da56fc5..d15b2b31d3c9 100644
--- a/include/linux/dma-buf.h
+++ b/include/linux/dma-buf.h
@@ -567,6 +567,7 @@ void dma_buf_unpin(struct dma_buf_attachment *attach);
struct dma_buf *dma_buf_export(const struct dma_buf_export_info *exp_info);
int dma_buf_fd(struct dma_buf *dmabuf, int flags);
+void dma_buf_fd_install(struct dma_buf *dmabuf, int fd);
struct dma_buf *dma_buf_get(int fd);
void dma_buf_put(struct dma_buf *dmabuf);
diff --git a/include/linux/dma-fence-array.h b/include/linux/dma-fence-array.h
index 1b1d87579c38..0c49d7ccefb6 100644
--- a/include/linux/dma-fence-array.h
+++ b/include/linux/dma-fence-array.h
@@ -28,7 +28,6 @@ struct dma_fence_array_cb {
/**
* struct dma_fence_array - fence to represent an array of fences
* @base: fence base class
- * @lock: spinlock for fence handling
* @num_fences: number of fences in the array
* @num_pending: fences in the array still pending
* @fences: array of the fences
diff --git a/include/linux/dma-fence-chain.h b/include/linux/dma-fence-chain.h
index df3beadf1515..705c4394ac0d 100644
--- a/include/linux/dma-fence-chain.h
+++ b/include/linux/dma-fence-chain.h
@@ -20,7 +20,6 @@
* @prev: previous fence of the chain
* @prev_seqno: original previous seqno before garbage collection
* @fence: encapsulated fence
- * @lock: spinlock for fence handling
*/
struct dma_fence_chain {
struct dma_fence base;
@@ -81,9 +80,8 @@ dma_fence_chain_contained(struct dma_fence *fence)
}
/**
- * dma_fence_chain_alloc
- *
- * Returns a new struct dma_fence_chain object or NULL on failure.
+ * dma_fence_chain_alloc - Returns a new &struct dma_fence_chain object or
+ * %NULL on failure.
*
* This specialized allocator has to be a macro for its allocations to be
* accounted separately (to have a separate alloc_tag). The typecast is
@@ -93,7 +91,8 @@ dma_fence_chain_contained(struct dma_fence *fence)
kmalloc_obj(struct dma_fence_chain)
/**
- * dma_fence_chain_free
+ * dma_fence_chain_free - Frees an allocated but not used
+ * &struct dma_fence_chain object.
* @chain: chain node to free
*
* Frees up an allocated but not used struct dma_fence_chain object. This
diff --git a/include/linux/dma-fence.h b/include/linux/dma-fence.h
index 158cd609f103..dd07d128adc3 100644
--- a/include/linux/dma-fence.h
+++ b/include/linux/dma-fence.h
@@ -141,6 +141,9 @@ struct dma_fence_ops {
* compute the name at runtime, without having it to store permanently
* for each fence, or build a cache of some sort.
*
+ * The returned string is RCU protected and can be freed after the fence
+ * signaled and a RCU grace period passed.
+ *
* This callback is mandatory.
*/
const char * (*get_driver_name)(struct dma_fence *fence);
@@ -153,6 +156,9 @@ struct dma_fence_ops {
* having it to store permanently for each fence, or build a cache of
* some sort.
*
+ * The returned string is RCU protected and can be freed after the fence
+ * signaled and a RCU grace period passed.
+ *
* This callback is mandatory.
*/
const char * (*get_timeline_name)(struct dma_fence *fence);
@@ -501,7 +507,7 @@ dma_fence_test_signaled_flag(struct dma_fence *fence)
* Returns true if the fence was already signaled, false if not. Since this
* function doesn't enable signaling, it is not guaranteed to ever return
* true if dma_fence_add_callback(), dma_fence_wait() or
- * dma_fence_enable_sw_signaling() haven't been called before.
+ * dma_fence_enable_signaling() haven't been called before.
*
* This function requires &dma_fence.lock to be held.
*
diff --git a/include/linux/dma-map-ops.h b/include/linux/dma-map-ops.h
index 8fae2b7deb20..afaf2c12189b 100644
--- a/include/linux/dma-map-ops.h
+++ b/include/linux/dma-map-ops.h
@@ -60,7 +60,7 @@ struct dma_map_ops {
int (*dma_supported)(struct device *dev, u64 mask);
u64 (*get_required_mask)(struct device *dev);
size_t (*max_mapping_size)(struct device *dev);
- size_t (*opt_mapping_size)(void);
+ size_t (*max_opt_mapping_size)(void);
unsigned long (*get_merge_boundary)(struct device *dev);
};
@@ -143,7 +143,7 @@ static inline struct page *dma_alloc_contiguous(struct device *dev, size_t size,
static inline void dma_free_contiguous(struct device *dev, struct page *page,
size_t size)
{
- __free_pages(page, get_order(size));
+ free_pages_exact(page_address(page), size);
}
#endif /* CONFIG_DMA_CMA*/
diff --git a/include/linux/dma-mapping.h b/include/linux/dma-mapping.h
index a3e880649fa4..efde0e5a58dc 100644
--- a/include/linux/dma-mapping.h
+++ b/include/linux/dma-mapping.h
@@ -207,7 +207,7 @@ int dma_set_coherent_mask(struct device *dev, u64 mask);
u64 dma_get_required_mask(struct device *dev);
bool dma_addressing_limited(struct device *dev);
size_t dma_max_mapping_size(struct device *dev);
-size_t dma_opt_mapping_size(struct device *dev);
+size_t dma_max_opt_mapping_size(struct device *dev);
unsigned long dma_get_merge_boundary(struct device *dev);
struct sg_table *dma_alloc_noncontiguous(struct device *dev, size_t size,
enum dma_data_direction dir, gfp_t gfp, unsigned long attrs);
@@ -326,7 +326,7 @@ static inline size_t dma_max_mapping_size(struct device *dev)
{
return 0;
}
-static inline size_t dma_opt_mapping_size(struct device *dev)
+static inline size_t dma_max_opt_mapping_size(struct device *dev)
{
return 0;
}
diff --git a/include/linux/dmaengine.h b/include/linux/dmaengine.h
index fe33a20abc61..f322e501ff62 100644
--- a/include/linux/dmaengine.h
+++ b/include/linux/dmaengine.h
@@ -1760,15 +1760,6 @@ void dma_run_dependencies(struct dma_async_tx_descriptor *tx);
#define dma_request_channel(mask, x, y) \
__dma_request_channel(&(mask), x, y, NULL)
-/* Deprecated, please use dma_request_chan() directly */
-static inline struct dma_chan * __deprecated
-dma_request_slave_channel(struct device *dev, const char *name)
-{
- struct dma_chan *ch = dma_request_chan(dev, name);
-
- return IS_ERR(ch) ? NULL : ch;
-}
-
static inline struct dma_chan
*dma_request_slave_channel_compat(const dma_cap_mask_t mask,
dma_filter_fn fn, void *fn_param,
diff --git a/include/linux/edac.h b/include/linux/edac.h
index e6b4e51130e5..f7a8218f9cc0 100644
--- a/include/linux/edac.h
+++ b/include/linux/edac.h
@@ -598,9 +598,6 @@ struct mem_ctl_info {
int op_state;
struct dentry *debugfs;
- u8 fake_inject_layer[EDAC_MAX_LAYERS];
- bool fake_inject_ue;
- u16 fake_inject_count;
/*
* Memory Controller hierarchy
diff --git a/include/linux/entry-common.h b/include/linux/entry-common.h
index 6574b7183c01..fa2854fed1f2 100644
--- a/include/linux/entry-common.h
+++ b/include/linux/entry-common.h
@@ -102,7 +102,14 @@ static __always_inline long syscall_trace_enter(struct pt_regs *regs, unsigned l
if (unlikely(work & SYSCALL_WORK_SYSCALL_TRACEPOINT))
trace_syscall_enter(regs);
- if (unlikely(audit_context()))
+ /*
+ * The config check works around broken compilers which fail to
+ * eliminate the dead code in case of CONFIG_AUDITSYSCALL=n as they
+ * insist on creating a always false runtime condition based on
+ * audit_context() which returns NULL in that case. The explicit
+ * IS_ENABLED() check makes that madness go away.
+ */
+ if (IS_ENABLED(CONFIG_AUDITSYSCALL) && unlikely(audit_context()))
syscall_enter_audit(regs);
return true;
diff --git a/include/linux/ethtool.h b/include/linux/ethtool.h
index 12683b5d125e..c3a41fbd5426 100644
--- a/include/linux/ethtool.h
+++ b/include/linux/ethtool.h
@@ -944,6 +944,7 @@ struct kernel_ethtool_ts_info {
#define ETHTOOL_OP_NEEDS_RTNL_SPAUSEPARAM BIT(6)
#define ETHTOOL_OP_NEEDS_RTNL_RSS BIT(7)
#define ETHTOOL_OP_NEEDS_RTNL_GLINK BIT(8)
+#define ETHTOOL_OP_NEEDS_RTNL_TEST BIT(9)
/**
* struct ethtool_ops - optional netdev operations
@@ -981,6 +982,7 @@ struct kernel_ethtool_ts_info {
* - netdev_update_features()
* - netif_set_real_num_tx_queues()
* - ethtool_op_get_link() (syncs link watch under rtnl_lock)
+ * - netif_open() / netif_close() (used by @self_test)
*
* @get_drvinfo: Report driver/device information. Modern drivers no
* longer have to implement this callback. Most fields are
@@ -1023,7 +1025,9 @@ struct kernel_ethtool_ts_info {
* types should be set in @supported_coalesce_params.
* Returns a negative error code or zero.
* @get_ringparam: Report ring sizes
- * @set_ringparam: Set ring sizes. Returns a negative error code or zero.
+ * @set_ringparam: Set ring sizes. The &struct ethtool_ringparam argument is
+ * also an output; drivers which normalize requested sizes must update it
+ * with the applied sizes. Returns a negative error code or zero.
* @get_pause_stats: Report pause frame statistics. Drivers must not zero
* statistics which they don't report. The stats structure is initialized
* to ETHTOOL_STAT_NOT_SET indicating driver does not report statistics.
@@ -1057,6 +1061,12 @@ struct kernel_ethtool_ts_info {
* @get_sset_count: Get number of strings that @get_strings will write.
* @get_rxnfc: Get RX flow classification rules. Returns a negative
* error code or zero.
+ * Note that for %ETHTOOL_GRXCLSRLALL rule_cnt and size of the arrays
+ * is user-provided, and not guaranteed to match what driver would
+ * have reported via %ETHTOOL_GRXCLSRLCNT. Drivers must return -%EMSGSIZE
+ * when rule_cnt is too small. rule_locs is %NULL when rule_cnt is zero.
+ * On success drivers must set rule_cnt to the number of locations they
+ * filled in, the core copies out exactly that many.
* @set_rxnfc: Set RX flow classification rules. Returns a negative
* error code or zero.
* @flash_device: Write a firmware image to device's flash memory.
diff --git a/include/linux/f2fs_fs.h b/include/linux/f2fs_fs.h
index bb2b6cd5d507..3081702b1ddb 100644
--- a/include/linux/f2fs_fs.h
+++ b/include/linux/f2fs_fs.h
@@ -14,9 +14,9 @@
#define F2FS_SUPER_OFFSET 1024 /* byte-size offset */
#define F2FS_MIN_LOG_SECTOR_SIZE 9 /* 9 bits for 512 bytes */
#define F2FS_MAX_LOG_SECTOR_SIZE PAGE_SHIFT /* Max is Block Size */
-#define F2FS_LOG_SECTORS_PER_BLOCK (PAGE_SHIFT - 9) /* log number for sector/blk */
-#define F2FS_BLKSIZE PAGE_SIZE /* support only block == page */
-#define F2FS_BLKSIZE_BITS PAGE_SHIFT /* bits for F2FS_BLKSIZE */
+#define F2FS_MIN_LOG_BLOCKSIZE 12
+#define F2FS_MIN_BLKSIZE 4096UL
+#define F2FS_MAX_BLKSIZE PAGE_SIZE
#define F2FS_MAX_EXTENSION 64 /* # of extension entries */
#define F2FS_EXTENSION_LEN 8 /* max size of extension */
@@ -24,19 +24,25 @@
#define NEW_ADDR ((block_t)-1) /* used as block_t addresses */
#define COMPRESS_ADDR ((block_t)-2) /* used as compressed data flag */
-#define F2FS_BLKSIZE_MASK (F2FS_BLKSIZE - 1)
-#define F2FS_BYTES_TO_BLK(bytes) ((unsigned long long)(bytes) >> F2FS_BLKSIZE_BITS)
-#define F2FS_BLK_TO_BYTES(blk) ((unsigned long long)(blk) << F2FS_BLKSIZE_BITS)
-#define F2FS_BLK_END_BYTES(blk) (F2FS_BLK_TO_BYTES(blk + 1) - 1)
-#define F2FS_BLK_ALIGN(x) (F2FS_BYTES_TO_BLK((x) + F2FS_BLKSIZE - 1))
+#define F2FS_BLKSIZE(sbi) ((sbi)->blocksize)
+#define F2FS_BLKSIZE_BITS(sbi) ((sbi)->log_blocksize)
+#define F2FS_BLKSIZE_MASK(sbi) (F2FS_BLKSIZE(sbi) - 1)
+#define F2FS_LOG_SECTORS_PER_BLOCK(sbi) (F2FS_BLKSIZE_BITS(sbi) - 9)
+#define F2FS_BLKS_PER_PAGE(sbi) (PAGE_SIZE / F2FS_BLKSIZE(sbi))
+#define F2FS_BYTES_TO_BLK(sbi, bytes) \
+ ((unsigned long long)(bytes) >> F2FS_BLKSIZE_BITS(sbi))
+#define F2FS_BLK_TO_BYTES(sbi, blk) \
+ ((unsigned long long)(blk) << F2FS_BLKSIZE_BITS(sbi))
+#define F2FS_BLK_END_BYTES(sbi, blk) \
+ (F2FS_BLK_TO_BYTES(sbi, (blk) + 1) - 1)
+#define F2FS_BLK_ALIGN(sbi, bytes) \
+ F2FS_BYTES_TO_BLK(sbi, (unsigned long long)(bytes) + \
+ F2FS_BLKSIZE(sbi) - 1)
/* 0, 1(node nid), 2(meta nid) are reserved node id */
#define F2FS_RESERVED_NODE_NUM 3
#define F2FS_ROOT_INO(sbi) ((sbi)->root_ino_num)
-#define F2FS_NODE_INO(sbi) ((sbi)->node_ino_num)
-#define F2FS_META_INO(sbi) ((sbi)->meta_ino_num)
-#define F2FS_COMPRESS_INO(sbi) (NM_I(sbi)->max_nid)
#define F2FS_MAX_QUOTAS 3
@@ -214,20 +220,27 @@ struct f2fs_checkpoint {
unsigned char sit_nat_version_bitmap[];
} __packed;
-#define CP_CHKSUM_OFFSET (F2FS_BLKSIZE - sizeof(__le32)) /* default chksum offset in checkpoint */
#define CP_MIN_CHKSUM_OFFSET \
(offsetof(struct f2fs_checkpoint, sit_nat_version_bitmap))
/*
* For orphan inode management
+ *
+ * The number of inode entries in an orphan block depends on the filesystem
+ * block size. Its exact on-disk layout is:
+ *
+ * 0 blocksize - 16 blocksize
+ * +--------------------------+--------------------------+
+ * | ino[0] ... ino[n - 1] | struct f2fs_orphan_footer |
+ * +--------------------------+--------------------------+
+ *
+ * n = (blocksize - sizeof(struct f2fs_orphan_footer)) / sizeof(__le32)
*/
-#define F2FS_ORPHANS_PER_BLOCK ((F2FS_BLKSIZE - 4 * sizeof(__le32)) / sizeof(__le32))
-
-#define GET_ORPHAN_BLOCKS(n) (((n) + F2FS_ORPHANS_PER_BLOCK - 1) / \
- F2FS_ORPHANS_PER_BLOCK)
-
struct f2fs_orphan_block {
- __le32 ino[F2FS_ORPHANS_PER_BLOCK]; /* inode numbers */
+ DECLARE_FLEX_ARRAY(__le32, ino);
+} __packed;
+
+struct f2fs_orphan_footer {
__le32 reserved; /* reserved */
__le16 blk_addr; /* block index in current CP */
__le16 blk_count; /* Number of orphan inode blocks in CP */
@@ -260,26 +273,14 @@ struct node_footer {
} __packed;
/* Address Pointers in an Inode */
-#define DEF_ADDRS_PER_INODE ((F2FS_BLKSIZE - OFFSET_OF_END_OF_I_EXT \
- - SIZE_OF_I_NID \
- - sizeof(struct node_footer)) / sizeof(__le32))
-#define CUR_ADDRS_PER_INODE(inode) (DEF_ADDRS_PER_INODE - \
- get_extra_isize(inode))
+#define F2FS_DEF_ADDRS_PER_INODE(blocksize) \
+ (((blocksize) - OFFSET_OF_END_OF_I_EXT - SIZE_OF_I_NID - \
+ sizeof(struct node_footer)) / sizeof(__le32))
#define DEF_NIDS_PER_INODE 5 /* Node IDs in an Inode */
#define ADDRS_PER_INODE(inode) addrs_per_page(inode, true)
/* Address Pointers in a Direct Block */
-#define DEF_ADDRS_PER_BLOCK ((F2FS_BLKSIZE - sizeof(struct node_footer)) / sizeof(__le32))
#define ADDRS_PER_BLOCK(inode) addrs_per_page(inode, false)
-/* Node IDs in an Indirect Block */
-#define NIDS_PER_BLOCK ((F2FS_BLKSIZE - sizeof(struct node_footer)) / sizeof(__le32))
-
-#define ADDRS_PER_PAGE(folio, inode) (addrs_per_page(inode, IS_INODE(folio)))
-
-#define NODE_DIR1_BLOCK (DEF_ADDRS_PER_INODE + 1)
-#define NODE_DIR2_BLOCK (DEF_ADDRS_PER_INODE + 2)
-#define NODE_IND1_BLOCK (DEF_ADDRS_PER_INODE + 3)
-#define NODE_IND2_BLOCK (DEF_ADDRS_PER_INODE + 4)
-#define NODE_DIND_BLOCK (DEF_ADDRS_PER_INODE + 5)
+#define ADDRS_PER_PAGE(folio, inode) (addrs_per_page(inode, IS_INODE(F2FS_I_SB(inode), folio)))
#define F2FS_INLINE_XATTR 0x01 /* file inline xattr flag */
#define F2FS_INLINE_DATA 0x02 /* file inline data flag */
@@ -339,18 +340,26 @@ struct f2fs_inode {
*/
__le32 i_extra_end[0]; /* for attribute size calculation */
} __packed;
- __le32 i_addr[DEF_ADDRS_PER_INODE]; /* Pointers to data blocks */
+ DECLARE_FLEX_ARRAY(__le32, i_addr); /* data block pointers */
};
- __le32 i_nid[DEF_NIDS_PER_INODE]; /* direct(2), indirect(2),
- double_indirect(1) node id */
+ /*
+ * __le32 i_nid[DEF_NIDS_PER_INODE];
+ * direct(2), indirect(2), double_indirect(1) node IDs
+ *
+ * It is stored immediately before the node footer at the end of the
+ * filesystem block. Its offset depends on the filesystem block size, so
+ * locate it dynamically with F2FS_INODE_NIDS().
+ */
} __packed;
struct direct_node {
- __le32 addr[DEF_ADDRS_PER_BLOCK]; /* array of data block address */
+ /* The address count depends on the filesystem block size. */
+ DECLARE_FLEX_ARRAY(__le32, addr); /* array of data block address */
} __packed;
struct indirect_node {
- __le32 nid[NIDS_PER_BLOCK]; /* array of data block address */
+ /* The node ID count depends on the filesystem block size. */
+ DECLARE_FLEX_ARRAY(__le32, nid); /* array of data block address */
} __packed;
enum {
@@ -369,14 +378,18 @@ struct f2fs_node {
struct direct_node dn;
struct indirect_node in;
};
- struct node_footer footer;
+ /*
+ * struct node_footer footer;
+ *
+ * It is stored at the end of the filesystem block, after the inode or
+ * direct/indirect node data. Its offset depends on the filesystem block
+ * size, so locate it dynamically with F2FS_NODE_FOOTER().
+ */
} __packed;
/*
* For NAT entries
*/
-#define NAT_ENTRY_PER_BLOCK (F2FS_BLKSIZE / sizeof(struct f2fs_nat_entry))
-
struct f2fs_nat_entry {
__u8 version; /* latest version of cached nat entry */
__le32 ino; /* inode number */
@@ -384,7 +397,8 @@ struct f2fs_nat_entry {
} __packed;
struct f2fs_nat_block {
- struct f2fs_nat_entry entries[NAT_ENTRY_PER_BLOCK];
+ /* The entry count depends on the filesystem block size. */
+ DECLARE_FLEX_ARRAY(struct f2fs_nat_entry, entries);
} __packed;
/*
@@ -396,8 +410,6 @@ struct f2fs_nat_block {
* Not allow to change this.
*/
#define SIT_VBLOCK_MAP_SIZE 64
-#define SIT_ENTRY_PER_BLOCK (F2FS_BLKSIZE / sizeof(struct f2fs_sit_entry))
-
/*
* F2FS uses 4 bytes to represent block address. As a result, supported size of
* disk is 16 TB for a 4K page size and 64 TB for a 16K page size and it equals
@@ -424,8 +436,13 @@ struct f2fs_sit_entry {
__le64 mtime; /* segment age for cleaning */
} __packed;
+/*
+ * The on-disk SIT block is a filesystem-block-sized array of SIT entries.
+ * Its entry count depends on the filesystem block size, so it must be
+ * calculated by the caller rather than implied by this C structure.
+ */
struct f2fs_sit_block {
- struct f2fs_sit_entry entries[SIT_ENTRY_PER_BLOCK];
+ DECLARE_FLEX_ARRAY(struct f2fs_sit_entry, entries);
} __packed;
/*
@@ -595,15 +612,7 @@ typedef __le32 f2fs_hash_t;
* dentry, when converting inline dentry we should handle this carefully.
*/
-/* the number of dentry in a block */
-#define NR_DENTRY_IN_BLOCK ((BITS_PER_BYTE * F2FS_BLKSIZE) / \
- ((SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * BITS_PER_BYTE + 1))
#define SIZE_OF_DIR_ENTRY 11 /* by byte */
-#define SIZE_OF_DENTRY_BITMAP ((NR_DENTRY_IN_BLOCK + BITS_PER_BYTE - 1) / \
- BITS_PER_BYTE)
-#define SIZE_OF_RESERVED (F2FS_BLKSIZE - ((SIZE_OF_DIR_ENTRY + \
- F2FS_SLOT_LEN) * \
- NR_DENTRY_IN_BLOCK + SIZE_OF_DENTRY_BITMAP))
#define MIN_INLINE_DENTRY_SIZE 40 /* just include '.' and '..' entries */
/* One directory entry slot representing F2FS_SLOT_LEN-sized file name */
@@ -614,14 +623,21 @@ struct f2fs_dir_entry {
__u8 file_type; /* file type */
} __packed;
-/* Block-sized directory entry block */
-struct f2fs_dentry_block {
- /* validity bitmap for directory entries in each block */
- __u8 dentry_bitmap[SIZE_OF_DENTRY_BITMAP];
- __u8 reserved[SIZE_OF_RESERVED];
- struct f2fs_dir_entry dentry[NR_DENTRY_IN_BLOCK];
- __u8 filename[NR_DENTRY_IN_BLOCK][F2FS_SLOT_LEN];
-} __packed;
+/*
+ * A dentry block is laid out as follows, where the number of entries and all
+ * offsets are determined by the filesystem block size at runtime:
+ *
+ * 0 blocksize
+ * +--------+----------+-------------------+-----------------------+
+ * | bitmap | reserved | dir_entry[entries]| filename[entries][8] |
+ * +--------+----------+-------------------+-----------------------+
+ *
+ * entries = (BITS_PER_BYTE * blocksize) /
+ * ((SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * BITS_PER_BYTE + 1)
+ * bitmap_size = DIV_ROUND_UP(entries, BITS_PER_BYTE)
+ * reserved_size = blocksize - bitmap_size -
+ * (SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * entries
+ */
#define F2FS_DEF_PROJID 0 /* default project ID */
diff --git a/include/linux/fault-inject.h b/include/linux/fault-inject.h
index 58fd14c82270..5c74748a53f3 100644
--- a/include/linux/fault-inject.h
+++ b/include/linux/fault-inject.h
@@ -19,6 +19,13 @@ enum fault_flags {
#include <linux/ratelimit.h>
/*
+ * Length of the debugfs directory name embedded in struct fault_attr.
+ * Chosen to accommodate every in-tree caller of fault_create_debugfs_attr()
+ * (the longest is "fail_dma_array_full", 19 chars) with generous headroom.
+ */
+#define FAULT_ATTR_DNAME_LEN 64
+
+/*
* For explanation of the elements of this struct, see
* Documentation/fault-injection/fault-injection.rst
*/
@@ -37,7 +44,7 @@ struct fault_attr {
unsigned long count;
struct ratelimit_state ratelimit_state;
- struct dentry *dname;
+ char dname[FAULT_ATTR_DNAME_LEN];
};
#define FAULT_ATTR_INITIALIZER { \
@@ -47,7 +54,6 @@ struct fault_attr {
.stacktrace_depth = 32, \
.ratelimit_state = RATELIMIT_STATE_INIT_DISABLED, \
.verbose = 2, \
- .dname = NULL, \
}
#define DECLARE_FAULT_ATTR(name) struct fault_attr name = FAULT_ATTR_INITIALIZER
diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h
index c45306a9f007..a46781058729 100644
--- a/include/linux/fdtable.h
+++ b/include/linux/fdtable.h
@@ -25,7 +25,7 @@
struct fdtable {
unsigned int max_fds;
- struct file __rcu **fd; /* current fd array */
+ struct file __rcu **fd __counted_by_ptr(max_fds); /* current fd array */
unsigned long *close_on_exec;
unsigned long *open_fds;
unsigned long *full_fds_bits;
@@ -101,11 +101,22 @@ struct task_struct;
void put_files_struct(struct files_struct *fs);
int unshare_files(void);
+void switch_files_struct(struct task_struct *tsk, struct files_struct *files);
+int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp);
+enum fd_range_flags {
+ /* Leave behind all descriptors outside of the specified range. */
+ FD_RANGE_EXCEPT = (1U << 0),
+
+ /* Only select descriptors that have close-on-exec set. */
+ FD_RANGE_CLOEXEC_ONLY = (1U << 1),
+};
+
struct fd_range {
unsigned int from, to;
+ enum fd_range_flags flags;
};
struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy;
-void do_close_on_exec(struct files_struct *);
+void close_cloexec_files(struct files_struct *);
int iterate_fd(struct files_struct *, unsigned,
int (*)(const void *, struct file *, unsigned),
const void *);
diff --git a/include/linux/file.h b/include/linux/file.h
index 27484b444d31..41c3c0be1064 100644
--- a/include/linux/file.h
+++ b/include/linux/file.h
@@ -12,6 +12,7 @@
#include <linux/errno.h>
#include <linux/cleanup.h>
#include <linux/err.h>
+#include <linux/vfsdebug.h>
struct file;
@@ -129,117 +130,84 @@ extern unsigned int sysctl_nr_open_min, sysctl_nr_open_max;
/*
* fd_prepare: Combined fd + file allocation cleanup class.
- * @err: Error code to indicate if allocation succeeded.
- * @__fd: Allocated fd (may not be accessed directly)
- * @__file: Allocated struct file pointer (may not be accessed directly)
+ * @fd: Allocated fd
+ * @file: Allocated struct file pointer
*
* Allocates an fd and a file together. On error paths, automatically cleans
* up whichever resource was successfully allocated. Allows flexible file
* allocation with different functions per usage.
*
- * Do not use directly.
+ * Do not declare directly, use FD_PREPARE().
*/
struct fd_prepare {
- s32 err;
- s32 __fd; /* do not access directly */
- struct file *__file; /* do not access directly */
+ int fd;
+ struct file *file;
};
-/* Typedef for fd_prepare cleanup guards. */
-typedef struct fd_prepare class_fd_prepare_t;
-
-/*
- * Accessors for fd_prepare class members.
- * _Generic() is used for zero-cost type safety.
- */
-#define fd_prepare_fd(_fdf) \
- (_Generic((_fdf), struct fd_prepare: (_fdf).__fd))
-
-#define fd_prepare_file(_fdf) \
- (_Generic((_fdf), struct fd_prepare: (_fdf).__file))
-
/* Do not use directly. */
-static inline void class_fd_prepare_destructor(const struct fd_prepare *fdf)
+static __always_inline void __fd_prepare_cleanup(const struct fd_prepare *fdf)
{
- if (unlikely(fdf->__fd >= 0))
- put_unused_fd(fdf->__fd);
- if (unlikely(!IS_ERR_OR_NULL(fdf->__file)))
- fput(fdf->__file);
+ if (unlikely(fdf->fd >= 0)) {
+ put_unused_fd(fdf->fd);
+ fput(fdf->file);
+ }
}
/* Do not use directly. */
-static inline int class_fd_prepare_lock_err(const struct fd_prepare *fdf)
+static __always_inline struct fd_prepare __fd_prepare(int fd, struct file *file)
{
- if (unlikely(fdf->err))
- return fdf->err;
- if (unlikely(fdf->__fd < 0))
- return fdf->__fd;
- if (unlikely(IS_ERR(fdf->__file)))
- return PTR_ERR(fdf->__file);
- if (unlikely(!fdf->__file))
- return -ENOMEM;
- return 0;
+ if (fd >= 0 && IS_ERR_OR_NULL(file)) {
+ int err = file ? PTR_ERR(file) : -ENOMEM;
+
+ put_unused_fd(fd);
+ fd = err;
+ file = NULL;
+ }
+ return (struct fd_prepare){ .fd = fd, .file = file };
}
/*
- * __FD_PREPARE_INIT - Helper to initialize fd_prepare class.
- * @_fd_flags: flags for get_unused_fd_flags()
- * @_file_owned: expression that returns struct file *
- *
- * Returns a struct fd_prepare with fd, file, and err set.
- * If fd allocation fails, fd will be negative and err will be set. If
- * fd succeeds but file_init_expr fails, file will be ERR_PTR and err
- * will be set. The err field is the single source of truth for error
- * checking.
- */
-#define __FD_PREPARE_INIT(_fd_flags, _file_owned) \
- ({ \
- struct fd_prepare fdf = { \
- .__fd = get_unused_fd_flags((_fd_flags)), \
- }; \
- if (likely(fdf.__fd >= 0)) \
- fdf.__file = (_file_owned); \
- fdf.err = ACQUIRE_ERR(fd_prepare, &fdf); \
- fdf; \
- })
-
-/*
- * FD_PREPARE - Macro to declare and initialize an fd_prepare variable.
+ * FD_PREPARE - Declare and initialize an fd_prepare instance.
*
- * Declares and initializes an fd_prepare variable with automatic
- * cleanup. No separate scope required - cleanup happens when variable
- * goes out of scope.
+ * This allocates a new fd and only evaluates @_file_owned if the
+ * allocation succeeded. Cleanup happens when the variable goes out of
+ * scope and the guard releases whichever of the descriptor and the file
+ * was allocated. If fd_publish() was called the fd and file are
+ * published and cleanup becomes a nop.
*
- * @_fdf: name of struct fd_prepare variable to define
+ * @_fdf: name of the const struct fd_prepare pointer to define
* @_fd_flags: flags for get_unused_fd_flags()
* @_file_owned: struct file to take ownership of (can be expression)
*/
+#define __FD_PREPARE(_guard, _fdf, _fd_flags, _file_owned) \
+ struct fd_prepare _guard __cleanup(__fd_prepare_cleanup) = ({ \
+ int __fd = get_unused_fd_flags(_fd_flags); \
+ __fd_prepare(__fd, __fd < 0 ? NULL : (_file_owned)); \
+ }); \
+ const struct fd_prepare *const _fdf = &_guard
+
#define FD_PREPARE(_fdf, _fd_flags, _file_owned) \
- CLASS_INIT(fd_prepare, _fdf, __FD_PREPARE_INIT(_fd_flags, _file_owned))
+ __FD_PREPARE(__UNIQUE_ID(fd_prepare), _fdf, _fd_flags, _file_owned)
/*
* fd_publish - Publish prepared fd and file to the fd table.
- * @_fdf: struct fd_prepare variable
+ * @fdf: struct fd_prepare pointer defined by FD_PREPARE()
*/
-#define fd_publish(_fdf) \
- ({ \
- struct fd_prepare *fdp = &(_fdf); \
- VFS_WARN_ON_ONCE(fdp->err); \
- VFS_WARN_ON_ONCE(fdp->__fd < 0); \
- VFS_WARN_ON_ONCE(IS_ERR_OR_NULL(fdp->__file)); \
- fd_install(fdp->__fd, fdp->__file); \
- retain_and_null_ptr(fdp->__file); \
- take_fd(fdp->__fd); \
- })
+static __always_inline int fd_publish(const struct fd_prepare *fdf)
+{
+ /* Callers only get a const view, the guard itself is writable. */
+ struct fd_prepare *guard = (struct fd_prepare *)fdf;
+
+ VFS_WARN_ON_ONCE(guard->fd < 0);
+ fd_install(guard->fd, guard->file);
+ return take_fd(guard->fd);
+}
/* Do not use directly. */
-#define __FD_ADD(_fdf, _fd_flags, _file_owned) \
- ({ \
- FD_PREPARE(_fdf, _fd_flags, _file_owned); \
- s32 ret = _fdf.err; \
- if (likely(!ret)) \
- ret = fd_publish(_fdf); \
- ret; \
+#define __FD_ADD(_fdf, _fd_flags, _file_owned) \
+ ({ \
+ FD_PREPARE(_fdf, _fd_flags, _file_owned); \
+ _fdf->fd < 0 ? _fdf->fd : fd_publish(_fdf); \
})
/*
diff --git a/include/linux/fileattr.h b/include/linux/fileattr.h
index 58044b598016..09e32b84e02a 100644
--- a/include/linux/fileattr.h
+++ b/include/linux/fileattr.h
@@ -74,7 +74,7 @@ static inline bool fileattr_has_fsx(const struct file_kattr *fa)
}
int vfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa);
-int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry,
+int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry,
struct file_kattr *fa);
int ioctl_getflags(struct file *file, unsigned int __user *argp);
int ioctl_setflags(struct file *file, unsigned int __user *argp);
diff --git a/include/linux/filter.h b/include/linux/filter.h
index 4a9bc6a848f2..e42eccb0990e 100644
--- a/include/linux/filter.h
+++ b/include/linux/filter.h
@@ -98,6 +98,11 @@ struct ctl_table_header;
/* BPF program can access up to 512 bytes of stack space. */
#define MAX_BPF_STACK 512
+/*
+ * Stack budget of a program on a JIT that lays out frames of that size.
+ * The interpreter and JITs without such support keep MAX_BPF_STACK.
+ */
+#define MAX_BPF_STACK_JIT 2048
/* Helper macros for filter block array initializers. */
@@ -1237,6 +1242,8 @@ bool bpf_jit_inlines_helper_call(s32 imm);
bool bpf_jit_supports_subprog_tailcalls(void);
bool bpf_jit_supports_percpu_insn(void);
bool bpf_jit_supports_kfunc_call(void);
+bool bpf_jit_supports_callx(void);
+bool bpf_jit_supports_kfunc_ret_reg_pair(void);
bool bpf_jit_supports_stack_args(void);
bool bpf_jit_supports_arena_args(void);
bool bpf_jit_supports_far_kfunc_call(void);
@@ -1245,8 +1252,41 @@ bool bpf_jit_supports_ptr_xchg(void);
bool bpf_jit_supports_arena(void);
bool bpf_jit_supports_insn(struct bpf_insn *insn, bool in_arena);
bool bpf_jit_supports_private_stack(void);
+bool bpf_jit_supports_large_stack(void);
bool bpf_jit_supports_timed_may_goto(void);
bool bpf_jit_supports_fsession(void);
+
+struct bpf_jit_arg_abi {
+ /* Argument registers of the kernel convention. */
+ u8 nr_arg_regs;
+ /* Round the register number up to an even one for 16-byte alignment. */
+ bool even_reg_align;
+ /* Round the stack slot up to an even one for 16-byte alignment. */
+ bool even_stack_align;
+ /* An argument may straddle the last register and the stack. */
+ bool split_at_boundary;
+ /* A later argument may reuse a register a stack-passed one skipped. */
+ bool backfill_after_stack;
+};
+
+const struct bpf_jit_arg_abi *bpf_jit_arg_abi(void);
+u32 bpf_jit_place_args(const struct bpf_jit_arg_abi *abi,
+ const struct btf_func_model *fm, u8 *pos_of_slot);
+
+/* The JIT's scratch register, in place of an argument slot. */
+#define BPF_JIT_ARG_TMP 0xff
+
+/* Every argument slot moves at most once, and the scratch goes out and back. */
+#define BPF_JIT_MAX_ARG_MOVES (MAX_BPF_FUNC_ARG_SLOTS + 2)
+
+struct bpf_jit_arg_move {
+ u8 dst;
+ u8 src;
+};
+
+u32 bpf_jit_plan_arg_moves(const struct bpf_jit_arg_abi *abi,
+ const struct btf_func_model *fm,
+ struct bpf_jit_arg_move *moves);
u64 bpf_arch_uaddress_limit(void);
void arch_bpf_stack_walk(bool (*consume_fn)(void *cookie, u64 ip, u64 sp, u64 bp), void *cookie);
u64 arch_bpf_timed_may_goto(void);
@@ -1376,7 +1416,6 @@ bpf_jit_binary_alloc(unsigned int proglen, u8 **image_ptr,
void bpf_jit_binary_free(struct bpf_binary_header *hdr);
u64 bpf_jit_alloc_exec_limit(void);
void *bpf_jit_alloc_exec(unsigned long size);
-void *bpf_jit_alloc_exec_rw(unsigned long size);
void bpf_jit_free_exec(void *addr);
void bpf_jit_free(struct bpf_prog *fp);
struct bpf_binary_header *
@@ -1898,6 +1937,11 @@ static __always_inline long __bpf_xdp_redirect_map(struct bpf_map *map, u64 inde
return XDP_REDIRECT;
}
+int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock,
+ void *search_res, u32 len, u64 flags);
+int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock,
+ const void *from, u32 len, u64 flags);
+
#ifdef CONFIG_NET
int __bpf_skb_load_bytes(const struct sk_buff *skb, u32 offset, void *to, u32 len);
int __bpf_skb_store_bytes(struct sk_buff *skb, u32 offset, const void *from,
diff --git a/include/linux/firewire.h b/include/linux/firewire.h
index fd35a6570cd8..dbb5c825a1cc 100644
--- a/include/linux/firewire.h
+++ b/include/linux/firewire.h
@@ -172,10 +172,10 @@ struct fw_attribute_group {
};
enum fw_device_quirk {
- // See afa1282a35d3 ("firewire: core: check for 1394a compliant IRM, fix inaccessibility of Sony camcorder").
+ // See 10389536742c ("firewire: core: check for 1394a compliant IRM, fix inaccessibility of Sony camcorder").
FW_DEVICE_QUIRK_IRM_IS_1394_1995_ONLY = BIT(0),
- // See a509e43ff338 ("firewire: core: fix unstable I/O with Canon camcorder").
+ // See 6044565af458 ("firewire: core: fix unstable I/O with Canon camcorder").
FW_DEVICE_QUIRK_IRM_IGNORES_BUS_MANAGER = BIT(1),
// MOTU Audio Express transfers acknowledge packet with 0x10 for pending state.
@@ -224,7 +224,7 @@ struct fw_device {
struct mutex client_list_mutex;
struct list_head client_list;
- const u32 *config_rom;
+ const u32 *config_rom __counted_by_ptr(config_rom_length);
size_t config_rom_length;
int config_rom_retries;
unsigned is_local:1;
@@ -298,16 +298,25 @@ union fw_transaction_callback {
fw_transaction_callback_with_tstamp_t with_tstamp;
};
-/*
- * This callback handles an inbound request subaction. It is called in
- * RCU read-side context, therefore must not sleep.
+/**
+ * typedef fw_address_callback_t - Function to handle the request of the asynchronous transaction.
+ * @card: the card instance which receives the request
+ * @request: the request instance.
+ * @tcode: the transaction code
+ * @destination: the destination node ID
+ * @source: the source node ID
+ * @generation: the bus generation in which the request was sent
+ * @offset: the destination offset in source node.
+ * @data: the request content if available.
+ * @length: the length of data.
+ * @callback_data: the data registered with this function.
*
- * The callback should not initiate outbound request subactions directly.
- * Otherwise there is a danger of recursion of inbound and outbound
- * transactions from and to the local node.
+ * This callback handles an inbound request subaction.
*
* The callback is responsible that fw_send_response() is called on the @request, except for FCP
* registers for which the core takes care of that.
+ *
+ * Context: Process context.
*/
typedef void (*fw_address_callback_t)(struct fw_card *card,
struct fw_request *request,
@@ -322,22 +331,24 @@ struct fw_packet {
int generation;
u32 header[4];
size_t header_length;
- void *payload;
+ void *payload __counted_by_ptr(payload_length);
size_t payload_length;
dma_addr_t payload_bus;
bool payload_mapped;
u32 timestamp;
+ // Used to handle the local-to-local packets in the AT request/response contexts.
+ struct list_head link_for_local;
+
/*
* This callback is called when the packet transmission has completed.
* For successful transmission, the status code is the ack received
* from the destination. Otherwise it is one of the juju-specific
* rcodes: RCODE_SEND_ERROR, _CANCELLED, _BUSY, _GENERATION, _NO_ACK.
- * The callback can be called from workqueue and thus must never block.
+ * The callback is called from a workqueue. It is not preferable to block it so long.
*/
fw_packet_callback_t callback;
int ack;
- struct list_head link;
void *driver_data;
};
@@ -359,6 +370,11 @@ struct fw_transaction {
union fw_transaction_callback callback;
bool with_tstamp;
void *callback_data;
+
+ // For some error cases.
+ struct work_struct error_work;
+ int rcode;
+ u32 response_timestamp;
};
struct fw_address_handler {
@@ -411,9 +427,8 @@ void __fw_send_request(struct fw_card *card, struct fw_transaction *t, int tcode
* A variation of __fw_send_request() to generate callback for response subaction without time
* stamp.
*
- * The callback is invoked in the workqueue context in most cases. However, if an error is detected
- * before queueing or the destination address refers to the local node, it is invoked in the
- * current context instead.
+ * After the transaction is completed successfully or unsuccessfully, the @callback will be called
+ * in process context.
*/
static inline void fw_send_request(struct fw_card *card, struct fw_transaction *t, int tcode,
int destination_id, int generation, int speed,
@@ -444,9 +459,8 @@ static inline void fw_send_request(struct fw_card *card, struct fw_transaction *
*
* A variation of __fw_send_request() to generate callback for response subaction with time stamp.
*
- * The callback is invoked in the workqueue context in most cases. However, if an error is detected
- * before queueing or the destination address refers to the local node, it is invoked in the current
- * context instead.
+ * After the transaction is completed successfully or unsuccessfully, the @callback will be called
+ * in process context.
*/
static inline void fw_send_request_with_tstamp(struct fw_card *card, struct fw_transaction *t,
int tcode, int destination_id, int generation, int speed, unsigned long long offset,
diff --git a/include/linux/firmware/cirrus/cs_dsp.h b/include/linux/firmware/cirrus/cs_dsp.h
index 4e3baa557068..6aa1e1b2b4a5 100644
--- a/include/linux/firmware/cirrus/cs_dsp.h
+++ b/include/linux/firmware/cirrus/cs_dsp.h
@@ -96,7 +96,7 @@ struct cs_dsp_alg_region {
struct cs_dsp_coeff_ctl {
struct list_head list;
struct cs_dsp *dsp;
- void *cache;
+ void *cache __counted_by_ptr(len);
const char *fw_name;
/* Subname is needed to match with firmware */
const char *subname;
diff --git a/include/linux/firmware/imx/se_api.h b/include/linux/firmware/imx/se_api.h
new file mode 100644
index 000000000000..b1c4c9115d7b
--- /dev/null
+++ b/include/linux/firmware/imx/se_api.h
@@ -0,0 +1,14 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * Copyright 2025 NXP
+ */
+
+#ifndef __SE_API_H__
+#define __SE_API_H__
+
+#include <linux/types.h>
+
+#define SOC_ID_OF_IMX8ULP 0x084d
+#define SOC_ID_OF_IMX93 0x9300
+
+#endif /* __SE_API_H__ */
diff --git a/include/linux/firmware/qcom/qcom_scm.h b/include/linux/firmware/qcom/qcom_scm.h
index 5747bd191bf1..a0a6bc0229c4 100644
--- a/include/linux/firmware/qcom/qcom_scm.h
+++ b/include/linux/firmware/qcom/qcom_scm.h
@@ -64,35 +64,6 @@ bool qcom_scm_is_available(void);
int qcom_scm_set_cold_boot_addr(void *entry);
int qcom_scm_set_warm_boot_addr(void *entry);
void qcom_scm_cpu_power_down(u32 flags);
-int qcom_scm_set_remote_state(u32 state, u32 id);
-
-struct qcom_scm_pas_context {
- struct device *dev;
- u32 pas_id;
- phys_addr_t mem_phys;
- size_t mem_size;
- void *ptr;
- dma_addr_t phys;
- ssize_t size;
- bool use_tzmem;
-};
-
-struct qcom_scm_pas_context *devm_qcom_scm_pas_context_alloc(struct device *dev,
- u32 pas_id,
- phys_addr_t mem_phys,
- size_t mem_size);
-int qcom_scm_pas_init_image(u32 pas_id, const void *metadata, size_t size,
- struct qcom_scm_pas_context *ctx);
-void qcom_scm_pas_metadata_release(struct qcom_scm_pas_context *ctx);
-int qcom_scm_pas_mem_setup(u32 pas_id, phys_addr_t addr, phys_addr_t size);
-int qcom_scm_pas_auth_and_reset(u32 pas_id);
-int qcom_scm_pas_shutdown(u32 pas_id);
-bool qcom_scm_pas_supported(u32 pas_id);
-struct resource_table *qcom_scm_pas_get_rsc_table(struct qcom_scm_pas_context *ctx,
- void *input_rt, size_t input_rt_size,
- size_t *output_rt_size);
-
-int qcom_scm_pas_prepare_and_auth_reset(struct qcom_scm_pas_context *ctx);
int qcom_scm_io_readl(phys_addr_t addr, unsigned int *val);
int qcom_scm_io_writel(phys_addr_t addr, unsigned int val);
diff --git a/include/linux/firmware/xlnx-zynqmp-ufs.h b/include/linux/firmware/xlnx-zynqmp-ufs.h
index d3538dd5822a..00383dd835f2 100644
--- a/include/linux/firmware/xlnx-zynqmp-ufs.h
+++ b/include/linux/firmware/xlnx-zynqmp-ufs.h
@@ -9,17 +9,17 @@
#define __FIRMWARE_XLNX_ZYNQMP_UFS_H__
#if IS_REACHABLE(CONFIG_ZYNQMP_FIRMWARE)
-int zynqmp_pm_is_mphy_tx_rx_config_ready(bool *is_ready);
-int zynqmp_pm_is_sram_init_done(bool *is_done);
+int zynqmp_pm_wait_mphy_tx_rx_config_ready(u32 timeout_us);
+int zynqmp_pm_wait_sram_init_done(u32 timeout_us);
int zynqmp_pm_set_sram_bypass(void);
int zynqmp_pm_get_ufs_calibration_values(u32 *val);
#else
-static inline int zynqmp_pm_is_mphy_tx_rx_config_ready(bool *is_ready)
+static inline int zynqmp_pm_wait_mphy_tx_rx_config_ready(u32 timeout_us)
{
return -ENODEV;
}
-static inline int zynqmp_pm_is_sram_init_done(bool *is_done)
+static inline int zynqmp_pm_wait_sram_init_done(u32 timeout_us)
{
return -ENODEV;
}
diff --git a/include/linux/firmware/xlnx-zynqmp.h b/include/linux/firmware/xlnx-zynqmp.h
index f4b3df614c77..41a1cd810bda 100644
--- a/include/linux/firmware/xlnx-zynqmp.h
+++ b/include/linux/firmware/xlnx-zynqmp.h
@@ -277,11 +277,6 @@ enum rpu_oper_mode {
PM_RPU_MODE_SPLIT = 1,
};
-enum rpu_boot_mem {
- PM_RPU_BOOTMEM_LOVEC = 0,
- PM_RPU_BOOTMEM_HIVEC = 1,
-};
-
enum rpu_tcm_comb {
PM_RPU_TCM_SPLIT = 0,
PM_RPU_TCM_COMB = 1,
diff --git a/include/linux/fs.h b/include/linux/fs.h
index f9d1e05e8ae6..3db90996756f 100644
--- a/include/linux/fs.h
+++ b/include/linux/fs.h
@@ -1436,10 +1436,10 @@ static inline void i_gid_write(struct inode *inode, gid_t gid)
* @idmap: idmap of the mount the inode was found from
* @inode: inode to map
*
- * Return: whe inode's i_uid mapped down according to @idmap.
+ * Return: the inode's i_uid mapped down according to @idmap.
* If the inode's i_uid has no mapping INVALID_VFSUID is returned.
*/
-static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap,
+static inline vfsuid_t i_uid_into_vfsuid(const struct mnt_idmap *idmap,
const struct inode *inode)
{
return make_vfsuid(idmap, i_user_ns(inode), inode->i_uid);
@@ -1456,7 +1456,7 @@ static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap,
*
* Return: true if @inode's i_uid field needs to be updated, false if not.
*/
-static inline bool i_uid_needs_update(struct mnt_idmap *idmap,
+static inline bool i_uid_needs_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
const struct inode *inode)
{
@@ -1474,7 +1474,7 @@ static inline bool i_uid_needs_update(struct mnt_idmap *idmap,
* Safely update @inode's i_uid field translating the vfsuid of any idmapped
* mount into the filesystem kuid.
*/
-static inline void i_uid_update(struct mnt_idmap *idmap,
+static inline void i_uid_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
struct inode *inode)
{
@@ -1491,7 +1491,7 @@ static inline void i_uid_update(struct mnt_idmap *idmap,
* Return: the inode's i_gid mapped down according to @idmap.
* If the inode's i_gid has no mapping INVALID_VFSGID is returned.
*/
-static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap,
+static inline vfsgid_t i_gid_into_vfsgid(const struct mnt_idmap *idmap,
const struct inode *inode)
{
return make_vfsgid(idmap, i_user_ns(inode), inode->i_gid);
@@ -1508,7 +1508,7 @@ static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap,
*
* Return: true if @inode's i_gid field needs to be updated, false if not.
*/
-static inline bool i_gid_needs_update(struct mnt_idmap *idmap,
+static inline bool i_gid_needs_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
const struct inode *inode)
{
@@ -1526,7 +1526,7 @@ static inline bool i_gid_needs_update(struct mnt_idmap *idmap,
* Safely update @inode's i_gid field translating the vfsgid of any idmapped
* mount into the filesystem kgid.
*/
-static inline void i_gid_update(struct mnt_idmap *idmap,
+static inline void i_gid_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
struct inode *inode)
{
@@ -1544,7 +1544,7 @@ static inline void i_gid_update(struct mnt_idmap *idmap,
* an idmapped mount map the caller's fsuid according to @idmap.
*/
static inline void inode_fsuid_set(struct inode *inode,
- struct mnt_idmap *idmap)
+ const struct mnt_idmap *idmap)
{
inode->i_uid = mapped_fsuid(idmap, i_user_ns(inode));
}
@@ -1558,7 +1558,7 @@ static inline void inode_fsuid_set(struct inode *inode,
* an idmapped mount map the caller's fsgid according to @idmap.
*/
static inline void inode_fsgid_set(struct inode *inode,
- struct mnt_idmap *idmap)
+ const struct mnt_idmap *idmap)
{
inode->i_gid = mapped_fsgid(idmap, i_user_ns(inode));
}
@@ -1575,7 +1575,7 @@ static inline void inode_fsgid_set(struct inode *inode,
* Return: true if fsuid and fsgid is mapped, false if not.
*/
static inline bool fsuidgid_has_mapping(struct super_block *sb,
- struct mnt_idmap *idmap)
+ const struct mnt_idmap *idmap)
{
struct user_namespace *fs_userns = sb->s_user_ns;
kuid_t kuid;
@@ -1755,25 +1755,25 @@ static inline bool file_write_not_started(const struct file *file)
return sb_write_not_started(file_inode(file)->i_sb);
}
-bool inode_owner_or_capable(struct mnt_idmap *idmap,
+bool inode_owner_or_capable(const struct mnt_idmap *idmap,
const struct inode *inode);
/*
* VFS helper functions..
*/
-int vfs_create(struct mnt_idmap *, struct dentry *, umode_t,
+int vfs_create(const struct mnt_idmap *, struct dentry *, umode_t,
struct delegated_inode *);
-struct dentry *vfs_mkdir(struct mnt_idmap *, struct inode *,
+struct dentry *vfs_mkdir(const struct mnt_idmap *, struct inode *,
struct dentry *, umode_t, struct delegated_inode *);
-int vfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *,
+int vfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *,
umode_t, dev_t, struct delegated_inode *);
-int vfs_symlink(struct mnt_idmap *, struct inode *,
+int vfs_symlink(const struct mnt_idmap *, struct inode *,
struct dentry *, const char *, struct delegated_inode *);
-int vfs_link(struct dentry *, struct mnt_idmap *, struct inode *,
+int vfs_link(struct dentry *, const struct mnt_idmap *, struct inode *,
struct dentry *, struct delegated_inode *);
-int vfs_rmdir(struct mnt_idmap *, struct inode *, struct dentry *,
+int vfs_rmdir(const struct mnt_idmap *, struct inode *, struct dentry *,
struct delegated_inode *);
-int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *,
+int vfs_unlink(const struct mnt_idmap *, struct inode *, struct dentry *,
struct delegated_inode *);
/**
@@ -1787,7 +1787,7 @@ int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *,
* @flags: rename flags
*/
struct renamedata {
- struct mnt_idmap *mnt_idmap;
+ const struct mnt_idmap *mnt_idmap;
struct dentry *old_parent;
struct dentry *old_dentry;
struct dentry *new_parent;
@@ -1798,14 +1798,14 @@ struct renamedata {
int vfs_rename(struct renamedata *);
-static inline int vfs_whiteout(struct mnt_idmap *idmap,
+static inline int vfs_whiteout(const struct mnt_idmap *idmap,
struct inode *dir, struct dentry *dentry)
{
return vfs_mknod(idmap, dir, dentry, S_IFCHR | WHITEOUT_MODE,
WHITEOUT_DEV, NULL);
}
-struct file *kernel_tmpfile_open(struct mnt_idmap *idmap,
+struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap,
const struct path *parentpath,
umode_t mode, int open_flag,
const struct cred *cred);
@@ -1830,12 +1830,12 @@ extern long compat_ptr_ioctl(struct file *file, unsigned int cmd,
/*
* VFS file helper functions.
*/
-void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode,
+void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode,
const struct inode *dir, umode_t mode);
extern bool may_open_dev(const struct path *path);
-umode_t mode_strip_sgid(struct mnt_idmap *idmap,
+umode_t mode_strip_sgid(const struct mnt_idmap *idmap,
const struct inode *dir, umode_t mode);
-bool in_group_or_capable(struct mnt_idmap *idmap,
+bool in_group_or_capable(const struct mnt_idmap *idmap,
const struct inode *inode, vfsgid_t vfsgid);
/*
@@ -1994,26 +1994,26 @@ enum fs_update_time {
struct inode_operations {
struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int);
const char * (*get_link) (struct dentry *, struct inode *, struct delayed_call *);
- int (*permission) (struct mnt_idmap *, struct inode *, int);
+ int (*permission) (const struct mnt_idmap *, struct inode *, int);
struct posix_acl * (*get_inode_acl)(struct inode *, int, bool);
int (*readlink) (struct dentry *, char __user *,int);
- int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,
+ int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *,
umode_t);
int (*link) (struct dentry *,struct inode *,struct dentry *);
int (*unlink) (struct inode *,struct dentry *);
- int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *,
+ int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *,
const char *);
- struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *,
+ struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *,
struct dentry *, umode_t);
int (*rmdir) (struct inode *,struct dentry *);
- int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *,
+ int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *,
umode_t,dev_t);
- int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *,
+ int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *,
struct inode *, struct dentry *, unsigned int);
- int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *);
- int (*getattr) (struct mnt_idmap *, const struct path *,
+ int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *);
+ int (*getattr) (const struct mnt_idmap *, const struct path *,
struct kstat *, u32, unsigned int);
ssize_t (*listxattr) (struct dentry *, char *, size_t);
int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start,
@@ -2024,13 +2024,13 @@ struct inode_operations {
int (*atomic_open)(struct inode *, struct dentry *,
struct file *, unsigned open_flag,
umode_t create_mode);
- int (*tmpfile) (struct mnt_idmap *, struct inode *,
+ int (*tmpfile) (const struct mnt_idmap *, struct inode *,
struct file *, umode_t);
- struct posix_acl *(*get_acl)(struct mnt_idmap *, struct dentry *,
+ struct posix_acl *(*get_acl)(const struct mnt_idmap *, struct dentry *,
int);
- int (*set_acl)(struct mnt_idmap *, struct dentry *,
+ int (*set_acl)(const struct mnt_idmap *, struct dentry *,
struct posix_acl *, int);
- int (*fileattr_set)(struct mnt_idmap *idmap,
+ int (*fileattr_set)(const struct mnt_idmap *idmap,
struct dentry *dentry, struct file_kattr *fa);
int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa);
struct offset_ctx *(*get_offset_ctx)(struct inode *inode);
@@ -2173,7 +2173,7 @@ extern loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos,
(inode)->i_rdev == WHITEOUT_DEV)
#define IS_ANON_FILE(inode) ((inode)->i_flags & S_ANON_INODE)
-static inline bool HAS_UNMAPPED_ID(struct mnt_idmap *idmap,
+static inline bool HAS_UNMAPPED_ID(const struct mnt_idmap *idmap,
struct inode *inode)
{
return !vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) ||
@@ -2459,7 +2459,7 @@ struct filename {
static_assert(offsetof(struct filename, iname) % sizeof(long) == 0);
static_assert(sizeof(struct filename) % 64 == 0);
-static inline struct mnt_idmap *file_mnt_idmap(const struct file *file)
+static inline const struct mnt_idmap *file_mnt_idmap(const struct file *file)
{
return mnt_idmap(file->f_path.mnt);
}
@@ -2483,7 +2483,7 @@ static inline bool is_idmapped_mnt(const struct vfsmount *mnt)
}
int vfs_truncate(const struct path *, loff_t);
-int do_truncate(struct mnt_idmap *, struct dentry *, loff_t start,
+int do_truncate(const struct mnt_idmap *, struct dentry *, loff_t start,
unsigned int time_attrs, struct file *filp);
extern int vfs_fallocate(struct file *file, int mode, loff_t offset,
loff_t len);
@@ -2707,10 +2707,10 @@ static inline int bmap(struct inode *inode, sector_t *block)
}
#endif
-int notify_change(struct mnt_idmap *, struct dentry *,
+int notify_change(const struct mnt_idmap *, struct dentry *,
struct iattr *, struct delegated_inode *);
-int inode_permission(struct mnt_idmap *, struct inode *, int);
-int generic_permission(struct mnt_idmap *, struct inode *, int);
+int inode_permission(const struct mnt_idmap *, struct inode *, int);
+int generic_permission(const struct mnt_idmap *, struct inode *, int);
static inline int file_permission(struct file *file, int mask)
{
return inode_permission(file_mnt_idmap(file),
@@ -2721,12 +2721,12 @@ static inline int path_permission(const struct path *path, int mask)
return inode_permission(mnt_idmap(path->mnt),
d_inode(path->dentry), mask);
}
-int __check_sticky(struct mnt_idmap *idmap, struct inode *dir,
+int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir,
struct inode *inode);
-int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir,
+int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir,
struct dentry *victim, bool isdir);
-int may_create_dentry(struct mnt_idmap *idmap,
+int may_create_dentry(const struct mnt_idmap *idmap,
struct inode *dir, struct dentry *child);
static inline bool execute_ok(struct inode *inode)
@@ -3045,9 +3045,9 @@ static inline struct inode *new_inode_pseudo(struct super_block *sb)
}
extern struct inode *new_inode(struct super_block *sb);
extern void free_inode_nonrcu(struct inode *inode);
-extern int setattr_should_drop_suidgid(struct mnt_idmap *, struct inode *);
+extern int setattr_should_drop_suidgid(const struct mnt_idmap *, struct inode *);
extern int file_remove_privs(struct file *);
-int setattr_should_drop_sgid(struct mnt_idmap *idmap,
+int setattr_should_drop_sgid(const struct mnt_idmap *idmap,
const struct inode *inode);
/*
@@ -3204,7 +3204,7 @@ extern int page_symlink(struct inode *inode, const char *symname, int len);
extern const struct inode_operations page_symlink_inode_operations;
extern void kfree_link(void *);
void fill_mg_cmtime(struct kstat *stat, u32 request_mask, struct inode *inode);
-void generic_fillattr(struct mnt_idmap *, u32, struct inode *, struct kstat *);
+void generic_fillattr(const struct mnt_idmap *, u32, struct inode *, struct kstat *);
void generic_fill_statx_attr(struct inode *inode, struct kstat *stat);
void generic_fill_statx_atomic_writes(struct kstat *stat,
unsigned int unit_min,
@@ -3261,9 +3261,9 @@ extern int dcache_dir_open(struct inode *, struct file *);
extern int dcache_dir_close(struct inode *, struct file *);
extern loff_t dcache_dir_lseek(struct file *, loff_t, int);
extern int dcache_readdir(struct file *, struct dir_context *);
-extern int simple_setattr(struct mnt_idmap *, struct dentry *,
+extern int simple_setattr(const struct mnt_idmap *, struct dentry *,
struct iattr *);
-extern int simple_getattr(struct mnt_idmap *, const struct path *,
+extern int simple_getattr(const struct mnt_idmap *, const struct path *,
struct kstat *, u32, unsigned int);
extern int simple_statfs(struct dentry *, struct kstatfs *);
extern int simple_open(struct inode *inode, struct file *file);
@@ -3276,7 +3276,7 @@ void simple_rename_timestamp(struct inode *old_dir, struct dentry *old_dentry,
struct inode *new_dir, struct dentry *new_dentry);
extern int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry,
struct inode *new_dir, struct dentry *new_dentry);
-extern int simple_rename(struct mnt_idmap *, struct inode *,
+extern int simple_rename(const struct mnt_idmap *, struct inode *,
struct dentry *, struct inode *, struct dentry *,
unsigned int);
extern void simple_recursive_removal(struct dentry *,
@@ -3397,11 +3397,11 @@ static inline bool generic_ci_validate_strict_name(struct inode *dir,
}
#endif
-int may_setattr(struct mnt_idmap *idmap, struct inode *inode,
+int may_setattr(const struct mnt_idmap *idmap, struct inode *inode,
unsigned int ia_valid);
-int setattr_prepare(struct mnt_idmap *, struct dentry *, struct iattr *);
+int setattr_prepare(const struct mnt_idmap *, struct dentry *, struct iattr *);
extern int inode_newsize_ok(const struct inode *, loff_t offset);
-void setattr_copy(struct mnt_idmap *, struct inode *inode,
+void setattr_copy(const struct mnt_idmap *, struct inode *inode,
const struct iattr *attr);
extern int file_update_time(struct file *file);
@@ -3578,7 +3578,7 @@ static inline bool is_sxid(umode_t mode)
return mode & (S_ISUID | S_ISGID);
}
-static inline int check_sticky(struct mnt_idmap *idmap,
+static inline int check_sticky(const struct mnt_idmap *idmap,
struct inode *dir, struct inode *inode)
{
if (!(dir->i_mode & S_ISVTX))
@@ -3653,23 +3653,6 @@ extern int vfs_fadvise(struct file *file, loff_t offset, loff_t len,
extern int generic_fadvise(struct file *file, loff_t offset, loff_t len,
int advice);
-static inline bool vfs_empty_path(int dfd, const char __user *path)
-{
- char c;
-
- if (dfd < 0)
- return false;
-
- /* We now allow NULL to be used for empty path. */
- if (!path)
- return true;
-
- if (unlikely(get_user(c, path)))
- return false;
-
- return !c;
-}
-
int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter);
static inline bool extensible_ioctl_valid(unsigned int cmd_a,
diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h
index 0d6c8a6d7be2..c920aba5177c 100644
--- a/include/linux/fs_context.h
+++ b/include/linux/fs_context.h
@@ -150,6 +150,10 @@ extern int vfs_parse_fs_param_source(struct fs_context *fc,
struct fs_parameter *param);
extern void fc_drop_locked(struct fs_context *fc);
+extern int get_tree_super(struct fs_context *fc,
+ int (*test)(struct super_block *, struct fs_context *),
+ int (*fill_super)(struct super_block *sb,
+ struct fs_context *fc));
extern int get_tree_nodev(struct fs_context *fc,
int (*fill_super)(struct super_block *sb,
struct fs_context *fc));
diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h
index 4c91a019972b..ee524c863fa9 100644
--- a/include/linux/fscache-cache.h
+++ b/include/linux/fscache-cache.h
@@ -67,7 +67,7 @@ struct fscache_cache_ops {
/* Change the size of a data object */
void (*resize_cookie)(struct netfs_cache_resources *cres,
- loff_t new_size);
+ uoff_t new_size);
/* Invalidate an object */
bool (*invalidate_cookie)(struct fscache_cookie *cookie);
diff --git a/include/linux/fscache.h b/include/linux/fscache.h
index 58fdb9605425..f2d958bd1f48 100644
--- a/include/linux/fscache.h
+++ b/include/linux/fscache.h
@@ -112,7 +112,7 @@ struct fscache_cookie {
struct list_head proc_link; /* Link in proc list */
struct list_head commit_link; /* Link in commit queue */
struct work_struct work; /* Commit/relinq/withdraw work */
- loff_t object_size; /* Size of the netfs object */
+ uoff_t object_size; /* Size of the netfs object */
unsigned long unused_at; /* Time at which unused (jiffies) */
unsigned long flags;
#define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */
@@ -147,6 +147,23 @@ struct fscache_cookie {
};
};
+enum fscache_extent_type {
+ FSCACHE_EXTENT_DATA,
+ FSCACHE_EXTENT_ZERO,
+} __mode(byte);
+
+/*
+ * Cache occupancy information.
+ */
+struct fscache_occupancy {
+ unsigned long long query_from; /* Point to query from */
+ unsigned long long query_to; /* Point to query to */
+ unsigned long long cached_from[2]; /* Point at which cache extents start */
+ unsigned long long cached_to[2]; /* Point at which cache extents end */
+ unsigned int granularity; /* Granularity desired */
+ enum fscache_extent_type cached_type[2]; /* Type of cache extent */
+};
+
/*
* slow-path functions for when there is actually caching available, and the
* netfs does actually have a valid token
@@ -163,22 +180,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie(
u8,
const void *, size_t,
const void *, size_t,
- loff_t);
+ uoff_t);
extern void __fscache_use_cookie(struct fscache_cookie *, bool);
-extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *);
+extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *);
extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool);
-extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t);
-extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int);
+extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t);
+extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int);
extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *);
extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *);
void __fscache_write_to_cache(struct fscache_cookie *cookie,
struct address_space *mapping,
- loff_t start, size_t len, loff_t i_size,
+ uoff_t start, size_t len, uoff_t i_size,
netfs_io_terminated_t term_func,
void *term_func_priv,
bool using_pgpriv2, bool cond);
-extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t);
+extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t);
/**
* fscache_acquire_volume - Register a volume as desiring caching services
@@ -249,7 +266,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume,
size_t index_key_len,
const void *aux_data,
size_t aux_data_len,
- loff_t object_size)
+ uoff_t object_size)
{
if (!fscache_volume_valid(volume))
return NULL;
@@ -286,7 +303,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie,
*/
static inline void fscache_unuse_cookie(struct fscache_cookie *cookie,
const void *aux_data,
- const loff_t *object_size)
+ const uoff_t *object_size)
{
if (fscache_cookie_valid(cookie))
__fscache_unuse_cookie(cookie, aux_data, object_size);
@@ -327,7 +344,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie)
*/
static inline
void fscache_update_aux(struct fscache_cookie *cookie,
- const void *aux_data, const loff_t *object_size)
+ const void *aux_data, const uoff_t *object_size)
{
void *p = fscache_get_aux(cookie);
@@ -343,7 +360,7 @@ extern atomic_t fscache_n_updates;
static inline
void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data,
- const loff_t *object_size)
+ const uoff_t *object_size)
{
#ifdef CONFIG_FSCACHE_STATS
atomic_inc(&fscache_n_updates);
@@ -369,7 +386,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data
*/
static inline
void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data,
- const loff_t *object_size)
+ const uoff_t *object_size)
{
if (fscache_cookie_enabled(cookie))
__fscache_update_cookie(cookie, aux_data, object_size);
@@ -386,7 +403,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data,
* description.
*/
static inline
-void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size)
+void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size)
{
if (fscache_cookie_enabled(cookie))
__fscache_resize_cookie(cookie, new_size);
@@ -413,7 +430,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size)
*/
static inline
void fscache_invalidate(struct fscache_cookie *cookie,
- const void *aux_data, loff_t size, unsigned int flags)
+ const void *aux_data, uoff_t size, unsigned int flags)
{
if (fscache_cookie_enabled(cookie))
__fscache_invalidate(cookie, aux_data, size, flags);
@@ -502,7 +519,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres)
*/
static inline
int fscache_read(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
enum netfs_read_from_hole read_hole,
netfs_io_terminated_t term_func,
@@ -561,7 +578,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres,
*/
static inline
int fscache_write(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
netfs_io_terminated_t term_func,
void *term_func_priv)
@@ -581,7 +598,7 @@ int fscache_write(struct netfs_cache_resources *cres,
* waiting.
*/
static inline void fscache_clear_page_bits(struct address_space *mapping,
- loff_t start, size_t len,
+ uoff_t start, size_t len,
bool caching)
{
if (caching)
@@ -615,7 +632,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping,
*/
static inline void fscache_write_to_cache(struct fscache_cookie *cookie,
struct address_space *mapping,
- loff_t start, size_t len, loff_t i_size,
+ uoff_t start, size_t len, uoff_t i_size,
netfs_io_terminated_t term_func,
void *term_func_priv,
bool using_pgpriv2, bool caching)
diff --git a/include/linux/ftrace.h b/include/linux/ftrace.h
index 02bc5027523a..bd76a16a63af 100644
--- a/include/linux/ftrace.h
+++ b/include/linux/ftrace.h
@@ -866,8 +866,9 @@ unsigned long ftrace_get_addr_new(struct dyn_ftrace *rec);
unsigned long ftrace_get_addr_curr(struct dyn_ftrace *rec);
extern ftrace_func_t ftrace_trace_function;
+struct trace_array;
-int ftrace_regex_open(struct ftrace_ops *ops, int flag,
+int ftrace_regex_open(struct trace_array *tr, struct ftrace_ops *ops, int flag,
struct inode *inode, struct file *file);
ssize_t ftrace_filter_write(struct file *file, const char __user *ubuf,
size_t cnt, loff_t *ppos);
@@ -1077,7 +1078,7 @@ static inline unsigned long ftrace_location(unsigned long ip)
* have them defined when ftrace is not enabled, but these
* functions may still be called. Use a macro instead of inline.
*/
-#define ftrace_regex_open(ops, flag, inod, file) ({ -ENODEV; })
+#define ftrace_regex_open(tr, ops, flag, inode, file) ({ -ENODEV; })
#define ftrace_set_early_filter(ops, buf, enable) do { } while (0)
#define ftrace_set_filter_ip(ops, ip, remove, reset) ({ -ENODEV; })
#define ftrace_set_filter_ips(ops, ips, cnt, remove, reset) ({ -ENODEV; })
diff --git a/include/linux/generic_pt/common.h b/include/linux/generic_pt/common.h
index 07ef1c8341a4..1b39b27f0cfa 100644
--- a/include/linux/generic_pt/common.h
+++ b/include/linux/generic_pt/common.h
@@ -213,4 +213,10 @@ enum {
PT_FEAT_X86_64_AMD_ENCRYPT_TABLES = PT_FEAT_FMT_START,
};
+struct pt_bcm2712 {
+ struct pt_common common;
+ u8 bigpage_lg2;
+ u8 superpage_lg2;
+};
+
#endif
diff --git a/include/linux/generic_pt/iommu.h b/include/linux/generic_pt/iommu.h
index dd0edd02a48a..13ebb72a67de 100644
--- a/include/linux/generic_pt/iommu.h
+++ b/include/linux/generic_pt/iommu.h
@@ -346,6 +346,18 @@ struct pt_iommu_x86_64_hw_info {
IOMMU_FORMAT(x86_64, x86_64_pt);
+struct pt_iommu_bcm2712_cfg {
+ struct pt_iommu_cfg common;
+ u8 bigpage_lg2;
+ u8 superpage_lg2;
+};
+
+struct pt_iommu_bcm2712_hw_info {
+ phys_addr_t pt_base;
+};
+
+IOMMU_FORMAT(bcm2712, bcm2712pt);
+
#undef IOMMU_PROTOTYPES
#undef IOMMU_FORMAT
#endif
diff --git a/include/linux/gfp.h b/include/linux/gfp.h
index 872bc53f32ec..aa4601805f21 100644
--- a/include/linux/gfp.h
+++ b/include/linux/gfp.h
@@ -329,7 +329,7 @@ void *alloc_pages_exact_noprof(size_t size, gfp_t gfp_mask) __alloc_size(1);
void free_pages_exact(void *virt, size_t size);
-__meminit void *alloc_pages_exact_nid_noprof(int nid, size_t size, gfp_t gfp_mask) __alloc_size(2);
+void *alloc_pages_exact_nid_noprof(int nid, size_t size, gfp_t gfp_mask) __alloc_size(2);
#define alloc_pages_exact_nid(...) \
alloc_hooks(alloc_pages_exact_nid_noprof(__VA_ARGS__))
diff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h
index 190191411009..bfd4c43ed777 100644
--- a/include/linux/gfp_types.h
+++ b/include/linux/gfp_types.h
@@ -244,7 +244,7 @@ enum {
* definitely preferable to use the flag rather than opencode endless
* loop around allocator.
* Allocating pages from the buddy with __GFP_NOFAIL and order > 1 is
- * not supported. Please consider using kvmalloc() instead.
+ * discouraged. Please consider using kvmalloc() instead if possible.
*/
#define __GFP_IO ((__force gfp_t)___GFP_IO)
#define __GFP_FS ((__force gfp_t)___GFP_FS)
diff --git a/include/linux/gpu_buddy.h b/include/linux/gpu_buddy.h
index 2c36124bb696..ddc4eca84175 100644
--- a/include/linux/gpu_buddy.h
+++ b/include/linux/gpu_buddy.h
@@ -43,8 +43,8 @@
/**
* GPU_BUDDY_CLEAR_ALLOCATION - Prefer pre-cleared (zeroed) memory
*
- * Attempt to allocate from the clear tree first. If insufficient clear
- * memory is available, falls back to dirty memory. Useful when the
+ * Attempt to allocate outside dirty-tracked ranges first. If insufficient
+ * clear memory is available, falls back to dirty memory. Useful when the
* caller needs zeroed memory and wants to avoid GPU clear operations.
*/
#define GPU_BUDDY_CLEAR_ALLOCATION BIT(3)
@@ -53,8 +53,8 @@
* GPU_BUDDY_CLEARED - Mark returned blocks as cleared
*
* Used with gpu_buddy_free_list() to indicate that the memory being
- * freed has been cleared (zeroed). The blocks will be placed in the
- * clear tree for future GPU_BUDDY_CLEAR_ALLOCATION requests.
+ * freed has been cleared (zeroed). The blocks will be removed from the
+ * dirty tracker for future GPU_BUDDY_CLEAR_ALLOCATION requests.
*/
#define GPU_BUDDY_CLEARED BIT(4)
@@ -67,15 +67,17 @@
*/
#define GPU_BUDDY_TRIM_DISABLE BIT(5)
-enum gpu_buddy_free_tree {
- GPU_BUDDY_CLEAR_TREE = 0,
- GPU_BUDDY_DIRTY_TREE,
- GPU_BUDDY_MAX_FREE_TREES,
+/*
+ * Clear/dirty state of a free block. Ordered so a numerically larger value
+ * is "more clear" (DIRTY < MIXED < CLEAR) which lets subtree_block_state be
+ * maintained as a simple max-augment over the per-order free tree.
+ */
+enum gpu_block_state {
+ GPU_BLOCK_DIRTY = 0,
+ GPU_BLOCK_MIXED = 1,
+ GPU_BLOCK_CLEAR = 2,
};
-#define for_each_free_tree(tree) \
- for ((tree) = 0; (tree) < GPU_BUDDY_MAX_FREE_TREES; (tree)++)
-
/**
* struct gpu_buddy_block - Block within a buddy allocator
*
@@ -103,6 +105,13 @@ struct gpu_buddy_block {
#define GPU_BUDDY_ALLOCATED (1 << 10)
#define GPU_BUDDY_FREE (2 << 10)
#define GPU_BUDDY_SPLIT (3 << 10)
+/*
+ * GPU_BUDDY_HEADER_CLEAR has two roles:
+ * - FREE state: set when the block's full range is cleared (dirty
+ * tracker confirmed no overlap).
+ * - ALLOCATED state: set when the block was served from cleared memory,
+ * informing the caller that no GPU clear pass is needed.
+ */
#define GPU_BUDDY_HEADER_CLEAR GENMASK_ULL(9, 9)
/* Free to be used, if needed in the future */
#define GPU_BUDDY_HEADER_UNUSED GENMASK_ULL(8, 6)
@@ -128,14 +137,45 @@ struct gpu_buddy_block {
struct list_head link;
};
/* private: */
- struct list_head tmp_link;
+ enum gpu_block_state subtree_block_state;
unsigned int subtree_max_alignment;
+ struct list_head tmp_link;
+ bool has_clear;
};
/* Order-zero must be at least SZ_4K */
#define GPU_BUDDY_MAX_ORDER (63 - 12)
/**
+ * struct gpu_dirty_extent - a contiguous dirty address range
+ *
+ * Tracks a single contiguous address range whose memory content is known
+ * to be dirty. Extents are non-overlapping and stored in an augmented
+ * red-black tree sorted by @start. The augmented value @subtree_max_size
+ * allows O(log N) search for an extent of at least a given size.
+ */
+struct gpu_dirty_extent {
+/* private: */
+ struct rb_node rb;
+ u64 start;
+ u64 end;
+ u64 subtree_max_size;
+};
+
+/**
+ * struct gpu_dirty_tracker - tracks dirty address intervals
+ *
+ * Maintains a set of non-overlapping dirty extents as an augmented
+ * red-black tree.
+ */
+struct gpu_dirty_tracker {
+/* private: */
+ struct rb_root root;
+ /* Total bytes of dirty memory currently tracked. */
+ u64 total_dirty;
+};
+
+/**
* struct gpu_buddy - GPU binary buddy allocator
*
* The buddy allocator provides efficient power-of-two memory allocation
@@ -152,20 +192,21 @@ struct gpu_buddy_block {
* @chunk_size: Minimum allocation granularity in bytes. Must be at least SZ_4K.
* @size: Total size of the address space managed by this allocator in bytes.
* @avail: Total free space currently available for allocation in bytes.
- * @clear_avail: Free space available in the clear tree (zeroed memory) in bytes.
- * This is a subset of @avail.
* @lock_dep_map: Annotates gpu_buddy API with a driver provided lock.
*/
struct gpu_buddy {
/* private: */
+ /* Tracker of dirty address ranges (decoupled from free_tree). */
+ struct gpu_dirty_tracker dirty;
/*
- * Array of red-black trees for free block management.
- * Indexed as free_trees[clear/dirty][order] where:
- * - Index 0 (GPU_BUDDY_CLEAR_TREE): blocks with zeroed content
- * - Index 1 (GPU_BUDDY_DIRTY_TREE): blocks with unknown content
- * Each tree holds free blocks of the corresponding order.
+ * One RB-tree per order containing all free blocks (clear and
+ * dirty alike). The augment field subtree_block_state (a max over
+ * the subtree of each block's state) lets clear allocations
+ * find the right-most fully-clear or mixed block in O(log N).
+ * Dirty free blocks coexist here but are also indexed by the
+ * @dirty tracker for fast dirty allocation lookups.
*/
- struct rb_root **free_trees;
+ struct rb_root *free_tree;
/*
* Array of root blocks representing the top-level blocks of the
* binary tree(s). Multiple roots exist when the total size is not
@@ -194,7 +235,6 @@ struct gpu_buddy {
u64 chunk_size;
u64 size;
u64 avail;
- u64 clear_avail;
#ifdef CONFIG_LOCKDEP
struct lockdep_map *lock_dep_map;
#endif
@@ -226,17 +266,32 @@ struct gpu_buddy {
*
* Ensure driver lock is held.
*/
-static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm)
+static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm)
{
if (mm->lock_dep_map)
lockdep_assert(lock_is_held_type(mm->lock_dep_map, 0));
}
#else
-static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm)
+static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm)
{
}
#endif
+/**
+ * gpu_buddy_clear_avail - free space that is clear (zeroed), in bytes
+ * @mm: gpu buddy allocator
+ *
+ * A subset of @mm->avail. Derived on demand as @mm->avail minus the bytes
+ * the dirty tracker records as dirty, so it is always consistent with the
+ * tracker without a cached field to keep in sync. Zero for a fresh pool,
+ * which is fully dirty.
+ */
+static inline u64 gpu_buddy_clear_avail(const struct gpu_buddy *mm)
+{
+ gpu_buddy_driver_lock_held(mm);
+ return mm->avail - mm->dirty.total_dirty;
+}
+
static inline u64
gpu_buddy_block_offset(const struct gpu_buddy_block *block)
{
diff --git a/include/linux/hazptr.h b/include/linux/hazptr.h
new file mode 100644
index 000000000000..d1670121947a
--- /dev/null
+++ b/include/linux/hazptr.h
@@ -0,0 +1,325 @@
+// SPDX-License-Identifier: LGPL-2.1-or-later
+//
+// SPDX-FileCopyrightText: 2024 Mathieu Desnoyers <mathieu.desnoyers@efficios.com>
+
+#ifndef _LINUX_HAZPTR_H
+#define _LINUX_HAZPTR_H
+
+/*
+ * hazptr: Hazard Pointers
+ *
+ * This API provides existence guarantees of objects through hazard
+ * pointers.
+ *
+ * Its main benefit over RCU is that it allows fast reclaim of
+ * HP-protected pointers without needing to wait for a grace period.
+ *
+ * References:
+ *
+ * [1]: M. M. Michael, "Hazard pointers: safe memory reclamation for
+ * lock-free objects," in IEEE Transactions on Parallel and
+ * Distributed Systems, vol. 15, no. 6, pp. 491-504, June 2004
+ */
+
+#include <linux/percpu.h>
+#include <linux/types.h>
+#include <linux/cleanup.h>
+#include <linux/sched.h>
+
+/* 4 slots (each sizeof(hazptr_slot_item)) fit in a single 64-byte cache line. */
+#define NR_HAZPTR_PERCPU_SLOTS 4
+
+/* The current hazard pointer wildcard. */
+extern void *hazptr_wildcard;
+
+/*
+ * Hazard pointer slot.
+ */
+struct hazptr_slot {
+ void *addr;
+};
+
+struct hazptr_overflow_list;
+
+struct hazptr_backup_slot {
+ struct hlist_node overflow_node;
+ struct hazptr_slot slot;
+ /* Overflow list where the backup slot is added. */
+ struct hazptr_overflow_list *overflow_list;
+};
+
+struct hazptr_ctx {
+ struct hazptr_slot *slot;
+ /* Backup slot in case all per-CPU slots are used. */
+ struct hazptr_backup_slot backup_slot;
+ struct hlist_node preempt_node;
+#ifdef CONFIG_HAZPTR_DEBUG
+ bool detach_task, detach_cpu; /* Whether the ctx has been detached from task/cpu. */
+ int acquire_pid, acquire_cpu; /* Note the task and cpu number at acquire. */
+ unsigned long acquire_caller; /* Acquire instruction pointer. */
+#endif
+};
+
+struct hazptr_slot_ctx {
+ struct hazptr_ctx *ctx;
+};
+
+struct hazptr_slot_item {
+ struct hazptr_slot slot;
+ struct hazptr_slot_ctx ctx;
+};
+
+struct hazptr_percpu_slots {
+ struct hazptr_slot_item items[NR_HAZPTR_PERCPU_SLOTS];
+} ____cacheline_aligned;
+
+DECLARE_PER_CPU(struct hazptr_percpu_slots, hazptr_percpu_slots);
+
+void *__hazptr_acquire(struct hazptr_ctx *ctx, void * const *addr_p);
+
+/**
+ * hazptr_synchronize: Wait for release from hazard-pointer protection
+ *
+ * @addr: The address to be released from hazard-pointer protection
+ *
+ * Wait for the specified @addr to be released from protection from all
+ * hazard pointers. The caller should make @addr inaccessible to all
+ * hazard-pointer readers before invoking this function.
+ *
+ * Must be called from preemptible context.
+ */
+void hazptr_synchronize(void *addr);
+
+/*
+ * hazptr_chain_backup_slot: Chain backup slot into overflow list.
+ *
+ * Set backup slot address to @addr, and chain it into the overflow
+ * list.
+ */
+struct hazptr_slot *hazptr_chain_backup_slot(struct hazptr_ctx *ctx);
+
+/*
+ * hazptr_unchain_backup_slot: Unchain backup slot from overflow list.
+ */
+void hazptr_unchain_backup_slot(struct hazptr_ctx *ctx);
+
+static inline
+bool hazptr_slot_is_backup(struct hazptr_ctx *ctx, struct hazptr_slot *slot)
+{
+ return slot == &ctx->backup_slot.slot;
+}
+
+/* Internal helper. */
+static inline
+void hazptr_promote_to_backup_slot(struct hazptr_ctx *ctx, struct hazptr_slot *slot)
+{
+ struct hazptr_slot *backup_slot;
+
+ backup_slot = hazptr_chain_backup_slot(ctx);
+ /*
+ * Move hazard pointer from the per-CPU slot to the
+ * backup slot. This requires hazard pointer
+ * synchronize to iterate on per-CPU slots with
+ * load-acquire before iterating on the overflow list.
+ */
+ WRITE_ONCE(backup_slot->addr, slot->addr);
+ /*
+ * store-release orders store to backup slot addr before
+ * store to per-CPU slot addr.
+ */
+ smp_store_release(&slot->addr, NULL);
+ /* Use the backup slot for context. */
+ ctx->slot = backup_slot;
+}
+
+/**
+ * hazptr_detach - Allow a hazard pointer to be released in some other context
+ *
+ * @ctx: The hazard-pointer context to be detached.
+ *
+ * By default, a given hazptr_acquire() and the corresponding
+ * hazptr_release() must run in a single execution context, for example,
+ * the context of a single task or a single interrupt handler. When you
+ * have acquired a hazard pointer in one context and need to release it
+ * in another, you must invoke hazptr_detach() on that hazard pointer's
+ * context. It is permissible to invoke hazptr_detach() multiple times
+ * on the same @ctx while it is protecting the same pointer, however,
+ * the first invocation absolutely must be in the same context that did
+ * the hazptr_acquire(), and must take place after the return from that
+ * hazptr_acquire().
+ *
+ * For example, if a hazard pointer is acquired by a task and released
+ * by a timer handler, that task would need to pass the hazard pointer's
+ * context to hazptr_detach() after return from the hazptr_acquire() and
+ * before arming the timer (or at least before the handler had a chance
+ * to access that hazard-pointer context).
+ */
+static inline
+void hazptr_detach(struct hazptr_ctx *ctx)
+{
+ struct hazptr_slot *slot;
+
+ guard(preempt)();
+ slot = ctx->slot;
+ if (!slot->addr)
+ return;
+#ifdef CONFIG_HAZPTR_DEBUG
+ ctx->detach_task = ctx->detach_cpu = true;
+#endif
+ if (unlikely(hazptr_slot_is_backup(ctx, slot)))
+ return;
+ hazptr_promote_to_backup_slot(ctx, slot);
+}
+
+static inline
+void hazptr_note_context_switch(void)
+{
+ struct hazptr_percpu_slots *percpu_slots = this_cpu_ptr(&hazptr_percpu_slots);
+ unsigned int idx;
+
+ for (idx = 0; idx < NR_HAZPTR_PERCPU_SLOTS; idx++) {
+ struct hazptr_slot_item *item = &percpu_slots->items[idx];
+ struct hazptr_slot *slot = &item->slot;
+
+ if (!slot->addr)
+ continue;
+#ifdef CONFIG_HAZPTR_DEBUG
+ item->ctx.ctx->detach_cpu = true;
+#endif
+ hazptr_promote_to_backup_slot(item->ctx.ctx, slot);
+ }
+}
+
+/**
+ * hazptr_acquire - Load pointer at address and protect with hazard pointer.
+ *
+ * @ctx: The hazard-pointer context to be passed to hazptr_release().
+ * @addr_p: Pointer to the pointer that is to be hazard-pointer protected.
+ *
+ * Load @addr_p, and protect the loaded pointer with hazard pointer.
+ * This protection is roughly similar to (but way faster than) that of a
+ * reference counter, and ends with a later call to hazptr_release().
+ *
+ * This protection is unconditional, and has limitations similar to
+ * that of unconditional reference-counter acquisition. In particular,
+ * although holding a hazard pointer prevents a hazard-pointer-protected
+ * object from being freed, it does not prevent that object from being
+ * removed from a linked data structure, and does not prevent other
+ * hazard-pointer-protected objects referenced by this object from being
+ * both removed and freed. At which point, invoking hazptr_acquire()
+ * on these dangling pointers would be a bug. On the other hand, use of
+ * hazptr_acquire() is safe for immortal pointers to objects that do not
+ * themselves contain pointers to hazard-pointer-protected objects.
+ * Other (more complex) use cases are also possible.
+ *
+ * By default, the call to hazptr_release() must be running in the same
+ * execution context as the corresponding hazptr_acquire(), for example,
+ * within the same task or interrupt handler. When it is necessary to
+ * instead call hazptr_release() from some other context, pass @ctx to
+ * hazptr_detach() in the original context after invoking hazptr_acquire()
+ * but before making the hazard pointer available to that other context.
+ *
+ * It is not permissible to invoke hazptr_acquire() twice on the same @ctx
+ * without an intervening hazptr_release().
+ *
+ * Returns a non-NULL protected address if the loaded pointer is non-NULL.
+ * Returns NULL if the loaded pointer is NULL.
+ *
+ * On success the protected hazptr slot is stored in @ctx->slot.
+ */
+static inline
+void *hazptr_acquire(struct hazptr_ctx *ctx, void * const *addr_p)
+{
+ struct hazptr_percpu_slots *percpu_slots;
+ struct hazptr_slot_item *slot_item;
+ struct hazptr_slot *slot;
+ void *addr;
+
+ guard(preempt)();
+ percpu_slots = this_cpu_ptr(&hazptr_percpu_slots);
+ slot_item = &percpu_slots->items[0];
+ slot = &slot_item->slot;
+#ifdef CONFIG_HAZPTR_DEBUG
+ ctx->detach_cpu = ctx->detach_task = false;
+ ctx->acquire_pid = current->pid;
+ ctx->acquire_cpu = smp_processor_id();
+ ctx->acquire_caller = _THIS_IP_;
+#endif
+ if (unlikely(slot->addr))
+ return __hazptr_acquire(ctx, addr_p);
+ WRITE_ONCE(slot->addr, READ_ONCE(hazptr_wildcard)); /* Store B */
+
+ /* Memory ordering: Store B before Load A. */
+ smp_mb();
+
+ /*
+ * Load @addr_p after storing wildcard to the hazard pointer slot.
+ */
+ addr = READ_ONCE(*addr_p); /* Load A */
+
+ /*
+ * We don't care about ordering of Store C. It will simply
+ * replace the wildcard by a more specific address. If addr is
+ * NULL, we simply store NULL into the slot.
+ */
+ WRITE_ONCE(slot->addr, addr); /* Store C */
+ slot_item->ctx.ctx = ctx;
+ ctx->slot = slot;
+ return addr;
+}
+
+#ifdef CONFIG_HAZPTR_DEBUG
+/* Called with preemption disabled. */
+static inline
+void hazptr_release_debug(struct hazptr_ctx *ctx, void *addr)
+{
+ int pid = current->pid, cpu = smp_processor_id();
+ bool warn_remote_cpu = !ctx->detach_cpu && ctx->acquire_cpu != cpu,
+ warn_remote_task = !ctx->detach_task && ctx->acquire_pid != pid;
+
+ WARN_ONCE(warn_remote_cpu || warn_remote_task,
+ "Hazard Pointer (addr=%p) released on remote %s without %s. Acquire: caller=%pS, pid=%d, cpu=%d. Release: pid=%d, cpu=%d.",
+ addr,
+ warn_remote_task ? "task" : "cpu",
+ warn_remote_task ? "being detached from task" : "context switch",
+ (void *) ctx->acquire_caller, ctx->acquire_pid, ctx->acquire_cpu, pid, cpu);
+}
+#else
+static inline void hazptr_release_debug(struct hazptr_ctx *ctx, void *addr) { }
+#endif
+
+/**
+ * hazptr_release - Release the specified hazard pointer
+ *
+ * @ctx: The hazard-pointer context that was passed to hazptr_acquire().
+ * @addr_p: The pointer that is to be hazard-pointer unprotected.
+ *
+ * Release the protected hazard pointer recorded in @ctx.
+ *
+ * By default, hazptr_release() must execute in the same execution context
+ * that invoked the corresponding hazptr_acquire(), for example, within the
+ * same task or the same interrupt handler. However, if this restriction
+ * is problematic for your use case, please see hazptr_detach().
+ *
+ * It is permissible (though unwise from a maintainability viewpoint)
+ * to invoke hazptr_release() twice on the same @ctx without an intervening
+ * hazptr_acquire().
+ */
+static inline
+void hazptr_release(struct hazptr_ctx *ctx, void *addr)
+{
+ struct hazptr_slot *slot;
+
+ if (!addr)
+ return;
+ guard(preempt)();
+ hazptr_release_debug(ctx, addr);
+ slot = ctx->slot;
+ smp_store_release(&slot->addr, NULL);
+ if (unlikely(hazptr_slot_is_backup(ctx, slot)))
+ hazptr_unchain_backup_slot(ctx);
+}
+
+void hazptr_init(void);
+
+#endif /* _LINUX_HAZPTR_H */
diff --git a/include/linux/hdmi.h b/include/linux/hdmi.h
index 8dab78e1f61b..b80a5ee63bb2 100644
--- a/include/linux/hdmi.h
+++ b/include/linux/hdmi.h
@@ -27,6 +27,18 @@
#include <linux/types.h>
#include <linux/device.h>
+enum hdmi_version {
+ HDMI_VERSION_UNKNOWN,
+ HDMI_VERSION_1_0,
+ HDMI_VERSION_1_1,
+ HDMI_VERSION_1_2,
+ HDMI_VERSION_1_3,
+ HDMI_VERSION_1_4,
+ HDMI_VERSION_2_0,
+ HDMI_VERSION_2_1,
+ HDMI_VERSION_2_2,
+};
+
enum hdmi_packet_type {
HDMI_PACKET_TYPE_NULL = 0x00,
HDMI_PACKET_TYPE_AUDIO_CLOCK_REGEN = 0x01,
diff --git a/include/linux/hiddev.h b/include/linux/hiddev.h
index 2164c03d2c72..8e9f8a33e359 100644
--- a/include/linux/hiddev.h
+++ b/include/linux/hiddev.h
@@ -13,6 +13,7 @@
#ifndef _HIDDEV_H
#define _HIDDEV_H
+#include <linux/refcount.h>
#include <uapi/linux/hiddev.h>
@@ -24,6 +25,7 @@ struct hiddev {
int minor;
int exist;
int open;
+ refcount_t refcount;
struct mutex existancelock;
wait_queue_head_t wait;
struct hid_device *hid;
diff --git a/include/linux/hrtimer.h b/include/linux/hrtimer.h
index 29072d89e5cb..cad8482337cb 100644
--- a/include/linux/hrtimer.h
+++ b/include/linux/hrtimer.h
@@ -72,7 +72,7 @@ enum hrtimer_mode {
*/
struct hrtimer_sleeper {
struct hrtimer timer;
- struct task_struct *task;
+ struct task_struct *__private task;
};
static inline void hrtimer_set_expires(struct hrtimer *timer, ktime_t time)
@@ -320,6 +320,14 @@ extern int schedule_hrtimeout_range_clock(ktime_t *expires,
const enum hrtimer_mode mode,
clockid_t clock_id);
extern int schedule_hrtimeout(ktime_t *expires, const enum hrtimer_mode mode);
+static inline struct task_struct *hrtimer_sleeper_task_get(struct hrtimer_sleeper *sl)
+{
+ return READ_ONCE(ACCESS_PRIVATE(sl, task));
+}
+static inline void hrtimer_sleeper_task_set(struct hrtimer_sleeper *sl, struct task_struct *t)
+{
+ WRITE_ONCE(ACCESS_PRIVATE(sl, task), t);
+}
/* Soft interrupt function to run the hrtimer queues: */
extern void hrtimer_run_queues(void);
diff --git a/include/linux/hrtimer_rearm.h b/include/linux/hrtimer_rearm.h
index 17a81826bd9a..12e42721c8e0 100644
--- a/include/linux/hrtimer_rearm.h
+++ b/include/linux/hrtimer_rearm.h
@@ -45,7 +45,7 @@ hrtimer_rearm_deferred_user_irq(unsigned long *tif_work, const unsigned long tif
clear_thread_flag(TIF_HRTIMER_REARM);
__hrtimer_rearm_deferred();
/* Don't go into the loop if HRTIMER_REARM was the only flag */
- *tif_work &= ~TIF_HRTIMER_REARM;
+ *tif_work &= ~_TIF_HRTIMER_REARM;
return !*tif_work;
}
return false;
diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h
index c745f7ad2298..8ca0fa3be2ac 100644
--- a/include/linux/huge_mm.h
+++ b/include/linux/huge_mm.h
@@ -510,8 +510,6 @@ change_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma,
int hugepage_madvise(struct vm_area_struct *vma, vm_flags_t *vm_flags,
int advice);
-int madvise_collapse(struct vm_area_struct *vma, unsigned long start,
- unsigned long end, bool *lock_dropped);
void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start,
unsigned long end, struct vm_area_struct *next);
spinlock_t *__pmd_trans_huge_lock(pmd_t *pmd, struct vm_area_struct *vma);
@@ -715,13 +713,6 @@ static inline int hugepage_madvise(struct vm_area_struct *vma,
return -EINVAL;
}
-static inline int madvise_collapse(struct vm_area_struct *vma,
- unsigned long start,
- unsigned long end, bool *lock_dropped)
-{
- return -EINVAL;
-}
-
static inline void vma_adjust_trans_huge(struct vm_area_struct *vma,
unsigned long start,
unsigned long end,
diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h
index 16c4c4caa126..5029c7241863 100644
--- a/include/linux/hugetlb.h
+++ b/include/linux/hugetlb.h
@@ -7,11 +7,11 @@
#include <linux/mm_types.h>
#include <linux/mmdebug.h>
#include <linux/fs.h>
-#include <linux/hugetlb_inline.h>
#include <linux/cgroup.h>
#include <linux/page_ref.h>
#include <linux/list.h>
#include <linux/kref.h>
+#include <linux/atomic.h>
#include <linux/pgtable.h>
#include <linux/gfp.h>
#include <linux/userfaultfd_k.h>
@@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio);
extern int movable_gigantic_pages __read_mostly;
extern int sysctl_hugetlb_shm_group __read_mostly;
-extern struct list_head huge_boot_pages[MAX_NUMNODES];
void hugetlb_bootmem_struct_page_init(void);
void hugetlb_bootmem_alloc(void);
@@ -253,14 +252,14 @@ extern void __hugetlb_zap_end(struct vm_area_struct *vma,
static inline void hugetlb_zap_begin(struct vm_area_struct *vma,
unsigned long *start, unsigned long *end)
{
- if (is_vm_hugetlb_page(vma))
+ if (vma_is_hugetlb(vma))
__hugetlb_zap_begin(vma, start, end);
}
static inline void hugetlb_zap_end(struct vm_area_struct *vma,
struct zap_details *details)
{
- if (is_vm_hugetlb_page(vma))
+ if (vma_is_hugetlb(vma))
__hugetlb_zap_end(vma, details);
}
@@ -509,6 +508,9 @@ struct hugetlbfs_inode_info {
struct inode vfs_inode;
struct resv_map *resv_map;
unsigned int seals;
+#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING
+ atomic64_t pmd_sharing_count;
+#endif
};
static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode)
@@ -516,6 +518,39 @@ static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode)
return container_of(inode, struct hugetlbfs_inode_info, vfs_inode);
}
+#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING
+static inline void hugetlbfs_pmd_sharing_init(struct inode *inode)
+{
+ atomic64_set(&HUGETLBFS_I(inode)->pmd_sharing_count, 0);
+}
+
+static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode)
+{
+ atomic64_inc(&HUGETLBFS_I(inode)->pmd_sharing_count);
+}
+
+static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode)
+{
+ atomic64_dec(&HUGETLBFS_I(inode)->pmd_sharing_count);
+}
+
+static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode)
+{
+ return atomic64_read(&HUGETLBFS_I(inode)->pmd_sharing_count) != 0;
+}
+#else
+static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) {}
+
+static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) {}
+
+static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) {}
+
+static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode)
+{
+ return false;
+}
+#endif
+
extern const struct vm_operations_struct hugetlb_vm_ops;
struct file *hugetlb_file_setup(const char *name, size_t size, vma_flags_t acct,
int creat_flags, int page_size_log);
@@ -676,10 +711,6 @@ struct hstate {
char name[HSTATE_NAME_LEN];
};
-#define HUGE_BOOTMEM_HVO 0x0001
-#define HUGE_BOOTMEM_ZONES_VALID 0x0002
-#define HUGE_BOOTMEM_CMA 0x0004
-
int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list);
int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn);
void wait_for_freed_hugetlb_folios(void);
@@ -699,7 +730,8 @@ enum hugetlb_alloc_flag {
#define HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS BIT(HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS_BIT)
struct folio *hugetlb_alloc_folio(struct hstate *h,
- struct mempolicy_interpreted *mpoli, u8 alloc_flags);
+ struct mempolicy_interpreted *mpoli, struct mm_struct *mm,
+ u8 alloc_flags);
struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
unsigned long addr, bool cow_from_owner);
struct folio *alloc_hugetlb_folio_nodemask(struct hstate *h, int preferred_nid,
diff --git a/include/linux/hugetlb_inline.h b/include/linux/hugetlb_inline.h
deleted file mode 100644
index 5c29cd3223a1..000000000000
--- a/include/linux/hugetlb_inline.h
+++ /dev/null
@@ -1,28 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0 */
-#ifndef _LINUX_HUGETLB_INLINE_H
-#define _LINUX_HUGETLB_INLINE_H
-
-#include <linux/mm.h>
-
-#ifdef CONFIG_HUGETLB_PAGE
-
-static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags)
-{
- return vma_flags_test(flags, VMA_HUGETLB_BIT);
-}
-
-#else
-
-static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags)
-{
- return false;
-}
-
-#endif
-
-static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma)
-{
- return is_vma_hugetlb_flags(&vma->flags);
-}
-
-#endif
diff --git a/include/linux/hwmon.h b/include/linux/hwmon.h
index dd713e193d0c..a3a7d27f3b5f 100644
--- a/include/linux/hwmon.h
+++ b/include/linux/hwmon.h
@@ -134,6 +134,8 @@ enum hwmon_in_attributes {
hwmon_in_max,
hwmon_in_lcrit,
hwmon_in_crit,
+ hwmon_in_lemergency,
+ hwmon_in_emergency,
hwmon_in_average,
hwmon_in_lowest,
hwmon_in_highest,
@@ -144,6 +146,8 @@ enum hwmon_in_attributes {
hwmon_in_max_alarm,
hwmon_in_lcrit_alarm,
hwmon_in_crit_alarm,
+ hwmon_in_lemergency_alarm,
+ hwmon_in_emergency_alarm,
hwmon_in_rated_min,
hwmon_in_rated_max,
hwmon_in_beep,
@@ -156,6 +160,8 @@ enum hwmon_in_attributes {
#define HWMON_I_MAX BIT(hwmon_in_max)
#define HWMON_I_LCRIT BIT(hwmon_in_lcrit)
#define HWMON_I_CRIT BIT(hwmon_in_crit)
+#define HWMON_I_LEMERGENCY BIT(hwmon_in_lemergency)
+#define HWMON_I_EMERGENCY BIT(hwmon_in_emergency)
#define HWMON_I_AVERAGE BIT(hwmon_in_average)
#define HWMON_I_LOWEST BIT(hwmon_in_lowest)
#define HWMON_I_HIGHEST BIT(hwmon_in_highest)
@@ -166,6 +172,8 @@ enum hwmon_in_attributes {
#define HWMON_I_MAX_ALARM BIT(hwmon_in_max_alarm)
#define HWMON_I_LCRIT_ALARM BIT(hwmon_in_lcrit_alarm)
#define HWMON_I_CRIT_ALARM BIT(hwmon_in_crit_alarm)
+#define HWMON_I_LEMERGENCY_ALARM BIT(hwmon_in_lemergency_alarm)
+#define HWMON_I_EMERGENCY_ALARM BIT(hwmon_in_emergency_alarm)
#define HWMON_I_RATED_MIN BIT(hwmon_in_rated_min)
#define HWMON_I_RATED_MAX BIT(hwmon_in_rated_max)
#define HWMON_I_BEEP BIT(hwmon_in_beep)
@@ -178,6 +186,7 @@ enum hwmon_curr_attributes {
hwmon_curr_max,
hwmon_curr_lcrit,
hwmon_curr_crit,
+ hwmon_curr_emergency,
hwmon_curr_average,
hwmon_curr_lowest,
hwmon_curr_highest,
@@ -188,6 +197,7 @@ enum hwmon_curr_attributes {
hwmon_curr_max_alarm,
hwmon_curr_lcrit_alarm,
hwmon_curr_crit_alarm,
+ hwmon_curr_emergency_alarm,
hwmon_curr_rated_min,
hwmon_curr_rated_max,
hwmon_curr_beep,
@@ -199,6 +209,7 @@ enum hwmon_curr_attributes {
#define HWMON_C_MAX BIT(hwmon_curr_max)
#define HWMON_C_LCRIT BIT(hwmon_curr_lcrit)
#define HWMON_C_CRIT BIT(hwmon_curr_crit)
+#define HWMON_C_EMERGENCY BIT(hwmon_curr_emergency)
#define HWMON_C_AVERAGE BIT(hwmon_curr_average)
#define HWMON_C_LOWEST BIT(hwmon_curr_lowest)
#define HWMON_C_HIGHEST BIT(hwmon_curr_highest)
@@ -209,6 +220,7 @@ enum hwmon_curr_attributes {
#define HWMON_C_MAX_ALARM BIT(hwmon_curr_max_alarm)
#define HWMON_C_LCRIT_ALARM BIT(hwmon_curr_lcrit_alarm)
#define HWMON_C_CRIT_ALARM BIT(hwmon_curr_crit_alarm)
+#define HWMON_C_EMERGENCY_ALARM BIT(hwmon_curr_emergency_alarm)
#define HWMON_C_RATED_MIN BIT(hwmon_curr_rated_min)
#define HWMON_C_RATED_MAX BIT(hwmon_curr_rated_max)
#define HWMON_C_BEEP BIT(hwmon_curr_beep)
diff --git a/include/linux/i2c.h b/include/linux/i2c.h
index 14ab4d3055af..6a847fc88d25 100644
--- a/include/linux/i2c.h
+++ b/include/linux/i2c.h
@@ -174,8 +174,23 @@ i2c_smbus_write_word_swapped(const struct i2c_client *client,
}
/* Returns the number of read bytes */
-s32 i2c_smbus_read_block_data(const struct i2c_client *client,
- u8 command, u8 *values);
+s32 __i2c_smbus_read_block_data(const struct i2c_client *client,
+ u8 command, u8 length, u8 *values);
+/*
+ * This monstrosity allows to call i2c_smbus_read_block_data() with either
+ * 3 or 4 arguments and will be removed once all users have been switched
+ * to the 4 argument version.
+ */
+#define __i2c_smbus_read_block_data_3arg(client, cmd, values) \
+ __i2c_smbus_read_block_data(client, cmd, I2C_SMBUS_BLOCK_MAX, values)
+#define __i2c_smbus_read_block_data_4arg(client, cmd, length, values) \
+ __i2c_smbus_read_block_data(client, cmd, length, values)
+#define __i2c_smbus_read_block_data_impl(_1, _2, _3, _4, impl, ...) impl
+#define i2c_smbus_read_block_data(client, cmd, varargs...) \
+ __i2c_smbus_read_block_data_impl(client, cmd, varargs, \
+ __i2c_smbus_read_block_data_4arg, \
+ __i2c_smbus_read_block_data_3arg) \
+ (client, cmd, varargs)
s32 i2c_smbus_write_block_data(const struct i2c_client *client,
u8 command, u8 length, const u8 *values);
/* Returns the number of read bytes */
@@ -742,6 +757,9 @@ struct i2c_adapter {
struct rt_mutex mux_lock;
int timeout; /* in jiffies */
+#ifdef CONFIG_I2C_DYNAMIC_TIMEOUT
+ int user_timeout; /* I2C_TIMEOUT ioctl value in jiffies */
+#endif
int retries;
struct device dev; /* the adapter device */
unsigned long locked_flags; /* owned by the I2C core */
@@ -913,14 +931,24 @@ unsigned int i2c_adapter_depth(struct i2c_adapter *adapter);
void i2c_parse_fw_timings(struct device *dev, struct i2c_timings *t, bool use_defaults);
+#ifdef CONFIG_I2C_DYNAMIC_TIMEOUT
+void i2c_update_timeout(struct i2c_adapter *adap, u32 bus_freq_hz,
+ size_t len, unsigned int safety_coeff,
+ unsigned int min_usec);
+#else
+static inline void i2c_update_timeout(struct i2c_adapter *adap, u32 bus_freq_hz,
+ size_t len, unsigned int safety_coeff,
+ unsigned int min_usec) {}
+#endif
+
/* Return the functionality mask */
static inline u32 i2c_get_functionality(struct i2c_adapter *adap)
{
return adap->algo->functionality(adap);
}
-/* Return 1 if adapter supports everything we need, 0 if not. */
-static inline int i2c_check_functionality(struct i2c_adapter *adap, u32 func)
+/* Return true if adapter supports everything we need, false if not. */
+static inline bool i2c_check_functionality(struct i2c_adapter *adap, u32 func)
{
return (func & i2c_get_functionality(adap)) == func;
}
diff --git a/include/linux/i3c/device.h b/include/linux/i3c/device.h
index 0f065b883ee0..e30ea2cb13cb 100644
--- a/include/linux/i3c/device.h
+++ b/include/linux/i3c/device.h
@@ -79,6 +79,10 @@ struct i3c_xfer {
enum i3c_error_code err;
};
+/* For splitting 'cmd' from struct i3c_xfer */
+#define I3C_HDR_CMD_RNW BIT(7)
+#define I3C_HDR_CMD_CODE GENMASK(6, 0)
+
/**
* enum i3c_dcr - I3C DCR values
* @I3C_DCR_GENERIC_DEVICE: generic I3C device
diff --git a/include/linux/i3c/hub.h b/include/linux/i3c/hub.h
new file mode 100644
index 000000000000..a368ea9e5ef7
--- /dev/null
+++ b/include/linux/i3c/hub.h
@@ -0,0 +1,92 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Copyright 2026 NXP
+ * Generic hub definitions and helper interfaces.
+ */
+#ifndef _LINUX_I3C_HUB_H
+#define _LINUX_I3C_HUB_H
+
+#include <linux/i3c/master.h>
+#include <linux/mutex.h>
+
+/**
+ * struct i3c_hub - Generic I3C hub context
+ * @ops: Vendor callbacks for port connection control
+ * @hub_dev: I3C device representing the hub on the parent bus
+ * @lock: Serializes hub port routing/forwarding; its lockdep class is keyed
+ * per hub nesting depth in i3c_hub_init().
+ */
+struct i3c_hub {
+ const struct i3c_hub_ops *ops;
+ struct i3c_device *hub_dev;
+ struct mutex lock; /* Serializes hub port routing. */
+};
+
+struct i3c_hub_controller {
+ struct i3c_master_controller *parent;
+ struct i3c_master_controller controller;
+ struct i3c_hub *hub;
+};
+
+struct i3c_hub_ops {
+ void (*enable_port)(struct i3c_master_controller *controller);
+ void (*disable_port)(struct i3c_master_controller *controller);
+};
+
+/**
+ * i3c_hub_enable_port() - Enable hub connection for a controller
+ * @controller: Virtual controller representing a hub port
+ *
+ * Retrieves hub context from controller drvdata and invokes the vendor
+ * callback to enable the associated port connection.
+ */
+static inline void i3c_hub_enable_port(struct i3c_master_controller *controller)
+{
+ struct i3c_hub_controller *hub_controller;
+ struct i3c_hub *hub;
+
+ hub_controller = dev_get_drvdata(&controller->dev);
+ if (!hub_controller || !hub_controller->hub)
+ return;
+
+ hub = hub_controller->hub;
+
+ if (hub && hub->ops && hub->ops->enable_port)
+ hub->ops->enable_port(controller);
+}
+
+/**
+ * i3c_hub_disable_port() - Disable hub connection for a controller
+ * @controller: Virtual controller representing a hub port
+ *
+ * Retrieves hub context from controller drvdata and invokes the vendor
+ * callback to disable the associated port connection.
+ */
+static inline void i3c_hub_disable_port(struct i3c_master_controller *controller)
+{
+ struct i3c_hub_controller *hub_controller;
+ struct i3c_hub *hub;
+
+ hub_controller = dev_get_drvdata(&controller->dev);
+ if (!hub_controller || !hub_controller->hub)
+ return;
+
+ hub = hub_controller->hub;
+
+ if (hub && hub->ops && hub->ops->disable_port)
+ hub->ops->disable_port(controller);
+}
+
+/*
+ * Controller operations used by the virtual controllers created for hub
+ * target ports. Hub drivers pass this to i3c_master_register_fwnode().
+ */
+extern const struct i3c_master_controller_ops i3c_hub_master_ops;
+
+void i3c_hub_init(struct i3c_hub *hub,
+ const struct i3c_hub_ops *ops,
+ struct i3c_device *hub_dev);
+
+int i3c_hub_reserve_parent_addrslots_from_dt(struct i3c_hub_controller *hubc,
+ struct device_node *node);
+#endif
diff --git a/include/linux/i3c/master.h b/include/linux/i3c/master.h
index f7ceec2b4477..d58070a72a49 100644
--- a/include/linux/i3c/master.h
+++ b/include/linux/i3c/master.h
@@ -536,6 +536,8 @@ struct i3c_master_controller_ops {
* Bit 0: SETDASA
* Bit 1: SETAASA
* All other bits are reserved.
+ * @instance: Zero-based instance number of the Bus Controller as defined by the
+ * DisCo specification I3C Target Address (_ADR) Encoding
* @wq: freezable workqueue which can be used by master
* drivers if they need to postpone operations that need to take place
* in a thread context. Typical examples are Hot Join processing which
@@ -573,6 +575,7 @@ struct i3c_master_controller {
} boardinfo;
struct i3c_bus bus;
u8 addr_method;
+ u8 instance;
struct workqueue_struct *wq;
struct work_struct hj_work;
struct work_struct reg_work;
@@ -649,9 +652,18 @@ DEFINE_FREE(i3c_master_dma_unmap_single, void *,
int i3c_master_reattach_i3c_dev_locked(struct i3c_dev_desc *dev,
u8 old_dyn_addr);
+int i3c_master_send_ccc_cmd(struct i3c_master_controller *master,
+ struct i3c_ccc_cmd *cmd);
+bool i3c_master_supports_ccc_cmd(struct i3c_master_controller *master,
+ const struct i3c_ccc_cmd *cmd);
int i3c_master_set_info(struct i3c_master_controller *master,
const struct i3c_device_info *info);
+int i3c_master_register_fwnode(struct i3c_master_controller *master,
+ struct device *parent,
+ struct fwnode_handle *fwnode,
+ const struct i3c_master_controller_ops *ops,
+ bool secondary);
int i3c_master_register(struct i3c_master_controller *master,
struct device *parent,
const struct i3c_master_controller_ops *ops,
@@ -775,4 +787,12 @@ void i3c_for_each_bus_locked(int (*fn)(struct i3c_bus *bus, void *data),
int i3c_register_notifier(struct notifier_block *nb);
int i3c_unregister_notifier(struct notifier_block *nb);
+enum i3c_addr_slot_status
+i3c_bus_get_addr_slot_status(struct i3c_bus *bus, u16 addr);
+
+void i3c_bus_set_addr_slot_status(struct i3c_bus *bus, u16 addr,
+ enum i3c_addr_slot_status status);
+
+void i3c_bus_maintenance_lock(struct i3c_bus *bus);
+void i3c_bus_maintenance_unlock(struct i3c_bus *bus);
#endif /* I3C_MASTER_H */
diff --git a/include/linux/idr.h b/include/linux/idr.h
index 789e23e67444..e2a4b6298511 100644
--- a/include/linux/idr.h
+++ b/include/linux/idr.h
@@ -16,6 +16,7 @@
#include <linux/gfp.h>
#include <linux/percpu.h>
#include <linux/cleanup.h>
+#include <linux/compiler.h>
struct idr {
struct radix_tree_root idr_rt;
@@ -269,7 +270,9 @@ struct ida {
#define IDA_INIT(name) { \
.xa = XARRAY_INIT(name, IDA_INIT_FLAGS) \
}
-#define DEFINE_IDA(name) struct ida name = IDA_INIT(name)
+#define DEFINE_IDA(name) \
+ struct ida name = IDA_INIT(name); \
+ ASSERT_STATIC_STORAGE(name)
int ida_alloc_range(struct ida *, unsigned int min, unsigned int max, gfp_t);
void ida_free(struct ida *, unsigned int id);
diff --git a/include/linux/ieee80211-eht.h b/include/linux/ieee80211-eht.h
index b62297a978e7..a699d108eee6 100644
--- a/include/linux/ieee80211-eht.h
+++ b/include/linux/ieee80211-eht.h
@@ -482,6 +482,7 @@ struct ieee80211_multi_link_elem {
#define IEEE80211_MLC_BASIC_PRES_MLD_ID 0x0200
#define IEEE80211_MLC_BASIC_PRES_EXT_MLD_CAPA_OP 0x0400
#define IEEE80211_MLC_BASIC_PRES_ENH_CRIT_UPD 0x0800
+#define IEEE80211_MLC_BASIC_PRES_AGE_OF_BSS_LOAD 0x1000
#define IEEE80211_MED_SYNC_DELAY_DURATION 0x00ff
#define IEEE80211_MED_SYNC_DELAY_SYNC_OFDM_ED_THRESH 0x0f00
@@ -550,6 +551,8 @@ struct ieee80211_mle_basic_common_info {
} __packed;
#define IEEE80211_MLC_PREQ_PRES_MLD_ID 0x0010
+#define IEEE80211_MLC_PREQ_PRES_MLD_MAC_ADDR 0x0020
+#define IEEE80211_MLC_PREQ_PRES_NEIGH_AP_MLD_MAC_ADDR 0x0040
struct ieee80211_mle_preq_common_info {
u8 len;
@@ -560,6 +563,7 @@ struct ieee80211_mle_preq_common_info {
#define IEEE80211_MLC_RECONF_PRES_EML_CAPA 0x0020
#define IEEE80211_MLC_RECONF_PRES_MLD_CAPA_OP 0x0040
#define IEEE80211_MLC_RECONF_PRES_EXT_MLD_CAPA_OP 0x0080
+#define IEEE80211_MLC_RECONF_PRES_TARGET_AP_MLD_ADDR 0x0100
/* no fixed fields in RECONF */
@@ -766,7 +770,7 @@ static inline u16 ieee80211_mle_get_mld_capa_op(const u8 *data)
#define IEEE80211_EHT_ML_EXT_MLD_CAPA_NSTR_UPDATE 0x0020
#define IEEE80211_EHT_ML_EXT_MLD_CAPA_EMLSR_ENA_ON_ONE_LINK 0x0040
#define IEEE80211_EHT_ML_EXT_MLD_CAPA_BTM_MLD_RECO_MULTI_AP 0x0080
-/* defined by UHR Draft P802.11bn_D1.3 Figure 9-1147 */
+/* defined by UHR Draft P802.11bn_D1.5 Figure 9-1168 */
#define IEEE80211_UHR_ML_EXT_MLD_CAPA_ML_PM 0x0100
/**
@@ -928,11 +932,17 @@ static inline bool ieee80211_mle_size_ok(const u8 *data, size_t len)
common += 2;
if (control & IEEE80211_MLC_BASIC_PRES_ENH_CRIT_UPD)
common += 1;
+ if (control & IEEE80211_MLC_BASIC_PRES_AGE_OF_BSS_LOAD)
+ common += 1;
break;
case IEEE80211_ML_CONTROL_TYPE_PREQ:
common += sizeof(struct ieee80211_mle_preq_common_info);
if (control & IEEE80211_MLC_PREQ_PRES_MLD_ID)
common += 1;
+ if (control & IEEE80211_MLC_PREQ_PRES_MLD_MAC_ADDR)
+ common += ETH_ALEN;
+ if (control & IEEE80211_MLC_PREQ_PRES_NEIGH_AP_MLD_MAC_ADDR)
+ common += ETH_ALEN;
break;
case IEEE80211_ML_CONTROL_TYPE_RECONF:
common += 1;
@@ -944,6 +954,8 @@ static inline bool ieee80211_mle_size_ok(const u8 *data, size_t len)
common += 2;
if (control & IEEE80211_MLC_RECONF_PRES_EXT_MLD_CAPA_OP)
common += 2;
+ if (control & IEEE80211_MLC_RECONF_PRES_TARGET_AP_MLD_ADDR)
+ common += ETH_ALEN;
break;
case IEEE80211_ML_CONTROL_TYPE_TDLS:
common += sizeof(struct ieee80211_mle_tdls_common_info);
@@ -1003,6 +1015,7 @@ enum ieee80211_mle_subelems {
#define IEEE80211_MLE_STA_CONTROL_BSS_PARAM_CHANGE_CNT_PRESENT 0x0800
#define IEEE80211_MLE_STA_CONTROL_ENH_CRIT_UPD_PRESENT 0x1000
#define IEEE80211_MLE_STA_CONTROL_AP_CONDUCTED_TX_PWR_PRESENT 0x2000
+#define IEEE80211_MLE_STA_CONTROL_AGE_OF_BSS_LOAD_PRESENT 0x4000
struct ieee80211_mle_per_sta_profile {
__le16 control;
@@ -1054,6 +1067,9 @@ static inline bool ieee80211_mle_basic_sta_prof_size_ok(const u8 *data,
if (control & IEEE80211_MLE_STA_CONTROL_AP_CONDUCTED_TX_PWR_PRESENT)
info_len += 1;
+ if (control & IEEE80211_MLE_STA_CONTROL_AGE_OF_BSS_LOAD_PRESENT)
+ info_len += 1;
+
return prof->sta_info_len >= info_len &&
fixed + prof->sta_info_len - 1 <= len;
}
diff --git a/include/linux/ieee80211-mesh.h b/include/linux/ieee80211-mesh.h
index 7eb15834531c..9e548b9173df 100644
--- a/include/linux/ieee80211-mesh.h
+++ b/include/linux/ieee80211-mesh.h
@@ -361,8 +361,7 @@ ieee80211_mesh_hwmp_perr_get_rcode(const u8 *ie, u8 dst_idx)
/* IEEE Std 802.11-2016 9.4.2.113 PREQ element */
static inline bool ieee80211_mesh_preq_size_ok(const u8 *pos, u8 elen)
{
- struct ieee80211_mesh_hwmp_preq_bottom *preq_elem_bottom =
- ieee80211_mesh_hwmp_preq_get_bottom(pos);
+ struct ieee80211_mesh_hwmp_preq_bottom *preq_elem_bottom;
u8 target_count;
int needed;
@@ -378,6 +377,7 @@ static inline bool ieee80211_mesh_preq_size_ok(const u8 *pos, u8 elen)
if (elen < needed)
return false;
+ preq_elem_bottom = ieee80211_mesh_hwmp_preq_get_bottom(pos);
target_count = preq_elem_bottom->target_count;
/* IEEE Std 802.11-2016 Table 14-10 to 14-16 */
if (target_count < 1)
diff --git a/include/linux/ieee80211-uhr.h b/include/linux/ieee80211-uhr.h
index 665d4b3a5b41..19e9260d0db8 100644
--- a/include/linux/ieee80211-uhr.h
+++ b/include/linux/ieee80211-uhr.h
@@ -17,32 +17,32 @@
#define IEEE80211_UHR_OPER_PARAMS_PEDCA_ENA 0x0004
#define IEEE80211_UHR_OPER_PARAMS_DBE_ENA 0x0008
#define IEEE80211_UHR_OPER_PARAMS_DBE_BW 0x0070
-#define IEEE80211_UHR_OPER_PARAMS_DUO_PRES 0x0080
-#define IEEE80211_UHR_OPER_PARAMS_DPS_PRES 0x0100
-#define IEEE80211_UHR_OPER_PARAMS_NPCA_PRES 0x0200
-#define IEEE80211_UHR_OPER_PARAMS_PEDCA_PRES 0x0400
-#define IEEE80211_UHR_OPER_PARAMS_DBE_PRES 0x0800
+#define IEEE80211_UHR_OPER_PARAMS_ELR_RX 0x0080
+#define IEEE80211_UHR_OPER_PARAMS_DUO_PRES 0x0100
+#define IEEE80211_UHR_OPER_PARAMS_DPS_PRES 0x0200
+#define IEEE80211_UHR_OPER_PARAMS_NPCA_PRES 0x0400
+#define IEEE80211_UHR_OPER_PARAMS_PEDCA_PRES 0x0800
+#define IEEE80211_UHR_OPER_PARAMS_DBE_PRES 0x1000
struct ieee80211_uhr_operation {
__le16 params;
- u8 basic_mcs_nss_set[4];
u8 variable[];
} __packed;
-#define IEEE80211_UHR_NPCA_PARAMS_PRIMARY_CHAN_OFFS 0x0000000F
-#define IEEE80211_UHR_NPCA_PARAMS_MIN_DUR_THRESH 0x000000F0
-#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_DELAY 0x00003F00
-#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_BACK_DELAY 0x000FC000
-#define IEEE80211_UHR_NPCA_PARAMS_INIT_QSRC 0x00300000
-#define IEEE80211_UHR_NPCA_PARAMS_MOPLEN 0x00400000
-#define IEEE80211_UHR_NPCA_PARAMS_DIS_SUBCH_BMAP_PRES 0x00800000
+#define IEEE80211_UHR_NPCA_PARAMS_PRIMARY_CHAN 0x000000FF
+#define IEEE80211_UHR_NPCA_PARAMS_MIN_DUR_THRESH 0x00000F00
+#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_DELAY 0x0003F000
+#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_BACK_DELAY 0x00FC0000
+#define IEEE80211_UHR_NPCA_PARAMS_INIT_QSRC 0x03000000
+#define IEEE80211_UHR_NPCA_PARAMS_MOPLEN 0x04000000
+#define IEEE80211_UHR_NPCA_PARAMS_DIS_SUBCH_BMAP_PRES 0x08000000
/**
* struct ieee80211_uhr_npca_info - npca operation information
*
* This structure is the "NPCA Operation Parameters field format" of "UHR
- * Operation Element" fields as described in P802.11bn_D1.3
- * subclause 9.4.2.353. See Figure 9-aa4.
+ * Operation Element" fields as described in P802.11bn_D1.5
+ * subclause 9.4.2.355.2, see Figure 9-aa6.
*
* Refer to IEEE80211_UHR_NPCA*
* @params:
@@ -97,7 +97,7 @@ struct ieee80211_uhr_npca_info {
* struct ieee80211_uhr_dps_info - DPS operation information
*
* This structure is the "DPS Operation Parameter field" of "UHR
- * Operation Element" fields as described in P802.11bn_D1.3
+ * Operation Element" fields as described in P802.11bn_D1.5
* subclause 9.4.1.87. See Figure 9-207u.
*
* Refer to IEEE80211_UHR_DPS*
@@ -211,8 +211,8 @@ static inline int ieee80211_uhr_dbe_bw_mhz(enum ieee80211_uhr_dbe_oper_bw bw)
* struct ieee80211_uhr_dbe_info - DBE operation information
*
* This structure is the "DBE Operation Parameters field" of
- * "UHR Operation Element" fields as described in P802.11bn_D1.3
- * subclause 9.4.2.353. See Figure 9-aa6.
+ * "UHR Operation Element" fields as described in P802.11bn_D1.5
+ * subclause 9.4.2.355.4, see Figure 9-aa9.
*
* Refer to IEEE80211_UHR_DBE_OPER*
* @params:
@@ -234,25 +234,23 @@ struct ieee80211_uhr_dbe_info {
__le16 dis_subch_bmap[];
} __packed;
-#define IEEE80211_UHR_P_EDCA_ECWMIN 0x0F
-#define IEEE80211_UHR_P_EDCA_ECWMAX 0xF0
-#define IEEE80211_UHR_P_EDCA_AIFSN 0x000F
-#define IEEE80211_UHR_P_EDCA_CW_DS 0x0030
-#define IEEE80211_UHR_P_EDCA_PSRC_THRESHOLD 0x01C0
-#define IEEE80211_UHR_P_EDCA_QSRC_THRESHOLD 0x0600
+#define IEEE80211_UHR_P_EDCA_ECWMIN 0x0007
+#define IEEE80211_UHR_P_EDCA_ECWMAX 0x0038
+#define IEEE80211_UHR_P_EDCA_AIFSN 0x01C0
+#define IEEE80211_UHR_P_EDCA_CW_DS 0x0600
+#define IEEE80211_UHR_P_EDCA_PSRC_THRESHOLD 0x3800
+#define IEEE80211_UHR_P_EDCA_QSRC_THRESHOLD 0xC000
/**
* struct ieee80211_uhr_p_edca_info - P-EDCA operation information
*
* This structure is the "P-EDCA Operation Parameters field" of
- * "UHR Operation Element" fields as described in P802.11bn_D1.3
- * subclause 9.4.2.353. See Figure 9-aa5.
+ * "UHR Operation Element" fields as described in P802.11bn_D1.5
+ * subclause 9.4.2.355.3, see Figure 9-aa8.
*
* Refer to IEEE80211_UHR_P_EDCA*
- * @p_edca_ec: P-EDCA ECWmin and ECWmax.
- * These fields indicate the CWmin and CWmax values used by a
- * P-EDCA STA during P-EDCA contention.
- * @params: AIFSN, CW DS, PSRC threshold, and QSRC threshold.
+ * @params: ECWmin, ECWmax, AIFSN, CW DS, PSRC threshold, and QSRC threshold.
+ * - CWmin/CWmax values used by a P-EDCA STA during P-EDCA contention.
* - The AIFSN field indicates the AIFSN value used by a P-EDCA STA
* during P-EDCA contention.
* - The CW DS field indicates the value used for randomization of the
@@ -266,7 +264,6 @@ struct ieee80211_uhr_dbe_info {
* value 0 is reserved.
*/
struct ieee80211_uhr_p_edca_info {
- u8 p_edca_ec;
__le16 params;
} __packed;
@@ -302,7 +299,7 @@ static inline bool ieee80211_uhr_oper_size_ok(const u8 *data, u8 len)
}
}
- /* P-EDCA Operation Parameters (fixed 3 bytes) */
+ /* P-EDCA Operation Parameters */
if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_PEDCA_PRES)) {
needed += sizeof(struct ieee80211_uhr_p_edca_info);
if (len < needed)
@@ -341,6 +338,9 @@ ieee80211_uhr_npca_info(const struct ieee80211_uhr_operation *oper)
if (!(oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_NPCA_ENA)))
return NULL;
+ if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DUO_PRES))
+ pos += 1;
+
if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DPS_PRES))
pos += sizeof(struct ieee80211_uhr_dps_info);
@@ -372,6 +372,9 @@ ieee80211_uhr_oper_dbe_info(const struct ieee80211_uhr_operation *oper)
if (!(oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DBE_ENA)))
return NULL;
+ if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DUO_PRES))
+ pos += 1;
+
if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DPS_PRES))
pos += sizeof(struct ieee80211_uhr_dps_info);
@@ -415,28 +418,42 @@ ieee80211_uhr_oper_dbe_info(const struct ieee80211_uhr_operation *oper)
#define IEEE80211_UHR_MAC_CAP2_UHR_OM_PU_TO_LOW 0xC0
#define IEEE80211_UHR_MAC_CAP3_UHR_OM_PU_TO_HIGH 0x03
-#define IEEE80211_UHR_MAC_CAP3_PARAM_UPD_ADV_NOTIF_INTV 0x1C
-#define IEEE80211_UHR_MAC_CAP3_UPD_IND_TIM_INTV_LOW 0xE0
+#define IEEE80211_UHR_MAC_CAP3_PARAM_UPD_ADV_NOTIF_INTV 0x7C
+#define IEEE80211_UHR_MAC_CAP3_UPD_IND_TIM_INTV_LOW 0x80
-#define IEEE80211_UHR_MAC_CAP4_UPD_IND_TIM_INTV_HIGH 0x03
-#define IEEE80211_UHR_MAC_CAP4_BOUNDED_ESS 0x04
-#define IEEE80211_UHR_MAC_CAP4_BTM_ASSURANCE 0x08
-#define IEEE80211_UHR_MAC_CAP4_CO_BF_SUPP 0x10
+#define IEEE80211_UHR_MAC_CAP4_UPD_IND_TIM_INTV_HIGH 0x0F
+#define IEEE80211_UHR_MAC_CAP4_BOUNDED_ESS 0x10
+#define IEEE80211_UHR_MAC_CAP4_BTM_ASSURANCE 0x20
+#define IEEE80211_UHR_MAC_CAP4_CO_BF_SUPP 0x40
+#define IEEE80211_UHR_MAC_CAP4_CO_SR_SUPP 0x80
+
+#define IEEE80211_UHR_MAC_CAP5_MAPC_ENH_MSRMT_SUPP 0x01
#define IEEE80211_UHR_MAC_CAP_DBE_MAX_BW 0x07
#define IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_160_PRES 0x08
#define IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_320_PRES 0x10
+/* struct ieee80211_uhr_cap_dbe::bwcap[].cap */
+#define IEEE80211_UHR_MAC_CAP_DBE_CAP_NUM_SND_DIMS 0x07
+#define IEEE80211_UHR_MAC_CAP_DBE_CAP_NON_OFDMA_UL_MUMIMO 0x08
+#define IEEE80211_UHR_MAC_CAP_DBE_CAP_MU_BEAMFORMER 0x10
+#define IEEE80211_UHR_MAC_CAP_DBE_CAP_BEAMFORMEE_SS 0xe0
+
struct ieee80211_uhr_cap_dbe {
u8 cap;
+ u8 max_switch_time_period;
+ u8 mode_change_intvl;
/* present 0, 1 or 2 times depending on _PRES bits */
- struct ieee80211_eht_mcs_nss_supp_bw eht_mcs_map[];
+ struct {
+ struct ieee80211_eht_mcs_nss_supp_bw eht_mcs_map;
+ u8 cap;
+ } __packed bwcap[];
} __packed;
/**
* enum ieee80211_uhr_dbe_max_supported_bw - DBE Maximum Supported Bandwidth
*
- * As per spec P802.11bn_D1.3 "Table 9-bb5—Encoding of the DBE Maximum
+ * As per spec P802.11bn_D1.5 Table 9-bb8 "Encoding of the DBE Maximum
* Supported Bandwidth field".
*
* @IEEE80211_UHR_DBE_MAX_BW_40: Indicates 40 MHz DBE max supported bw
@@ -520,10 +537,10 @@ static inline bool ieee80211_uhr_capa_size_ok(const u8 *data, u8 len,
dbe = (const void *)cap->variable;
if (dbe->cap & IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_160_PRES)
- needed += sizeof(dbe->eht_mcs_map[0]);
+ needed += sizeof(dbe->bwcap[0]);
if (dbe->cap & IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_320_PRES)
- needed += sizeof(dbe->eht_mcs_map[0]);
+ needed += sizeof(dbe->bwcap[0]);
}
return len >= needed;
diff --git a/include/linux/ieee80211.h b/include/linux/ieee80211.h
index 26e674038865..73ac5336a17a 100644
--- a/include/linux/ieee80211.h
+++ b/include/linux/ieee80211.h
@@ -1828,6 +1828,7 @@ enum ieee80211_eid_ext {
WLAN_EID_EXT_BANDWIDTH_INDICATION = 135,
WLAN_EID_EXT_KNOWN_STA_IDENTIFCATION = 136,
WLAN_EID_EXT_NON_AP_STA_REG_CON = 137,
+ WLAN_EID_EXT_CIP_CAPA = 150,
WLAN_EID_EXT_UHR_OPER = 151,
WLAN_EID_EXT_UHR_CAPA = 152,
WLAN_EID_EXT_MACP = 153,
@@ -2421,8 +2422,7 @@ static inline bool ieee80211_is_bufferable_mmpdu(struct sk_buff *skb)
/* action frame - additionally check for non-bufferable FTM */
- if (mgmt->u.action.category != WLAN_CATEGORY_PUBLIC &&
- mgmt->u.action.category != WLAN_CATEGORY_PROTECTED_DUAL_OF_ACTION)
+ if (mgmt->u.action.category != WLAN_CATEGORY_PUBLIC)
return true;
if (mgmt->u.action.action_code == WLAN_PUB_ACTION_FTM_REQUEST ||
@@ -2900,4 +2900,25 @@ static inline bool ieee80211_check_tim(const struct ieee80211_tim_ie *tim,
__ieee80211_check_tim(tim, tim_len, aid);
}
+/**
+ * enum ieee80211_cip_cap_fields - CIP Capabilities element fields
+ * @IEEE80211_CIP_CAP_MIC_PADDING: The MIC padding subfield in the CIP
+ * capabilities
+ * @IEEE80211_CIP_CAP_PROTECTED_CTRL_FRAME_ONLY: The MIC Padding For Protected
+ * Control Frames Only bit in CIP capabilities
+ */
+enum ieee80211_cip_cap_fields {
+ IEEE80211_CIP_CAP_MIC_PADDING = 0x0F,
+ IEEE80211_CIP_CAP_PROTECTED_CTRL_FRAME_ONLY = 0x10,
+};
+
+/**
+ * struct ieee80211_cip_cap - CIP Capabilities element
+ *
+ * @v: The value of the Control Integrity Protocol Capability element
+ */
+struct ieee80211_cip_cap {
+ u8 v;
+} __packed;
+
#endif /* LINUX_IEEE80211_H */
diff --git a/include/linux/if_vlan.h b/include/linux/if_vlan.h
index 20cc16ea4e5a..4846032bf4ff 100644
--- a/include/linux/if_vlan.h
+++ b/include/linux/if_vlan.h
@@ -365,6 +365,9 @@ static inline int __vlan_insert_inner_tag(struct sk_buff *skb,
const u8 meta_len = mac_len > ETH_TLEN ? skb_metadata_len(skb) : 0;
struct vlan_ethhdr *veth;
+ if (unlikely(!pskb_may_pull(skb, mac_len)))
+ return -EINVAL;
+
if (skb_cow_head(skb, meta_len + VLAN_HLEN) < 0)
return -ENOMEM;
diff --git a/include/linux/igmp.h b/include/linux/igmp.h
index 3a2d35a9f307..e075611344ef 100644
--- a/include/linux/igmp.h
+++ b/include/linux/igmp.h
@@ -14,6 +14,7 @@
#include <linux/timer.h>
#include <linux/in.h>
#include <linux/ip.h>
+#include <linux/net.h>
#include <linux/refcount.h>
#include <linux/sockptr.h>
#include <uapi/linux/igmp.h>
@@ -57,20 +58,21 @@ struct ip_mc_socklist {
};
struct ip_sf_list {
- struct ip_sf_list *sf_next;
+ struct ip_sf_list __rcu *sf_next;
unsigned long sf_count[2]; /* include/exclude counts */
__be32 sf_inaddr;
unsigned char sf_gsresp; /* include in g & s response? */
unsigned char sf_oldin; /* change state */
unsigned char sf_crcount; /* retrans. left to send */
+ struct rcu_head rcu;
};
struct ip_mc_list {
struct in_device *interface;
__be32 multiaddr;
unsigned int sfmode;
- struct ip_sf_list *sources;
- struct ip_sf_list *tomb;
+ struct ip_sf_list __rcu *sources;
+ struct ip_sf_list __rcu *tomb;
unsigned long sfcount[2];
union {
struct ip_mc_list *next;
@@ -272,7 +274,7 @@ extern int ip_mc_source(int add, int omode, struct sock *sk,
struct ip_mreq_source *mreqs, int ifindex);
extern int ip_mc_msfilter(struct sock *sk, struct ip_msfilter *msf,int ifindex);
extern int ip_mc_msfget(struct sock *sk, struct ip_msfilter *msf,
- sockptr_t optval, sockptr_t optlen);
+ sockopt_t *opt);
extern int ip_mc_gsfget(struct sock *sk, struct group_filter *gsf,
sockptr_t optval, size_t offset);
extern int ip_mc_sf_allow(const struct sock *sk, __be32 local, __be32 rmt,
diff --git a/include/linux/input/samsung-keypad.h b/include/linux/input/samsung-keypad.h
deleted file mode 100644
index ab6b97114c08..000000000000
--- a/include/linux/input/samsung-keypad.h
+++ /dev/null
@@ -1,39 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0-or-later */
-/*
- * Samsung Keypad platform data definitions
- *
- * Copyright (C) 2010 Samsung Electronics Co.Ltd
- * Author: Joonyoung Shim <jy0922.shim@samsung.com>
- */
-
-#ifndef __SAMSUNG_KEYPAD_H
-#define __SAMSUNG_KEYPAD_H
-
-#include <linux/input/matrix_keypad.h>
-
-#define SAMSUNG_MAX_ROWS 8
-#define SAMSUNG_MAX_COLS 8
-
-/**
- * struct samsung_keypad_platdata - Platform device data for Samsung Keypad.
- * @keymap_data: pointer to &matrix_keymap_data.
- * @rows: number of keypad row supported.
- * @cols: number of keypad col supported.
- * @no_autorepeat: disable key autorepeat.
- * @wakeup: controls whether the device should be set up as wakeup source.
- * @cfg_gpio: configure the GPIO.
- *
- * Initialisation data specific to either the machine or the platform
- * for the device driver to use or call-back when configuring gpio.
- */
-struct samsung_keypad_platdata {
- const struct matrix_keymap_data *keymap_data;
- unsigned int rows;
- unsigned int cols;
- bool no_autorepeat;
- bool wakeup;
-
- void (*cfg_gpio)(unsigned int rows, unsigned int cols);
-};
-
-#endif /* __SAMSUNG_KEYPAD_H */
diff --git a/include/linux/intel_vsec.h b/include/linux/intel_vsec.h
index 843cda8f8644..917d9397a993 100644
--- a/include/linux/intel_vsec.h
+++ b/include/linux/intel_vsec.h
@@ -90,13 +90,25 @@ enum intel_vsec_quirks {
* @read_telem: when specified, called by client driver to access PMT
* data (instead of direct copy).
* * dev: device reference for the callback's use
- * * guid: ID of data to acccss
+ * * guid: ID of data to access
* * data: buffer for the data to be copied
* * off: offset into the requested buffer
* * count: size of buffer
+ * @read_reg: when specified called by client driver to read PMT state
+ * * dev: device reference for the callback's use
+ * * guid: ID of data to access
+ * * reg_data: register data
+ * * offset: offset of register to read
+ * @write_reg: when specified called by client driver to write PMT state
+ * * dev: device reference for the callback's use
+ * * guid: ID of data to access
+ * * reg_data: register data
+ * * offset: offset of register to write
*/
struct pmt_callbacks {
int (*read_telem)(struct device *dev, u32 guid, u64 *data, loff_t off, u32 count);
+ int (*read_reg)(struct device *dev, u32 guid, u32 *reg_data, u32 offset);
+ int (*write_reg)(struct device *dev, u32 guid, u32 reg_data, u32 offset);
};
struct vsec_feature_dependency {
diff --git a/include/linux/interrupt.h b/include/linux/interrupt.h
index 3bf969ad8fe0..52bb684090c0 100644
--- a/include/linux/interrupt.h
+++ b/include/linux/interrupt.h
@@ -573,9 +573,11 @@ enum
* _ IRQ_POLL: irq_poll_cpu_dead() migrates the queue
*
* _ (HR)TIMER_SOFTIRQ: (hr)timers_dead_cpu() migrates the queue
+ *
+ * _ BLOCK_SOFTIRQ: blk_softirq_cpu_dead() completes the remaining requests
*/
-#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(IRQ_POLL_SOFTIRQ) |\
- BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ))
+#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(BLOCK_SOFTIRQ) |\
+ BIT(IRQ_POLL_SOFTIRQ) | BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ))
/* map softirq index to softirq name. update 'softirq_to_name' in
diff --git a/include/linux/interrupt_rc.h b/include/linux/interrupt_rc.h
index b9a7f05ecf42..a9ed937a80e7 100644
--- a/include/linux/interrupt_rc.h
+++ b/include/linux/interrupt_rc.h
@@ -20,11 +20,8 @@
/* Per-CPU interrupt disabling state for local_interrupt_{disable,enable}(). */
DECLARE_PER_CPU(unsigned long, local_interrupt_disable_state);
-static __always_inline void __local_interrupt_disable(void)
+static __always_inline void __local_interrupt_save_state(unsigned long flags)
{
- unsigned long flags;
-
- local_irq_save(flags);
raw_cpu_write(local_interrupt_disable_state, flags);
}
@@ -36,9 +33,9 @@ static __always_inline void __local_interrupt_enable(void)
}
#ifndef INSTANTIATE_EXPORTED_INTERRUPT_DISABLE
-static __always_inline void _local_interrupt_disable(void)
+static __always_inline void _local_interrupt_save_state(unsigned long flags)
{
- __local_interrupt_disable();
+ __local_interrupt_save_state(flags);
}
static __always_inline void _local_interrupt_enable(void)
@@ -46,27 +43,30 @@ static __always_inline void _local_interrupt_enable(void)
__local_interrupt_enable();
}
#else
-extern void _local_interrupt_disable(void);
+extern void _local_interrupt_save_state(unsigned long flags);
extern void _local_interrupt_enable(void);
#endif
#else /* !MODULE */
-extern void _local_interrupt_disable(void);
+extern void _local_interrupt_save_state(unsigned long flags);
extern void _local_interrupt_enable(void);
#endif /* !MODULE */
+#define hardirq_disable_enter() __preempt_count_add_return(HARDIRQ_DISABLE_OFFSET)
+#define hardirq_disable_exit() __preempt_count_sub_return(HARDIRQ_DISABLE_OFFSET)
+
static inline void local_interrupt_disable(void)
{
int new_count;
+ unsigned long flags;
WARN_ON_ONCE(in_nmi());
+ local_irq_save(flags);
new_count = hardirq_disable_enter();
- /* Interrupts can happen here, but it's OK, see __irq_exit_rcu(). */
-
if ((new_count & HARDIRQ_DISABLE_MASK) == HARDIRQ_DISABLE_OFFSET)
- _local_interrupt_disable();
+ _local_interrupt_save_state(flags);
}
static inline void local_interrupt_enable(void)
diff --git a/include/linux/io_uring.h b/include/linux/io_uring.h
index d1aa4edfc2a5..505caa99c621 100644
--- a/include/linux/io_uring.h
+++ b/include/linux/io_uring.h
@@ -2,10 +2,42 @@
#ifndef _LINUX_IO_URING_H
#define _LINUX_IO_URING_H
+#include <linux/io_uring_types.h>
#include <linux/sched.h>
#include <linux/xarray.h>
#include <uapi/linux/io_uring.h>
+static inline void req_set_fail(struct io_kiocb *req)
+{
+ req->flags |= REQ_F_FAIL;
+ if (req->flags & REQ_F_CQE_SKIP) {
+ req->flags &= ~REQ_F_CQE_SKIP;
+ req->flags |= REQ_F_SKIP_LINK_CQES;
+ }
+}
+
+static inline void io_req_set_res(struct io_kiocb *req, s32 res, u32 cflags)
+{
+ req->cqe.res = res;
+ req->cqe.flags = cflags;
+}
+
+static inline u32 ctx_cqe32_flags(struct io_ring_ctx *ctx)
+{
+ if (ctx->flags & IORING_SETUP_CQE_MIXED)
+ return IORING_CQE_F_32;
+ return 0;
+}
+
+static inline void io_req_set_res32(struct io_kiocb *req, s32 res, u32 cflags,
+ __u64 extra1, __u64 extra2)
+{
+ req->cqe.res = res;
+ req->cqe.flags = cflags | ctx_cqe32_flags(req->ctx);
+ req->big_cqe.extra1 = extra1;
+ req->big_cqe.extra2 = extra2;
+}
+
#if defined(CONFIG_IO_URING)
void __io_uring_cancel(bool cancel_all);
void __io_uring_free(struct task_struct *tsk);
diff --git a/include/linux/io_uring/cmd.h b/include/linux/io_uring/cmd.h
index 42801f0b6456..e18d4af62e1a 100644
--- a/include/linux/io_uring/cmd.h
+++ b/include/linux/io_uring/cmd.h
@@ -3,6 +3,7 @@
#define _LINUX_IO_URING_CMD_H
#include <uapi/linux/io_uring.h>
+#include <linux/io_uring.h>
#include <linux/io_uring_types.h>
#include <linux/blk-mq.h>
@@ -41,6 +42,24 @@ static inline void io_uring_cmd_private_sz_check(size_t cmd_sz)
((pdu_type *)&(cmd)->pdu) \
)
+static inline void io_uring_cmd_set_res(struct io_uring_cmd *cmd, s32 ret)
+{
+ struct io_kiocb *req = cmd_to_io_kiocb(cmd);
+
+ if (ret < 0)
+ req_set_fail(req);
+ io_req_set_res(req, ret, 0);
+}
+
+static inline void io_uring_cmd_set_res32(struct io_uring_cmd *cmd, s32 ret, u64 res2)
+{
+ struct io_kiocb *req = cmd_to_io_kiocb(cmd);
+
+ if (ret < 0)
+ req_set_fail(req);
+ io_req_set_res32(req, ret, 0, res2, 0);
+}
+
#if defined(CONFIG_IO_URING)
int io_uring_cmd_import_fixed(u64 ubuf, unsigned long len, int rw,
struct iov_iter *iter,
@@ -56,11 +75,12 @@ int io_uring_cmd_import_fixed_vec(struct io_uring_cmd *ioucmd,
* Completes the request, i.e. posts an io_uring CQE and deallocates @ioucmd
* and the corresponding io_uring request.
*
+ * io_uring_cmd_set_res()/io_uring_cmd_set_res32() must be called first.
+ *
* Note: the caller should never hard code @issue_flags and is only allowed
* to pass the mask provided by the core io_uring code.
*/
-void __io_uring_cmd_done(struct io_uring_cmd *cmd, s32 ret, u64 res2,
- unsigned issue_flags, bool is_cqe32);
+void __io_uring_cmd_done(struct io_uring_cmd *, unsigned issue_flags);
void __io_uring_cmd_do_in_task(struct io_uring_cmd *ioucmd,
io_req_tw_func_t task_work_cb,
@@ -116,8 +136,8 @@ static inline int io_uring_cmd_import_fixed_vec(struct io_uring_cmd *ioucmd,
{
return -EOPNOTSUPP;
}
-static inline void __io_uring_cmd_done(struct io_uring_cmd *cmd, s32 ret,
- u64 ret2, unsigned issue_flags, bool is_cqe32)
+static inline void __io_uring_cmd_done(struct io_uring_cmd *cmd,
+ unsigned issue_flags)
{
}
static inline void __io_uring_cmd_do_in_task(struct io_uring_cmd *ioucmd,
@@ -205,13 +225,15 @@ static inline void *io_uring_cmd_ctx_handle(struct io_uring_cmd *cmd)
static inline void io_uring_cmd_done(struct io_uring_cmd *ioucmd, s32 ret,
unsigned issue_flags)
{
- return __io_uring_cmd_done(ioucmd, ret, 0, issue_flags, false);
+ io_uring_cmd_set_res(ioucmd, ret);
+ __io_uring_cmd_done(ioucmd, issue_flags);
}
static inline void io_uring_cmd_done32(struct io_uring_cmd *ioucmd, s32 ret,
u64 res2, unsigned issue_flags)
{
- return __io_uring_cmd_done(ioucmd, ret, res2, issue_flags, true);
+ io_uring_cmd_set_res32(ioucmd, ret, res2);
+ __io_uring_cmd_done(ioucmd, issue_flags);
}
#endif /* _LINUX_IO_URING_CMD_H */
diff --git a/include/linux/io_uring_types.h b/include/linux/io_uring_types.h
index 39629ee77b91..90fea94ad202 100644
--- a/include/linux/io_uring_types.h
+++ b/include/linux/io_uring_types.h
@@ -522,6 +522,8 @@ struct io_ring_ctx {
/* protected by ->completion_lock */
unsigned nr_req_allocated;
+ /* pending SEND_ZC notifications, protected by ->uring_lock */
+ unsigned nr_notifs;
#ifdef CONFIG_NET_RX_BUSY_POLL
struct list_head napi_list; /* track busy poll napi_id */
diff --git a/include/linux/iomap.h b/include/linux/iomap.h
index bc7ae6327dbf..59718f73c15a 100644
--- a/include/linux/iomap.h
+++ b/include/linux/iomap.h
@@ -483,13 +483,35 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno,
#define IOMAP_IOEND_BOUNDARY (1U << 2)
/* is direct I/O */
#define IOMAP_IOEND_DIRECT (1U << 3)
+/* generate integrity (PI) information */
+#ifdef CONFIG_BLK_DEV_INTEGRITY
+#define IOMAP_IOEND_INTEGRITY (1U << 4)
+#else
+#define IOMAP_IOEND_INTEGRITY 0
+#endif /* CONFIG_BLK_DEV_INTEGRITY */
/*
* Flags that if set on either ioend prevent the merge of two ioends.
* (IOMAP_IOEND_BOUNDARY also prevents merges, but only one-way)
*/
#define IOMAP_IOEND_NOMERGE_FLAGS \
- (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT)
+ (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT | \
+ IOMAP_IOEND_INTEGRITY)
+
+/* ioend flags directly implied by iomap flags */
+static inline u16 iomap_ioend_flags(const struct iomap *iomap)
+{
+ unsigned int flags = 0;
+
+ if (iomap->type == IOMAP_UNWRITTEN)
+ flags |= IOMAP_IOEND_UNWRITTEN;
+ if (iomap->flags & IOMAP_F_SHARED)
+ flags |= IOMAP_IOEND_SHARED;
+ if (iomap->flags & IOMAP_F_INTEGRITY)
+ flags |= IOMAP_IOEND_INTEGRITY;
+
+ return flags;
+}
/*
* Structure for writeback I/O completions.
@@ -500,6 +522,7 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno,
struct iomap_ioend {
struct list_head io_list; /* next ioend in chain */
u16 io_flags; /* IOMAP_IOEND_* */
+ u32 io_bvec_offset; /* offset into first bvec */
struct inode *io_inode; /* file being written to */
size_t io_size; /* size of the extent */
atomic_t io_remaining; /* completetion defer count */
@@ -517,6 +540,13 @@ static inline struct iomap_ioend *iomap_ioend_from_bio(struct bio *bio)
return container_of(bio, struct iomap_ioend, io_bio);
}
+#define BVEC_ITER_IOEND(_ioend) \
+{ \
+ .bi_sector = (_ioend)->io_sector, \
+ .bi_size = (_ioend)->io_size, \
+ .bi_offset = (_ioend)->io_bvec_offset, \
+}
+
struct iomap_writeback_ops {
/*
* Performs writeback on the passed in range
@@ -565,6 +595,7 @@ void iomap_finish_ioends(struct iomap_ioend *ioend, int error);
void iomap_ioend_try_merge(struct iomap_ioend *ioend,
struct list_head *more_ioends);
void iomap_sort_ioends(struct list_head *ioend_list);
+int iomap_ioend_integrity_verify(struct iomap_ioend *ioend);
ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio,
loff_t pos, loff_t end_pos, unsigned int dirty_len);
int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error);
@@ -577,6 +608,11 @@ void iomap_finish_folio_write(struct inode *inode, struct folio *folio,
int iomap_writeback_folio(struct iomap_writepage_ctx *wpc, struct folio *folio);
int iomap_writepages(struct iomap_writepage_ctx *wpc);
+void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize,
+ void (*submit_ioend)(struct iomap_ioend *ioend));
+void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio,
+ int error);
+
struct iomap_read_folio_ctx {
const struct iomap_read_ops *ops;
struct folio *cur_folio;
diff --git a/include/linux/iommu-dma.h b/include/linux/iommu-dma.h
index 060f6e23ab3c..fae5d50e4f27 100644
--- a/include/linux/iommu-dma.h
+++ b/include/linux/iommu-dma.h
@@ -39,7 +39,7 @@ int iommu_dma_get_sgtable(struct device *dev, struct sg_table *sgt,
void *cpu_addr, dma_addr_t dma_addr, size_t size,
unsigned long attrs);
unsigned long iommu_dma_get_merge_boundary(struct device *dev);
-size_t iommu_dma_opt_mapping_size(void);
+size_t iommu_dma_max_opt_mapping_size(void);
size_t iommu_dma_max_mapping_size(struct device *dev);
void iommu_dma_free(struct device *dev, size_t size, void *cpu_addr,
dma_addr_t handle, unsigned long attrs);
diff --git a/include/linux/irq-entry-common.h b/include/linux/irq-entry-common.h
index 0bb6c03481fa..2be273bb68f0 100644
--- a/include/linux/irq-entry-common.h
+++ b/include/linux/irq-entry-common.h
@@ -346,22 +346,7 @@ typedef struct irqentry_state {
*
* Conditional reschedule with additional sanity checks.
*/
-void raw_irqentry_exit_cond_resched(void);
-
-#ifdef CONFIG_PREEMPT_DYNAMIC
-#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched
-#define irqentry_exit_cond_resched_dynamic_disabled NULL
-DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
-#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)()
-#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
-void dynamic_irqentry_exit_cond_resched(void);
-#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched()
-#endif
-#else /* CONFIG_PREEMPT_DYNAMIC */
-#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched()
-#endif /* CONFIG_PREEMPT_DYNAMIC */
+void irqentry_exit_cond_resched(void);
/**
* irqentry_enter_from_kernel_mode - Establish state before invoking the irq handler
diff --git a/include/linux/irqchip/arm-gic-v5.h b/include/linux/irqchip/arm-gic-v5.h
index 2c2fb39f049c..299e75bc9525 100644
--- a/include/linux/irqchip/arm-gic-v5.h
+++ b/include/linux/irqchip/arm-gic-v5.h
@@ -63,20 +63,34 @@
#define GICV5_OUTER_SHARE 0b10
#define GICV5_INNER_SHARE 0b11
+#define GICV5_AIDR_COMPONENT_IRS 0b00
+#define GICV5_AIDR_COMPONENT_ITS 0b01
+#define GICV5_AIDR_COMPONENT_IWB 0b10
+
+#define GICV5_AIDR_ARCH_MAJ_REV_V5 0
+#define GICV5_AIDR_ARCH_MIN_REV_V0 0
+
/*
* IRS registers and tables structures
*/
#define GICV5_IRS_IDR0 0x0000
#define GICV5_IRS_IDR1 0x0004
#define GICV5_IRS_IDR2 0x0008
+#define GICV5_IRS_IDR3 0x000c
+#define GICV5_IRS_IDR4 0x0010
#define GICV5_IRS_IDR5 0x0014
#define GICV5_IRS_IDR6 0x0018
#define GICV5_IRS_IDR7 0x001c
+#define GICV5_IRS_IIDR 0x0040
+#define GICV5_IRS_AIDR 0x0044
#define GICV5_IRS_CR0 0x0080
#define GICV5_IRS_CR1 0x0084
#define GICV5_IRS_SYNCR 0x00c0
#define GICV5_IRS_SYNC_STATUSR 0x00c4
+#define GICV5_IRS_SPI_VMR 0x0100
#define GICV5_IRS_SPI_SELR 0x0108
+#define GICV5_IRS_SPI_DOMAINR 0x010c
+#define GICV5_IRS_SPI_RESAMPLER 0x0110
#define GICV5_IRS_SPI_CFGR 0x0114
#define GICV5_IRS_SPI_STATUSR 0x0118
#define GICV5_IRS_PE_SELR 0x0140
@@ -86,11 +100,51 @@
#define GICV5_IRS_IST_CFGR 0x0190
#define GICV5_IRS_IST_STATUSR 0x0194
#define GICV5_IRS_MAP_L2_ISTR 0x01c0
-
+#define GICV5_IRS_VMT_BASER 0x0200
+#define GICV5_IRS_VMT_CFGR 0x0210
+#define GICV5_IRS_VMT_STATUSR 0x0214
+#define GICV5_IRS_VPE_SELR 0x0240
+#define GICV5_IRS_VPE_DBR 0x0248
+#define GICV5_IRS_VPE_HPPIR 0x0250
+#define GICV5_IRS_VPE_CR0 0x0258
+#define GICV5_IRS_VPE_STATUSR 0x025c
+#define GICV5_IRS_VM_DBR 0x0280
+#define GICV5_IRS_VM_SELR 0x0288
+#define GICV5_IRS_VM_STATUSR 0x028c
+#define GICV5_IRS_VMAP_L2_VMTR 0x02c0
+#define GICV5_IRS_VMAP_VMR 0x02c8
+#define GICV5_IRS_VMAP_VISTR 0x02d0
+#define GICV5_IRS_VMAP_L2_VISTR 0x02d8
+#define GICV5_IRS_VMAP_VPER 0x02e0
+#define GICV5_IRS_SAVE_VMR 0x0300
+#define GICV5_IRS_SAVE_VM_STATUSR 0x0308
+#define GICV5_IRS_MEC_IDR 0x0340
+#define GICV5_IRS_MEC_MECID_R 0x0344
+#define GICV5_IRS_MPAM_IDR 0x0380
+#define GICV5_IRS_MPAM_PARTID_R 0x0384
+#define GICV5_IRS_SWERR_STATUSR 0x03c0
+#define GICV5_IRS_SWERR_SYNDROMER0 0x03c8
+#define GICV5_IRS_SWERR_SYNDROMER1 0x03d0
+
+#define GICV5_IRS_IDR0_IRSID GENMASK(31, 16)
+#define GICV5_IRS_IDR0_SWE BIT(12)
+#define GICV5_IRS_IDR0_MPAM BIT(11)
+#define GICV5_IRS_IDR0_MEC BIT(10)
+#define GICV5_IRS_IDR0_SETLPI BIT(9)
+#define GICV5_IRS_IDR0_VIRT_ONE_N BIT(8)
+#define GICV5_IRS_IDR0_ONE_N BIT(7)
#define GICV5_IRS_IDR0_VIRT BIT(6)
+#define GICV5_IRS_IDR0_PA_RANGE GENMASK(5, 2)
+#define GICV5_IRS_IDR0_INT_DOM GENMASK(1, 0)
+
+#define GICV5_IRS_IDR0_INT_DOM_SECURE 0b00
+#define GICV5_IRS_IDR0_INT_DOM_NON_SECURE 0b01
+#define GICV5_IRS_IDR0_INT_DOM_EL3 0b10
+#define GICV5_IRS_IDR0_INT_DOM_REALM 0b11
#define GICV5_IRS_IDR1_PRIORITY_BITS GENMASK(22, 20)
#define GICV5_IRS_IDR1_IAFFID_BITS GENMASK(19, 16)
+#define GICV5_IRS_IDR1_PE_CNT GENMASK(15, 0)
#define GICV5_IRS_IDR1_PRIORITY_BITS_1BITS 0b000
#define GICV5_IRS_IDR1_PRIORITY_BITS_2BITS 0b001
@@ -106,13 +160,30 @@
#define GICV5_IRS_IDR2_LPI BIT(5)
#define GICV5_IRS_IDR2_ID_BITS GENMASK(4, 0)
+#define GICV5_IRS_IST_L2SZ_SUPPORT_4KB(r) FIELD_GET(BIT(0), (r))
+#define GICV5_IRS_IST_L2SZ_SUPPORT_16KB(r) FIELD_GET(BIT(1), (r))
+#define GICV5_IRS_IST_L2SZ_SUPPORT_64KB(r) FIELD_GET(BIT(2), (r))
+
+#define GICV5_IRS_IDR3_VMT_LEVELS BIT(10)
+#define GICV5_IRS_IDR3_VM_ID_BITS GENMASK(9, 5)
+#define GICV5_IRS_IDR3_VMD_SZ GENMASK(4, 1)
+#define GICV5_IRS_IDR3_VMD BIT(0)
+
+#define GICV5_IRS_IDR4_VPE_ID_BITS GENMASK(9, 6)
+#define GICV5_IRS_IDR4_VPED_SZ GENMASK(5, 0)
+
#define GICV5_IRS_IDR5_SPI_RANGE GENMASK(24, 0)
#define GICV5_IRS_IDR6_SPI_IRS_RANGE GENMASK(24, 0)
#define GICV5_IRS_IDR7_SPI_BASE GENMASK(23, 0)
-#define GICV5_IRS_IST_L2SZ_SUPPORT_4KB(r) FIELD_GET(BIT(11), (r))
-#define GICV5_IRS_IST_L2SZ_SUPPORT_16KB(r) FIELD_GET(BIT(12), (r))
-#define GICV5_IRS_IST_L2SZ_SUPPORT_64KB(r) FIELD_GET(BIT(13), (r))
+#define GICV5_IRS_IIDR_PRODUCT_ID GENMASK(31, 20)
+#define GICV5_IRS_IIDR_VARIANT GENMASK(19, 16)
+#define GICV5_IRS_IIDR_REVISION GENMASK(15, 12)
+#define GICV5_IRS_IIDR_IMPLEMENTER GENMASK(11, 0)
+
+#define GICV5_IRS_AIDR_COMPONENT GENMASK(11, 8)
+#define GICV5_IRS_AIDR_ARCHMAJORREV GENMASK(7, 4)
+#define GICV5_IRS_AIDR_ARCHMINORREV GENMASK(3, 0)
#define GICV5_IRS_CR0_IDLE BIT(1)
#define GICV5_IRS_CR0_IRSEN BIT(0)
@@ -135,21 +206,39 @@
#define GICV5_IRS_SYNC_STATUSR_IDLE BIT(0)
-#define GICV5_IRS_SPI_STATUSR_V BIT(1)
-#define GICV5_IRS_SPI_STATUSR_IDLE BIT(0)
+#define GICV5_IRS_SPI_VMR_VIRT BIT_ULL(63)
+#define GICV5_IRS_SPI_VMR_VM_ID GENMASK_ULL(15, 0)
#define GICV5_IRS_SPI_SELR_ID GENMASK(23, 0)
+#define GICV5_IRS_SPI_DOMAINR_DOMAIN GENMASK(1, 0)
+
+#define GICV5_IRS_SPI_DOMAINR_DOMAIN_SECURE 0b00
+#define GICV5_IRS_SPI_DOMAINR_DOMAIN_NON_SECURE 0b01
+#define GICV5_IRS_SPI_DOMAINR_DOMAIN_EL3 0b10
+#define GICV5_IRS_SPI_DOMAINR_DOMAIN_REALM 0b11
+
+#define GICV5_IRS_SPI_RESAMPLER_ID GENMASK(23, 0)
+
#define GICV5_IRS_SPI_CFGR_TM BIT(0)
+#define GICV5_IRS_SPI_CFGR_TM_EDGE 0b0
+#define GICV5_IRS_SPI_CFGR_TM_LEVEL 0b1
+
+#define GICV5_IRS_SPI_STATUSR_V BIT(1)
+#define GICV5_IRS_SPI_STATUSR_IDLE BIT(0)
+
#define GICV5_IRS_PE_SELR_IAFFID GENMASK(15, 0)
+#define GICV5_IRS_PE_STATUSR_ONLINE BIT(2)
#define GICV5_IRS_PE_STATUSR_V BIT(1)
#define GICV5_IRS_PE_STATUSR_IDLE BIT(0)
#define GICV5_IRS_PE_CR0_DPS BIT(0)
-#define GICV5_IRS_IST_STATUSR_IDLE BIT(0)
+#define GICV5_IRS_IST_BASER_ADDR_MASK GENMASK_ULL(55, 6)
+#define GICV5_IRS_IST_BASER_VALID BIT_ULL(0)
+#define GICV5_IRS_IST_BASER_ADDR_SHIFT 6ULL
#define GICV5_IRS_IST_CFGR_STRUCTURE BIT(16)
#define GICV5_IRS_IST_CFGR_ISTSZ GENMASK(8, 7)
@@ -167,15 +256,111 @@
#define GICV5_IRS_IST_CFGR_L2SZ_16K 0b01
#define GICV5_IRS_IST_CFGR_L2SZ_64K 0b10
-#define GICV5_IRS_IST_BASER_ADDR_MASK GENMASK_ULL(55, 6)
-#define GICV5_IRS_IST_BASER_VALID BIT_ULL(0)
+#define GICV5_IRS_IST_STATUSR_IDLE BIT(0)
#define GICV5_IRS_MAP_L2_ISTR_ID GENMASK(23, 0)
+#define GICV5_IRS_VMT_BASER_ADDR GENMASK_ULL(55, 3)
+#define GICV5_IRS_VMT_BASER_ADDR_SHIFT 3ULL
+#define GICV5_IRS_VMT_BASER_VALID BIT_ULL(0)
+
+#define GICV5_IRS_VMT_CFGR_STRUCTURE_TWO_LEVEL 0b1
+#define GICV5_IRS_VMT_CFGR_STRUCTURE_LINEAR 0b0
+
+#define GICV5_IRS_VMT_CFGR_STRUCTURE BIT(16)
+#define GICV5_IRS_VMT_CFGR_VM_ID_BITS GENMASK(4, 0)
+
+#define GICV5_IRS_VMT_STATUSR_IDLE BIT(0)
+
+#define GICV5_IRS_VPE_SELR_S BIT_ULL(63)
+#define GICV5_IRS_VPE_SELR_VPE_ID GENMASK_ULL(47, 32)
+#define GICV5_IRS_VPE_SELR_VM_ID GENMASK_ULL(15, 0)
+
+#define GICV5_IRS_VPE_DBR_DBV BIT_ULL(63)
+#define GICV5_IRS_VPE_DBR_REQ_DB BIT_ULL(62)
+#define GICV5_IRS_VPE_DBR_DBPM GENMASK_ULL(36, 32)
+#define GICV5_IRS_VPE_DBR_INTID GENMASK_ULL(23, 0)
+
+#define GICV5_IRS_VPE_HPPIR_HPPIV BIT_ULL(32)
+#define GICV5_IRS_VPE_HPPIR_TYPE GENMASK_ULL(31, 29)
+#define GICV5_IRS_VPE_HPPIR_ID GENMASK_ULL(23, 0)
+
+#define GICV5_IRS_VPE_CR0_DPS BIT(0)
+
+#define GICV5_IRS_VPE_STATUSR_V BIT(1)
+#define GICV5_IRS_VPE_STATUSR_IDLE BIT(0)
+
+#define GICV5_IRS_VM_DBR_EN BIT_ULL(63)
+#define GICV5_IRS_VM_DBR_VPE_ID GENMASK_ULL(15, 0)
+
+#define GICV5_IRS_VM_SELR_VM_ID GENMASK(15, 0)
+
+#define GICV5_IRS_VM_STATUSR_V BIT(1)
+#define GICV5_IRS_VM_STATUSR_IDLE BIT(0)
+
+#define GICV5_IRS_VMAP_L2_VMTR_M BIT_ULL(63)
+#define GICV5_IRS_VMAP_L2_VMTR_VM_ID GENMASK_ULL(15, 0)
+
+#define GICV5_IRS_VMAP_VMR_M BIT_ULL(63)
+#define GICV5_IRS_VMAP_VMR_U BIT_ULL(62)
+#define GICV5_IRS_VMAP_VMR_VM_ID GENMASK_ULL(15, 0)
+
+#define GICV5_IRS_VMAP_VISTR_M BIT_ULL(63)
+#define GICV5_IRS_VMAP_VISTR_U BIT_ULL(62)
+#define GICV5_IRS_VMAP_VISTR_VM_ID GENMASK_ULL(47, 32)
+#define GICV5_IRS_VMAP_VISTR_TYPE GENMASK_ULL(31, 29)
+
+#define GICV5_IRS_VMAP_L2_VISTR_M BIT_ULL(63)
+#define GICV5_IRS_VMAP_L2_VISTR_VM_ID GENMASK_ULL(47, 32)
+#define GICV5_IRS_VMAP_L2_VISTR_TYPE GENMASK_ULL(31, 29)
+#define GICV5_IRS_VMAP_L2_VISTR_ID GENMASK_ULL(23, 0)
+
+#define GICV5_IRS_VMAP_VPER_M BIT_ULL(63)
+#define GICV5_IRS_VMAP_VPER_VM_ID GENMASK_ULL(47, 32)
+#define GICV5_IRS_VMAP_VPER_VPE_ID GENMASK_ULL(15, 0)
+
+#define GICV5_IRS_SAVE_VMR_VM_ID GENMASK_ULL(15, 0)
+#define GICV5_IRS_SAVE_VMR_Q BIT_ULL(62)
+#define GICV5_IRS_SAVE_VMR_S BIT_ULL(63)
+
+#define GICV5_IRS_SAVE_VM_STATUSR_IDLE BIT(0)
+#define GICV5_IRS_SAVE_VM_STATUSR_Q BIT(1)
+
+#define GICV5_IRS_MEC_IDR_MECIDSIZE GENMASK(3, 0)
+
+#define GICV5_IRS_MEC_MECID_R_MECID GENMASK(15, 0)
+
+#define GICV5_IRS_MPAM_IDR_HAS_MPAM_SP BIT(24)
+#define GICV5_IRS_MPAM_IDR_PMG_MAX GENMASK(23, 16)
+#define GICV5_IRS_MPAM_IDR_PARTID_MAX GENMASK(15, 0)
+
+#define GICV5_IRS_MPAM_PARTID_R_IDLE BIT(31)
+#define GICV5_IRS_MPAM_PARTID_R_MPAM_SP GENMASK(25, 24)
+#define GICV5_IRS_MPAM_PARTID_R_PMG GENMASK(23, 16)
+#define GICV5_IRS_MPAM_PARTID_R_PARTID GENMASK(15, 0)
+
+#define GICV5_IRS_SWERR_STATUSR_IMP_EC GENMASK_ULL(31, 24)
+#define GICV5_IRS_SWERR_STATUSR_EC GENMASK_ULL(23, 16)
+#define GICV5_IRS_SWERR_STATUSR_OF BIT_ULL(3)
+#define GICV5_IRS_SWERR_STATUSR_S1V BIT_ULL(2)
+#define GICV5_IRS_SWERR_STATUSR_S0V BIT_ULL(1)
+#define GICV5_IRS_SWERR_STATUSR_V BIT_ULL(0)
+
+#define GICV5_IRS_SWERR_SYNDROMER0_VIRTUAL BIT_ULL(63)
+#define GICV5_IRS_SWERR_SYNDROMER0_TYPE GENMASK_ULL(62, 60)
+#define GICV5_IRS_SWERR_SYNDROMER0_ID GENMASK_ULL(55, 32)
+#define GICV5_IRS_SWERR_SYNDROMER0_VM_ID GENMASK_ULL(15, 0)
+
+#define GICV5_IRS_SWERR_SYNDROMER1_ADDR GENMASK_ULL(55, 3)
+
#define GICV5_ISTL1E_VALID BIT_ULL(0)
+#define GICV5_IRS_ISTL1E_SIZE 8UL
#define GICV5_ISTL1E_L2_ADDR_MASK GENMASK_ULL(55, 12)
+#define GICV5_IRS_SETLPIR 0x0000
+#define GICV5_IRS_SETLPIR_ID GENMASK(23, 0)
+
/*
* ITS registers and tables structures
*/
@@ -298,6 +483,42 @@
#define GICV5_GSI_IWB_WIRE GENMASK(15, 0)
/*
+ * CoreSight identification registers - ordered by increasing offset.
+ */
+#define GICV5_CORESIGHT_DEVARCH 0xffbc
+#define GICV5_CORESIGHT_PIDR4 0xffd0
+#define GICV5_CORESIGHT_PIDR5 0xffd4
+#define GICV5_CORESIGHT_PIDR6 0xffd8
+#define GICV5_CORESIGHT_PIDR7 0xffdc
+#define GICV5_CORESIGHT_PIDR0 0xffe0
+#define GICV5_CORESIGHT_PIDR1 0xffe4
+#define GICV5_CORESIGHT_PIDR2 0xffe8
+#define GICV5_CORESIGHT_PIDR3 0xffec
+#define GICV5_CORESIGHT_CIDR0 0xfff0
+#define GICV5_CORESIGHT_CIDR1 0xfff4
+#define GICV5_CORESIGHT_CIDR2 0xfff8
+#define GICV5_CORESIGHT_CIDR3 0xfffc
+
+#define GICV5_CORESIGHT_DEVARCH_VAL \
+ (FIELD_PREP(GENMASK(31, 21), 0x23b) | \
+ BIT(20) | \
+ 0x5a19)
+
+#define GICV5_CORESIGHT_PIDR4_JEP106_CONT 0x04
+#define GICV5_CORESIGHT_PIDR5_RES0 0x00
+#define GICV5_CORESIGHT_PIDR6_RES0 0x00
+#define GICV5_CORESIGHT_PIDR7_RES0 0x00
+#define GICV5_CORESIGHT_PIDR0_PART_0 0x4b
+#define GICV5_CORESIGHT_PIDR1_DES_0_PART_1 0xb0
+#define GICV5_CORESIGHT_PIDR2_DES_1 0x0b
+#define GICV5_CORESIGHT_PIDR3_REVAND_CMOD 0x00
+
+#define GICV5_CORESIGHT_CIDR0_VAL 0x0d
+#define GICV5_CORESIGHT_CIDR1_VAL 0xf0
+#define GICV5_CORESIGHT_CIDR2_VAL 0x05
+#define GICV5_CORESIGHT_CIDR3_VAL 0xb1
+
+/*
* Global Data structures and functions
*/
struct gicv5_chip_data {
@@ -332,6 +553,8 @@ struct gicv5_irs_chip_data {
raw_spinlock_t spi_config_lock;
};
+#define IRS_FLAGS_NON_COHERENT BIT(0)
+
static inline int gicv5_wait_for_op_s_atomic(void __iomem *addr, u32 offset,
const char *reg_s, u32 mask,
u32 *val)
@@ -379,6 +602,7 @@ void __init gicv5_free_lpi_domain(void);
int gicv5_irs_of_probe(struct device_node *parent);
int gicv5_irs_acpi_probe(void);
+struct gicv5_irs_chip_data *gicv5_irs_get_chip_data(void);
void gicv5_irs_remove(void);
int gicv5_irs_enable(void);
void gicv5_irs_its_probe(void);
@@ -387,10 +611,13 @@ int gicv5_irs_cpu_to_iaffid(int cpu_id, u16 *iaffid);
struct gicv5_irs_chip_data *gicv5_irs_lookup_by_spi_id(u32 spi_id);
int gicv5_spi_irq_set_type(struct irq_data *d, unsigned int type);
int gicv5_irs_iste_alloc(u32 lpi);
+unsigned int gicv5_irs_l2_sz(u32 l2sz);
void gicv5_irs_syncr(void);
/* Embedded in kvm.arch */
struct gicv5_vpe {
+ int db;
+ bool db_fired;
bool resident;
};
@@ -429,4 +656,15 @@ void gicv5_deinit_lpis(void);
void __init gicv5_its_of_probe(struct device_node *parent);
void __init gicv5_its_acpi_probe(void);
+
+enum gicv5_vcpu_cmd {
+ VMT_L2_MAP, /* Map in a L2 VMT - *may* happen on VM init */
+ VMTE_MAKE_VALID, /* Make the VMTE valid */
+ VMTE_MAKE_INVALID, /* Make the VMTE (et al.) invalid */
+ VPE_MAKE_VALID, /* No corresponding invalid */
+ SPI_VIST_MAKE_VALID, /* No corresponding invalid */
+ LPI_VIST_MAKE_VALID, /* Triggered by a guest */
+ LPI_VIST_MAKE_INVALID, /* Triggered by a guest */
+};
+
#endif
diff --git a/include/linux/irqchip/arm-vgic-info.h b/include/linux/irqchip/arm-vgic-info.h
index 67d9d960273b..f05370e2debf 100644
--- a/include/linux/irqchip/arm-vgic-info.h
+++ b/include/linux/irqchip/arm-vgic-info.h
@@ -38,6 +38,11 @@ struct gic_kvm_info {
bool has_v4_1;
/* Deactivation impared, subpar stuff */
bool no_hw_deactivation;
+ /* GICv5 IRS base */
+ struct {
+ void __iomem *base;
+ bool non_coherent;
+ } gicv5_irs;
};
#ifdef CONFIG_KVM
diff --git a/include/linux/irqdomain.h b/include/linux/irqdomain.h
index 73c25d40846c..3ba75a4ed3da 100644
--- a/include/linux/irqdomain.h
+++ b/include/linux/irqdomain.h
@@ -752,24 +752,6 @@ static inline void msi_device_domain_free_wired(struct irq_domain *domain, unsig
}
#endif
-static inline struct irq_domain *irq_domain_add_linear(struct device_node *of_node,
- unsigned int size,
- const struct irq_domain_ops *ops,
- void *host_data)
-{
- struct irq_domain_info info = {
- .fwnode = of_fwnode_handle(of_node),
- .size = size,
- .hwirq_max = size,
- .ops = ops,
- .host_data = host_data,
- };
- struct irq_domain *d;
-
- d = irq_domain_instantiate(&info);
- return IS_ERR(d) ? NULL : d;
-}
-
#else /* CONFIG_IRQ_DOMAIN */
static inline void irq_dispose_mapping(unsigned int virq) { }
static inline struct irq_domain *irq_find_matching_fwnode(struct fwnode_handle *fwnode,
diff --git a/include/linux/kdev_t.h b/include/linux/kdev_t.h
index 4856706fbfeb..2dbbd47f1e68 100644
--- a/include/linux/kdev_t.h
+++ b/include/linux/kdev_t.h
@@ -9,7 +9,7 @@
#define MAJOR(dev) ((unsigned int) ((dev) >> MINORBITS))
#define MINOR(dev) ((unsigned int) ((dev) & MINORMASK))
-#define MKDEV(ma,mi) (((ma) << MINORBITS) | (mi))
+#define MKDEV(ma, mi) (((dev_t)(ma) << MINORBITS) | (mi))
#define print_dev_t(buffer, dev) \
sprintf((buffer), "%u:%u\n", MAJOR(dev), MINOR(dev))
diff --git a/include/linux/kernel-page-flags.h b/include/linux/kernel-page-flags.h
index 196778a087c4..fe5ab6e50bd7 100644
--- a/include/linux/kernel-page-flags.h
+++ b/include/linux/kernel-page-flags.h
@@ -11,7 +11,6 @@
#define KPF_RESERVED 32
#define KPF_MLOCKED 33
#define KPF_OWNER_2 34
-#define KPF_PRIVATE 35
#define KPF_PRIVATE_2 36
#define KPF_OWNER_PRIVATE 37
#define KPF_ARCH 38
diff --git a/include/linux/kernel.h b/include/linux/kernel.h
index 24414c79e59a..4b11d1dc0a67 100644
--- a/include/linux/kernel.h
+++ b/include/linux/kernel.h
@@ -43,30 +43,10 @@ struct completion;
struct user;
#ifdef CONFIG_PREEMPT_VOLUNTARY_BUILD
-
extern int __cond_resched(void);
# define might_resched() __cond_resched()
-
-#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-
-extern int __cond_resched(void);
-
-DECLARE_STATIC_CALL(might_resched, __cond_resched);
-
-static __always_inline void might_resched(void)
-{
- static_call_mod(might_resched)();
-}
-
-#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-
-extern int dynamic_might_resched(void);
-# define might_resched() dynamic_might_resched()
-
#else
-
# define might_resched() do { } while (0)
-
#endif /* CONFIG_PREEMPT_* */
#ifdef CONFIG_DEBUG_ATOMIC_SLEEP
diff --git a/include/linux/kernel_stat.h b/include/linux/kernel_stat.h
index 9ca6c2259dfe..c1e85550bf12 100644
--- a/include/linux/kernel_stat.h
+++ b/include/linux/kernel_stat.h
@@ -196,6 +196,17 @@ static inline void kcpustat_cpu_fetch(struct kernel_cpustat *dst, int cpu)
}
#endif /* !CONFIG_VIRT_CPU_ACCOUNTING_GEN */
+static inline u64 kcpustat_field_total(enum cpu_usage_stat usage, const struct cpumask *cpus)
+{
+ u64 total = 0;
+ int cpu;
+
+ for_each_cpu(cpu, cpus)
+ total += kcpustat_field(usage, cpu);
+
+ return total;
+}
+
extern void account_user_time(struct task_struct *, u64);
extern void account_guest_time(struct task_struct *, u64);
extern void account_system_time(struct task_struct *, int, u64);
diff --git a/include/linux/kexec.h b/include/linux/kexec.h
index 0af8ae4fdd08..f970ca2c8ce9 100644
--- a/include/linux/kexec.h
+++ b/include/linux/kexec.h
@@ -259,21 +259,6 @@ int kexec_kernel_verify_pe_sig(const char *kernel, unsigned long kernel_len);
extern int kexec_add_buffer(struct kexec_buf *kbuf);
int kexec_locate_mem_hole(struct kexec_buf *kbuf);
-#ifndef arch_kexec_locate_mem_hole
-/**
- * arch_kexec_locate_mem_hole - Find free memory to place the segments.
- * @kbuf: Parameters for the memory search.
- *
- * On success, kbuf->mem will have the start address of the memory region found.
- *
- * Return: 0 on success, negative errno on error.
- */
-static inline int arch_kexec_locate_mem_hole(struct kexec_buf *kbuf)
-{
- return kexec_locate_mem_hole(kbuf);
-}
-#endif
-
#ifndef arch_kexec_apply_relocations_add
/*
* arch_kexec_apply_relocations_add - apply relocations of type RELA
@@ -418,7 +403,7 @@ struct kimage {
#endif
struct {
- struct kexec_segment *scratch;
+ struct kexec_segment *bootmem;
phys_addr_t fdt;
} kho;
diff --git a/include/linux/kexec_handover.h b/include/linux/kexec_handover.h
index 46de86dc343e..ef2b188c8ed0 100644
--- a/include/linux/kexec_handover.h
+++ b/include/linux/kexec_handover.h
@@ -36,15 +36,15 @@ int kho_retrieve_subtree(const char *name, phys_addr_t *phys, size_t *size);
void kho_memory_init(void);
void kho_memory_init_early(void);
-void kho_populate(phys_addr_t fdt_phys, u64 fdt_len, phys_addr_t scratch_phys,
- u64 scratch_len);
+void kho_populate(phys_addr_t fdt_phys, u64 fdt_len, phys_addr_t bootmem_phys,
+ u64 bootmem_len);
-bool kho_scratch_overlap(phys_addr_t phys, size_t size);
+bool kho_bootmem_overlap(phys_addr_t phys, size_t size);
-static inline enum migratetype kho_scratch_migratetype(unsigned long pfn,
+static inline enum migratetype kho_bootmem_migratetype(unsigned long pfn,
enum migratetype mt)
{
- if (kho_scratch_overlap(PFN_PHYS(pfn), pageblock_nr_pages << PAGE_SHIFT))
+ if (kho_bootmem_overlap(PFN_PHYS(pfn), pageblock_nr_pages << PAGE_SHIFT))
return MIGRATE_CMA;
return mt;
}
@@ -123,16 +123,16 @@ static inline void kho_memory_init(void) { }
static inline void kho_memory_init_early(void) { }
static inline void kho_populate(phys_addr_t fdt_phys, u64 fdt_len,
- phys_addr_t scratch_phys, u64 scratch_len)
+ phys_addr_t bootmem_phys, u64 bootmem_len)
{
}
-static inline bool kho_scratch_overlap(phys_addr_t phys, size_t size)
+static inline bool kho_bootmem_overlap(phys_addr_t phys, size_t size)
{
return false;
}
-static inline enum migratetype kho_scratch_migratetype(unsigned long pfn,
+static inline enum migratetype kho_bootmem_migratetype(unsigned long pfn,
enum migratetype mt)
{
return mt;
diff --git a/include/linux/key.h b/include/linux/key.h
index 81b8f05c6898..bd10fe45819d 100644
--- a/include/linux/key.h
+++ b/include/linux/key.h
@@ -440,6 +440,8 @@ extern key_ref_t keyring_search(key_ref_t keyring,
extern int keyring_restrict(key_ref_t keyring, const char *type,
const char *restriction);
+extern void key_register_bpf_keyring(struct key *keyring);
+
extern struct key *key_lookup(key_serial_t id);
static inline key_serial_t key_serial(const struct key *key)
diff --git a/include/linux/kho/abi/pci.h b/include/linux/kho/abi/pci.h
new file mode 100644
index 000000000000..9485ed73c351
--- /dev/null
+++ b/include/linux/kho/abi/pci.h
@@ -0,0 +1,66 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+
+/*
+ * Copyright (c) 2026, Google LLC.
+ * David Matlack <dmatlack@google.com>
+ */
+
+#ifndef _LINUX_KHO_ABI_PCI_H
+#define _LINUX_KHO_ABI_PCI_H
+
+#include <linux/bug.h>
+#include <linux/compiler.h>
+#include <linux/types.h>
+
+/**
+ * DOC: PCI File-Lifecycle Bound (FLB) Live Update ABI
+ *
+ * This header defines the ABI for preserving core PCI state across kexec using
+ * Live Update File-Lifecycle Bound (FLB) data.
+ *
+ * This interface is a contract. Any modification to any of the serialization
+ * structs defined here constitutes a breaking change. Such changes require
+ * incrementing the version number in the PCI_LUO_FLB_VERSION number.
+ */
+
+#define PCI_LUO_FLB_COMPATIBLE "pci"
+#define PCI_LUO_FLB_VERSION 2
+
+/**
+ * struct pci_dev_ser - Serialized state about a single PCI device.
+ *
+ * @domain: The device's PCI domain number (segment).
+ * @bdf: The device's PCI bus, device, and function number.
+ * @refcount: Reference count used by the PCI core to keep track of whether it
+ * is done using a device's struct pci_dev_ser. The value of the
+ * refcount is equal to the number of preserved devices at or below
+ * it in the PCI hierarchy when the struct pci_dev_ser is in use, and
+ * 0 otherwise.
+ */
+struct pci_dev_ser {
+ u32 domain;
+ u16 bdf;
+ u16 refcount;
+} __packed;
+
+/**
+ * struct pci_ser - PCI Subsystem Live Update State
+ *
+ * This struct tracks state about all devices that are being preserved across
+ * a Live Update for the next kernel. It contains only state owned by the PCI
+ * core; the state a driver needs to resume its device is preserved separately
+ * by that driver.
+ *
+ * @version: The version of the "pci" FLB struct. This field must never be
+ * deleted, moved, or resized, as the kernel depends on always being
+ * able to check the struct pci_ser version number.
+ * @nr_devices: The number of devices that were preserved.
+ * @devices: Physical address of the first KHO block containing pci_dev_ser.
+ */
+struct pci_ser {
+ u32 version;
+ u32 nr_devices;
+ u64 devices;
+} __packed;
+
+#endif /* _LINUX_KHO_ABI_PCI_H */
diff --git a/include/linux/kmod.h b/include/linux/kmod.h
index 9a07c3215389..b9474a62a568 100644
--- a/include/linux/kmod.h
+++ b/include/linux/kmod.h
@@ -2,17 +2,9 @@
#ifndef __LINUX_KMOD_H__
#define __LINUX_KMOD_H__
-/*
- * include/linux/kmod.h
- */
-
-#include <linux/umh.h>
-#include <linux/gfp.h>
-#include <linux/stddef.h>
+#include <linux/compiler_attributes.h>
#include <linux/errno.h>
-#include <linux/compiler.h>
-#include <linux/workqueue.h>
-#include <linux/sysctl.h>
+#include <linux/types.h>
#ifdef CONFIG_MODULES
/* modprobe exit status on success, -ve on error. Return value
diff --git a/include/linux/kprobes.h b/include/linux/kprobes.h
index 8c4f3bb24429..e6de7ae55bda 100644
--- a/include/linux/kprobes.h
+++ b/include/linux/kprobes.h
@@ -181,6 +181,7 @@ struct kprobe_blacklist_entry {
struct list_head list;
unsigned long start_addr;
unsigned long end_addr;
+ struct rcu_head rcu;
};
#ifdef CONFIG_KPROBES
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index 03bfc92864b6..c0748f7e40ea 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -722,27 +722,6 @@ static inline int kvm_arch_vcpu_memslots_id(struct kvm_vcpu *vcpu)
}
#endif
-#ifndef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES
-static inline bool kvm_arch_has_private_mem(struct kvm *kvm)
-{
- return false;
-}
-#endif
-
-#ifdef CONFIG_KVM_GUEST_MEMFD
-bool kvm_arch_supports_gmem_init_shared(struct kvm *kvm);
-
-static inline u64 kvm_gmem_get_supported_flags(struct kvm *kvm)
-{
- u64 flags = GUEST_MEMFD_FLAG_MMAP;
-
- if (!kvm || kvm_arch_supports_gmem_init_shared(kvm))
- flags |= GUEST_MEMFD_FLAG_INIT_SHARED;
-
- return flags;
-}
-#endif
-
#ifndef kvm_arch_has_readonly_mem
static inline bool kvm_arch_has_readonly_mem(struct kvm *kvm)
{
@@ -791,7 +770,6 @@ struct kvm {
/* The current active memslot set for each address space */
struct kvm_memslots __rcu *memslots[KVM_MAX_NR_ADDRESS_SPACES];
struct xarray vcpu_array;
- DECLARE_BITMAP(vcpu_ids, KVM_MAX_VCPU_IDS);
/*
* Protected by slots_lock, but can be read outside if an
* incorrect answer is acceptable.
@@ -855,6 +833,8 @@ struct kvm {
gfn_t mmu_invalidate_range_start;
gfn_t mmu_invalidate_range_end;
+ unsigned long gpc_invalidate_seq;
+
struct list_head devices;
u64 manual_dirty_log_protect;
struct dentry *debugfs_dentry;
@@ -872,7 +852,7 @@ struct kvm {
#ifdef CONFIG_HAVE_KVM_PM_NOTIFIER
struct notifier_block pm_notifier;
#endif
-#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES
+#ifdef CONFIG_KVM_VM_MEMORY_ATTRIBUTES
/* Protected by slots_lock (for writes) and RCU (for reads) */
struct xarray mem_attr_array;
#endif
@@ -2324,13 +2304,18 @@ static inline bool kvm_test_request(int req, struct kvm_vcpu *vcpu)
return test_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests);
}
-static inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu)
+static __always_inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu)
{
+ BUILD_BUG_ON(req == KVM_REQ_VM_DEAD);
+
clear_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests);
}
-static inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu)
+static __always_inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu)
{
+ /* Once a VM is dead, it needs to stay dead. */
+ BUILD_BUG_ON(req == KVM_REQ_VM_DEAD);
+
if (kvm_test_request(req, vcpu)) {
kvm_clear_request(req, vcpu);
@@ -2560,48 +2545,80 @@ static inline bool kvm_memslot_is_gmem_only(const struct kvm_memory_slot *slot)
return slot->flags & KVM_MEMSLOT_GMEM_ONLY;
}
-#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES
-static inline unsigned long kvm_get_memory_attributes(struct kvm *kvm, gfn_t gfn)
+#ifdef CONFIG_KVM_VM_MEMORY_ATTRIBUTES
+static inline unsigned long kvm_get_vm_memory_attributes(struct kvm *kvm, gfn_t gfn)
{
return xa_to_value(xa_load(&kvm->mem_attr_array, gfn));
}
-bool kvm_range_has_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
- unsigned long mask, unsigned long attrs);
-bool kvm_arch_pre_set_memory_attributes(struct kvm *kvm,
- struct kvm_gfn_range *range);
-bool kvm_arch_post_set_memory_attributes(struct kvm *kvm,
- struct kvm_gfn_range *range);
+bool kvm_range_has_vm_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
+ unsigned long mask, unsigned long attrs);
+bool kvm_arch_pre_set_vm_memory_attributes(struct kvm *kvm,
+ struct kvm_gfn_range *range);
+bool kvm_arch_post_set_vm_memory_attributes(struct kvm *kvm,
+ struct kvm_gfn_range *range);
-static inline bool kvm_mem_is_private(struct kvm *kvm, gfn_t gfn)
+static inline bool kvm_vm_is_private_gfn(struct kvm *kvm, gfn_t gfn)
{
- return kvm_get_memory_attributes(kvm, gfn) & KVM_MEMORY_ATTRIBUTE_PRIVATE;
+ return kvm_get_vm_memory_attributes(kvm, gfn) & KVM_MEMORY_ATTRIBUTE_PRIVATE;
+}
+#endif /* CONFIG_KVM_VM_MEMORY_ATTRIBUTES */
+
+#ifdef kvm_arch_has_private_mem
+extern bool gmem_in_place_conversion;
+
+typedef bool (kvm_is_private_gfn_t)(struct kvm *kvm, gfn_t gfn);
+DECLARE_STATIC_CALL(__kvm_is_private_gfn, kvm_is_private_gfn_t);
+
+static inline bool kvm_is_private_gfn(struct kvm *kvm, gfn_t gfn)
+{
+ return static_call(__kvm_is_private_gfn)(kvm, gfn);
}
#else
-static inline bool kvm_mem_is_private(struct kvm *kvm, gfn_t gfn)
+#define gmem_in_place_conversion false
+
+static inline bool kvm_arch_has_private_mem(struct kvm *kvm)
+{
+ return false;
+}
+
+static inline bool kvm_is_private_gfn(struct kvm *kvm, gfn_t gfn)
{
return false;
}
-#endif /* CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES */
+#endif /* kvm_arch_has_private_mem */
#ifdef CONFIG_KVM_GUEST_MEMFD
+bool kvm_gmem_is_private_gfn(struct kvm *kvm, gfn_t gfn);
+bool kvm_arch_supports_gmem_init_shared(struct kvm *kvm);
+
+static inline u64 kvm_gmem_get_supported_flags(struct kvm *kvm)
+{
+ u64 flags = GUEST_MEMFD_FLAG_MMAP;
+
+ if (!kvm || kvm_arch_supports_gmem_init_shared(kvm))
+ flags |= GUEST_MEMFD_FLAG_INIT_SHARED;
+
+ return flags;
+}
+
int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot,
- gfn_t gfn, kvm_pfn_t *pfn, struct page **page,
- int *max_order);
+ gfn_t gfn, kvm_pfn_t *pfn, int *max_order);
#else
static inline int kvm_gmem_get_pfn(struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
- kvm_pfn_t *pfn, struct page **page,
- int *max_order)
+ kvm_pfn_t *pfn, int *max_order)
{
KVM_BUG_ON(1, kvm);
return -EIO;
}
#endif /* CONFIG_KVM_GUEST_MEMFD */
-#ifdef CONFIG_HAVE_KVM_ARCH_GMEM_CONVERT
int kvm_arch_gmem_make_private(struct kvm *kvm, gfn_t gfn, kvm_pfn_t pfn,
kvm_pfn_t nr_pages);
+void kvm_arch_gmem_make_shared(kvm_pfn_t pfn, kvm_pfn_t nr_pages);
+#ifndef CONFIG_HAVE_KVM_ARCH_GMEM_CONVERT
+#define kvm_arch_has_gmem_convert() false
#endif
#ifdef CONFIG_HAVE_KVM_ARCH_GMEM_POPULATE
@@ -2643,6 +2660,7 @@ void kvm_arch_gmem_invalidate_range(struct kvm *kvm, struct kvm_gfn_range *range
#endif
#ifdef CONFIG_KVM_GENERIC_PRE_FAULT_MEMORY
+int kvm_arch_pre_fault_allowed(struct kvm_vcpu *vcpu);
long kvm_arch_vcpu_pre_fault_memory(struct kvm_vcpu *vcpu,
struct kvm_pre_fault_memory *range);
#endif
diff --git a/include/linux/libata.h b/include/linux/libata.h
index 313e96173b19..3bd74d79f812 100644
--- a/include/linux/libata.h
+++ b/include/linux/libata.h
@@ -76,7 +76,7 @@ enum ata_quirks {
__ATA_QUIRK_MAX_SEC, /* Limit max sectors */
__ATA_QUIRK_MAX_TRIM_128M, /* Limit max trim size to 128M */
__ATA_QUIRK_NO_NCQ_ON_ATI, /* Disable NCQ on ATI chipset */
- __ATA_QUIRK_NO_LPM_ON_ATI, /* Disable LPM on ATI chipset */
+ __ATA_QUIRK_NO_LPM_ON_ATI_AND_AMD, /* Disable LPM on ATI and AMD chipsets */
__ATA_QUIRK_NO_ID_DEV_LOG, /* Identify device log missing */
__ATA_QUIRK_NO_LOG_DIR, /* Do not read log directory */
__ATA_QUIRK_NO_FUA, /* Do not use FUA */
@@ -115,7 +115,7 @@ enum {
ATA_QUIRK_MAX_SEC = BIT_ULL(__ATA_QUIRK_MAX_SEC),
ATA_QUIRK_MAX_TRIM_128M = BIT_ULL(__ATA_QUIRK_MAX_TRIM_128M),
ATA_QUIRK_NO_NCQ_ON_ATI = BIT_ULL(__ATA_QUIRK_NO_NCQ_ON_ATI),
- ATA_QUIRK_NO_LPM_ON_ATI = BIT_ULL(__ATA_QUIRK_NO_LPM_ON_ATI),
+ ATA_QUIRK_NO_LPM_ON_ATI_AND_AMD = BIT_ULL(__ATA_QUIRK_NO_LPM_ON_ATI_AND_AMD),
ATA_QUIRK_NO_ID_DEV_LOG = BIT_ULL(__ATA_QUIRK_NO_ID_DEV_LOG),
ATA_QUIRK_NO_LOG_DIR = BIT_ULL(__ATA_QUIRK_NO_LOG_DIR),
ATA_QUIRK_NO_FUA = BIT_ULL(__ATA_QUIRK_NO_FUA),
@@ -1147,6 +1147,7 @@ extern struct ata_host *ata_host_alloc_pinfo(struct device *dev,
extern void ata_host_get(struct ata_host *host);
extern void ata_host_put(struct ata_host *host);
extern int ata_host_start(struct ata_host *host);
+void ata_host_undo_start(struct ata_host *host);
extern int ata_host_register(struct ata_host *host,
const struct scsi_host_template *sht);
extern int ata_host_activate(struct ata_host *host, int irq,
diff --git a/include/linux/list.h b/include/linux/list.h
index 77fb62f79928..e3753695e76c 100644
--- a/include/linux/list.h
+++ b/include/linux/list.h
@@ -1172,7 +1172,7 @@ static inline void hlist_move_list(struct hlist_head *old,
{
new->first = old->first;
if (new->first)
- new->first->pprev = &new->first;
+ WRITE_ONCE(new->first->pprev, &new->first);
old->first = NULL;
}
@@ -1189,10 +1189,10 @@ static inline void hlist_splice_init(struct hlist_head *from,
struct hlist_head *to)
{
if (to->first)
- to->first->pprev = &last->next;
+ WRITE_ONCE(to->first->pprev, &last->next);
last->next = to->first;
to->first = from->first;
- from->first->pprev = &to->first;
+ WRITE_ONCE(from->first->pprev, &to->first);
from->first = NULL;
}
diff --git a/include/linux/liveupdate.h b/include/linux/liveupdate.h
index 63ea5417de84..6051abc0612c 100644
--- a/include/linux/liveupdate.h
+++ b/include/linux/liveupdate.h
@@ -25,6 +25,7 @@ struct file;
/**
* struct liveupdate_file_op_args - Arguments for file operation callbacks.
* @handler: The file handler being called.
+ * @session: The session this file belongs to.
* @retrieve_status: The retrieve status for the 'can_finish / finish'
* operation. A value of 0 means the retrieve has not been
* attempted, a positive value means the retrieve was
@@ -45,6 +46,7 @@ struct file;
*/
struct liveupdate_file_op_args {
struct liveupdate_file_handler *handler;
+ struct liveupdate_session *session;
int retrieve_status;
struct file *file;
u64 serialized_data;
@@ -247,6 +249,14 @@ void liveupdate_flb_put_incoming(struct liveupdate_flb *flb);
int liveupdate_flb_get_outgoing(struct liveupdate_flb *flb, void **objp);
void liveupdate_flb_put_outgoing(struct liveupdate_flb *flb);
+/* kernel can internally retrieve files */
+int liveupdate_get_file_incoming(struct liveupdate_session *s, u64 token,
+ struct file **filep);
+
+/* Get a token for an outgoing file, or -ENOENT if file is not preserved */
+int liveupdate_get_token_outgoing(struct liveupdate_session *s,
+ struct file *file, u64 *tokenp);
+
#else /* CONFIG_LIVEUPDATE */
static inline bool liveupdate_enabled(void)
@@ -299,5 +309,17 @@ static inline void liveupdate_flb_put_outgoing(struct liveupdate_flb *flb)
{
}
+static inline int liveupdate_get_file_incoming(struct liveupdate_session *s,
+ u64 token, struct file **filep)
+{
+ return -EOPNOTSUPP;
+}
+
+static inline int liveupdate_get_token_outgoing(struct liveupdate_session *s,
+ struct file *file, u64 *tokenp)
+{
+ return -EOPNOTSUPP;
+}
+
#endif /* CONFIG_LIVEUPDATE */
#endif /* _LINUX_LIVEUPDATE_H */
diff --git a/include/linux/llc.h b/include/linux/llc.h
index 944e9e213112..c03bb12bcc9c 100644
--- a/include/linux/llc.h
+++ b/include/linux/llc.h
@@ -9,9 +9,4 @@
#include <uapi/linux/llc.h>
-#define LLC_SAP_DYN_START 0xC0
-#define LLC_SAP_DYN_STOP 0xDE
-#define LLC_SAP_DYN_TRIES 4
-
-#define llc_ui_skb_cb(__skb) ((struct sockaddr_llc *)&((__skb)->cb[0]))
#endif /* __LINUX_LLC_H */
diff --git a/include/linux/lsm_audit.h b/include/linux/lsm_audit.h
index 584db296e43b..5cf0b4795065 100644
--- a/include/linux/lsm_audit.h
+++ b/include/linux/lsm_audit.h
@@ -44,7 +44,7 @@ struct lsm_network_audit {
struct lsm_ioctlop_audit {
struct path path;
- u16 cmd;
+ unsigned int cmd;
};
struct lsm_ibpkey_audit {
@@ -78,6 +78,7 @@ struct common_audit_data {
#define LSM_AUDIT_DATA_NOTIFICATION 16
#define LSM_AUDIT_DATA_ANONINODE 17
#define LSM_AUDIT_DATA_NLMSGTYPE 18
+#define LSM_AUDIT_DATA_NS 19
union {
struct path path;
struct dentry *dentry;
@@ -100,6 +101,10 @@ struct common_audit_data {
int reason;
const char *anonclass;
u16 nlmsg_type;
+ struct {
+ u32 ns_type;
+ u64 ns_id;
+ } ns;
} u;
/* this union contains LSM specific data */
union {
diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h
index 65c9609ec207..ef16e5343e09 100644
--- a/include/linux/lsm_hook_defs.h
+++ b/include/linux/lsm_hook_defs.h
@@ -36,6 +36,7 @@ LSM_HOOK(int, 0, binder_transfer_file, const struct cred *from,
LSM_HOOK(int, 0, ptrace_access_check, struct task_struct *child,
unsigned int mode)
LSM_HOOK(int, 0, ptrace_traceme, struct task_struct *parent)
+LSM_HOOK(int, 0, mem_foll_force, const struct cred *subject, bool opened_by_owner)
LSM_HOOK(int, 0, capget, const struct task_struct *target, kernel_cap_t *effective,
kernel_cap_t *inheritable, kernel_cap_t *permitted)
LSM_HOOK(int, 0, capset, struct cred *new, const struct cred *old,
@@ -94,7 +95,7 @@ LSM_HOOK(int, 0, path_mkdir, const struct path *dir, struct dentry *dentry,
LSM_HOOK(int, 0, path_rmdir, const struct path *dir, struct dentry *dentry)
LSM_HOOK(int, 0, path_mknod, const struct path *dir, struct dentry *dentry,
umode_t mode, unsigned int dev)
-LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, const struct mnt_idmap *idmap,
struct dentry *dentry)
LSM_HOOK(int, 0, path_truncate, const struct path *path)
LSM_HOOK(int, 0, path_symlink, const struct path *dir, struct dentry *dentry,
@@ -120,59 +121,60 @@ LSM_HOOK(int, -EOPNOTSUPP, inode_init_security, struct inode *inode,
int *xattr_count)
LSM_HOOK(int, 0, inode_init_security_anon, struct inode *inode,
const struct qstr *name, const struct inode *context_inode)
-LSM_HOOK(int, 0, inode_create, struct inode *dir, struct dentry *dentry,
- umode_t mode)
-LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_create, const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, umode_t mode)
+LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, const struct mnt_idmap *idmap,
struct inode *inode)
-LSM_HOOK(int, 0, inode_link, struct dentry *old_dentry, struct inode *dir,
- struct dentry *new_dentry)
+LSM_HOOK(int, 0, inode_link, const struct mnt_idmap *idmap,
+ struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry)
LSM_HOOK(int, 0, inode_unlink, struct inode *dir, struct dentry *dentry)
-LSM_HOOK(int, 0, inode_symlink, struct inode *dir, struct dentry *dentry,
- const char *old_name)
-LSM_HOOK(int, 0, inode_mkdir, struct inode *dir, struct dentry *dentry,
- umode_t mode)
+LSM_HOOK(int, 0, inode_symlink, const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, const char *old_name)
+LSM_HOOK(int, 0, inode_mkdir, const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, umode_t mode)
LSM_HOOK(int, 0, inode_rmdir, struct inode *dir, struct dentry *dentry)
-LSM_HOOK(int, 0, inode_mknod, struct inode *dir, struct dentry *dentry,
- umode_t mode, dev_t dev)
+LSM_HOOK(int, 0, inode_mknod, const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, umode_t mode, dev_t dev)
LSM_HOOK(int, 0, inode_rename, struct inode *old_dir, struct dentry *old_dentry,
struct inode *new_dir, struct dentry *new_dentry)
LSM_HOOK(int, 0, inode_readlink, struct dentry *dentry)
LSM_HOOK(int, 0, inode_follow_link, struct dentry *dentry, struct inode *inode,
bool rcu)
-LSM_HOOK(int, 0, inode_permission, struct inode *inode, int mask)
-LSM_HOOK(int, 0, inode_setattr, struct mnt_idmap *idmap, struct dentry *dentry,
+LSM_HOOK(int, 0, inode_permission, const struct mnt_idmap *idmap,
+ struct inode *inode, int mask)
+LSM_HOOK(int, 0, inode_setattr, const struct mnt_idmap *idmap, struct dentry *dentry,
struct iattr *attr)
-LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, const struct mnt_idmap *idmap,
struct dentry *dentry, int ia_valid)
LSM_HOOK(int, 0, inode_getattr, const struct path *path)
LSM_HOOK(int, 0, inode_xattr_skipcap, const char *name)
-LSM_HOOK(int, 0, inode_setxattr, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_setxattr, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name, const void *value,
size_t size, int flags)
LSM_HOOK(void, LSM_RET_VOID, inode_post_setxattr, struct dentry *dentry,
const char *name, const void *value, size_t size, int flags)
LSM_HOOK(int, 0, inode_getxattr, struct dentry *dentry, const char *name)
LSM_HOOK(int, 0, inode_listxattr, struct dentry *dentry)
-LSM_HOOK(int, 0, inode_removexattr, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_removexattr, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name)
LSM_HOOK(void, LSM_RET_VOID, inode_post_removexattr, struct dentry *dentry,
const char *name)
LSM_HOOK(int, 0, inode_file_setattr, struct dentry *dentry, struct file_kattr *fa)
LSM_HOOK(int, 0, inode_file_getattr, struct dentry *dentry, struct file_kattr *fa)
-LSM_HOOK(int, 0, inode_set_acl, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_set_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name, struct posix_acl *kacl)
LSM_HOOK(void, LSM_RET_VOID, inode_post_set_acl, struct dentry *dentry,
const char *acl_name, struct posix_acl *kacl)
-LSM_HOOK(int, 0, inode_get_acl, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_get_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
-LSM_HOOK(int, 0, inode_remove_acl, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_remove_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
-LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
LSM_HOOK(int, 0, inode_need_killpriv, struct dentry *dentry)
-LSM_HOOK(int, 0, inode_killpriv, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_killpriv, const struct mnt_idmap *idmap,
struct dentry *dentry)
-LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, struct mnt_idmap *idmap,
+LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, const struct mnt_idmap *idmap,
struct inode *inode, const char *name, void **buffer, bool alloc)
LSM_HOOK(int, -EOPNOTSUPP, inode_setsecurity, struct inode *inode,
const char *name, const void *value, size_t size, int flags)
@@ -187,7 +189,7 @@ LSM_HOOK(int, 0, inode_setintegrity, const struct inode *inode,
enum lsm_integrity_type type, const void *value, size_t size)
LSM_HOOK(int, 0, kernfs_init_security, struct kernfs_node *kn_dir,
struct kernfs_node *kn)
-LSM_HOOK(int, 0, file_permission, struct file *file, int mask)
+LSM_HOOK(int, 0, file_permission, const struct file *file, int mask)
LSM_HOOK(int, 0, file_alloc_security, struct file *file)
LSM_HOOK(void, LSM_RET_VOID, file_release, struct file *file)
LSM_HOOK(void, LSM_RET_VOID, file_free_security, struct file *file)
@@ -265,6 +267,9 @@ LSM_HOOK(int, -ENOSYS, task_prctl, int option, unsigned long arg2,
LSM_HOOK(void, LSM_RET_VOID, task_to_inode, struct task_struct *p,
struct inode *inode)
LSM_HOOK(int, 0, userns_create, const struct cred *cred)
+LSM_HOOK(int, 0, namespace_init, struct ns_common *ns)
+LSM_HOOK(void, LSM_RET_VOID, namespace_free, struct ns_common *ns)
+LSM_HOOK(int, 0, namespace_install, const struct nsset *nsset, struct ns_common *ns)
LSM_HOOK(int, 0, ipc_permission, struct kern_ipc_perm *ipcp, short flag)
LSM_HOOK(void, LSM_RET_VOID, ipc_getlsmprop, struct kern_ipc_perm *ipcp,
struct lsm_prop *prop)
diff --git a/include/linux/lsm_hooks.h b/include/linux/lsm_hooks.h
index c4488c4a6d8a..13621e9e233e 100644
--- a/include/linux/lsm_hooks.h
+++ b/include/linux/lsm_hooks.h
@@ -112,6 +112,7 @@ struct lsm_blob_sizes {
unsigned int lbs_ipc;
unsigned int lbs_key;
unsigned int lbs_msg_msg;
+ unsigned int lbs_ns;
unsigned int lbs_perf_event;
unsigned int lbs_task;
unsigned int lbs_xattr_count; /* num xattr slots in new_xattrs array */
diff --git a/include/linux/mailbox/riscv-rpmi-message.h b/include/linux/mailbox/riscv-rpmi-message.h
index e135c6564d0c..d5362b5821f9 100644
--- a/include/linux/mailbox/riscv-rpmi-message.h
+++ b/include/linux/mailbox/riscv-rpmi-message.h
@@ -93,6 +93,7 @@ static inline int rpmi_to_linux_error(int rpmi_error)
/* RPMI service group IDs */
#define RPMI_SRVGRP_SYSTEM_MSI 0x00002
#define RPMI_SRVGRP_CLOCK 0x00008
+#define RPMI_SRVGRP_DEVICE_POWER 0x00009
/* RPMI clock service IDs */
enum rpmi_clock_service_id {
@@ -119,6 +120,16 @@ enum rpmi_sysmsi_service_id {
RPMI_SYSMSI_SRV_ID_MAX_COUNT
};
+/* RPMI device power service IDs */
+enum rpmi_device_power_service_id {
+ RPMI_DP_SRV_ENABLE_NOTIFICATION = 0x01,
+ RPMI_DP_SRV_GET_NUM_DOMAINS = 0x02,
+ RPMI_DP_SRV_GET_ATTRS = 0x03,
+ RPMI_DP_SRV_SET_STATE = 0x04,
+ RPMI_DP_SRV_GET_STATE = 0x05,
+ RPMI_DP_SRV_ID_MAX_COUNT,
+};
+
/* RPMI Linux mailbox attribute IDs */
enum rpmi_mbox_attribute_id {
RPMI_MBOX_ATTR_SPEC_VERSION,
diff --git a/include/linux/maple_tree.h b/include/linux/maple_tree.h
index e595ae5cd0ee..e30e57f5f3df 100644
--- a/include/linux/maple_tree.h
+++ b/include/linux/maple_tree.h
@@ -9,6 +9,7 @@
*/
#include <linux/kernel.h>
+#include <linux/compiler.h>
#include <linux/rcupdate.h>
#include <linux/spinlock.h>
@@ -297,7 +298,8 @@ struct maple_tree {
#endif
#define DEFINE_MTREE(name) \
- struct maple_tree name = MTREE_INIT(name, 0)
+ struct maple_tree name = MTREE_INIT(name, 0); \
+ ASSERT_STATIC_STORAGE(name)
#define mtree_lock(mt) spin_lock((&(mt)->ma_lock))
#define mtree_lock_nested(mas, subclass) \
diff --git a/include/linux/mdio-mux.h b/include/linux/mdio-mux.h
index a5d58f221939..e2af9cb8e017 100644
--- a/include/linux/mdio-mux.h
+++ b/include/linux/mdio-mux.h
@@ -1,5 +1,5 @@
/*
- * MDIO bus multiplexer framwork.
+ * MDIO bus multiplexer framework.
*
* This file is subject to the terms and conditions of the GNU General Public
* License. See the file "COPYING" in the main directory of this archive
diff --git a/include/linux/memblock.h b/include/linux/memblock.h
index d62db9e776cf..ea288cf21400 100644
--- a/include/linux/memblock.h
+++ b/include/linux/memblock.h
@@ -46,11 +46,11 @@ extern unsigned long long max_possible_pfn;
* @MEMBLOCK_RSRV_KERN: memory region that is reserved for kernel use,
* either explictitly with memblock_reserve_kern() or via memblock
* allocation APIs. All memblock allocations set this flag.
- * @MEMBLOCK_KHO_SCRATCH: memory region that kexec can pass to the next
- * kernel in handover mode. During early boot, we do not know about all
- * memory reservations yet, so we get scratch memory from the previous
- * kernel that we know is good to use. It is the only memory that
- * allocations may happen from in this phase.
+ * @MEMBLOCK_KHO_NOPRSRV: memory region with no preservation from kexec
+ * handover. During early boot, we do not know about all memory preservations
+ * yet, so we get memory with no preservations from the previous kernel that we
+ * know is good to use. It is the only memory that allocations may happen from
+ * in this phase.
* @MEMBLOCK_RSRV_HUGETLB: memory is reserved for hugetlb pages
*/
enum memblock_flags {
@@ -61,7 +61,7 @@ enum memblock_flags {
MEMBLOCK_DRIVER_MANAGED = 0x8, /* always detected via a driver */
MEMBLOCK_RSRV_NOINIT = 0x10, /* don't initialize struct pages */
MEMBLOCK_RSRV_KERN = 0x20, /* memory reserved for kernel use */
- MEMBLOCK_KHO_SCRATCH = 0x40, /* scratch memory for kexec handover */
+ MEMBLOCK_KHO_NOPRSRV = 0x40, /* memory with no KHO preservations */
MEMBLOCK_RSRV_HUGETLB = 0x80, /* memory reserved for hugetlb pages */
};
@@ -158,11 +158,10 @@ int memblock_mark_nomap(phys_addr_t base, phys_addr_t size);
int memblock_clear_nomap(phys_addr_t base, phys_addr_t size);
int memblock_reserved_mark_noinit(phys_addr_t base, phys_addr_t size);
int memblock_reserved_mark_kern(phys_addr_t base, phys_addr_t size);
-int memblock_mark_kho_scratch(phys_addr_t base, phys_addr_t size);
-int memblock_clear_kho_scratch(phys_addr_t base, phys_addr_t size);
+int memblock_mark_kho_noprsrv(phys_addr_t base, phys_addr_t size);
+int memblock_clear_kho_noprsrv(phys_addr_t base, phys_addr_t size);
void memblock_free(void *ptr, size_t size);
-void reset_all_zones_managed_pages(void);
/* Low level functions */
void __next_mem_range(u64 *idx, int nid, enum memblock_flags flags,
@@ -301,9 +300,9 @@ static inline bool memblock_is_driver_managed(struct memblock_region *m)
return m->flags & MEMBLOCK_DRIVER_MANAGED;
}
-static inline bool memblock_is_kho_scratch(struct memblock_region *m)
+static inline bool memblock_is_kho_noprsrv(struct memblock_region *m)
{
- return m->flags & MEMBLOCK_KHO_SCRATCH;
+ return m->flags & MEMBLOCK_KHO_NOPRSRV;
}
int memblock_search_pfn_nid(unsigned long pfn, unsigned long *start_pfn,
@@ -614,12 +613,12 @@ static inline void early_memtest(phys_addr_t start, phys_addr_t end) { }
static inline void memtest_report_meminfo(struct seq_file *m) { }
#endif
-#ifdef CONFIG_MEMBLOCK_KHO_SCRATCH
-void memblock_set_kho_scratch_only(void);
-void memblock_clear_kho_scratch_only(void);
+#ifdef CONFIG_KEXEC_HANDOVER
+void memblock_set_kho_noprsrv_only(void);
+void memblock_clear_kho_noprsrv_only(void);
#else
-static inline void memblock_set_kho_scratch_only(void) { }
-static inline void memblock_clear_kho_scratch_only(void) { }
+static inline void memblock_set_kho_noprsrv_only(void) { }
+static inline void memblock_clear_kho_noprsrv_only(void) { }
#endif
#endif /* _LINUX_MEMBLOCK_H */
diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
index 7d1c0ce189a8..74110a324f9e 100644
--- a/include/linux/memcontrol.h
+++ b/include/linux/memcontrol.h
@@ -23,6 +23,7 @@
#include <linux/writeback.h>
#include <linux/page-flags.h>
#include <linux/shrinker.h>
+#include <linux/irq_work_types.h>
struct mem_cgroup;
struct obj_cgroup;
@@ -66,11 +67,6 @@ struct mem_cgroup_reclaim_cookie {
#define MEM_CGROUP_ID_SHIFT 16
-struct mem_cgroup_private_id {
- int id;
- refcount_t ref;
-};
-
struct memcg_vmstats_percpu;
struct memcg1_events_percpu;
struct memcg_vmstats;
@@ -87,50 +83,48 @@ struct mem_cgroup_reclaim_iter {
* per-node information in memory controller.
*/
struct mem_cgroup_per_node {
- /* Keep the read-only fields at the start */
+ /* Set when the memcg is created, then only read. */
+ __cacheline_group_begin_aligned(memcg_pn_read_mostly);
struct mem_cgroup *memcg; /* Back pointer, we cannot */
/* use container_of */
struct lruvec_stats_percpu __percpu *lruvec_stats_percpu;
struct lruvec_stats *lruvec_stats;
struct shrinker_info __rcu *shrinker_info;
+ struct obj_cgroup __rcu *objcg;
+
+ __cacheline_group_end_aligned(memcg_pn_read_mostly);
-#ifdef CONFIG_MEMCG_V1
/*
- * Memcg-v1 only stuff in middle as buffer between read mostly fields
- * and update often fields to avoid false sharing. If v1 stuff is
- * not present, an explicit padding is needed.
+ * Keep lruvec on its own lines. Sharing them with lru_zone_size[]
+ * regressed, see commit f59adcf59332 ("mm: memcg: add cacheline
+ * padding after lruvec in mem_cgroup_per_node").
*/
-
- struct rb_node tree_node; /* RB tree node */
- unsigned long usage_in_excess;/* Set to the value by which */
- /* the soft limit is exceeded*/
- bool on_tree;
-#else
- CACHELINE_PADDING(_pad1_);
-#endif
-
- /* Fields which get updated often at the end. */
+ __cacheline_group_begin_aligned(memcg_pn_lruvec);
struct lruvec lruvec;
- CACHELINE_PADDING(_pad2_);
- unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
+ __cacheline_group_end_aligned(memcg_pn_lruvec);
+
+ /* Written on every LRU update and on every reclaim iteration. */
+ __cacheline_group_begin_aligned(memcg_pn_write_hot);
+ long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
struct mem_cgroup_reclaim_iter iter;
+#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC
+ /* slab stats for nmi context */
+ atomic_t slab_reclaimable;
+ atomic_t slab_unreclaimable;
+#endif
+ __cacheline_group_end_aligned(memcg_pn_write_hot);
+ /* Touched only when the memcg is reparented or freed. */
+ __cacheline_group_begin_aligned(memcg_pn_cold);
/*
- * objcg is wiped out as a part of the objcg repaprenting process.
* orig_objcg preserves a pointer (and a reference) to the original
- * objcg until the end of live of memcg.
+ * objcg until the end of life of memcg.
*/
- struct obj_cgroup __rcu *objcg;
struct obj_cgroup *orig_objcg;
/* list of inherited objcgs, protected by objcg_lock */
struct list_head objcg_list;
-
-#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC
- /* slab stats for nmi context */
- atomic_t slab_reclaimable;
- atomic_t slab_unreclaimable;
-#endif
+ __cacheline_group_end_aligned(memcg_pn_cold);
};
struct mem_cgroup_threshold {
@@ -186,6 +180,7 @@ struct obj_cgroup {
struct percpu_ref refcnt;
struct mem_cgroup *memcg;
atomic_t nr_charged_bytes;
+ refcount_t private_id_ref;
union {
struct list_head list; /* protected by objcg_lock */
struct rcu_head rcu;
@@ -202,9 +197,6 @@ struct obj_cgroup {
struct mem_cgroup {
struct cgroup_subsys_state css;
- /* Private memcg ID. Used to ID objects that outlive the cgroup */
- struct mem_cgroup_private_id id;
-
/* Accounted resources */
struct page_counter memory; /* Both v1 & v2 */
@@ -213,70 +205,59 @@ struct mem_cgroup {
struct page_counter memsw; /* v1 only */
};
- /* registered local peak watchers */
- struct list_head memory_peaks;
- struct list_head swap_peaks;
- spinlock_t peaks_lock;
-
- /* Range enforcement for interrupt charges */
- struct work_struct high_work;
-
-#ifdef CONFIG_ZSWAP
- unsigned long zswap_max;
-
+ /* Written on the charge, reclaim and socket paths. */
+ __cacheline_group_begin_aligned(memcg_write_hot);
/*
- * Prevent pages from this memcg from being written back from zswap to
- * swap, and from being swapped out on zswap store failures.
+ * Hint of reclaim pressure for socket memory management. Note
+ * that this indicator should NOT be used in legacy cgroup mode
+ * where socket memory is accounted/charged separately.
*/
- bool zswap_writeback;
+ u64 socket_pressure;
+#if BITS_PER_LONG < 64
+ seqlock_t socket_pressure_seqlock;
#endif
-
- /* vmpressure notifications */
- struct vmpressure vmpressure;
-
/*
- * Should the OOM killer kill all belonging tasks, had it kill one?
+ * memory.events is bumped for this memcg and all its ancestors, so a
+ * busy child dirties every ancestor.
*/
- bool oom_group;
-
- /* memory.events and memory.events.local */
- struct cgroup_file events_file;
- struct cgroup_file events_local_file;
-
- /* handle for "memory.swap.events" */
- struct cgroup_file swap_events_file;
-
- /* memory.stat */
- struct memcg_vmstats *vmstats;
-
- /* memory.events */
atomic_long_t memory_events[MEMCG_NR_MEMORY_EVENTS];
atomic_long_t memory_events_local[MEMCG_NR_MEMORY_EVENTS];
+ /* vmpressure notifications. Written on every reclaim iteration. */
+ struct vmpressure vmpressure;
+
#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC
/* MEMCG_KMEM for nmi context */
atomic_t kmem_stat;
#endif
+
+ /* Range enforcement for interrupt charges */
+ struct irq_work high_irq_work;
+ struct work_struct high_work;
+
+ __cacheline_group_end_aligned(memcg_write_hot);
+
/*
- * Hint of reclaim pressure for socket memroy management. Note
- * that this indicator should NOT be used in legacy cgroup mode
- * where socket memory is accounted/charged separately.
+ * Off the charge and fault paths. Not write free: cgwb_domain is
+ * written on every writeout completion and mm_list on fork, exit and
+ * MGLRU aging. They are grouped here so those writes cannot land on
+ * a line that the fast paths read.
*/
- u64 socket_pressure;
-#if BITS_PER_LONG < 64
- seqlock_t socket_pressure_seqlock;
-#endif
- int kmemcg_id;
+ __cacheline_group_begin_aligned(memcg_cold);
+ /* registered local peak watchers */
+ struct list_head memory_peaks;
+ struct list_head swap_peaks;
+ spinlock_t peaks_lock;
-#ifdef CONFIG_CGROUP_WRITEBACK
- struct list_head cgwb_list;
-#endif
+ /* memory.events and memory.events.local */
+ struct cgroup_file events_file;
+ struct cgroup_file events_local_file;
- /* Keep the hot per-CPU stats pointer away from memory event counters. */
- struct memcg_vmstats_percpu __percpu *vmstats_percpu
- ____cacheline_aligned_in_smp;
+ /* handle for "memory.swap.events" */
+ struct cgroup_file swap_events_file;
#ifdef CONFIG_CGROUP_WRITEBACK
+ struct list_head cgwb_list;
struct wb_domain cgwb_domain;
struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT];
#endif
@@ -285,16 +266,17 @@ struct mem_cgroup {
/* per-memcg mm_struct list */
struct lru_gen_mm_list mm_list;
#endif
+ __cacheline_group_end_aligned(memcg_cold);
#ifdef CONFIG_MEMCG_V1
+ /* v1 only. Not grouped: v1 is legacy, sorting it is not worth it. */
+
/* Legacy consumer-oriented counters */
struct page_counter kmem; /* v1 only */
struct page_counter tcpmem; /* v1 only */
struct memcg1_events_percpu __percpu *events_percpu;
- unsigned long soft_limit;
-
/* protected by memcg_oom_lock */
bool oom_lock;
int under_oom;
@@ -325,6 +307,44 @@ struct mem_cgroup {
int swappiness;
#endif /* CONFIG_MEMCG_V1 */
+ /*
+ * Set when the memcg is created and cleared when it is offlined.
+ * Never written on a hot path.
+ */
+ __cacheline_group_begin_aligned(memcg_read_mostly);
+ /* Read on every stat update */
+ struct memcg_vmstats_percpu __percpu *vmstats_percpu;
+
+ /* memory.stat */
+ struct memcg_vmstats *vmstats;
+
+#ifdef CONFIG_ZSWAP
+ unsigned long zswap_max;
+#endif
+
+ /* The objcg holding private memcg ID. */
+ struct obj_cgroup *private_id_objcg;
+
+ /* Private memcg ID. Used to ID objects that outlive the cgroup */
+ int private_id;
+
+ int kmemcg_id;
+
+ /*
+ * Should the OOM killer kill all belonging tasks, had it kill one?
+ */
+ bool oom_group;
+
+#ifdef CONFIG_ZSWAP
+ /*
+ * Prevent pages from this memcg from being written back from zswap to
+ * swap, and from being swapped out on zswap store failures.
+ */
+ bool zswap_writeback;
+#endif
+ /* Not padded: nodeinfo[] is read-mostly too, let it share the line. */
+ __cacheline_group_end(memcg_read_mostly);
+
struct mem_cgroup_per_node *nodeinfo[];
};
@@ -377,10 +397,10 @@ enum objext_flags {
*
* The caller must ensure that the returned memcg won't be released.
*/
-static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg)
+static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg)
{
lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex));
- return READ_ONCE(objcg->memcg);
+ return objcg ? READ_ONCE(objcg->memcg) : NULL;
}
/*
@@ -391,7 +411,7 @@ static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg)
* or NULL. This function assumes that the folio is known to have a
* proper object cgroup pointer.
*/
-static inline struct obj_cgroup *folio_objcg(struct folio *folio)
+static inline struct obj_cgroup *folio_objcg(const struct folio *folio)
{
unsigned long memcg_data = folio->memcg_data;
@@ -429,11 +449,11 @@ static inline struct obj_cgroup *folio_objcg(struct folio *folio)
* Note: The caller should hold an rcu read lock or cgroup_mutex to protect
* memcg associated with a folio from being released.
*/
-static inline struct mem_cgroup *folio_memcg(struct folio *folio)
+static inline struct mem_cgroup *folio_memcg(const struct folio *folio)
{
struct obj_cgroup *objcg = folio_objcg(folio);
- return objcg ? obj_cgroup_memcg(objcg) : NULL;
+ return obj_cgroup_memcg(objcg);
}
/*
@@ -442,7 +462,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio)
*
* Returns true if folio is charged to a memory cgroup, otherwise returns false.
*/
-static inline bool folio_memcg_charged(struct folio *folio)
+static inline bool folio_memcg_charged(const struct folio *folio)
{
return folio->memcg_data != 0;
}
@@ -462,7 +482,7 @@ static inline bool folio_memcg_charged(struct folio *folio)
* A caller should hold an rcu read lock to protect memcg associated with a
* page from being released.
*/
-static inline struct mem_cgroup *folio_memcg_check(struct folio *folio)
+static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio)
{
/*
* Because folio->memcg_data might be changed asynchronously
@@ -476,17 +496,17 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio)
objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK);
- return objcg ? obj_cgroup_memcg(objcg) : NULL;
+ return obj_cgroup_memcg(objcg);
}
-static inline struct mem_cgroup *page_memcg_check(struct page *page)
+static inline struct mem_cgroup *page_memcg_check(const struct page *page)
{
if (PageTail(page))
return NULL;
- return folio_memcg_check((struct folio *)page);
+ return folio_memcg_check((const struct folio *)page);
}
-static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg)
+static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg)
{
struct mem_cgroup *memcg;
@@ -508,19 +528,19 @@ retry:
* that the folio has an associated memory cgroup. It's not safe to call
* this function against some types of folios, e.g. slab folios.
*/
-static inline bool folio_memcg_kmem(struct folio *folio)
+static inline bool folio_memcg_kmem(const struct folio *folio)
{
VM_BUG_ON_PGFLAGS(PageTail(&folio->page), &folio->page);
VM_BUG_ON_FOLIO(folio->memcg_data & MEMCG_DATA_OBJEXTS, folio);
return folio->memcg_data & MEMCG_DATA_KMEM;
}
-static inline bool PageMemcgKmem(struct page *page)
+static inline bool PageMemcgKmem(const struct page *page)
{
return folio_memcg_kmem(page_folio(page));
}
-static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg)
+static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg)
{
return (memcg == root_mem_cgroup);
}
@@ -536,7 +556,7 @@ static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg)
* and do not honour sc->memcg can use this to early-return 0 in per-memcg
* contexts.
*/
-static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc)
+static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc)
{
return !sc->memcg || mem_cgroup_is_root(sc->memcg);
}
@@ -551,8 +571,8 @@ static inline bool mem_cgroup_disabled(void)
return !cgroup_subsys_enabled(memory_cgrp_subsys);
}
-static inline void mem_cgroup_protection(struct mem_cgroup *root,
- struct mem_cgroup *memcg,
+static inline void mem_cgroup_protection(const struct mem_cgroup *root,
+ const struct mem_cgroup *memcg,
unsigned long *min,
unsigned long *low,
unsigned long *usage)
@@ -606,8 +626,8 @@ static inline void mem_cgroup_protection(struct mem_cgroup *root,
void mem_cgroup_calculate_protection(struct mem_cgroup *root,
struct mem_cgroup *memcg);
-static inline bool mem_cgroup_unprotected(struct mem_cgroup *target,
- struct mem_cgroup *memcg)
+static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target,
+ const struct mem_cgroup *memcg)
{
/*
* The root memcg doesn't account charges, and doesn't support
@@ -618,8 +638,8 @@ static inline bool mem_cgroup_unprotected(struct mem_cgroup *target,
memcg == target;
}
-static inline bool mem_cgroup_below_low(struct mem_cgroup *target,
- struct mem_cgroup *memcg)
+static inline bool mem_cgroup_below_low(const struct mem_cgroup *target,
+ const struct mem_cgroup *memcg)
{
if (mem_cgroup_unprotected(target, memcg))
return false;
@@ -628,8 +648,8 @@ static inline bool mem_cgroup_below_low(struct mem_cgroup *target,
page_counter_read(&memcg->memory);
}
-static inline bool mem_cgroup_below_min(struct mem_cgroup *target,
- struct mem_cgroup *memcg)
+static inline bool mem_cgroup_below_min(const struct mem_cgroup *target,
+ const struct mem_cgroup *memcg)
{
if (mem_cgroup_unprotected(target, memcg))
return false;
@@ -662,7 +682,8 @@ static inline int mem_cgroup_charge(struct folio *folio, struct mm_struct *mm,
return __mem_cgroup_charge(folio, mm, gfp);
}
-int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp);
+int mem_cgroup_charge_hugetlb(struct folio *folio, struct mm_struct *mm,
+ gfp_t gfp);
int mem_cgroup_swapin_charge_folio(struct folio *folio, unsigned short id,
struct mm_struct *mm, gfp_t gfp);
@@ -702,7 +723,7 @@ void mem_cgroup_migrate(struct folio *old, struct folio *new);
* @pgdat combination. This can be the node lruvec, if the memory
* controller is disabled.
*/
-static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg,
+static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg,
struct pglist_data *pgdat)
{
struct mem_cgroup_per_node *mz;
@@ -743,7 +764,7 @@ out:
* their binding is stable if the returned lruvec matches the one the caller has
* locked. Useful for lock batching.
*/
-static inline struct lruvec *folio_lruvec(struct folio *folio)
+static inline struct lruvec *folio_lruvec(const struct folio *folio)
{
struct mem_cgroup *memcg = folio_memcg(folio);
@@ -757,11 +778,11 @@ struct mem_cgroup *get_mem_cgroup_from_mm(struct mm_struct *mm);
struct mem_cgroup *get_mem_cgroup_from_current(void);
-struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio);
+struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio);
-struct lruvec *folio_lruvec_lock(struct folio *folio);
-struct lruvec *folio_lruvec_lock_irq(struct folio *folio);
-struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio,
+struct lruvec *folio_lruvec_lock(const struct folio *folio);
+struct lruvec *folio_lruvec_lock_irq(const struct folio *folio);
+struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio,
unsigned long *flags);
static inline
@@ -820,16 +841,16 @@ void mem_cgroup_iter_break(struct mem_cgroup *, struct mem_cgroup *);
void mem_cgroup_scan_tasks(struct mem_cgroup *memcg,
int (*)(struct task_struct *, void *), void *arg);
-static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg)
+static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg)
{
if (mem_cgroup_disabled())
return 0;
- return memcg->id.id;
+ return memcg->private_id;
}
struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id);
-static inline u64 mem_cgroup_id(struct mem_cgroup *memcg)
+static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg)
{
return memcg ? cgroup_id(memcg->css.cgroup) : 0;
}
@@ -841,14 +862,14 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m)
return mem_cgroup_from_css(seq_css(m));
}
-static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec)
+static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec)
{
- struct mem_cgroup_per_node *mz;
+ const struct mem_cgroup_per_node *mz;
if (mem_cgroup_disabled())
return NULL;
- mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
+ mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec);
return mz->memcg;
}
@@ -858,13 +879,13 @@ static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec)
*
* Returns the parent memcg, or NULL if this is the root.
*/
-static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg)
+static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg)
{
return mem_cgroup_from_css(memcg->css.parent);
}
-static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg,
- struct mem_cgroup *root)
+static inline bool mem_cgroup_is_descendant(const struct mem_cgroup *memcg,
+ const struct mem_cgroup *root)
{
if (root == memcg)
return true;
@@ -872,7 +893,7 @@ static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg,
}
static inline bool mm_match_cgroup(struct mm_struct *mm,
- struct mem_cgroup *memcg)
+ const struct mem_cgroup *memcg)
{
struct mem_cgroup *task_memcg;
bool match = false;
@@ -885,8 +906,8 @@ static inline bool mm_match_cgroup(struct mm_struct *mm,
return match;
}
-struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio);
-ino_t page_cgroup_ino(struct page *page);
+struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(const struct folio *folio);
+ino_t page_cgroup_ino(const struct page *page);
static inline bool mem_cgroup_online(struct mem_cgroup *memcg)
{
@@ -899,13 +920,18 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru,
int zid, long nr_pages);
static inline
-unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec,
+unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec,
enum lru_list lru, int zone_idx)
{
- struct mem_cgroup_per_node *mz;
+ long val;
+ const struct mem_cgroup_per_node *mz;
+
+ mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec);
+ val = READ_ONCE(mz->lru_zone_size[zone_idx][lru]);
+ if (WARN_ON_ONCE(val < 0))
+ return 0;
- mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
- return READ_ONCE(mz->lru_zone_size[zone_idx][lru]);
+ return val;
}
void __mem_cgroup_handle_over_high(gfp_t gfp_mask);
@@ -916,9 +942,9 @@ static inline void mem_cgroup_handle_over_high(gfp_t gfp_mask)
__mem_cgroup_handle_over_high(gfp_mask);
}
-unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg);
+unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg);
-void mem_cgroup_print_oom_context(struct mem_cgroup *memcg,
+void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg,
struct task_struct *p);
void mem_cgroup_print_oom_meminfo(struct mem_cgroup *memcg);
@@ -931,7 +957,7 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg);
void mod_memcg_state(struct mem_cgroup *memcg,
enum memcg_stat_item idx, int val);
-static inline void mod_memcg_page_state(struct page *page,
+static inline void mod_memcg_page_state(const struct page *page,
enum memcg_stat_item idx, int val)
{
struct mem_cgroup *memcg;
@@ -946,17 +972,21 @@ static inline void mod_memcg_page_state(struct page *page,
rcu_read_unlock();
}
-unsigned long memcg_events(struct mem_cgroup *memcg, int event);
-unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx);
-unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item);
+unsigned long memcg_events(const struct mem_cgroup *memcg, int event);
+unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx);
+unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item);
bool memcg_stat_item_valid(int idx);
bool memcg_vm_event_item_valid(enum vm_event_item idx);
-unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx);
-unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
+unsigned long lruvec_page_state(const struct lruvec *lruvec,
+ enum node_stat_item idx);
+unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec,
enum node_stat_item idx);
-unsigned long lruvec_page_state_local(struct lruvec *lruvec,
+unsigned long lruvec_page_state_local(const struct lruvec *lruvec,
enum node_stat_item idx);
+void mod_memcg_lruvec_state(struct lruvec *lruvec,
+ enum node_stat_item idx, int val);
+
void mem_cgroup_flush_stats(struct mem_cgroup *memcg);
void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg);
@@ -965,7 +995,7 @@ void mod_lruvec_kmem_state(void *p, enum node_stat_item idx, int val);
void count_memcg_events(struct mem_cgroup *memcg, enum vm_event_item idx,
unsigned long count);
-static inline void count_memcg_folio_events(struct folio *folio,
+static inline void count_memcg_folio_events(const struct folio *folio,
enum vm_event_item idx, unsigned long nr)
{
struct mem_cgroup *memcg;
@@ -1050,51 +1080,56 @@ void mem_cgroup_flush_workqueue(void);
extern int mem_cgroup_init(void);
#else /* CONFIG_MEMCG */
+static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg)
+{
+ return NULL;
+}
+
#define MEM_CGROUP_ID_SHIFT 0
#define root_mem_cgroup (NULL)
-static inline struct mem_cgroup *folio_memcg(struct folio *folio)
+static inline struct mem_cgroup *folio_memcg(const struct folio *folio)
{
return NULL;
}
-static inline bool folio_memcg_charged(struct folio *folio)
+static inline bool folio_memcg_charged(const struct folio *folio)
{
return false;
}
-static inline struct mem_cgroup *folio_memcg_check(struct folio *folio)
+static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio)
{
return NULL;
}
-static inline struct mem_cgroup *page_memcg_check(struct page *page)
+static inline struct mem_cgroup *page_memcg_check(const struct page *page)
{
return NULL;
}
-static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg)
+static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg)
{
return NULL;
}
-static inline bool folio_memcg_kmem(struct folio *folio)
+static inline bool folio_memcg_kmem(const struct folio *folio)
{
return false;
}
-static inline bool PageMemcgKmem(struct page *page)
+static inline bool PageMemcgKmem(const struct page *page)
{
return false;
}
-static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg)
+static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg)
{
return true;
}
-static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc)
+static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc)
{
return true;
}
@@ -1119,8 +1154,8 @@ static inline void memcg_memory_event_mm(struct mm_struct *mm,
{
}
-static inline void mem_cgroup_protection(struct mem_cgroup *root,
- struct mem_cgroup *memcg,
+static inline void mem_cgroup_protection(const struct mem_cgroup *root,
+ const struct mem_cgroup *memcg,
unsigned long *min,
unsigned long *low,
unsigned long *usage)
@@ -1133,19 +1168,19 @@ static inline void mem_cgroup_calculate_protection(struct mem_cgroup *root,
{
}
-static inline bool mem_cgroup_unprotected(struct mem_cgroup *target,
- struct mem_cgroup *memcg)
+static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target,
+ const struct mem_cgroup *memcg)
{
return true;
}
-static inline bool mem_cgroup_below_low(struct mem_cgroup *target,
- struct mem_cgroup *memcg)
+static inline bool mem_cgroup_below_low(const struct mem_cgroup *target,
+ const struct mem_cgroup *memcg)
{
return false;
}
-static inline bool mem_cgroup_below_min(struct mem_cgroup *target,
- struct mem_cgroup *memcg)
+static inline bool mem_cgroup_below_min(const struct mem_cgroup *target,
+ const struct mem_cgroup *memcg)
{
return false;
}
@@ -1156,9 +1191,10 @@ static inline int mem_cgroup_charge(struct folio *folio,
return 0;
}
-static inline int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp)
+static inline int mem_cgroup_charge_hugetlb(struct folio *folio,
+ struct mm_struct *mm, gfp_t gfp)
{
- return 0;
+ return 0;
}
static inline int mem_cgroup_swapin_charge_folio(struct folio *folio,
@@ -1184,25 +1220,25 @@ static inline void mem_cgroup_migrate(struct folio *old, struct folio *new)
{
}
-static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg,
+static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg,
struct pglist_data *pgdat)
{
return &pgdat->__lruvec;
}
-static inline struct lruvec *folio_lruvec(struct folio *folio)
+static inline struct lruvec *folio_lruvec(const struct folio *folio)
{
struct pglist_data *pgdat = folio_pgdat(folio);
return &pgdat->__lruvec;
}
-static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg)
+static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg)
{
return NULL;
}
static inline bool mm_match_cgroup(struct mm_struct *mm,
- struct mem_cgroup *memcg)
+ const struct mem_cgroup *memcg)
{
return true;
}
@@ -1217,7 +1253,7 @@ static inline struct mem_cgroup *get_mem_cgroup_from_current(void)
return NULL;
}
-static inline struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio)
+static inline struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio)
{
return NULL;
}
@@ -1250,7 +1286,7 @@ static inline void mem_cgroup_put(struct mem_cgroup *memcg)
{
}
-static inline struct lruvec *folio_lruvec_lock(struct folio *folio)
+static inline struct lruvec *folio_lruvec_lock(const struct folio *folio)
{
struct pglist_data *pgdat = folio_pgdat(folio);
@@ -1259,7 +1295,7 @@ static inline struct lruvec *folio_lruvec_lock(struct folio *folio)
return &pgdat->__lruvec;
}
-static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio)
+static inline struct lruvec *folio_lruvec_lock_irq(const struct folio *folio)
{
struct pglist_data *pgdat = folio_pgdat(folio);
@@ -1268,7 +1304,7 @@ static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio)
return &pgdat->__lruvec;
}
-static inline struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio,
+static inline struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio,
unsigned long *flagsp)
{
struct pglist_data *pgdat = folio_pgdat(folio);
@@ -1296,7 +1332,7 @@ static inline void mem_cgroup_scan_tasks(struct mem_cgroup *memcg,
{
}
-static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg)
+static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg)
{
return 0;
}
@@ -1308,7 +1344,7 @@ static inline struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id)
return NULL;
}
-static inline u64 mem_cgroup_id(struct mem_cgroup *memcg)
+static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg)
{
return 0;
}
@@ -1323,7 +1359,7 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m)
return NULL;
}
-static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec)
+static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec)
{
return NULL;
}
@@ -1334,19 +1370,20 @@ static inline bool mem_cgroup_online(struct mem_cgroup *memcg)
}
static inline
-unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec,
+unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec,
enum lru_list lru, int zone_idx)
{
return 0;
}
-static inline unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg)
+static inline unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg)
{
return 0;
}
static inline void
-mem_cgroup_print_oom_context(struct mem_cgroup *memcg, struct task_struct *p)
+mem_cgroup_print_oom_context(const struct mem_cgroup *memcg,
+ struct task_struct *p)
{
}
@@ -1375,17 +1412,17 @@ static inline void mod_memcg_state(struct mem_cgroup *memcg,
{
}
-static inline void mod_memcg_page_state(struct page *page,
+static inline void mod_memcg_page_state(const struct page *page,
enum memcg_stat_item idx, int val)
{
}
-static inline unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx)
+static inline unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx)
{
return 0;
}
-static inline unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item)
+static inline unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item)
{
return 0;
}
@@ -1400,24 +1437,29 @@ static inline bool memcg_vm_event_item_valid(enum vm_event_item idx)
return false;
}
-static inline unsigned long lruvec_page_state(struct lruvec *lruvec,
+static inline unsigned long lruvec_page_state(const struct lruvec *lruvec,
enum node_stat_item idx)
{
return node_page_state(lruvec_pgdat(lruvec), idx);
}
-static inline unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
+static inline unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec,
enum node_stat_item idx)
{
return node_page_state_monotonic(lruvec_pgdat(lruvec), idx);
}
-static inline unsigned long lruvec_page_state_local(struct lruvec *lruvec,
+static inline unsigned long lruvec_page_state_local(const struct lruvec *lruvec,
enum node_stat_item idx)
{
return node_page_state(lruvec_pgdat(lruvec), idx);
}
+static inline void mod_memcg_lruvec_state(struct lruvec *lruvec,
+ enum node_stat_item idx, int val)
+{
+}
+
static inline void mem_cgroup_flush_stats(struct mem_cgroup *memcg)
{
}
@@ -1440,7 +1482,7 @@ static inline void count_memcg_events(struct mem_cgroup *memcg,
{
}
-static inline void count_memcg_folio_events(struct folio *folio,
+static inline void count_memcg_folio_events(const struct folio *folio,
enum vm_event_item idx, unsigned long nr)
{
}
@@ -1474,7 +1516,7 @@ static inline void mem_cgroup_flush_workqueue(void) { }
static inline int mem_cgroup_init(void) { return 0; }
#endif /* CONFIG_MEMCG */
-static inline struct lruvec *parent_lruvec(struct lruvec *lruvec)
+static inline struct lruvec *parent_lruvec(const struct lruvec *lruvec)
{
struct mem_cgroup *memcg;
@@ -1537,15 +1579,15 @@ static inline void lruvec_unlock_irqrestore(struct lruvec *lruvec, unsigned long
}
/* Test requires a stable folio->memcg binding, see folio_memcg() */
-static inline bool folio_matches_lruvec(struct folio *folio,
- struct lruvec *lruvec)
+static inline bool folio_matches_lruvec(const struct folio *folio,
+ const struct lruvec *lruvec)
{
return lruvec_pgdat(lruvec) == folio_pgdat(folio) &&
lruvec_memcg(lruvec) == folio_memcg(folio);
}
/* Don't lock again iff page's lruvec locked */
-static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio,
+static inline struct lruvec *folio_lruvec_relock_irq(const struct folio *folio,
struct lruvec *locked_lruvec)
{
if (locked_lruvec) {
@@ -1559,7 +1601,7 @@ static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio,
}
/* Don't lock again iff folio's lruvec locked */
-static inline void folio_lruvec_relock_irqsave(struct folio *folio,
+static inline void folio_lruvec_relock_irqsave(const struct folio *folio,
struct lruvec **lruvecp, unsigned long *flags)
{
if (*lruvecp) {
@@ -1651,7 +1693,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg)
write_sequnlock_irqrestore(&memcg->socket_pressure_seqlock, flags);
}
-static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg)
+static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg)
{
unsigned int seq;
u64 val;
@@ -1669,7 +1711,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg)
WRITE_ONCE(memcg->socket_pressure, jiffies + HZ);
}
-static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg)
+static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg)
{
return READ_ONCE(memcg->socket_pressure);
}
@@ -1736,7 +1778,7 @@ void __memcg_kmem_uncharge_page(struct page *page, int order);
* needs to be used outside of the local scope.
*/
struct obj_cgroup *current_obj_cgroup(void);
-struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio);
+struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio);
static inline struct obj_cgroup *get_obj_cgroup_from_current(void)
{
@@ -1782,7 +1824,7 @@ static inline void memcg_kmem_uncharge_page(struct page *page, int order)
* A helper for accessing memcg's kmem_id, used for getting
* corresponding LRU lists.
*/
-static inline int memcg_kmem_id(struct mem_cgroup *memcg)
+static inline int memcg_kmem_id(const struct mem_cgroup *memcg)
{
return memcg ? memcg->kmemcg_id : -1;
}
@@ -1839,7 +1881,7 @@ static inline void __memcg_kmem_uncharge_page(struct page *page, int order)
{
}
-static inline struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio)
+static inline struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio)
{
return NULL;
}
@@ -1854,7 +1896,7 @@ static inline bool memcg_kmem_online(void)
return false;
}
-static inline int memcg_kmem_id(struct mem_cgroup *memcg)
+static inline int memcg_kmem_id(const struct mem_cgroup *memcg)
{
return -1;
}
@@ -1870,7 +1912,7 @@ static inline void count_objcg_events(struct obj_cgroup *objcg,
{
}
-static inline ino_t page_cgroup_ino(struct page *page)
+static inline ino_t page_cgroup_ino(const struct page *page)
{
return 0;
}
@@ -1890,11 +1932,21 @@ static inline bool memcg_is_dying(struct mem_cgroup *memcg)
}
#endif /* CONFIG_MEMCG */
+#if defined(CONFIG_MEMCG) && defined(CONFIG_LRU_GEN)
+void mem_cgroup_calculate_protection_path(struct mem_cgroup *root,
+ struct mem_cgroup *memcg);
+#else
+static inline void mem_cgroup_calculate_protection_path(struct mem_cgroup *root,
+ struct mem_cgroup *memcg)
+{
+}
+#endif
+
#if defined(CONFIG_MEMCG) && defined(CONFIG_ZSWAP)
bool obj_cgroup_may_zswap(struct obj_cgroup *objcg);
void obj_cgroup_charge_zswap(struct obj_cgroup *objcg, size_t size);
void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size);
-bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg);
+bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg);
#else
static inline bool obj_cgroup_may_zswap(struct obj_cgroup *objcg)
{
@@ -1908,7 +1960,7 @@ static inline void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg,
size_t size)
{
}
-static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg)
+static inline bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg)
{
/* if zswap is disabled, do not block pages going to the swapping device */
return true;
@@ -1919,13 +1971,9 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg)
/* Cgroup v1-related declarations */
#ifdef CONFIG_MEMCG_V1
-unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order,
- gfp_t gfp_mask,
- unsigned long *total_scanned);
-
bool mem_cgroup_oom_synchronize(bool wait);
-static inline bool task_in_memcg_oom(struct task_struct *p)
+static inline bool task_in_memcg_oom(const struct task_struct *p)
{
return p->memcg_in_oom;
}
@@ -1943,15 +1991,7 @@ static inline void mem_cgroup_exit_user_fault(void)
}
#else /* CONFIG_MEMCG_V1 */
-static inline
-unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order,
- gfp_t gfp_mask,
- unsigned long *total_scanned)
-{
- return 0;
-}
-
-static inline bool task_in_memcg_oom(struct task_struct *p)
+static inline bool task_in_memcg_oom(const struct task_struct *p)
{
return false;
}
diff --git a/include/linux/mempool.h b/include/linux/mempool.h
index a0fa6d43e0dc..6da502aef2f7 100644
--- a/include/linux/mempool.h
+++ b/include/linux/mempool.h
@@ -70,6 +70,13 @@ int mempool_alloc_bulk_noprof(struct mempool *pool, void **elem,
#define mempool_alloc_bulk(...) \
alloc_hooks(mempool_alloc_bulk_noprof(__VA_ARGS__))
+/*
+ * Allocate a new element without dipping into the pool's reserves or
+ * waiting. Returns NULL on failure.
+ */
+#define mempool_alloc_noreserve(_pool, _gfp) \
+ alloc_hooks((_pool)->alloc(_gfp, (_pool)->pool_data))
+
void *mempool_alloc_preallocated(struct mempool *pool) __malloc;
void mempool_free(void *element, struct mempool *pool);
unsigned int mempool_free_bulk(struct mempool *pool, void **elem,
diff --git a/include/linux/mfd/88pm886.h b/include/linux/mfd/88pm886.h
index 2c24dd3032ab..9e96d2cb92f5 100644
--- a/include/linux/mfd/88pm886.h
+++ b/include/linux/mfd/88pm886.h
@@ -2,6 +2,7 @@
#ifndef __MFD_88PM886_H
#define __MFD_88PM886_H
+#include <linux/bits.h>
#include <linux/i2c.h>
#include <linux/regmap.h>
@@ -130,6 +131,12 @@
#define PM886_GPADC_INDEX_TO_BIAS_uA(i) (1 + (i) * 5)
/* Battery block register definitions */
+#define PM886_REG_BATTERY_CONFIG1 0x28
+#define PM886_REG_VBUS_EN BIT(7)
+
+#define PM886_REG_BOOST_CONFIG1 0x6b
+#define PM886_REG_BOOST_MASK GENMASK(2, 0)
+
#define PM886_REG_CLS_CONFIG1 0x71
struct pm886_chip {
diff --git a/include/linux/mfd/cs42l43-regs.h b/include/linux/mfd/cs42l43-regs.h
index 68831f113589..4c00ceae8b46 100644
--- a/include/linux/mfd/cs42l43-regs.h
+++ b/include/linux/mfd/cs42l43-regs.h
@@ -1183,6 +1183,7 @@
/* CS42L43B VARIANT REGISTERS */
#define CS42L43B_DEVID_VAL 0x0042A43B
+#define CS42L44_DEVID_VAL 0x00042A44
#define CS42L43B_DECIM_VOL_CTRL_CH1_CH2 0x00008280
#define CS42L43B_DECIM_VOL_CTRL_CH3_CH4 0x00008284
diff --git a/include/linux/mfd/da9150/core.h b/include/linux/mfd/da9150/core.h
index d116d5f3ef56..369698036d1a 100644
--- a/include/linux/mfd/da9150/core.h
+++ b/include/linux/mfd/da9150/core.h
@@ -65,6 +65,7 @@ struct da9150 {
struct regmap_irq_chip_data *regmap_irq_data;
int irq;
int irq_base;
+ bool irq_wake_enabled;
};
/* Device I/O - Query Interface for FG and standard register access */
diff --git a/include/linux/mfd/khadas-mcu.h b/include/linux/mfd/khadas-mcu.h
index a99ba2ed0e4e..acd3291061b4 100644
--- a/include/linux/mfd/khadas-mcu.h
+++ b/include/linux/mfd/khadas-mcu.h
@@ -70,6 +70,13 @@
#define KHADAS_MCU_WOL_INIT_START_REG 0x87 /* WO */
#define KHADAS_MCU_CMD_FAN_STATUS_CTRL_REG 0x88 /* WO */
+/* VIM4 specific registers */
+#define KHADAS_MCU_VIM4_REST_CONF_REG 0x2c /* WO - reset EEPROM */
+#define KHADAS_MCU_VIM4_LED_ON_RAM_REG 0x89 /* WO - LED volatile */
+#define KHADAS_MCU_VIM4_FAN_CTRL_REG 0x8a /* WO */
+#define KHADAS_MCU_VIM4_WDT_EN_REG 0x8b /* WO */
+#define KHADAS_MCU_VIM4_SYS_RST_REG 0x91 /* WO */
+
enum {
KHADAS_BOARD_VIM1 = 0x1,
KHADAS_BOARD_VIM2,
@@ -80,7 +87,7 @@ enum {
/**
* struct khadas_mcu - Khadas MCU structure
- * @device: device reference used for logs
+ * @dev: device reference used for logs
* @regmap: register map
*/
struct khadas_mcu {
@@ -88,4 +95,14 @@ struct khadas_mcu {
struct regmap *regmap;
};
+/**
+ * enum khadas_mcu_type - Khadas MCU hardware variant
+ * @KHADAS_MCU_GENERIC: VIM1, VIM2, VIM3, Edge, Edge-V (shared register map)
+ * @KHADAS_MCU_VIM4: VIM4 (extended register map, distinct fan/LED/WDT regs)
+ */
+enum khadas_mcu_type {
+ KHADAS_MCU_GENERIC = 1,
+ KHADAS_MCU_VIM4,
+};
+
#endif /* MFD_KHADAS_MCU_H */
diff --git a/include/linux/mfd/lm3533.h b/include/linux/mfd/lm3533.h
index 69059a7a2ce5..8f72dd41e8f0 100644
--- a/include/linux/mfd/lm3533.h
+++ b/include/linux/mfd/lm3533.h
@@ -15,9 +15,14 @@
#define LM3533_ATTR_RW(_name) \
DEVICE_ATTR(_name, S_IRUGO | S_IWUSR , show_##_name, store_##_name)
+#define LM3533_MAX_CURRENT_MIN 5000
+#define LM3533_MAX_CURRENT_MAX 29800
+#define LM3533_MAX_CURRENT_STEP 800
+
struct device;
struct gpio_desc;
struct regmap;
+struct regulator;
struct lm3533 {
struct device *dev;
@@ -25,7 +30,10 @@ struct lm3533 {
struct regmap *regmap;
struct gpio_desc *hwen;
- int irq;
+ struct regulator *vin_supply;
+
+ u32 boost_ovp;
+ u32 boost_freq;
unsigned have_als:1;
unsigned have_backlights:1;
@@ -33,67 +41,18 @@ struct lm3533 {
};
struct lm3533_ctrlbank {
- struct lm3533 *lm3533;
+ struct regmap *regmap;
struct device *dev;
int id;
};
-struct lm3533_als_platform_data {
- unsigned pwm_mode:1; /* PWM input mode (default analog) */
- u8 r_select; /* 1 - 127 (ignored in PWM-mode) */
-};
-
-struct lm3533_bl_platform_data {
- char *name;
- u16 max_current; /* 5000 - 29800 uA (800 uA step) */
- u8 default_brightness; /* 0 - 255 */
- u8 pwm; /* 0 - 0x3f */
-};
-
-struct lm3533_led_platform_data {
- char *name;
- const char *default_trigger;
- u16 max_current; /* 5000 - 29800 uA (800 uA step) */
- u8 pwm; /* 0 - 0x3f */
-};
-
-enum lm3533_boost_freq {
- LM3533_BOOST_FREQ_500KHZ,
- LM3533_BOOST_FREQ_1000KHZ,
-};
-
-enum lm3533_boost_ovp {
- LM3533_BOOST_OVP_16V,
- LM3533_BOOST_OVP_24V,
- LM3533_BOOST_OVP_32V,
- LM3533_BOOST_OVP_40V,
-};
-
-struct lm3533_platform_data {
- enum lm3533_boost_ovp boost_ovp;
- enum lm3533_boost_freq boost_freq;
-
- struct lm3533_als_platform_data *als;
-
- struct lm3533_bl_platform_data *backlights;
- int num_backlights;
-
- struct lm3533_led_platform_data *leds;
- int num_leds;
-};
-
-extern int lm3533_ctrlbank_enable(struct lm3533_ctrlbank *cb);
-extern int lm3533_ctrlbank_disable(struct lm3533_ctrlbank *cb);
-
-extern int lm3533_ctrlbank_set_brightness(struct lm3533_ctrlbank *cb, u8 val);
-extern int lm3533_ctrlbank_get_brightness(struct lm3533_ctrlbank *cb, u8 *val);
-extern int lm3533_ctrlbank_set_max_current(struct lm3533_ctrlbank *cb,
- u16 imax);
-extern int lm3533_ctrlbank_set_pwm(struct lm3533_ctrlbank *cb, u8 val);
-extern int lm3533_ctrlbank_get_pwm(struct lm3533_ctrlbank *cb, u8 *val);
+int lm3533_ctrlbank_enable(struct lm3533_ctrlbank *cb);
+int lm3533_ctrlbank_disable(struct lm3533_ctrlbank *cb);
-extern int lm3533_read(struct lm3533 *lm3533, u8 reg, u8 *val);
-extern int lm3533_write(struct lm3533 *lm3533, u8 reg, u8 val);
-extern int lm3533_update(struct lm3533 *lm3533, u8 reg, u8 val, u8 mask);
+int lm3533_ctrlbank_set_brightness(struct lm3533_ctrlbank *cb, u32 val);
+int lm3533_ctrlbank_get_brightness(struct lm3533_ctrlbank *cb, u32 *val);
+int lm3533_ctrlbank_set_max_current(struct lm3533_ctrlbank *cb, u32 imax);
+int lm3533_ctrlbank_set_pwm(struct lm3533_ctrlbank *cb, u32 val);
+int lm3533_ctrlbank_get_pwm(struct lm3533_ctrlbank *cb, u32 *val);
#endif /* __LINUX_MFD_LM3533_H */
diff --git a/include/linux/mfd/motorola-cpcap.h b/include/linux/mfd/motorola-cpcap.h
index 981e5777deb7..bb23363eeccd 100644
--- a/include/linux/mfd/motorola-cpcap.h
+++ b/include/linux/mfd/motorola-cpcap.h
@@ -25,6 +25,13 @@
#define CPCAP_REVISION_2_0 0x10
#define CPCAP_REVISION_2_1 0x11
+enum cpcap_variant {
+ CPCAP_DEFAULT = 1,
+ CPCAP_MAPPHONE,
+ CPCAP_MOT,
+ CPCAP_MAX
+};
+
/* CPCAP registers */
#define CPCAP_REG_INT1 0x0000 /* Interrupt 1 */
#define CPCAP_REG_INT2 0x0004 /* Interrupt 2 */
diff --git a/include/linux/mfd/rohm-bd73800.h b/include/linux/mfd/rohm-bd73800.h
new file mode 100644
index 000000000000..c6c0d453a40d
--- /dev/null
+++ b/include/linux/mfd/rohm-bd73800.h
@@ -0,0 +1,306 @@
+/* SPDX-License-Identifier: GPL-2.0-or-later */
+/*
+ * Copyright 2026 ROHM Semiconductors.
+ *
+ * Author: Matti Vaittinen <matti.vaittinen@fi.rohmeurope.com>
+ */
+
+#ifndef _MFD_BD73800_H
+#define _MFD_BD73800_H
+
+#include <linux/regmap.h>
+
+enum {
+ BD73800_BUCK1 = 0,
+ BD73800_BUCK2,
+ BD73800_BUCK3,
+ BD73800_BUCK4,
+ BD73800_BUCK5,
+ BD73800_BUCK6,
+ BD73800_BUCK7,
+ BD73800_BUCK8,
+ BD73800_LDO1,
+ BD73800_LDO2,
+ BD73800_LDO3,
+ BD73800_LDO4,
+};
+
+/*
+ * All regulators except BUCK 5 have full 8-bit register of valid voltage
+ * values, including 0.
+ */
+#define BD73800_NUM_VOLTS (0xff + 1)
+/*
+ * BUCK 5 has two sets of voltage ranges, both having valid voltage selectors
+ * from 0x0 to 0x7f
+ */
+#define BD73800_BUCK5_VOLTS (0x80 + 0x80)
+
+/* BD73800 interrupts */
+enum {
+ /* INT_STAT_1 register IRQs, ADC and RTC */
+ BD73800_INT_ADC_ACCUM_DONE,
+ BD73800_INT_ADC_ACCUM_OVF,
+ BD73800_INT_ADC_ACCUM_VAL,
+ BD73800_INT_ADC_ACCUM_TW,
+ BD73800_INT_ADC_POW_VAL,
+ BD73800_INT_RTC0,
+ BD73800_INT_RTC1,
+ BD73800_INT_RTC2,
+
+ /* BUCK reg interrupts */
+ /* INT_STAT_2 IRQs */
+ BD73800_INT_BUCK1_DVS_DONE,
+ BD73800_INT_BUCK2_DVS_DONE,
+ BD73800_INT_BUCK3_DVS_DONE,
+ BD73800_INT_BUCK4_DVS_DONE,
+ BD73800_INT_BUCK5_DVS_DONE,
+ BD73800_INT_BUCK6_DVS_DONE,
+ BD73800_INT_BUCK7_DVS_DONE,
+ BD73800_INT_BUCK8_DVS_DONE,
+ /* INT_STAT_3 IRQs */
+ BD73800_INT_BUCK1_OCP,
+ BD73800_INT_BUCK2_OCP,
+ BD73800_INT_BUCK3_OCP,
+ BD73800_INT_BUCK4_OCP,
+ BD73800_INT_BUCK5_OCP,
+ BD73800_INT_BUCK6_OCP,
+ BD73800_INT_BUCK7_OCP,
+ BD73800_INT_BUCK8_OCP,
+
+ /* INT_STAT_4 IRQs, power-button, WDG and reset */
+ BD73800_INT_PBTN_LONG_PRESS,
+ BD73800_INT_PBTN_MID_PRESS,
+ /*
+ * The SHORT_PUSH is generated when button is first pressed (longer
+ * than configured time limit), and then released before the MID_PRESS
+ * time limit. The SHORT_PRESS is generated immediately when button is
+ * pressed for longer than configured limit, whether it is released or
+ * not.
+ */
+ BD73800_INT_PBTN_SHORT_PUSH,
+ BD73800_INT_PBTN_SHORT_PRESS,
+ BD73800_INT_WDG,
+ BD73800_INT_SWRESET,
+ BD73800_INT_SEQ_DONE,
+
+ /* INT_STAT_5 IRQs, GPIO */
+ BD73800_INT_GPIO1,
+ BD73800_INT_GPIO2,
+ BD73800_INT_GPIO3,
+ BD73800_INT_GPIO4,
+};
+
+#define BD73800_MASK_RUN_EN BIT(2)
+#define BD73800_MASK_SUSP_EN BIT(1)
+#define BD73800_MASK_IDLE_EN BIT(0)
+#define BD73800_MASK_VOLT GENMASK(7, 0)
+#define BD73800_MASK_BUCK5_VOLT GENMASK(6, 0)
+#define BD73800_MASK_RAMP_DELAY GENMASK(2, 1)
+#define BD73800_BUCK5_RANGE_MASK BIT(7)
+
+/* BD73800 registers */
+enum {
+ BD73800_REG_PRODUCT_ID = 0x0,
+ BD73800_REG_MANUFACTURER_ID,
+ BD73800_REG_REVISION,
+ BD73800_REG_NVMVERSION,
+ BD73800_REG_POR_REASON,
+ BD73800_REG_RESET_REASON1,
+ BD73800_REG_RESET_REASON2,
+ BD73800_REG_RESET_REASON3,
+ BD73800_REG_POW_STATE,
+ BD73800_REG_WRST_SEL,
+ BD73800_REG_PS_CTRL_1,
+ BD73800_REG_PS_CTRL_2,
+ BD73800_REG_RCVCFG,
+ BD73800_REG_RCVNUM,
+ BD73800_REG_CRDCFG, /* 0x0f, followed by undocumented reg */
+
+ BD73800_REG_BUCK1_ON = 0x11,
+ BD73800_REG_BUCK1_MODE,
+ BD73800_REG_BUCK1_VOLT_RUN,
+ BD73800_REG_BUCK1_VOLT_IDLE,
+ BD73800_REG_BUCK1_VOLT_SUSP, /* 0x15, followed by undocumented reg */
+
+ BD73800_REG_BUCK2_ON = 0x17,
+ BD73800_REG_BUCK2_MODE,
+ BD73800_REG_BUCK2_VOLT_RUN,
+ BD73800_REG_BUCK2_VOLT_IDLE,
+ BD73800_REG_BUCK2_VOLT_SUSP, /* 0x1b */
+
+ BD73800_REG_BUCK3_ON = 0x1d,
+ BD73800_REG_BUCK3_MODE,
+ BD73800_REG_BUCK3_VOLT_RUN,
+ BD73800_REG_BUCK3_VOLT_IDLE,
+ BD73800_REG_BUCK3_VOLT_SUSP, /* 0x21 */
+
+ BD73800_REG_BUCK4_ON = 0x23,
+ BD73800_REG_BUCK4_MODE,
+ BD73800_REG_BUCK4_VOLT_RUN,
+ BD73800_REG_BUCK4_VOLT_IDLE,
+ BD73800_REG_BUCK4_VOLT_SUSP, /* 0x27 */
+
+ BD73800_REG_BUCK5_ON = 0x29,
+ BD73800_REG_BUCK5_MODE,
+ BD73800_REG_BUCK5_VOLT_RUN,
+ BD73800_REG_BUCK5_VOLT_IDLE,
+ BD73800_REG_BUCK5_VOLT_SUSP, /* 0x2d */
+
+ BD73800_REG_BUCK6_ON = 0x2f,
+ BD73800_REG_BUCK6_MODE,
+ BD73800_REG_BUCK6_VOLT_RUN,
+ BD73800_REG_BUCK6_VOLT_IDLE,
+ BD73800_REG_BUCK6_VOLT_SUSP, /* 0x33 */
+
+ BD73800_REG_BUCK7_ON = 0x35,
+ BD73800_REG_BUCK7_MODE,
+ BD73800_REG_BUCK7_VOLT_RUN,
+ BD73800_REG_BUCK7_VOLT_IDLE,
+ BD73800_REG_BUCK7_VOLT_SUSP, /* 0x39 */
+
+ BD73800_REG_BUCK8_ON = 0x3b,
+ BD73800_REG_BUCK8_MODE,
+ BD73800_REG_BUCK8_VOLT_RUN,
+ BD73800_REG_BUCK8_VOLT_IDLE,
+ BD73800_REG_BUCK8_VOLT_SUSP, /* 0x3f */
+
+ BD73800_REG_LDO1_ON = 0x41,
+ BD73800_REG_LDO1_VOLT,
+ BD73800_REG_LDO1_MODE,
+ BD73800_REG_LDO2_ON,
+ BD73800_REG_LDO2_VOLT,
+ BD73800_REG_LDO2_MODE,
+ BD73800_REG_LDO3_ON,
+ BD73800_REG_LDO3_VOLT,
+ BD73800_REG_LDO3_MODE,
+ BD73800_REG_LDO4_ON,
+ BD73800_REG_LDO4_VOLT,
+ BD73800_REG_LDO4_MODE, /* 0x4c */
+
+ BD73800_REG_GPO_OUT = 0x4e,
+ BD73800_REG_OUT32K = 0x50,
+ BD73800_REG_RTC_SEC,
+ BD73800_REG_RTC_MIN,
+ BD73800_REG_RTC_HOUR,
+ BD73800_REG_RTC_WEEK,
+ BD73800_REG_RTC_DAY,
+ BD73800_REG_RTC_MONTH,
+ BD73800_REG_RTC_YEAR,
+ BD73800_REG_RTC_ALM0_SEC,
+ BD73800_REG_RTC_ALM0_MIN,
+ BD73800_REG_RTC_ALM0_HOUR,
+ BD73800_REG_RTC_ALM0_WEEK,
+ BD73800_REG_RTC_ALM0_DAY,
+ BD73800_REG_RTC_ALM0_MONTH,
+ BD73800_REG_RTC_ALM0_YEAR,
+ BD73800_REG_RTC_ALM1_SEC,
+ BD73800_REG_RTC_ALM1_MIN,
+ BD73800_REG_RTC_ALM1_HOUR,
+ BD73800_REG_RTC_ALM1_WEEK,
+ BD73800_REG_RTC_ALM1_DAY,
+ BD73800_REG_RTC_ALM1_MONTH,
+ BD73800_REG_RTC_ALM1_YEAR,
+ BD73800_REG_RTC_ALM2,
+ BD73800_REG_RTC_CONF, /* 0x69 */
+
+ BD73800_REG_ADC_CTRL_1 = 0x6b,
+ BD73800_REG_ADC_CTRL_2,
+ BD73800_REG_ADC_ACCUM_NUM2,
+ BD73800_REG_ADC_ACCUM_NUM1,
+ BD73800_REG_ADC_ACCUM_NUM0,
+ BD73800_REG_ADC_ACCUM_KICK,
+ BD73800_REG_ADC_ACCUM_CNT2,
+ BD73800_REG_ADC_ACCUM_CNT1,
+ BD73800_REG_ADC_ACCUM_CNT0,
+ BD73800_REG_ADC_ACCUM_VAL2,
+ BD73800_REG_ADC_ACCUM_VAL1,
+ BD73800_REG_ADC_ACCUM_VAL0,
+ BD73800_REG_ADC_VOL_VAL1,
+ BD73800_REG_ADC_VOL_VAL0,
+ BD73800_REG_ADC_CUR_VAL1,
+ BD73800_REG_ADC_CUR_VAL0,
+ BD73800_REG_ADC_POW_VAL1,
+ BD73800_REG_ADC_POW_VAL0,
+ BD73800_REG_ADC_TEMP_VAL1,
+ BD73800_REG_ADC_TEMP_VAL0,
+ BD73800_REG_ADC_ACCUM_VAL_INT_TH4,
+ BD73800_REG_ADC_ACCUM_VAL_INT_TH3,
+ BD73800_REG_ADC_ACCUM_VAL_INT_TH2,
+ BD73800_REG_ADC_ACCUM_VAL_INT_TH1,
+ BD73800_REG_ADC_ACCUM_VAL_INT_TH0,
+ BD73800_REG_ADC_WARN_TEMP_INT_TH1,
+ BD73800_REG_ADC_WARN_TEMP_INT_TH0,
+ BD73800_REG_ADC_POW_VAL_INT_TH1,
+ BD73800_REG_ADC_POW_VAL_INT_TH0, /* 0x89 */
+
+ BD73800_REG_PBTN_CONF = 0x8b,
+
+ BD73800_REG_INT_MAIN_EN = 0x8f,
+ BD73800_REG_INT_1_EN,
+ BD73800_REG_INT_2_EN,
+ BD73800_REG_INT_3_EN,
+ BD73800_REG_INT_4_EN,
+ BD73800_REG_INT_5_EN, /* 0x94 */
+
+ BD73800_REG_INT_MAIN_STAT = 0x96,
+ BD73800_REG_INT_1_STAT,
+ BD73800_REG_INT_2_STAT,
+ BD73800_REG_INT_3_STAT,
+ BD73800_REG_INT_4_STAT,
+ BD73800_REG_INT_5_STAT, /* 0x9b */
+
+ BD73800_REG_INT_MAIN_SRC = 0x9d,
+ BD73800_REG_INT_1_SRC,
+ BD73800_REG_INT_2_SRC,
+ BD73800_REG_INT_3_SRC,
+ BD73800_REG_INT_4_SRC,
+ BD73800_REG_INT_5_SRC, /* 0xa2 */
+
+ BD73800_REG_RST_MASK = 0xaf,
+ BD73800_MAX_REGISTER,
+};
+
+#define BD73800_REG_RTC_START BD73800_REG_RTC_SEC
+#define BD73800_REG_RTC_ALM_START BD73800_REG_RTC_ALM0_SEC
+
+/* BD73800 IRQ register masks */
+
+#define BD73800_INT_MAIN_EN_ALL GENMASK(4, 0)
+#define BD73800_INT_ADC_ACCUM_DONE_MASK BIT(0)
+#define BD73800_INT_ADC_ACCUM_OVF_MASK BIT(1)
+#define BD73800_INT_ADC_ACCUM_VAL_MASK BIT(2)
+#define BD73800_INT_ADC_ACCUM_TW_MASK BIT(3)
+#define BD73800_INT_ADC_POW_VAL_MASK BIT(4)
+#define BD73800_INT_RTC0_MASK BIT(5)
+#define BD73800_INT_RTC1_MASK BIT(6)
+#define BD73800_INT_RTC2_MASK BIT(7)
+#define BD73800_INT_BUCK1_DVS_DONE_MASK BIT(0)
+#define BD73800_INT_BUCK2_DVS_DONE_MASK BIT(1)
+#define BD73800_INT_BUCK3_DVS_DONE_MASK BIT(2)
+#define BD73800_INT_BUCK4_DVS_DONE_MASK BIT(3)
+#define BD73800_INT_BUCK5_DVS_DONE_MASK BIT(4)
+#define BD73800_INT_BUCK6_DVS_DONE_MASK BIT(5)
+#define BD73800_INT_BUCK7_DVS_DONE_MASK BIT(6)
+#define BD73800_INT_BUCK8_DVS_DONE_MASK BIT(7)
+#define BD73800_INT_BUCK1_OCP_MASK BIT(0)
+#define BD73800_INT_BUCK2_OCP_MASK BIT(1)
+#define BD73800_INT_BUCK3_OCP_MASK BIT(2)
+#define BD73800_INT_BUCK4_OCP_MASK BIT(3)
+#define BD73800_INT_BUCK5_OCP_MASK BIT(4)
+#define BD73800_INT_BUCK6_OCP_MASK BIT(5)
+#define BD73800_INT_BUCK7_OCP_MASK BIT(6)
+#define BD73800_INT_BUCK8_OCP_MASK BIT(7)
+#define BD73800_INT_PBTN_LONG_PRESS_MASK BIT(0)
+#define BD73800_INT_PBTN_MID_PRESS_MASK BIT(1)
+#define BD73800_INT_PBTN_SHORT_PUSH_MASK BIT(2)
+#define BD73800_INT_PBTN_SHORT_PRESS_MASK BIT(3)
+#define BD73800_INT_WDG_MASK BIT(4)
+#define BD73800_INT_SWRESET_MASK BIT(5)
+#define BD73800_INT_SEQ_DONE_MASK BIT(6)
+#define BD73800_INT_GPIO1_MASK BIT(0)
+#define BD73800_INT_GPIO2_MASK BIT(1)
+#define BD73800_INT_GPIO3_MASK BIT(2)
+#define BD73800_INT_GPIO4_MASK BIT(3)
+
+#endif /* _MFD_BD73800_H */
diff --git a/include/linux/mfd/rohm-generic.h b/include/linux/mfd/rohm-generic.h
index 0a284919a6c3..3ec87428ee97 100644
--- a/include/linux/mfd/rohm-generic.h
+++ b/include/linux/mfd/rohm-generic.h
@@ -17,6 +17,7 @@ enum rohm_chip_type {
ROHM_CHIP_TYPE_BD71837,
ROHM_CHIP_TYPE_BD71847,
ROHM_CHIP_TYPE_BD72720,
+ ROHM_CHIP_TYPE_BD73800,
ROHM_CHIP_TYPE_BD96801,
ROHM_CHIP_TYPE_BD96802,
ROHM_CHIP_TYPE_BD96805,
diff --git a/include/linux/micrel_phy.h b/include/linux/micrel_phy.h
index 9c6f9817383f..bb71b2510c2c 100644
--- a/include/linux/micrel_phy.h
+++ b/include/linux/micrel_phy.h
@@ -47,11 +47,6 @@
#define MICREL_PHY_FXEN BIT(1)
#define MICREL_KSZ8_P1_ERRATA BIT(2)
-#define MICREL_KSZ9021_EXTREG_CTRL 0xB
-#define MICREL_KSZ9021_EXTREG_DATA_WRITE 0xC
-#define MICREL_KSZ9021_RGMII_CLK_CTRL_PAD_SCEW 0x104
-#define MICREL_KSZ9021_RGMII_RX_DATA_PAD_SCEW 0x105
-
/* Device specific MII_BMCR (Reg 0) bits */
/* 1 = HP Auto MDI/MDI-X mode, 0 = Microchip Auto MDI/MDI-X mode */
#define KSZ886X_BMCR_HP_MDIX BIT(5)
diff --git a/include/linux/minmax.h b/include/linux/minmax.h
index a0158db54a04..5ef4d58c0c42 100644
--- a/include/linux/minmax.h
+++ b/include/linux/minmax.h
@@ -38,9 +38,9 @@
* Note that 'x' is the original expression, and 'ux' is the unique variable
* that contains the value.
*
- * We use 'ux' for pure type checking, and 'x' for when we need to look at the
- * value (but without evaluating it for side effects!
- * Careful to only ever evaluate it with sizeof() or __builtin_constant_p() etc).
+ * We use 'ux' for both the type and the value checks, so 'x' itself is only
+ * expanded twice: once to initialise 'ux', and once quoted in the error
+ * message.
*
* Pointers end up being checked by the normal C type rules at the actual
* comparison, and these expressions only need to be careful to not cause
diff --git a/include/linux/mlx5/eswitch.h b/include/linux/mlx5/eswitch.h
index a0dd162baa78..03d3620141c8 100644
--- a/include/linux/mlx5/eswitch.h
+++ b/include/linux/mlx5/eswitch.h
@@ -222,6 +222,9 @@ static inline bool is_mdev_switchdev_mode(struct mlx5_core_dev *dev)
/* The returned number is valid only when the dev is eswitch manager. */
static inline u16 mlx5_eswitch_manager_vport(struct mlx5_core_dev *dev)
{
+ if (MLX5_CAP_ESW(dev, esw_manager_vport_number_valid))
+ return MLX5_CAP_ESW(dev, esw_manager_vport_number);
+
return mlx5_core_is_ecpf_esw_manager(dev) ?
MLX5_VPORT_ECPF : MLX5_VPORT_HOST_PF;
}
diff --git a/include/linux/mm.h b/include/linux/mm.h
index dd09c438fa23..3d135b09feeb 100644
--- a/include/linux/mm.h
+++ b/include/linux/mm.h
@@ -38,6 +38,7 @@
#include <linux/bitops.h>
#include <linux/iommu-debug-pagealloc.h>
#include <linux/kcsan-checks.h>
+#include <linux/vmemmap-optimization.h>
struct mempolicy;
struct anon_vma;
@@ -57,16 +58,6 @@ static inline unsigned long totalram_pages(void)
return (unsigned long)atomic_long_read(&_totalram_pages);
}
-static inline void totalram_pages_inc(void)
-{
- atomic_long_inc(&_totalram_pages);
-}
-
-static inline void totalram_pages_dec(void)
-{
- atomic_long_dec(&_totalram_pages);
-}
-
static inline void totalram_pages_add(long count)
{
atomic_long_add(count, &_totalram_pages);
@@ -587,14 +578,6 @@ enum {
#define VMA_ACCESS_FLAGS mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, VMA_EXEC_BIT)
/*
- * Special vmas that are non-mergable, non-mlock()able.
- */
-
-#define VMA_SPECIAL_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_DONTEXPAND_BIT, \
- VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT)
-#define VM_SPECIAL vma_flags_to_legacy(VMA_SPECIAL_FLAGS)
-
-/*
* Physically remapped pages are special. Tell the
* rest of the world about it:
* IO tells people not to look at these pages
@@ -610,9 +593,6 @@ enum {
#define VMA_REMAP_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_PFNMAP_BIT, \
VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT)
-/* This mask prevents VMA from being scanned with khugepaged */
-#define VM_NO_KHUGEPAGED (VM_SPECIAL | VM_HUGETLB)
-
/* This mask defines which mm->def_flags a process can inherit its parent */
#define VM_INIT_DEF_MASK VM_NOHUGEPAGE
@@ -928,7 +908,6 @@ static inline void vma_numab_state_free(struct vm_area_struct *vma) {}
* These must be here rather than mmap_lock.h as dependent on vm_fault type,
* declared in this header.
*/
-#ifdef CONFIG_PER_VMA_LOCK
static inline void release_fault_lock(struct vm_fault *vmf)
{
if (vmf->flags & FAULT_FLAG_VMA_LOCK)
@@ -944,17 +923,6 @@ static inline void assert_fault_locked(const struct vm_fault *vmf)
else
mmap_assert_locked(vmf->vma->vm_mm);
}
-#else
-static inline void release_fault_lock(struct vm_fault *vmf)
-{
- mmap_read_unlock(vmf->vma->vm_mm);
-}
-
-static inline void assert_fault_locked(const struct vm_fault *vmf)
-{
- mmap_assert_locked(vmf->vma->vm_mm);
-}
-#endif /* CONFIG_PER_VMA_LOCK */
static inline bool mm_flags_test(int flag, const struct mm_struct *mm)
{
@@ -1551,11 +1519,6 @@ static inline void vma_set_anonymous(struct vm_area_struct *vma)
vma->vm_ops = NULL;
}
-static inline void vma_desc_set_anonymous(struct vm_area_desc *desc)
-{
- desc->vm_ops = NULL;
-}
-
static inline bool vma_is_anonymous(const struct vm_area_struct *vma)
{
return !vma->vm_ops;
@@ -1640,6 +1603,211 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma)
}
/**
+ * vma_flags_is_hugetlb() - Do the specified VMA flags indicate that the
+ * VMA is a hugetlb mapping?
+ * @flags: The VMA flags to test.
+ *
+ * Returns: true if the flags indicate a hugetlb mapping, false otherwise.
+ */
+static inline bool vma_flags_is_hugetlb(const vma_flags_t *flags)
+{
+ return IS_ENABLED(CONFIG_HUGETLB_PAGE) &&
+ vma_flags_test(flags, VMA_HUGETLB_BIT);
+}
+
+/**
+ * vma_is_hugetlb() - Is @vma a hugetlb mapping?
+ * @vma: The VMA to test.
+ *
+ * Returns: true if @vma is a hugetlb mapping, false otherwise.
+ */
+static inline bool vma_is_hugetlb(const struct vm_area_struct *vma)
+{
+ return vma_flags_is_hugetlb(&vma->flags);
+}
+
+/**
+ * vma_flags_is_kernel_owned() - Do the specified VMA flags indicate that the
+ * contents of the VMA are owned by the kernel rather than the core mm?
+ * @flags: The VMA flags to test.
+ *
+ * A kernel-owned mapping is one whose contents are established and controlled
+ * by the kernel, typically a driver, rather than by the core mm's fault and
+ * rmap machinery.
+ *
+ * The mapping may be memory-mapped I/O, kernel-allocated pages or ordinary
+ * pages the owner has chosen to map itself (shmem via a PFN map, for instance).
+ *
+ * In all cases the core mm must not populate, reclaim, migrate, copy-on-write
+ * or merge it of its own accord.
+ *
+ * Pages mapped this way are not necessarily reference counted or map counted.
+ *
+ * Returns: true if the flags indicate a kernel-owned mapping.
+ */
+static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags)
+{
+ return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT);
+}
+
+/**
+ * vma_is_kernel_owned() - Are the contents of @vma owned by the kernel?
+ * @vma: The VMA to test.
+ *
+ * See vma_flags_is_kernel_owned() for a description of this property.
+ *
+ * Returns: true if the VMA is kernel-owned.
+ */
+static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma)
+{
+ return vma_flags_is_kernel_owned(&vma->flags);
+}
+
+/**
+ * vma_flags_is_fixed_mapping() - Do the specified VMA flags indicate that this
+ * is a fixed mapping that cannot be expanded or merged?
+ * @flags: The VMA flags to test.
+ *
+ * Fixed mappings are those whose size is set at the point of mmap (for
+ * instance, a kernel-owned mapping of a fixed range of memory), and thus
+ * cannot be expanded or merged.
+ *
+ * Returns: true if the flags indicate a fixed mapping.
+ */
+static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags)
+{
+ /*
+ * VMA_PFNMAP_BIT should imply VMA_DONTEXPAND_BIT, but some callers set
+ * only the former.
+ */
+ return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_DONTEXPAND_BIT);
+}
+
+/**
+ * vma_is_fixed_mapping() - Is this VMA a fixed mapping that cannot be
+ * expanded or merged?
+ * @vma: The VMA to test.
+ *
+ * See vma_flags_is_fixed_mapping() for a description of this property.
+ *
+ * Returns: true if the VMA maps a fixed mapping.
+ */
+static inline bool vma_is_fixed_mapping(const struct vm_area_struct *vma)
+{
+ return vma_flags_is_fixed_mapping(&vma->flags);
+}
+
+/**
+ * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be
+ * merged with another?
+ * @flags: The VMA flags to test.
+ * Returns: true if the flags permit merging, false otherwise.
+ */
+static inline bool vma_flags_can_merge(const vma_flags_t *flags)
+{
+ /*
+ * VMA merging assumes that a VMA's flags and fields completely describe
+ * its state.
+ *
+ * However, kernel-owned mappings may have established state upon mapping
+ * not embodied in any attribute of the VMA.
+ *
+ * Additionally, private (CoW) PFN maps encode the source PFN of the
+ * range in vma->vm_pgoff, which may otherwise cause spurious merges.
+ */
+ if (vma_flags_is_kernel_owned(flags))
+ return false;
+ /* VMA explicitly marked as being unmergeable. */
+ if (vma_flags_is_fixed_mapping(flags))
+ return false;
+
+ return true;
+}
+
+/**
+ * vma_can_merge() - Do @vma's flags permit it to be merged with another VMA?
+ * @vma: The VMA to test.
+ * Returns: true if the flags permit merging, otherwise false.
+ */
+static inline bool vma_can_merge(const struct vm_area_struct *vma)
+{
+ return vma_flags_can_merge(&vma->flags);
+}
+
+/**
+ * vma_flags_is_persistent() - Do the specified VMA flags imply that the VMA
+ * contains persistent data?
+ * @flags: The VMA flags to test.
+ *
+ * Persistent in the sense that - if you write bytes to the mapping - do they
+ * stay written?
+ *
+ * If the kernel or a device could write to the memory independently of
+ * userland, or the kernel could arbitrarily discard it, then it is not
+ * persistent.
+ *
+ * Returns: true if the flags imply this VMA is persistent, otherwise false.
+ */
+static inline bool vma_flags_is_persistent(const vma_flags_t *flags)
+{
+ /* hugetlb is a fixed mapping, but its contents are the user's own. */
+ if (vma_flags_is_hugetlb(flags))
+ return true;
+ /*
+ * MMIO mappings may not store what is written and may be changed by the
+ * device. Kernel-owned and fixed mappings may be changed by their owner
+ * without the user having initiated it.
+ */
+ if (vma_flags_is_kernel_owned(flags) ||
+ vma_flags_is_fixed_mapping(flags))
+ return false;
+ /* Droppable memory is discardable by definition. */
+ return !vma_flags_test_single_mask(flags, VMA_DROPPABLE);
+}
+
+/**
+ * vma_is_persistent() - Does the VMA contain persistent data?
+ * @vma: The VMA to test.
+ *
+ * See vma_flags_is_persistent() for details.
+ *
+ * Returns: true if the VMA is persistent, otherwise false.
+ */
+static inline bool vma_is_persistent(const struct vm_area_struct *vma)
+{
+ return vma_flags_is_persistent(&vma->flags);
+}
+
+/**
+ * vma_flags_can_gup() - Do the specified VMA flags permit GUP to access the
+ * mapping's pages?
+ * @flags: The VMA flags to test.
+ *
+ * GUP cannot obtain pages from a PFN map (VMA_PFNMAP_BIT), which may have no
+ * struct pages behind it, and must not provide access to memory-mapped I/O
+ * (VMA_IO_BIT).
+ *
+ * Returns: true if GUP may access pages from the mapping, otherwise false.
+ */
+static inline bool vma_flags_can_gup(const vma_flags_t *flags)
+{
+ return !vma_flags_test_any(flags, VMA_IO_BIT, VMA_PFNMAP_BIT);
+}
+
+/**
+ * vma_can_gup() - May GUP obtain pages from @vma?
+ * @vma: The VMA to test.
+ *
+ * See vma_flags_can_gup() for details.
+ *
+ * Returns: true if GUP may access pages from the mapping, otherwise false.
+ */
+static inline bool vma_can_gup(const struct vm_area_struct *vma)
+{
+ return vma_flags_can_gup(&vma->flags);
+}
+
+/**
* vma_kernel_pagesize - Default page size granularity for this VMA.
* @vma: The user mapping.
*
@@ -2075,20 +2243,21 @@ vm_fault_t finish_fault(struct vm_fault *vmf);
*
* A pagecache page contains an opaque `private' member, which belongs to the
* page's address_space. Usually, this is the address of a circular list of
- * the page's disk buffers. PG_private must be set to tell the VM to call
- * into the filesystem to release these pages.
+ * the page's disk buffers. It tells the VM to call into the filesystem to
+ * release these pages.
*
* A folio may belong to an inode's memory mapping. In this case,
* folio->mapping points to the inode, and folio->index is the file
* offset of the folio, in units of PAGE_SIZE.
*
- * If pagecache pages are not associated with an inode, they are said to be
- * anonymous pages. These may become associated with the swapcache, and in that
- * case PG_swapcache is set, and page->private is an offset into the swapcache.
+ * If pagecache folios are not associated with an inode, they are said to be
+ * anonymous folios. These may become associated with the swapcache, and in that
+ * case PG_swapcache is set, and folio->private is an offset into the swapcache.
*
* In either case (swapcache or inode backed), the pagecache itself holds one
- * reference to the page. Setting PG_private should also increment the
- * refcount. The each user mapping also has a reference to the page.
+ * reference to the folio. Attaching filesystem private data via
+ * folio_attach_private() also increments the refcount. Each user mapping also
+ * has a reference to the folio.
*
* The pagecache pages are stored in a per-mapping radix tree, which is
* rooted at mapping->i_pages, and indexed by offset.
@@ -2644,12 +2813,23 @@ static inline void set_page_section(struct page *page, unsigned long section)
page->flags.f |= (section & SECTIONS_MASK) << SECTIONS_PGSHIFT;
}
+static inline void set_page_section_from_pfn(struct page *page,
+ unsigned long pfn)
+{
+ set_page_section(page, pfn_to_section_nr(pfn));
+}
+
static inline unsigned long memdesc_section(const memdesc_flags_t *mdf)
{
ASSERT_EXCLUSIVE_BITS(mdf->f, SECTIONS_MASK << SECTIONS_PGSHIFT);
return (mdf->f >> SECTIONS_PGSHIFT) & SECTIONS_MASK;
}
#else /* !SECTION_IN_PAGE_FLAGS */
+static inline void set_page_section_from_pfn(struct page *page,
+ unsigned long pfn)
+{
+}
+
static inline unsigned long memdesc_section(const memdesc_flags_t *mdf)
{
return 0;
@@ -2872,9 +3052,7 @@ static inline void set_page_links(struct page *page, enum zone_type zone,
{
set_page_zone(page, zone);
set_page_node(page, node);
-#ifdef SECTION_IN_PAGE_FLAGS
- set_page_section(page, pfn_to_section_nr(pfn));
-#endif
+ set_page_section_from_pfn(page, pfn);
}
/**
@@ -3022,9 +3200,9 @@ static inline bool folio_maybe_mapped_shared(struct folio *folio)
* @folio: the folio
*
* Calculate the expected folio refcount, taking references from the pagecache,
- * swapcache, PG_private and page table mappings into account. Useful in
- * combination with folio_ref_count() to detect unexpected references (e.g.,
- * GUP or other temporary references).
+ * swapcache, private data (folio->private != NULL) and page table mappings into
+ * account. Useful in combination with folio_ref_count() to detect unexpected
+ * references (e.g., GUP or other temporary references).
*
* Does currently not consider references from the LRU cache. If the folio
* was isolated from the LRU (which is the case during migration or split),
@@ -3062,10 +3240,15 @@ static inline int folio_expected_ref_count(const struct folio *folio)
ref_count += folio_test_swapcache(folio) << order;
if (!folio_test_anon(folio)) {
- /* One reference per page from the pagecache. */
- ref_count += !!folio->mapping << order;
- /* One reference from PG_private. */
- ref_count += folio_test_private(folio);
+ /*
+ * One reference per page from the pagecache.
+ * Use data_race() since folio might not be locked.
+ */
+ ref_count += !!data_race(folio->mapping) << order;
+ /*
+ * One reference from filesystem private data.
+ */
+ ref_count += folio_has_attached_private(folio);
}
/* One reference per page table mapping. */
@@ -3325,10 +3508,10 @@ extern int access_process_vm(struct task_struct *tsk, unsigned long addr,
extern int access_remote_vm(struct mm_struct *mm, unsigned long addr,
void *buf, int len, unsigned int gup_flags);
-#ifdef CONFIG_BPF_SYSCALL
-extern int copy_remote_vm_str(struct task_struct *tsk, unsigned long addr,
- void *buf, int len, unsigned int gup_flags);
-#endif
+int copy_remote_mm_str(struct mm_struct *mm, unsigned long addr,
+ void *buf, int len, unsigned int gup_flags);
+int copy_remote_vm_str(struct task_struct *tsk, unsigned long addr,
+ void *buf, int len, unsigned int gup_flags);
long get_user_pages_remote(struct mm_struct *mm,
unsigned long start, unsigned long nr_pages,
@@ -4083,12 +4266,6 @@ static inline void free_reserved_page(struct page *page)
free_reserved_pages(page, 0);
}
-static inline void mark_page_reserved(struct page *page)
-{
- SetPageReserved(page);
- adjust_managed_page_count(page, -1);
-}
-
static inline void free_reserved_ptdesc(struct ptdesc *pt)
{
free_reserved_page(ptdesc_page(pt));
@@ -4181,6 +4358,12 @@ void mapping_rmap_tree_insert_after(struct vm_area_struct *vma,
struct address_space *mapping);
void mapping_rmap_tree_remove(struct vm_area_struct *vma,
struct address_space *mapping);
+void mapping_rmap_tree_pre_update(struct vm_area_struct *vma,
+ struct address_space *mapping,
+ bool pgoff_unchanged);
+void mapping_rmap_tree_post_update(struct vm_area_struct *vma,
+ struct address_space *mapping,
+ bool pgoff_unchanged);
struct vm_area_struct *
mapping_rmap_tree_iter_first(struct address_space *mapping,
pgoff_t pgoff_start, pgoff_t pgoff_last);
@@ -4198,6 +4381,10 @@ void anon_rmap_tree_insert(struct anon_vma_chain *avc,
struct anon_vma *anon_vma);
void anon_rmap_tree_remove(struct anon_vma_chain *avc,
struct anon_vma *anon_vma);
+void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma,
+ bool anon_pgoff_unchanged);
+void anon_rmap_tree_post_update_vma(struct vm_area_struct *vma,
+ bool anon_pgoff_unchanged);
struct anon_vma_chain *
anon_rmap_tree_iter_first(struct anon_vma *anon_vma,
pgoff_t pgoff_start, pgoff_t pgoff_last);
@@ -4408,9 +4595,8 @@ static inline unsigned long vma_pages(const struct vm_area_struct *vma)
* If @vma is a MAP_PRIVATE file-backed mapping, then this returns the
* page offset within the file.
*
- * Edge cases: nommu does not abide by these, MAP_PRIVATE-/dev/zero satisfies
- * vma_is_anonymous() but has file-backed page offset, and MAP_PRIVATE-pfnmap
- * regions have their page offset set to the first PFN in the range.
+ * Edge cases: nommu does not abide by these and CoW MAP_PRIVATE-pfnmap regions
+ * have their page offset set to the first PFN in the range.
*
* Returns: The page offset of the start of @vma.
*/
@@ -4619,7 +4805,7 @@ static inline void mmap_action_simple_ioremap(struct vm_area_desc *desc,
* @desc: The VMA descriptor for the VMA requiring kernel pags to be mapped.
* @start: The virtual address from which to map them.
* @pages: An array of struct page pointers describing the memory to map.
- * @nr_pages: The number of entries in the @pages aray.
+ * @nr_pages: The number of entries in the @pages array.
*/
static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc,
unsigned long start, struct page **pages,
@@ -4627,7 +4813,7 @@ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc,
{
struct mmap_action *action = &desc->action;
- action->type = MMAP_MAP_KERNEL_PAGES;
+ action->type = MMAP_KERNEL_PAGES;
action->map_kernel.start = start;
action->map_kernel.pages = pages;
action->map_kernel.nr_pages = nr_pages;
@@ -4651,10 +4837,55 @@ static inline void mmap_action_map_kernel_pages_full(struct vm_area_desc *desc,
vma_desc_pages(desc));
}
+static inline
+void mmap_action_map_discontig_kernel_pages(struct vm_area_desc *desc,
+ void *init_private, const struct discontig_kernel_page_ops *ops)
+{
+ struct mmap_action *action = &desc->action;
+
+ action->type = MMAP_DISCONTIG_KERNEL_PAGES;
+ action->map_kernel_discontig.init_private = init_private;
+ action->map_kernel_discontig.ops = ops;
+}
+
int mmap_action_prepare(struct vm_area_desc *desc);
int mmap_action_complete(struct vm_area_struct *vma,
struct mmap_action *action, bool is_compat);
+static inline void
+discontig_kernel_map_abort(struct discontig_kernel_page_state *state)
+{
+ state->action = DISCONTIG_KERNEL_PAGE_ABORT;
+}
+
+static inline void
+discontig_kernel_map_page(struct discontig_kernel_page_state *state,
+ struct page *page)
+{
+ struct folio *folio = page_folio(page);
+
+ if (folio_test_large(folio)) {
+ VM_WARN_ON_ONCE(page != folio_page(folio, 0));
+ state->action = DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE;
+ state->__folio = folio;
+ state->__nr_pages = min(state->nr_pages_remain,
+ folio_nr_pages(folio));
+ } else {
+ state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE;
+ state->__page = page;
+ state->__nr_pages = 1;
+ }
+}
+
+static inline void
+discontig_kernel_map_page_range(struct discontig_kernel_page_state *state,
+ struct page **page_arr, unsigned long nr_pages)
+{
+ state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE;
+ state->__page_arr = page_arr;
+ state->__nr_pages = nr_pages;
+}
+
/* Look up the first VMA which exactly match the interval vm_start ... vm_end */
static inline struct vm_area_struct *find_exact_vma(struct mm_struct *mm,
unsigned long vm_start, unsigned long vm_end)
@@ -4772,9 +5003,6 @@ int remap_pfn_range(struct vm_area_struct *vma, unsigned long addr,
int vm_insert_page(struct vm_area_struct *, unsigned long addr, struct page *);
int vm_insert_pages(struct vm_area_struct *vma, unsigned long addr,
struct page **pages, unsigned long *num);
-int map_kernel_pages_prepare(struct vm_area_desc *desc);
-int map_kernel_pages_complete(struct vm_area_struct *vma,
- struct mmap_action *action);
int vm_map_pages(struct vm_area_struct *vma, struct page **pages,
unsigned long num);
int vm_map_pages_zero(struct vm_area_struct *vma, struct page **pages,
@@ -4787,8 +5015,6 @@ vm_fault_t vmf_insert_pfn_prot(struct vm_area_struct *vma, unsigned long addr,
unsigned long pfn, pgprot_t pgprot);
vm_fault_t vmf_insert_mixed(struct vm_area_struct *vma, unsigned long addr,
unsigned long pfn);
-vm_fault_t vmf_insert_mixed_mkwrite(struct vm_area_struct *vma,
- unsigned long addr, unsigned long pfn);
int vm_iomap_memory(struct vm_area_struct *vma, phys_addr_t start, unsigned long len);
static inline vm_fault_t vmf_insert_page(struct vm_area_struct *vma,
@@ -5140,7 +5366,6 @@ static inline void print_vma_addr(char *prefix, unsigned long rip)
}
#endif
-unsigned long section_map_size(void);
struct page * __populate_section_memmap(unsigned long pfn,
unsigned long nr_pages, int nid, struct vmem_altmap *altmap,
struct dev_pagemap *pgmap);
@@ -5159,9 +5384,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end,
int node, struct vmem_altmap *altmap);
int vmemmap_populate(unsigned long start, unsigned long end, int node,
struct vmem_altmap *altmap);
-int vmemmap_populate_hvo(unsigned long start, unsigned long end,
- unsigned int order, struct zone *zone,
- unsigned long headsize);
void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node,
unsigned long headsize);
void vmemmap_populate_print_last(void);
@@ -5196,7 +5418,6 @@ static inline void vmem_altmap_free(struct vmem_altmap *altmap,
}
#endif
-#define VMEMMAP_RESERVE_NR 2
#ifdef CONFIG_ARCH_WANT_OPTIMIZE_DAX_VMEMMAP
static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap,
struct dev_pagemap *pgmap)
@@ -5204,6 +5425,9 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap,
unsigned long nr_pages;
unsigned long nr_vmemmap_pages;
+ if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION))
+ return false;
+
if (!pgmap || !is_power_of_2(sizeof(struct page)))
return false;
@@ -5213,7 +5437,7 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap,
* For vmemmap optimization with DAX we need minimum 2 vmemmap
* pages. See layout diagram in Documentation/mm/vmemmap_dedup.rst
*/
- return !altmap && (nr_vmemmap_pages > VMEMMAP_RESERVE_NR);
+ return !altmap && (nr_vmemmap_pages > VMEMMAP_OPTIMIZATION_PAGES);
}
/*
* If we don't have an architecture override, use the generic rule
@@ -5254,6 +5478,8 @@ extern const struct attribute_group memory_failure_attr_group;
extern void memory_failure_queue(unsigned long pfn, int flags);
void num_poisoned_pages_inc(unsigned long pfn);
void num_poisoned_pages_sub(unsigned long pfn, long i);
+phys_addr_t range_first_hwpoison(phys_addr_t start, unsigned long size);
+phys_addr_t range_last_hwpoison(phys_addr_t start, unsigned long size);
#else
static inline void memory_failure_queue(unsigned long pfn, int flags)
{
@@ -5266,6 +5492,18 @@ static inline void num_poisoned_pages_inc(unsigned long pfn)
static inline void num_poisoned_pages_sub(unsigned long pfn, long i)
{
}
+
+static inline phys_addr_t range_first_hwpoison(phys_addr_t start,
+ unsigned long size)
+{
+ return PHYS_ADDR_MAX;
+}
+
+static inline phys_addr_t range_last_hwpoison(phys_addr_t start,
+ unsigned long size)
+{
+ return PHYS_ADDR_MAX;
+}
#endif
#if defined(CONFIG_MEMORY_FAILURE) && defined(CONFIG_MEMORY_HOTPLUG)
diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h
index 621c8653d8f7..ab69b9930893 100644
--- a/include/linux/mm_inline.h
+++ b/include/linux/mm_inline.h
@@ -30,6 +30,17 @@ static inline int folio_is_file_lru(const struct folio *folio)
return !folio_test_swapbacked(folio);
}
+/**
+ * folio_flags_is_file_lru - Should the folio be on a file LRU or anon LRU?
+ * @flags: The folio's flags.
+ *
+ * Just like folio_is_file_lru but take the folio flags directly instead.
+ */
+static inline int folio_flags_is_file_lru(const unsigned long *flags)
+{
+ return !test_bit(PG_swapbacked, flags);
+}
+
static __always_inline void __update_lru_size(struct lruvec *lruvec,
enum lru_list lru, enum zone_type zid,
long nr_pages)
@@ -142,10 +153,66 @@ static inline int lru_tier_from_refs(int refs, bool workingset)
return workingset ? MAX_NR_TIERS - 1 : order_base_2(refs);
}
-static inline int folio_lru_refs(const struct folio *folio)
+/**
+ * lru_set_gen_flags - Set the LRU generation number to specified folio flags.
+ * @flags: pointer to the folio flags
+ * @gen: generation number, between 0 and (MAX_NR_GENS - 1), inclusive.
+ */
+static inline void lru_set_gen_flags(unsigned long *flags, int gen)
+{
+ BUILD_BUG_ON(LRU_GEN_MASK & LRU_REFS_MASK);
+ VM_WARN_ON_ONCE(gen >= MAX_NR_GENS || gen < 0);
+ /* Store gen offset by 1, zero means the folio is off-list. */
+ *flags &= ~LRU_GEN_MASK;
+ *flags |= (gen + 1UL) << LRU_GEN_PGOFF;
+}
+
+/**
+ * lru_get_gen_flags - Return the LRU generation number from folio flags.
+ * @flags: folio flags
+ *
+ * Returns: A number between 0 and (MAX_NR_GENS - 1), inclusive. Returns
+ * -1 if the flags indicate the folio is off the list (e.g., isolated).
+ */
+static inline int lru_get_gen_flags(unsigned long flags)
+{
+ int gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
+
+ /* Exclude the legal -1 from the unsigned MAX_NR_GENS comparison */
+ VM_WARN_ON_ONCE(gen != -1 && gen >= MAX_NR_GENS);
+ return gen;
+}
+
+/**
+ * lru_set_refs_flags - Set the LRU referenced count to folio flags.
+ * @flags: pointer to the folio flags
+ * @refs: referenced / access count number, between 0 and LRU_REFS_MAX, inclusive.
+ *
+ * For MGLRU, PG_referenced holds the first ref, and the extra bits hold the
+ * remaining refs. For classical LRU the extra bits are not used, so it can
+ * also be seen as the refs count never exceeds 1. In both cases, refs == 1
+ * means PG_referenced is set and the extra bits are zero, and refs == 0 means
+ * PG_referenced and the extra bits are all unset.
+ */
+static inline void lru_set_refs_flags(unsigned long *flags, unsigned int refs)
{
- unsigned long flags = READ_ONCE(folio->flags.f);
+ VM_WARN_ON_ONCE(refs > LRU_REFS_MAX);
+ BUILD_BUG_ON(LRU_REFS_MAX != (LRU_REFS_MASK >> LRU_REFS_PGOFF) + 1);
+ *flags &= ~LRU_REFS_FLAGS;
+ if (!refs)
+ return;
+ *flags |= (BIT(PG_referenced) | ((refs - 1UL) << LRU_REFS_PGOFF));
+}
+
+/**
+ * lru_get_refs_flags - Return LRU referenced / access count from folio flags.
+ * @flags: folio flags
+ *
+ * Reads the LRU referenced count set by lru_set_refs_flags().
+ */
+static inline int lru_get_refs_flags(unsigned long flags)
+{
if (!(flags & BIT(PG_referenced)))
return 0;
/*
@@ -155,11 +222,24 @@ static inline int folio_lru_refs(const struct folio *folio)
return ((flags & LRU_REFS_MASK) >> LRU_REFS_PGOFF) + 1;
}
-static inline int folio_lru_gen(const struct folio *folio)
+static inline int folio_lru_refs(const struct folio *folio)
{
- unsigned long flags = READ_ONCE(folio->flags.f);
+ return lru_get_refs_flags(READ_ONCE(*const_folio_flags(folio, 0)));
+}
- return ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
+static inline void folio_set_lru_refs(struct folio *folio, unsigned int refs)
+{
+ unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0));
+
+ do {
+ new_flags = old_flags;
+ lru_set_refs_flags(&new_flags, refs);
+ } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags));
+}
+
+static inline int folio_lru_gen(const struct folio *folio)
+{
+ return lru_get_gen_flags(READ_ONCE(*const_folio_flags(folio, 0)));
}
static inline bool lru_gen_is_active(const struct lruvec *lruvec, int gen)
@@ -270,7 +350,7 @@ static inline bool lru_gen_add_folio(struct lruvec *lruvec, struct folio *folio,
gen = lru_gen_from_seq(seq);
flags = (gen + 1UL) << LRU_GEN_PGOFF;
/* see the comment on MIN_NR_GENS about PG_active */
- set_mask_bits(&folio->flags.f, LRU_GEN_MASK | BIT(PG_active), flags);
+ set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK | BIT(PG_active), flags);
lru_gen_update_size(lruvec, folio, -1, gen);
/* for folio_rotate_reclaimable() */
@@ -295,7 +375,7 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
/* for folio_migrate_flags() */
flags = !reclaiming && lru_gen_is_active(lruvec, gen) ? BIT(PG_active) : 0;
- flags = set_mask_bits(&folio->flags.f, LRU_GEN_MASK, flags);
+ flags = set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK, flags);
gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
lru_gen_update_size(lruvec, folio, gen, -1);
@@ -304,11 +384,19 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
return true;
}
-static inline void folio_migrate_refs(struct folio *new, const struct folio *old)
+/**
+ * folio_migrate_lru_refs - copy the reference state to a new folio
+ * @new: the destination folio
+ * @old: the source folio
+ *
+ * Transfer the reference state to @new during migration: the MGLRU
+ * refs count, including PG_referenced, or just PG_referenced for the
+ * active/inactive LRU.
+ */
+static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old)
{
- unsigned long refs = READ_ONCE(old->flags.f) & LRU_REFS_MASK;
-
- set_mask_bits(&new->flags.f, LRU_REFS_MASK, refs);
+ BUILD_BUG_ON(LRU_REFS_MASK & BIT(PG_referenced));
+ folio_set_lru_refs(new, folio_lru_refs(old));
}
#else /* !CONFIG_LRU_GEN */
@@ -337,9 +425,10 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
return false;
}
-static inline void folio_migrate_refs(struct folio *new, const struct folio *old)
+static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old)
{
-
+ if (folio_test_referenced(old))
+ folio_set_referenced(new);
}
#endif /* CONFIG_LRU_GEN */
diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h
index 6d815f6440c9..6141160ec652 100644
--- a/include/linux/mm_types.h
+++ b/include/linux/mm_types.h
@@ -108,7 +108,7 @@ struct page {
};
/**
* @private: Mapping-private opaque data.
- * Usually used for buffer_heads if PagePrivate.
+ * Usually used for buffer_heads.
* Used for swp_entry_t if swapcache flag set.
* Indicates order in the buddy system if PageBuddy
* or on pcp_llist.
@@ -675,7 +675,7 @@ static inline void ptdesc_pmd_pts_init(struct ptdesc *ptdesc)
#define STRUCT_PAGE_MAX_SHIFT (order_base_2(sizeof(struct page)))
/*
- * page_private can be used on tail pages. However, PagePrivate is only
+ * page_private can be used on tail pages. However, it is only
* checked by the VM on the head page. So page_private on the tail pages
* should be used for data that's ancillary to the head page (eg attaching
* buffer heads to tail pages after attaching buffer heads to the head page)
@@ -759,7 +759,7 @@ static inline struct anon_vma_name *anon_vma_name_alloc(const char *name)
/*
* While __vma_enter_locked() is working to ensure are no read-locks held on a
* VMA (either while acquiring a VMA write lock or marking a VMA detached) we
- * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indiciate to
+ * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indicate to
* vma_start_read() that the reference count should be left alone.
*
* See the comment describing vm_refcnt in vm_area_struct for details as to
@@ -815,11 +815,47 @@ struct pfnmap_track_ctx {
/* What action should be taken after an .mmap_prepare call is complete? */
enum mmap_action_type {
- MMAP_NOTHING, /* Mapping is complete, no further action. */
- MMAP_REMAP_PFN, /* Remap PFN range. */
- MMAP_IO_REMAP_PFN, /* I/O remap PFN range. */
- MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */
- MMAP_MAP_KERNEL_PAGES, /* Map kernel page range from array. */
+ MMAP_NOTHING,
+ MMAP_REMAP_PFN,
+ MMAP_IO_REMAP_PFN,
+ MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */
+ MMAP_KERNEL_PAGES, /* Map kernel page range from array. */
+ MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */
+};
+
+enum discontig_kernel_page_action {
+ DISCONTIG_KERNEL_PAGE_ABORT,
+ DISCONTIG_KERNEL_PAGE_MAP_PAGE,
+ DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE,
+ DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE,
+};
+
+struct discontig_kernel_page_state {
+ /* Map state. */
+ const unsigned long start; /* Start address of VMA. */
+ const unsigned long end; /* End address of VMA. */
+ unsigned long addr; /* The current address to be mapped. */
+ pgoff_t pgoff; /* The current pgoff to be mapped. */
+ unsigned long nr_pages_mapped; /* The number of pages mapped. */
+ unsigned long nr_pages_remain; /* The number of pages remaining. */
+
+ /* User-defined state. */
+ void *vm_private_data; /* VMA private data. */
+ void *private; /* Mapping private data. */
+
+ /* Users should not touch these, use discontig_kernel_map_*() helpers. */
+ enum discontig_kernel_page_action action;
+ union {
+ struct page *__page;
+ struct folio *__folio;
+ struct page **__page_arr;
+ };
+ unsigned long __nr_pages;
+};
+
+struct discontig_kernel_page_ops {
+ int (*init)(void *vm_private_data, void **private);
+ int (*get)(struct discontig_kernel_page_state *state);
};
/*
@@ -844,6 +880,10 @@ struct mmap_action {
unsigned long nr_pages;
pgoff_t pgoff;
} map_kernel;
+ struct {
+ void *init_private;
+ const struct discontig_kernel_page_ops *ops;
+ } map_kernel_discontig;
};
enum mmap_action_type type;
@@ -950,7 +990,6 @@ struct vm_area_struct {
vma_flags_t flags;
};
-#ifdef CONFIG_PER_VMA_LOCK
/*
* Can only be written (using WRITE_ONCE()) while holding both:
* - mmap_lock (in write mode)
@@ -966,7 +1005,7 @@ struct vm_area_struct {
* slowpath.
*/
unsigned int vm_lock_seq;
-#endif
+
/*
* Low 32-bits of anonymous page offset.
* See vma_start_anon_pgoff() comment for details.
@@ -1003,7 +1042,6 @@ struct vm_area_struct {
#ifdef CONFIG_NUMA_BALANCING
struct vma_numab_state *numab_state; /* NUMA Balancing state */
#endif
-#ifdef CONFIG_PER_VMA_LOCK
/*
* Used to keep track of firstly, whether the VMA is attached, secondly,
* if attached, how many read locks are taken, and thirdly, if the
@@ -1046,7 +1084,6 @@ struct vm_area_struct {
#ifdef CONFIG_DEBUG_LOCK_ALLOC
struct lockdep_map vmlock_dep_map;
#endif
-#endif
#ifdef CONFIG_64BIT
/*
* High 32-bits of anonymous page offset.
@@ -1226,7 +1263,7 @@ struct mm_struct {
struct mm_mm_cid mm_cid;
/* sched_cache related statistics */
- struct sched_cache_stat sc_stat;
+ struct sched_cache_group *sched_cache_grp;
#ifdef CONFIG_MMU
atomic_long_t pgtables_bytes; /* size of all page tables */
#endif
@@ -1254,7 +1291,6 @@ struct mm_struct {
* init_mm.mmlist, and are protected
* by mmlist_lock
*/
-#ifdef CONFIG_PER_VMA_LOCK
struct rcuwait vma_writer_wait;
/*
* This field has lock-like semantics, meaning it is sometimes
@@ -1274,7 +1310,7 @@ struct mm_struct {
* mmap_lock.
*/
seqcount_t mm_lock_seq;
-#endif
+
struct futex_mm_data futex;
unsigned long hiwater_rss; /* High-watermark of RSS usage */
@@ -1624,8 +1660,9 @@ static inline unsigned int mm_cid_size(void)
#endif /* CONFIG_SCHED_MM_CID */
#ifdef CONFIG_SCHED_CACHE
-void mm_init_sched(struct mm_struct *mm,
- struct sched_cache_time __percpu *pcpu_sched);
+int mm_init_sched(struct mm_struct *mm,
+ struct sched_cache_time __percpu *pcpu_sched);
+void mm_destroy_sched(struct mm_struct *mm);
static inline int mm_alloc_sched_noprof(struct mm_struct *mm)
{
@@ -1635,17 +1672,11 @@ static inline int mm_alloc_sched_noprof(struct mm_struct *mm)
if (!pcpu_sched)
return -ENOMEM;
- mm_init_sched(mm, pcpu_sched);
- return 0;
+ return mm_init_sched(mm, pcpu_sched);
}
#define mm_alloc_sched(...) alloc_hooks(mm_alloc_sched_noprof(__VA_ARGS__))
-static inline void mm_destroy_sched(struct mm_struct *mm)
-{
- free_percpu(mm->sc_stat.pcpu_sched);
- mm->sc_stat.pcpu_sched = NULL;
-}
#else /* !CONFIG_SCHED_CACHE */
static inline int mm_alloc_sched(struct mm_struct *mm) { return 0; }
@@ -1980,7 +2011,7 @@ enum {
/*
* MMF_HAS_PINNED: Whether this mm has pinned any pages. This can be either
* replaced in the future by mm.pinned_vm when it becomes stable, or grow into
- * a counter on its own. We're aggresive on this bit for now: even if the
+ * a counter on its own. We're aggressive on this bit for now: even if the
* pinned pages were unpinned later on, we'll still keep this bit set for the
* lifecycle of this mm, just for simplicity.
*/
diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h
index bec0eab6ef03..e5553f4a414c 100644
--- a/include/linux/mmap_lock.h
+++ b/include/linux/mmap_lock.h
@@ -76,8 +76,6 @@ static inline void mmap_assert_write_locked(const struct mm_struct *mm)
rwsem_assert_held_write(&mm->mmap_lock);
}
-#ifdef CONFIG_PER_VMA_LOCK
-
#ifdef CONFIG_LOCKDEP
#define __vma_lockdep_map(vma) (&vma->vmlock_dep_map)
#else
@@ -230,10 +228,14 @@ static inline void vma_refcount_put(struct vm_area_struct *vma)
}
/*
- * Use only while holding mmap read lock which guarantees that locking will not
- * fail (nobody can concurrently write-lock the vma). vma_start_read() should
+ * Use only while holding mmap read lock which guarantees that vma lock is not
+ * contended (nobody can concurrently write-lock the vma). vma_start_read() should
* not be used in such cases because it might fail due to mm_lock_seq overflow.
* This functionality is used to obtain vma read lock and drop the mmap read lock.
+ *
+ * VMA can't be detached while we are holding mmap lock, therefore in practice this
+ * function can fail only when there are so many readers that vm_refcnt overflows.
+ * The failure case is very unlikely and is already annotated as such internally.
*/
static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int subclass)
{
@@ -249,22 +251,29 @@ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int
}
/*
- * Use only while holding mmap read lock which guarantees that locking will not
- * fail (nobody can concurrently write-lock the vma). vma_start_read() should
+ * Use only while holding mmap read lock which guarantees that vma lock is not
+ * contended (nobody can concurrently write-lock the vma). vma_start_read() should
* not be used in such cases because it might fail due to mm_lock_seq overflow.
* This functionality is used to obtain vma read lock and drop the mmap read lock.
+ *
+ * VMA can't be detached while we are holding mmap lock, therefore in practice this
+ * function can fail only when there are so many readers that vm_refcnt overflows.
+ * The failure case is very unlikely and is already annotated as such internally.
*/
static inline bool vma_start_read_locked(struct vm_area_struct *vma)
{
return vma_start_read_locked_nested(vma, 0);
}
+struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm,
+ unsigned long address);
+
static inline void vma_end_read(struct vm_area_struct *vma)
{
vma_refcount_put(vma);
}
-static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma)
+static inline unsigned int __vma_raw_mm_seqnum(const struct vm_area_struct *vma)
{
const struct mm_struct *mm = vma->vm_mm;
@@ -279,7 +288,7 @@ static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma)
*
* Returns true if write-locked, otherwise false.
*/
-static inline bool __is_vma_write_locked(struct vm_area_struct *vma)
+static inline bool __is_vma_write_locked(const struct vm_area_struct *vma)
{
/*
* current task is holding mmap_write_lock, both vma->vm_lock_seq and
@@ -297,6 +306,9 @@ int __vma_start_write(struct vm_area_struct *vma, int state);
*/
static inline void vma_start_write(struct vm_area_struct *vma)
{
+ if (!IS_ENABLED(CONFIG_MMU))
+ return;
+
if (__is_vma_write_locked(vma))
return;
@@ -319,6 +331,9 @@ static inline void vma_start_write(struct vm_area_struct *vma)
static inline __must_check
int vma_start_write_killable(struct vm_area_struct *vma)
{
+ if (!IS_ENABLED(CONFIG_MMU))
+ return 0;
+
if (__is_vma_write_locked(vma))
return 0;
@@ -329,8 +344,13 @@ int vma_start_write_killable(struct vm_area_struct *vma)
* vma_assert_write_locked() - assert that @vma holds a VMA write lock.
* @vma: The VMA to assert.
*/
-static inline void vma_assert_write_locked(struct vm_area_struct *vma)
+static inline void vma_assert_write_locked(const struct vm_area_struct *vma)
{
+ if (!IS_ENABLED(CONFIG_MMU)) {
+ mmap_assert_write_locked(vma->vm_mm);
+ return;
+ }
+
VM_WARN_ON_ONCE_VMA(!__is_vma_write_locked(vma), vma);
}
@@ -339,10 +359,15 @@ static inline void vma_assert_write_locked(struct vm_area_struct *vma)
* lock and is not detached.
* @vma: The VMA to assert.
*/
-static inline void vma_assert_locked(struct vm_area_struct *vma)
+static inline void vma_assert_locked(const struct vm_area_struct *vma)
{
unsigned int refcnt;
+ if (!IS_ENABLED(CONFIG_MMU)) {
+ mmap_assert_locked(vma->vm_mm);
+ return;
+ }
+
if (IS_ENABLED(CONFIG_LOCKDEP)) {
if (!lock_is_held(__vma_lockdep_map(vma)))
vma_assert_write_locked(vma);
@@ -385,7 +410,7 @@ static inline void vma_assert_locked(struct vm_area_struct *vma)
* With lockdep disabled we may sometimes race with other threads acquiring the
* mmap read lock simultaneous with our VMA read lock.
*/
-static inline void vma_assert_stabilised(struct vm_area_struct *vma)
+static inline void vma_assert_stabilised(const struct vm_area_struct *vma)
{
/*
* If another thread owns an mmap lock, it may go away at any time, and
@@ -420,7 +445,7 @@ static inline void vma_assert_stabilised(struct vm_area_struct *vma)
vma_assert_locked(vma);
}
-static inline bool vma_is_attached(struct vm_area_struct *vma)
+static inline bool vma_is_attached(const struct vm_area_struct *vma)
{
return refcount_read(&vma->vm_refcnt);
}
@@ -432,6 +457,9 @@ static inline bool vma_is_attached(struct vm_area_struct *vma)
*/
static inline void vma_assert_attached(struct vm_area_struct *vma)
{
+ if (!IS_ENABLED(CONFIG_MMU))
+ return;
+
WARN_ON_ONCE(!vma_is_attached(vma));
}
@@ -442,6 +470,9 @@ static inline void vma_assert_detached(struct vm_area_struct *vma)
static inline void vma_mark_attached(struct vm_area_struct *vma)
{
+ if (!IS_ENABLED(CONFIG_MMU))
+ return;
+
vma_assert_write_locked(vma);
vma_assert_detached(vma);
refcount_set_release(&vma->vm_refcnt, 1);
@@ -451,6 +482,9 @@ void __vma_exclude_readers_for_detach(struct vm_area_struct *vma);
static inline void vma_mark_detached(struct vm_area_struct *vma)
{
+ if (!IS_ENABLED(CONFIG_MMU))
+ return;
+
vma_assert_write_locked(vma);
vma_assert_attached(vma);
@@ -484,54 +518,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm,
struct vma_iterator *iter,
unsigned long address);
-#else /* CONFIG_PER_VMA_LOCK */
-
-static inline void mm_lock_seqcount_init(struct mm_struct *mm) {}
-static inline void mm_lock_seqcount_begin(struct mm_struct *mm) {}
-static inline void mm_lock_seqcount_end(struct mm_struct *mm) {}
-
-static inline bool mmap_lock_speculate_try_begin(struct mm_struct *mm, unsigned int *seq)
-{
- return false;
-}
-
-static inline bool mmap_lock_speculate_retry(struct mm_struct *mm, unsigned int seq)
-{
- return true;
-}
-static inline void vma_lock_init(struct vm_area_struct *vma, bool reset_refcnt) {}
-static inline void vma_end_read(struct vm_area_struct *vma) {}
-static inline void vma_start_write(struct vm_area_struct *vma) {}
-static inline __must_check
-int vma_start_write_killable(struct vm_area_struct *vma) { return 0; }
-static inline void vma_assert_write_locked(struct vm_area_struct *vma)
- { mmap_assert_write_locked(vma->vm_mm); }
-static inline bool vma_is_attached(struct vm_area_struct *vma)
- { return true; }
-static inline void vma_assert_attached(struct vm_area_struct *vma) {}
-static inline void vma_assert_detached(struct vm_area_struct *vma) {}
-static inline void vma_mark_attached(struct vm_area_struct *vma) {}
-static inline void vma_mark_detached(struct vm_area_struct *vma) {}
-
-static inline struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm,
- unsigned long address)
-{
- return NULL;
-}
-
-static inline void vma_assert_locked(struct vm_area_struct *vma)
-{
- mmap_assert_locked(vma->vm_mm);
-}
-
-static inline void vma_assert_stabilised(struct vm_area_struct *vma)
-{
- /* If no VMA locks, then either mmap lock suffices to stabilise. */
- mmap_assert_locked(vma->vm_mm);
-}
-
-#endif /* CONFIG_PER_VMA_LOCK */
-
static inline void vma_assert_can_modify(struct vm_area_struct *vma)
{
if (vma_is_attached(vma))
@@ -630,6 +616,8 @@ static inline void mmap_read_unlock(struct mm_struct *mm)
DEFINE_GUARD(mmap_read_lock, struct mm_struct *,
mmap_read_lock(_T), mmap_read_unlock(_T))
DEFINE_GUARD_COND(mmap_read_lock, _try, mmap_read_trylock(_T))
+DEFINE_GUARD(mmap_write_lock, struct mm_struct *,
+ mmap_write_lock(_T), mmap_write_unlock(_T))
static inline void mmap_read_unlock_non_owner(struct mm_struct *mm)
{
diff --git a/include/linux/mmc/host.h b/include/linux/mmc/host.h
index ba84f02c2a10..032c018650e0 100644
--- a/include/linux/mmc/host.h
+++ b/include/linux/mmc/host.h
@@ -23,6 +23,7 @@ struct mmc_ios {
unsigned int clock; /* clock rate */
unsigned short vdd;
unsigned int power_delay_ms; /* waiting for stable power */
+ unsigned int power_off_delay_us; /* waiting for power discharge */
/* vdd stores the bit number of the selected voltage range from below. */
@@ -463,6 +464,7 @@ struct mmc_host {
#define MMC_CAP2_CRYPTO 0
#endif
#define MMC_CAP2_ALT_GPT_TEGRA (1 << 28) /* Host with eMMC that has GPT entry at a non-standard location */
+#define MMC_CAP2_CRYPTO_NO_REPROG (1 << 29) /* Host handles inline crypto key reprogramming */
bool uhs2_sd_tran; /* UHS-II flag for SD_TRAN state */
bool uhs2_app_cmd; /* UHS-II flag for APP command */
@@ -583,11 +585,9 @@ struct mmc_host {
struct device_node;
-struct mmc_host *mmc_alloc_host(int extra, struct device *);
struct mmc_host *devm_mmc_alloc_host(struct device *dev, int extra);
int mmc_add_host(struct mmc_host *);
void mmc_remove_host(struct mmc_host *);
-void mmc_free_host(struct mmc_host *);
void mmc_of_parse_clk_phase(struct device *dev,
struct mmc_clk_phase_map *map);
int mmc_of_parse(struct mmc_host *host);
@@ -753,6 +753,8 @@ static inline int mmc_card_uhs2_hd_mode(struct mmc_host *host)
int mmc_sd_switch(struct mmc_card *card, bool mode, int group,
u8 value, u8 *resp);
int mmc_send_status(struct mmc_card *card, u32 *status);
+int mmc_send_tuning_timeout(struct mmc_host *host, u32 opcode, int *cmd_error,
+ u32 timeout_ms);
int mmc_send_tuning(struct mmc_host *host, u32 opcode, int *cmd_error);
int mmc_send_abort_tuning(struct mmc_host *host, u32 opcode);
int mmc_get_ext_csd(struct mmc_card *card, u8 **new_ext_csd);
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
index 94f9c3ff5416..ecd1ebd49756 100644
--- a/include/linux/mmzone.h
+++ b/include/linux/mmzone.h
@@ -96,25 +96,6 @@
#define MAX_FOLIO_NR_PAGES (1UL << MAX_FOLIO_ORDER)
-/*
- * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to
- * be naturally aligned with regard to the folio size.
- *
- * HVO which is only active if the size of struct page is a power of 2.
- */
-#define MAX_FOLIO_VMEMMAP_ALIGN \
- (IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP) && \
- is_power_of_2(sizeof(struct page)) ? \
- MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0)
-
-/*
- * vmemmap optimization (like HVO) is only possible for page orders that fill
- * two or more pages with struct pages.
- */
-#define VMEMMAP_TAIL_MIN_ORDER (ilog2(2 * PAGE_SIZE / sizeof(struct page)))
-#define __NR_VMEMMAP_TAILS (MAX_FOLIO_ORDER - VMEMMAP_TAIL_MIN_ORDER + 1)
-#define NR_VMEMMAP_TAILS (__NR_VMEMMAP_TAILS > 0 ? __NR_VMEMMAP_TAILS : 0)
-
enum migratetype {
MIGRATE_UNMOVABLE,
MIGRATE_MOVABLE,
@@ -492,11 +473,14 @@ enum lruvec_flags {
* folio->flags, masked by LRU_REFS_MASK.
*/
#define MAX_NR_TIERS 4U
+#define LRU_TIER_MIN 0U
+#define LRU_TIER_MAX (MAX_NR_TIERS - 1)
#ifndef __GENERATING_BOUNDS_H
#define LRU_GEN_MASK ((BIT(LRU_GEN_WIDTH) - 1) << LRU_GEN_PGOFF)
#define LRU_REFS_MASK ((BIT(LRU_REFS_WIDTH) - 1) << LRU_REFS_PGOFF)
+#define LRU_REFS_MAX BIT(LRU_REFS_WIDTH)
/*
* For folios accessed multiple times through file descriptors,
@@ -635,35 +619,32 @@ struct lru_gen_mm_walk {
* For each node, memcgs are divided into two generations: the old and the
* young. For each generation, memcgs are randomly sharded into multiple bins
* to improve scalability. For each bin, the hlist_nulls is virtually divided
- * into three segments: the head, the tail and the default.
+ * into two segments: the tail and the default.
*
* An onlining memcg is added to the tail of a random bin in the old generation.
* The eviction starts at the head of a random bin in the old generation. The
* per-node memcg generation counter, whose reminder (mod MEMCG_NR_GENS) indexes
* the old generation, is incremented when all its bins become empty.
*
- * There are four operations:
- * 1. MEMCG_LRU_HEAD, which moves a memcg to the head of a random bin in its
- * current generation (old or young) and updates its "seg" to "head";
- * 2. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its
+ * There are three operations:
+ * 1. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its
* current generation (old or young) and updates its "seg" to "tail";
- * 3. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old
+ * 2. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old
* generation, updates its "gen" to "old" and resets its "seg" to "default";
- * 4. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the
+ * 3. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the
* young generation, updates its "gen" to "young" and resets its "seg" to
* "default".
*
* The events that trigger the above operations are:
- * 1. Exceeding the soft limit, which triggers MEMCG_LRU_HEAD;
- * 2. The first attempt to reclaim a memcg below low, which triggers
+ * 1. The first attempt to reclaim a memcg below low, which triggers
* MEMCG_LRU_TAIL;
- * 3. The first attempt to reclaim a memcg offlined or below reclaimable size
+ * 2. The first attempt to reclaim a memcg offlined or below reclaimable size
* threshold, which triggers MEMCG_LRU_TAIL;
- * 4. The second attempt to reclaim a memcg offlined or below reclaimable size
+ * 3. The second attempt to reclaim a memcg offlined or below reclaimable size
* threshold, which triggers MEMCG_LRU_YOUNG;
- * 5. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG;
- * 6. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG;
- * 7. Offlining a memcg, which triggers MEMCG_LRU_OLD.
+ * 4. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG;
+ * 5. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG;
+ * 6. Offlining a memcg, which triggers MEMCG_LRU_OLD.
*
* Notes:
* 1. Memcg LRU only applies to global reclaim, and the round-robin incrementing
@@ -696,7 +677,6 @@ void lru_gen_exit_memcg(struct mem_cgroup *memcg);
void lru_gen_online_memcg(struct mem_cgroup *memcg);
void lru_gen_offline_memcg(struct mem_cgroup *memcg);
void lru_gen_release_memcg(struct mem_cgroup *memcg);
-void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid);
void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid);
bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid);
void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid);
@@ -737,10 +717,6 @@ static inline void lru_gen_release_memcg(struct mem_cgroup *memcg)
{
}
-static inline void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid)
-{
-}
-
static inline void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid)
{
}
@@ -1043,6 +1019,42 @@ struct zone {
* cma pages is present pages that are assigned for CMA use
* (MIGRATE_CMA).
*
+ * pages_with_online_memmap tracks pages within the zone that have
+ * an online memory map: present pages and memory holes whose
+ * memory map has been initialized and pfn_to_online_page()
+ * succeeds. When spanned_pages == pages_with_online_memmap,
+ * pfn_to_page() can be performed without further checks on any
+ * PFN within the zone span.
+ *
+ * Note: this counter may temporarily undercount when pages with an
+ * online memory map exist outside the current zone span. Such pages
+ * are only created during boot, when initializing the memory map of
+ * pages that do not fall into any zone span. The undercount itself
+ * can only happen after boot, during memory hotplug, when growing
+ * the zone to cover such pages and later shrinking it back, which
+ * may result in a "too small" value. This is safe: it merely
+ * prevents detecting a contiguous zone.
+ *
+ * Here is an example (page numbers are just for illustration
+ * purposes):
+ * after boot:
+ * [ zone span ]
+ * [ zone pages ]
+ * spanned=10, initialized=15, online=10
+ * online == spanned -> contiguous
+ *
+ * growing after hotplug (hotplug 5):
+ * [ zone span ]
+ * [ zone pages ] [ zone pages ]
+ * spanned=30, initialized=20, online=15
+ * online != spanned -> not contiguous
+ *
+ * shrinking after hotunplug (hotunplug 5 again):
+ * [ zone span ]
+ * [ zone pages ]
+ * spanned=15, initialized=15, online=10
+ * online != spanned -> not contiguous although contiguous
+ *
* So present_pages may be used by memory hotplug or memory power
* management logic to figure out unmanaged pages by checking
* (present_pages - managed_pages). And managed_pages should be used
@@ -1067,6 +1079,7 @@ struct zone {
atomic_long_t managed_pages;
unsigned long spanned_pages;
unsigned long present_pages;
+ unsigned long pages_with_online_memmap;
#if defined(CONFIG_MEMORY_HOTPLUG)
unsigned long present_early_pages;
#endif
@@ -1101,9 +1114,6 @@ struct zone {
#ifdef CONFIG_UNACCEPTED_MEMORY
/* Pages to be accepted. All pages on the list are MAX_PAGE_ORDER */
struct list_head unaccepted_pages;
-
- /* To be called once the last page in the zone is accepted */
- struct work_struct unaccepted_cleanup;
#endif
/* zone flags, see below */
@@ -1157,8 +1167,8 @@ struct zone {
/* Zone statistics */
atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS];
atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS];
-#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP
- struct page *vmemmap_tails[NR_VMEMMAP_TAILS];
+#ifdef CONFIG_VMEMMAP_OPTIMIZATION
+ struct page **vmemmap_tails;
#endif
} ____cacheline_internodealigned_in_smp;
@@ -1662,7 +1672,7 @@ extern void init_currently_empty_zone(struct zone *zone, unsigned long start_pfn
extern void lruvec_init(struct lruvec *lruvec);
-static inline struct pglist_data *lruvec_pgdat(struct lruvec *lruvec)
+static inline struct pglist_data *lruvec_pgdat(const struct lruvec *lruvec)
{
#ifdef CONFIG_MEMCG
return lruvec->pgdat;
@@ -1694,6 +1704,38 @@ static inline bool zone_is_zone_device(const struct zone *zone)
}
#endif
+/**
+ * zone_is_contiguous - test whether a zone is contiguous
+ * @zone: the zone to test.
+ *
+ * In a contiguous zone, it is valid to call pfn_to_page() on any PFN in the
+ * spanned zone without requiring pfn_valid() or pfn_to_online_page() checks.
+ *
+ * Note that missing synchronization with memory offlining makes any PFN
+ * traversal prone to races.
+ *
+ * ZONE_DEVICE zones are always marked non-contiguous.
+ *
+ * Return: true if contiguous, otherwise false.
+ */
+static inline bool zone_is_contiguous(const struct zone *zone)
+{
+ return READ_ONCE(zone->contiguous);
+}
+
+static inline void set_zone_contiguous(struct zone *zone)
+{
+ if (zone_is_zone_device(zone))
+ return;
+ if (zone->spanned_pages == zone->pages_with_online_memmap)
+ WRITE_ONCE(zone->contiguous, true);
+}
+
+static inline void clear_zone_contiguous(struct zone *zone)
+{
+ WRITE_ONCE(zone->contiguous, false);
+}
+
/*
* Returns true if a zone has pages managed by the buddy allocator.
* All the reclaim decisions have to use this function rather than
@@ -2021,19 +2063,23 @@ struct mem_section {
unsigned long section_mem_map;
struct mem_section_usage *usage;
+#ifdef CONFIG_VMEMMAP_OPTIMIZATION
+ /*
+ * Normally, sections hold regular (order-0) pages. However, for
+ * sections with HVO enabled, this tracks the compound page order
+ * to enable deduplication of redundant vmemmap pages.
+ */
+ unsigned int compound_page_order;
+#endif
#ifdef CONFIG_PAGE_EXTENSION
/*
* If SPARSEMEM, pgdat doesn't have page_ext pointer. We use
* section. (see page_ext.h about this.)
*/
struct page_ext *page_ext;
- unsigned long pad;
#endif
- /*
- * WARNING: mem_section must be a power-of-2 in size for the
- * calculation and use of SECTION_ROOT_MASK to make sense.
- */
-};
+/* Sacrifice minor padding space for efficient lookup. */
+} __aligned(2 * sizeof(unsigned long));
#ifdef CONFIG_SPARSEMEM_EXTREME
#define SECTIONS_PER_ROOT (PAGE_SIZE / sizeof (struct mem_section))
@@ -2043,7 +2089,6 @@ struct mem_section {
#define SECTION_NR_TO_ROOT(sec) ((sec) / SECTIONS_PER_ROOT)
#define NR_SECTION_ROOTS DIV_ROUND_UP(NR_MEM_SECTIONS, SECTIONS_PER_ROOT)
-#define SECTION_ROOT_MASK (SECTIONS_PER_ROOT - 1)
#ifdef CONFIG_SPARSEMEM_EXTREME
extern struct mem_section **mem_section;
@@ -2067,7 +2112,7 @@ static inline struct mem_section *__nr_to_section(unsigned long nr)
if (!mem_section || !mem_section[root])
return NULL;
#endif
- return &mem_section[root][nr & SECTION_ROOT_MASK];
+ return &mem_section[root][nr % SECTIONS_PER_ROOT];
}
/*
@@ -2083,29 +2128,21 @@ static inline struct mem_section *__nr_to_section(unsigned long nr)
* accommodate SECTION_MAP_LAST_BIT. We use BUILD_BUG_ON() to ensure this.
*/
enum {
- SECTION_MARKED_PRESENT_BIT,
SECTION_HAS_MEM_MAP_BIT,
SECTION_IS_ONLINE_BIT,
SECTION_IS_EARLY_BIT,
#ifdef CONFIG_ZONE_DEVICE
SECTION_TAINT_ZONE_DEVICE_BIT,
#endif
-#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
- SECTION_IS_VMEMMAP_PREINIT_BIT,
-#endif
SECTION_MAP_LAST_BIT,
};
-#define SECTION_MARKED_PRESENT BIT(SECTION_MARKED_PRESENT_BIT)
#define SECTION_HAS_MEM_MAP BIT(SECTION_HAS_MEM_MAP_BIT)
#define SECTION_IS_ONLINE BIT(SECTION_IS_ONLINE_BIT)
#define SECTION_IS_EARLY BIT(SECTION_IS_EARLY_BIT)
#ifdef CONFIG_ZONE_DEVICE
#define SECTION_TAINT_ZONE_DEVICE BIT(SECTION_TAINT_ZONE_DEVICE_BIT)
#endif
-#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
-#define SECTION_IS_VMEMMAP_PREINIT BIT(SECTION_IS_VMEMMAP_PREINIT_BIT)
-#endif
#define SECTION_MAP_MASK (~(BIT(SECTION_MAP_LAST_BIT) - 1))
#define SECTION_NID_SHIFT SECTION_MAP_LAST_BIT
@@ -2116,16 +2153,6 @@ static inline struct page *__section_mem_map_addr(struct mem_section *section)
return (struct page *)map;
}
-static inline int present_section(const struct mem_section *section)
-{
- return (section && (section->section_mem_map & SECTION_MARKED_PRESENT));
-}
-
-static inline int present_section_nr(unsigned long nr)
-{
- return present_section(__nr_to_section(nr));
-}
-
static inline int valid_section(const struct mem_section *section)
{
return (section && (section->section_mem_map & SECTION_HAS_MEM_MAP));
@@ -2141,6 +2168,11 @@ static inline int valid_section_nr(unsigned long nr)
return valid_section(__nr_to_section(nr));
}
+static inline int early_section_nr(unsigned long nr)
+{
+ return early_section(__nr_to_section(nr));
+}
+
static inline int online_section(const struct mem_section *section)
{
return (section && (section->section_mem_map & SECTION_IS_ONLINE));
@@ -2153,28 +2185,20 @@ static inline int online_device_section(const struct mem_section *section)
return section && ((section->section_mem_map & flags) == flags);
}
-#else
-static inline int online_device_section(const struct mem_section *section)
-{
- return 0;
-}
-#endif
-#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
-static inline int preinited_vmemmap_section(const struct mem_section *section)
+static inline struct zone *device_zone(int nid)
{
- return (section &&
- (section->section_mem_map & SECTION_IS_VMEMMAP_PREINIT));
+ return &NODE_DATA(nid)->node_zones[ZONE_DEVICE];
}
-
-void sparse_vmemmap_init_nid_early(int nid);
#else
-static inline int preinited_vmemmap_section(const struct mem_section *section)
+static inline int online_device_section(const struct mem_section *section)
{
return 0;
}
-static inline void sparse_vmemmap_init_nid_early(int nid)
+
+static inline struct zone *device_zone(int nid)
{
+ return NULL;
}
#endif
@@ -2193,7 +2217,7 @@ static inline struct mem_section *__pfn_to_section(unsigned long pfn)
return __nr_to_section(pfn_to_section_nr(pfn));
}
-extern unsigned long __highest_present_section_nr;
+extern unsigned long __highest_used_section_nr;
static inline int subsection_map_index(unsigned long pfn)
{
@@ -2241,9 +2265,6 @@ static inline bool pfn_section_first_valid(struct mem_section *ms, unsigned long
}
#endif
-void sparse_init_early_section(int nid, struct page *map, unsigned long pnum,
- unsigned long flags);
-
#ifndef CONFIG_HAVE_ARCH_PFN_VALID
/**
* pfn_valid - check if there is a valid memory map entry for a PFN
@@ -2295,7 +2316,7 @@ static inline unsigned long first_valid_pfn(unsigned long pfn, unsigned long end
rcu_read_lock_sched();
- while (nr <= __highest_present_section_nr && pfn < end_pfn) {
+ while (nr <= __highest_used_section_nr && pfn < end_pfn) {
struct mem_section *ms = __pfn_to_section(pfn);
if (valid_section(ms) &&
@@ -2341,27 +2362,20 @@ static inline unsigned long next_valid_pfn(unsigned long pfn, unsigned long end_
#endif
-static inline int pfn_in_present_section(unsigned long pfn)
-{
- if (pfn_to_section_nr(pfn) >= NR_MEM_SECTIONS)
- return 0;
- return present_section(__pfn_to_section(pfn));
-}
-
-static inline unsigned long next_present_section_nr(unsigned long section_nr)
+static inline unsigned long next_early_section_nr(unsigned long section_nr)
{
- while (++section_nr <= __highest_present_section_nr) {
- if (present_section_nr(section_nr))
+ while (++section_nr <= __highest_used_section_nr) {
+ if (early_section_nr(section_nr))
return section_nr;
}
return -1;
}
-#define for_each_present_section_nr(start, section_nr) \
- for (section_nr = next_present_section_nr(start - 1); \
+#define for_each_early_section_nr(start, section_nr) \
+ for (section_nr = next_early_section_nr(start - 1); \
section_nr != -1; \
- section_nr = next_present_section_nr(section_nr))
+ section_nr = next_early_section_nr(section_nr))
/*
* These are _only_ used during initialisation, therefore they
@@ -2377,10 +2391,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr)
#else
#define pfn_to_nid(pfn) (0)
#endif
-
-#else
-#define sparse_vmemmap_init_nid_early(_nid) do {} while (0)
-#define pfn_in_present_section pfn_valid
#endif /* CONFIG_SPARSEMEM */
/*
diff --git a/include/linux/mnt_idmapping.h b/include/linux/mnt_idmapping.h
index e71a6070a8f8..78eeef4c2996 100644
--- a/include/linux/mnt_idmapping.h
+++ b/include/linux/mnt_idmapping.h
@@ -8,8 +8,8 @@
struct mnt_idmap;
struct user_namespace;
-extern struct mnt_idmap nop_mnt_idmap;
-extern struct mnt_idmap invalid_mnt_idmap;
+extern const struct mnt_idmap nop_mnt_idmap;
+extern const struct mnt_idmap invalid_mnt_idmap;
extern struct user_namespace init_user_ns;
typedef struct {
@@ -121,19 +121,19 @@ static inline bool vfsgid_eq_kgid(vfsgid_t vfsgid, kgid_t kgid)
int vfsgid_in_group_p(vfsgid_t vfsgid);
-struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap);
-void mnt_idmap_put(struct mnt_idmap *idmap);
+const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap);
+void mnt_idmap_put(const struct mnt_idmap *idmap);
-vfsuid_t make_vfsuid(struct mnt_idmap *idmap,
+vfsuid_t make_vfsuid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, kuid_t kuid);
-vfsgid_t make_vfsgid(struct mnt_idmap *idmap,
+vfsgid_t make_vfsgid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, kgid_t kgid);
-kuid_t from_vfsuid(struct mnt_idmap *idmap,
+kuid_t from_vfsuid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, vfsuid_t vfsuid);
-kgid_t from_vfsgid(struct mnt_idmap *idmap,
+kgid_t from_vfsgid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, vfsgid_t vfsgid);
/**
@@ -148,7 +148,7 @@ kgid_t from_vfsgid(struct mnt_idmap *idmap,
*
* Return: true if @vfsuid has a mapping in the filesystem, false if not.
*/
-static inline bool vfsuid_has_fsmapping(struct mnt_idmap *idmap,
+static inline bool vfsuid_has_fsmapping(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns,
vfsuid_t vfsuid)
{
@@ -186,7 +186,7 @@ static inline kuid_t vfsuid_into_kuid(vfsuid_t vfsuid)
*
* Return: true if @vfsgid has a mapping in the filesystem, false if not.
*/
-static inline bool vfsgid_has_fsmapping(struct mnt_idmap *idmap,
+static inline bool vfsgid_has_fsmapping(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns,
vfsgid_t vfsgid)
{
@@ -225,7 +225,7 @@ static inline kgid_t vfsgid_into_kgid(vfsgid_t vfsgid)
*
* Return: the caller's current fsuid mapped up according to @idmap.
*/
-static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap,
+static inline kuid_t mapped_fsuid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns)
{
return from_vfsuid(idmap, fs_userns, VFSUIDT_INIT(current_fsuid()));
@@ -244,7 +244,7 @@ static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap,
*
* Return: the caller's current fsgid mapped up according to @idmap.
*/
-static inline kgid_t mapped_fsgid(struct mnt_idmap *idmap,
+static inline kgid_t mapped_fsgid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns)
{
return from_vfsgid(idmap, fs_userns, VFSGIDT_INIT(current_fsgid()));
diff --git a/include/linux/mod_devicetable.h b/include/linux/mod_devicetable.h
index a397213bedac..d1581bb64729 100644
--- a/include/linux/mod_devicetable.h
+++ b/include/linux/mod_devicetable.h
@@ -15,7 +15,7 @@
#include "device-id/acpi.h"
#include "device-id/amba.h"
#include "device-id/ap.h"
-#include "device-id/apr.h"
+#include "device-id/arm_smccc.h"
#include "device-id/auxiliary.h"
#include "device-id/bcma.h"
#include "device-id/ccw.h"
diff --git a/include/linux/module_symbol.h b/include/linux/module_symbol.h
index 574609aced99..698d3db2b37e 100644
--- a/include/linux/module_symbol.h
+++ b/include/linux/module_symbol.h
@@ -7,8 +7,11 @@ enum ksym_flags {
KSYM_FLAG_GPL_ONLY = 1 << 0,
};
-/* This ignores the intensely annoying "mapping symbols" found in ELF files. */
-static inline bool is_mapping_symbol(const char *str)
+/*
+ * Ignore local labels (.L*, L0*) and mapping symbols ($*). These symbols are
+ * not useful for the kernel, for example, they should not appear in kallsyms.
+ */
+static inline bool is_ignored_kernel_symbol(const char *str)
{
if (str[0] == '.' && str[1] == 'L')
return true;
diff --git a/include/linux/mount.h b/include/linux/mount.h
index acfe7ef86a1b..e90ccafef281 100644
--- a/include/linux/mount.h
+++ b/include/linux/mount.h
@@ -59,10 +59,10 @@ struct vfsmount {
struct dentry *mnt_root; /* root of the mounted tree */
struct super_block *mnt_sb; /* pointer to superblock */
int mnt_flags;
- struct mnt_idmap *mnt_idmap;
+ const struct mnt_idmap *mnt_idmap;
} __randomize_layout;
-static inline struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt)
+static inline const struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt)
{
/* Pairs with smp_store_release() in do_idmap_mount(). */
return READ_ONCE(mnt->mnt_idmap);
diff --git a/include/linux/mtd/bbm.h b/include/linux/mtd/bbm.h
index d890805f5494..67929fdf2dff 100644
--- a/include/linux/mtd/bbm.h
+++ b/include/linux/mtd/bbm.h
@@ -75,7 +75,7 @@ struct nand_bbt_descr {
* with NAND_BBT_CREATE.
*/
#define NAND_BBT_CREATE_EMPTY 0x00000400
-/* Write bbt if neccecary */
+/* Write bbt if necessary */
#define NAND_BBT_WRITE 0x00002000
/* Read and write back block contents when writing bbt */
#define NAND_BBT_SAVECONTENT 0x00004000
diff --git a/include/linux/mtd/lpc32xx_mlc.h b/include/linux/mtd/lpc32xx_mlc.h
deleted file mode 100644
index 35e971be0950..000000000000
--- a/include/linux/mtd/lpc32xx_mlc.h
+++ /dev/null
@@ -1,17 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0-only */
-/*
- * Platform data for LPC32xx SoC MLC NAND controller
- *
- * Copyright © 2012 Roland Stigge
- */
-
-#ifndef __LINUX_MTD_LPC32XX_MLC_H
-#define __LINUX_MTD_LPC32XX_MLC_H
-
-#include <linux/dmaengine.h>
-
-struct lpc32xx_mlc_platform_data {
- dma_filter_fn dma_filter;
-};
-
-#endif /* __LINUX_MTD_LPC32XX_MLC_H */
diff --git a/include/linux/mtd/lpc32xx_slc.h b/include/linux/mtd/lpc32xx_slc.h
deleted file mode 100644
index a044b806566b..000000000000
--- a/include/linux/mtd/lpc32xx_slc.h
+++ /dev/null
@@ -1,17 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0-only */
-/*
- * Platform data for LPC32xx SoC SLC NAND controller
- *
- * Copyright © 2012 Roland Stigge
- */
-
-#ifndef __LINUX_MTD_LPC32XX_SLC_H
-#define __LINUX_MTD_LPC32XX_SLC_H
-
-#include <linux/dmaengine.h>
-
-struct lpc32xx_slc_platform_data {
- dma_filter_fn dma_filter;
-};
-
-#endif /* __LINUX_MTD_LPC32XX_SLC_H */
diff --git a/include/linux/mtd/nand-qpic-common.h b/include/linux/mtd/nand-qpic-common.h
index 437448995187..0f40964eefe7 100644
--- a/include/linux/mtd/nand-qpic-common.h
+++ b/include/linux/mtd/nand-qpic-common.h
@@ -262,7 +262,6 @@ struct bam_transaction {
u32 tx_sgl_start;
u32 rx_sgl_pos;
u32 rx_sgl_start;
-
);
};
diff --git a/include/linux/mtd/nand.h b/include/linux/mtd/nand.h
index 09c8c93e4dba..6936180b6ea5 100644
--- a/include/linux/mtd/nand.h
+++ b/include/linux/mtd/nand.h
@@ -305,6 +305,8 @@ int nand_ecc_prepare_io_req(struct nand_device *nand,
struct nand_page_io_req *req);
int nand_ecc_finish_io_req(struct nand_device *nand,
struct nand_page_io_req *req);
+bool nand_ecc_is_pipelined(const struct nand_device *nand);
+
bool nand_ecc_is_strong_enough(struct nand_device *nand);
#if IS_REACHABLE(CONFIG_MTD_NAND_CORE)
diff --git a/include/linux/mtd/rawnand.h b/include/linux/mtd/rawnand.h
index 5c70e7bd3ed5..43fefff02025 100644
--- a/include/linux/mtd/rawnand.h
+++ b/include/linux/mtd/rawnand.h
@@ -1592,7 +1592,7 @@ void nand_wait_ready(struct nand_chip *chip);
/*
* Free resources held by the NAND device, must be called on error after a
- * sucessful nand_scan().
+ * successful nand_scan().
*/
void nand_cleanup(struct nand_chip *chip);
diff --git a/include/linux/mtd/sh_flctl.h b/include/linux/mtd/sh_flctl.h
index 78fc2d4218c8..15adf2474fa0 100644
--- a/include/linux/mtd/sh_flctl.h
+++ b/include/linux/mtd/sh_flctl.h
@@ -60,8 +60,8 @@
* Some hardware uses bits called PULSEx instead of FCKSEL_E and QTSEL_E
* to control the clock divider used between the High-Speed Peripheral Clock
* and the FLCTL internal clock. If so, use CLK_8_BIT_xxx for connecting 8 bit
- * and CLK_16_BIT_xxx for connecting 16 bit bus bandwith NAND chips. For the 16
- * bit version the divider is seperate for the pulse width of high and low
+ * and CLK_16_BIT_xxx for connecting 16 bit bus bandwidth NAND chips. For the 16
+ * bit version the divider is separate for the pulse width of high and low
* signals.
*/
#define PULSE3 (0x1 << 27)
diff --git a/include/linux/mtd/spi-nor.h b/include/linux/mtd/spi-nor.h
index 4b92494827b1..b3e3c6b10186 100644
--- a/include/linux/mtd/spi-nor.h
+++ b/include/linux/mtd/spi-nor.h
@@ -25,6 +25,7 @@
#define SPINOR_OP_WRSR 0x01 /* Write status register 1 */
#define SPINOR_OP_RDSR2 0x3f /* Read status register 2 */
#define SPINOR_OP_WRSR2 0x3e /* Write status register 2 */
+#define SPINOR_OP_WRSR2_ALT 0x31 /* Write status register 2 (alternative) */
#define SPINOR_OP_READ 0x03 /* Read data bytes (low frequency) */
#define SPINOR_OP_READ_FAST 0x0b /* Read data bytes (high frequency) */
#define SPINOR_OP_READ_1_1_2 0x3b /* Read data bytes (Dual Output SPI) */
@@ -113,20 +114,16 @@
#define SR_E_ERR BIT(5)
#define SR_P_ERR BIT(6)
-#define SR1_QUAD_EN_BIT6 BIT(6)
-
#define SR_BP_SHIFT 2
/* Enhanced Volatile Configuration Register bits */
#define EVCR_QUAD_EN_MICRON BIT(7) /* Micron Quad I/O */
/* Status Register 2 bits. */
-#define SR2_QUAD_EN_BIT1 BIT(1)
#define SR2_LB1 BIT(3) /* Security Register Lock Bit 1 */
#define SR2_LB2 BIT(4) /* Security Register Lock Bit 2 */
#define SR2_LB3 BIT(5) /* Security Register Lock Bit 3 */
#define SR2_CMP_BIT6 BIT(6)
-#define SR2_QUAD_EN_BIT7 BIT(7)
/* Supported SPI protocols */
#define SNOR_PROTO_INST_MASK GENMASK(23, 16)
@@ -365,8 +362,6 @@ struct spi_nor_flash_parameter;
* @read_dummy: the dummy needed by the read operation
* @program_opcode: the program opcode
* @sst_write_second: used by the SST write operation
- * @flags: flag options for the current SPI NOR (SNOR_F_*)
- * @cmd_ext_type: the command opcode extension type for DTR mode.
* @read_proto: the SPI protocol for read operations
* @write_proto: the SPI protocol for write operations
* @reg_proto: the SPI protocol for read_reg/write_reg/erase operations
@@ -398,6 +393,7 @@ struct spi_nor {
u8 *id;
const struct flash_info *info;
const struct spi_nor_manufacturer *manufacturer;
+ const char *partname;
u8 addr_nbytes;
u8 erase_opcode;
u8 read_opcode;
@@ -407,8 +403,6 @@ struct spi_nor {
enum spi_nor_protocol write_proto;
enum spi_nor_protocol reg_proto;
bool sst_write_second;
- u32 flags;
- enum spi_nor_cmd_ext cmd_ext_type;
struct sfdp *sfdp;
struct dentry *debugfs_root;
u8 dfs_sr_cache[2];
diff --git a/include/linux/mtd/spinand.h b/include/linux/mtd/spinand.h
index 5f4c00ae72a7..d7c99fcfb85e 100644
--- a/include/linux/mtd/spinand.h
+++ b/include/linux/mtd/spinand.h
@@ -399,7 +399,7 @@ struct spinand_devid {
};
/**
- * struct manufacurer_ops - SPI NAND manufacturer specific operations
+ * struct spinand_manufacturer_ops - SPI NAND manufacturer specific operations
* @init: initialize a SPI NAND device
* @cleanup: cleanup a SPI NAND device
*
@@ -438,6 +438,7 @@ extern const struct spinand_manufacturer fmsh_spinand_manufacturer;
extern const struct spinand_manufacturer foresee_spinand_manufacturer;
extern const struct spinand_manufacturer gigadevice_spinand_manufacturer;
extern const struct spinand_manufacturer heyangtek_spinand_manufacturer;
+extern const struct spinand_manufacturer issi_spinand_manufacturer;
extern const struct spinand_manufacturer macronix_spinand_manufacturer;
extern const struct spinand_manufacturer micron_spinand_manufacturer;
extern const struct spinand_manufacturer paragon_spinand_manufacturer;
diff --git a/include/linux/namei.h b/include/linux/namei.h
index 86d657b24fc6..c4436e5c2ba6 100644
--- a/include/linux/namei.h
+++ b/include/linux/namei.h
@@ -32,8 +32,9 @@ enum { MAX_NESTED_LINKS = 8 };
#define LOOKUP_CREATE BIT(17) /* ... in object creation */
#define LOOKUP_EXCL BIT(18) /* ... in target must not exist */
#define LOOKUP_RENAME_TARGET BIT(19) /* ... in destination of rename() */
+#define LOOKUP_SHARED BIT(20) /* Parent lock is held shared */
-/* 4 spare bits for intent */
+/* 3 spare bits for intent */
/* Scoping flags for lookup. */
#define LOOKUP_NO_SYMLINKS BIT(24) /* No symlink crossing. */
@@ -70,24 +71,24 @@ extern struct dentry *try_lookup_noperm(struct qstr *, struct dentry *);
extern struct dentry *lookup_noperm(struct qstr *, struct dentry *);
extern struct dentry *lookup_noperm_unlocked(struct qstr *, struct dentry *);
extern struct dentry *lookup_noperm_positive_unlocked(struct qstr *, struct dentry *);
-struct dentry *lookup_one(struct mnt_idmap *, struct qstr *, struct dentry *);
-struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap,
+struct dentry *lookup_one(const struct mnt_idmap *, struct qstr *, struct dentry *);
+struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap,
struct qstr *name, struct dentry *base);
-struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap,
+struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap,
struct qstr *name,
struct dentry *base);
-struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap,
+struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap,
struct qstr *name,
struct dentry *base);
-struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent,
+struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent,
struct qstr *name);
-struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent,
+struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent,
struct qstr *name);
-struct dentry *start_creating_killable(struct mnt_idmap *idmap,
+struct dentry *start_creating_killable(const struct mnt_idmap *idmap,
struct dentry *parent,
struct qstr *name);
-struct dentry *start_removing_killable(struct mnt_idmap *idmap,
+struct dentry *start_removing_killable(const struct mnt_idmap *idmap,
struct dentry *parent,
struct qstr *name);
struct dentry *start_creating_noperm(struct dentry *parent, struct qstr *name);
diff --git a/include/linux/net.h b/include/linux/net.h
index 3d82966e2243..e2a866fbcfa5 100644
--- a/include/linux/net.h
+++ b/include/linux/net.h
@@ -47,6 +47,9 @@ typedef struct sockopt {
int optlen;
} sockopt_t;
+int sockptr_to_sockopt(sockopt_t *opt, sockptr_t optval, sockptr_t optlen,
+ struct kvec *kvec);
+
/*
* Initialize a user-backed sockopt_t from the (optval, optlen) __user pair of
* a getsockopt() callback. Used by transitional __user getsockopt wrappers
@@ -70,6 +73,35 @@ static inline int sockopt_init_user(sockopt_t *opt, char __user *optval,
return 0;
}
+/*
+ * Grow optval to @size, for the options whose reply is sized by a count the
+ * caller left in optval rather than by optlen. Those write past optlen today
+ * and userspace relies on it.
+ *
+ * Call it before writing through opt->iter_out: it re-anchors the iterator at
+ * the head of optval. Only a user buffer can be longer than the optlen the
+ * caller declared, so a kernel-backed optval is refused with -EINVAL.
+ */
+static inline int sockopt_expand_out(sockopt_t *opt, size_t size)
+{
+ if (size <= (size_t)opt->optlen)
+ return 0;
+
+ if (size > INT_MAX)
+ return -EINVAL;
+
+ /* Re-anchoring reads iter_out.ubuf, so the iterator has to be a user
+ * buffer that nothing has written through yet.
+ */
+ if (WARN_ON_ONCE(!iter_is_ubuf(&opt->iter_out) ||
+ iov_iter_count(&opt->iter_out) != (size_t)opt->optlen))
+ return -EINVAL;
+
+ iov_iter_ubuf(&opt->iter_out, ITER_DEST, opt->iter_out.ubuf, size);
+
+ return 0;
+}
+
struct poll_table_struct;
struct pipe_inode_info;
struct inode;
@@ -166,7 +198,7 @@ struct socket {
struct file *file;
struct sock *sk;
- const struct proto_ops *ops; /* Might change with IPV6_ADDRFORM or MPTCP. */
+ const struct proto_ops *ops; /* Might change with MPTCP. */
struct socket_wq wq;
};
diff --git a/include/linux/netdevice.h b/include/linux/netdevice.h
index 87cafc932e9e..f61c97231309 100644
--- a/include/linux/netdevice.h
+++ b/include/linux/netdevice.h
@@ -887,6 +887,7 @@ enum net_device_path_type {
DEV_PATH_DSA,
DEV_PATH_MTK_WDMA,
DEV_PATH_TUN,
+ DEV_PATH_IEEE80211,
};
struct net_device_path {
@@ -953,6 +954,8 @@ struct net_device_path_ctx {
u16 id;
__be16 proto;
} vlan[NET_DEVICE_PATH_VLAN_MAX];
+
+ bool ieee80211;
};
enum tc_setup_type {
@@ -1135,13 +1138,16 @@ struct netdev_net_notifier {
* struct netdev_hw_addr_list *uc,
* struct netdev_hw_addr_list *mc);
* Async version of ndo_set_rx_mode which runs in process context
- * with rtnl_lock and netdev_lock_ops(dev) held. The uc/mc parameters
+ * under the netdev instance lock for "ops locked" drivers, or
+ * rtnl_lock for all other drivers. The uc/mc parameters
* are snapshots of the address lists - iterate with
* netdev_hw_addr_list_for_each(ha, uc). Return 0 on success or a
* negative errno to request a retry via the core backoff.
*
* void (*ndo_work)(struct net_device *dev, unsigned long events);
* Run deferred work scheduled with netdev_work_sched(@events).
+ * Runs in process context under the netdev instance lock for "ops
+ * locked" drivers, or rtnl_lock for all other drivers.
*
* int (*ndo_set_mac_address)(struct net_device *dev, void *addr);
* This function is called when the Media Access Control address
@@ -1151,11 +1157,6 @@ struct netdev_net_notifier {
* int (*ndo_validate_addr)(struct net_device *dev);
* Test if Media Access Control address is valid for the device.
*
- * int (*ndo_do_ioctl)(struct net_device *dev, struct ifreq *ifr, int cmd);
- * Old-style ioctl entry point. This is used internally by the
- * ieee802154 subsystem but is no longer called by the device
- * ioctl handler.
- *
* int (*ndo_siocbond)(struct net_device *dev, struct ifreq *ifr, int cmd);
* Used by the bonding driver for its device specific ioctls:
* SIOCBONDENSLAVE, SIOCBONDRELEASE, SIOCBONDSETHWADDR, SIOCBONDCHANGEACTIVE,
@@ -1477,8 +1478,6 @@ struct net_device_ops {
int (*ndo_set_mac_address)(struct net_device *dev,
void *addr);
int (*ndo_validate_addr)(struct net_device *dev);
- int (*ndo_do_ioctl)(struct net_device *dev,
- struct ifreq *ifr, int cmd);
int (*ndo_eth_ioctl)(struct net_device *dev,
struct ifreq *ifr, int cmd);
int (*ndo_siocbond)(struct net_device *dev,
@@ -1837,6 +1836,7 @@ enum netdev_reg_state {
* drivers. Mainly used by logical interfaces, such as
* bonding and tunnels
* @netmem_tx: device netmem TX mode
+ * @pacing_offload: enable EDT pacing offload.
*
* @name: This is the first field of the "visible" part of this structure
* (i.e. as seen by users in the "Space.c" file). It is the name
@@ -2167,6 +2167,7 @@ struct net_device {
unsigned long priv_flags:32;
unsigned long lltx:1;
unsigned long netmem_tx:2;
+ unsigned long pacing_offload:1;
);
const struct net_device_ops *netdev_ops;
const struct header_ops *header_ops;
@@ -3669,7 +3670,11 @@ struct page_pool_bh {
};
DECLARE_PER_CPU(struct page_pool_bh, system_page_pool);
+#ifdef CONFIG_KASAN
+#define XMIT_RECURSION_LIMIT 4
+#else
#define XMIT_RECURSION_LIMIT 8
+#endif
#ifndef CONFIG_PREEMPT_RT
static inline int dev_recursion_level(void)
@@ -5321,7 +5326,6 @@ void netdev_lower_state_changed(struct net_device *lower_dev,
void *lower_state_info);
#define NETDEV_RSS_KEY_LEN 256
-extern u8 netdev_rss_key[NETDEV_RSS_KEY_LEN] __read_mostly;
void netdev_rss_key_fill(void *buffer, size_t len);
int skb_checksum_help(struct sk_buff *skb);
@@ -5606,12 +5610,12 @@ static inline bool netif_has_l3_rx_handler(const struct net_device *dev)
static inline bool netif_is_l3_master(const struct net_device *dev)
{
- return dev->priv_flags & IFF_L3MDEV_MASTER;
+ return IS_ENABLED(CONFIG_NET_VRF) && (dev->priv_flags & IFF_L3MDEV_MASTER);
}
static inline bool netif_is_l3_slave(const struct net_device *dev)
{
- return dev->priv_flags & IFF_L3MDEV_SLAVE;
+ return IS_ENABLED(CONFIG_NET_VRF) && (dev->priv_flags & IFF_L3MDEV_SLAVE);
}
static inline int dev_sdif(const struct net_device *dev)
diff --git a/include/linux/netfilter_arp/arp_tables.h b/include/linux/netfilter_arp/arp_tables.h
index 05631a25e622..8b8d472eff34 100644
--- a/include/linux/netfilter_arp/arp_tables.h
+++ b/include/linux/netfilter_arp/arp_tables.h
@@ -56,23 +56,4 @@ void arpt_unregister_table(struct net *net, const char *name);
extern unsigned int arpt_do_table(void *priv, struct sk_buff *skb,
const struct nf_hook_state *state);
-#ifdef CONFIG_NETFILTER_XTABLES_COMPAT
-#include <net/compat.h>
-
-struct compat_arpt_entry {
- struct arpt_arp arp;
- __u16 target_offset;
- __u16 next_offset;
- compat_uint_t comefrom;
- struct compat_xt_counters counters;
- unsigned char elems[];
-};
-
-static inline struct xt_entry_target *
-compat_arpt_get_target(struct compat_arpt_entry *e)
-{
- return (void *)e + e->target_offset;
-}
-
-#endif /* CONFIG_COMPAT */
#endif /* _ARPTABLES_H */
diff --git a/include/linux/netfilter_netdev.h b/include/linux/netfilter_netdev.h
index 3175073a66ba..2a854a0bbd4b 100644
--- a/include/linux/netfilter_netdev.h
+++ b/include/linux/netfilter_netdev.h
@@ -133,7 +133,7 @@ static inline struct sk_buff *nf_hook_egress(struct sk_buff *skb, int *rc,
static inline void nf_skip_egress(struct sk_buff *skb, bool skip)
{
-#ifdef CONFIG_NETFILTER_SKIP_EGRESS
+#ifdef CONFIG_NET_EGRESS
skb->nf_skip_egress = skip;
#endif
}
diff --git a/include/linux/netfs.h b/include/linux/netfs.h
index f837a501008c..e5961450f3f1 100644
--- a/include/linux/netfs.h
+++ b/include/linux/netfs.h
@@ -22,6 +22,7 @@
enum netfs_sreq_ref_trace;
typedef struct mempool mempool_t;
+struct fscache_occupancy;
struct folio_queue;
/**
@@ -62,8 +63,8 @@ struct netfs_inode {
struct fscache_cookie *cache;
#endif
struct list_head wb_queue; /* Queue of processes wanting to do writeback */
- loff_t _remote_i_size; /* Size of the remote file */
- loff_t _zero_point; /* Size after which we assume there's no data
+ uoff_t _remote_i_size; /* Size of the remote file */
+ uoff_t _zero_point; /* Size after which we assume there's no data
* on the server */
spinlock_t lock; /* Lock covering wb_queue */
atomic_t io_count; /* Number of outstanding reqs */
@@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio)
return priv;
}
+enum netfs_cache_collect {
+ NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */
+ NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */
+ NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */
+};
+
/*
* Stream of I/O subrequests going to a particular destination, such as the
* server or the local cache. This is mainly intended for writing where we may
@@ -142,7 +149,7 @@ struct netfs_io_stream {
void (*issue_write)(struct netfs_io_subrequest *subreq);
/* Collection tracking */
struct list_head subrequests; /* Contributory I/O operations */
- unsigned long long collected_to; /* Position we've collected results to */
+ uoff_t collected_to; /* Position we've collected results to */
size_t transferred; /* The amount transferred from this stream */
unsigned short error; /* Aggregate error for the stream */
enum netfs_io_source source; /* Where to read from/write to */
@@ -152,6 +159,7 @@ struct netfs_io_stream {
bool need_retry; /* T if this stream needs retrying */
bool failed; /* T if this stream failed */
bool transferred_valid; /* T is ->transferred is valid */
+ enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */
};
/*
@@ -161,8 +169,11 @@ struct netfs_cache_resources {
const struct netfs_cache_ops *ops;
void *cache_priv;
void *cache_priv2;
- unsigned int debug_id; /* Cookie debug ID */
+ uoff_t cache_i_size; /* Initial size of cache file */
+ unsigned int cookie_id; /* Cache cookie debug ID */
+ unsigned int object_id; /* Cache object debug ID */
unsigned int inval_counter; /* object->inval_counter at begin_op */
+ unsigned int dio_size; /* DIO block size */
};
/*
@@ -177,7 +188,7 @@ struct netfs_io_subrequest {
struct work_struct work;
struct list_head rreq_link; /* Link in rreq->subrequests */
struct iov_iter io_iter; /* Iterator for this subrequest */
- unsigned long long start; /* Where to start the I/O */
+ uoff_t start; /* Where to start the I/O */
size_t len; /* Size of the I/O */
size_t transferred; /* Amount of data transferred */
refcount_t ref;
@@ -196,6 +207,7 @@ struct netfs_io_subrequest {
#define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */
#define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */
#define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */
+#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */
};
enum netfs_io_origin {
@@ -208,7 +220,6 @@ enum netfs_io_origin {
NETFS_DIO_READ, /* This is a direct I/O read */
NETFS_WRITEBACK, /* This write was triggered by writepages */
NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */
- NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */
NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */
NETFS_DIO_WRITE, /* This is a direct I/O write */
NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */
@@ -243,16 +254,18 @@ struct netfs_io_request {
void *netfs_priv; /* Private data for the netfs */
void *netfs_priv2; /* Private data for the netfs */
struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */
- unsigned long long submitted; /* Amount submitted for I/O so far */
- unsigned long long len; /* Length of the request */
+ uoff_t submitted; /* Amount submitted for I/O so far */
+ uoff_t len; /* Length of the request */
size_t transferred; /* Amount to be indicated as transferred */
+ size_t progress_at; /* Report read progress when hit this much read */
long error; /* 0 or error that occurred */
- unsigned long long i_size; /* Size of the file */
- unsigned long long start; /* Start position */
+ uoff_t i_size; /* Size of the file */
+ uoff_t start; /* Start position */
atomic64_t issued_to; /* Write issuer folio cursor */
- unsigned long long collected_to; /* Point we've collected to */
- unsigned long long cleaned_to; /* Position we've cleaned folios to */
- unsigned long long abandon_to; /* Position to abandon folios to */
+ uoff_t collected_to; /* Point we've collected to */
+ uoff_t cache_coll_to; /* Point the cache has collected to */
+ uoff_t cleaned_to; /* Position we've cleaned folios to */
+ uoff_t abandon_to; /* Position to abandon folios to */
const struct folio *no_unlock_folio; /* Don't unlock this folio after read */
gfp_t gfp; /* GFP flags to use */
unsigned int direct_bv_count; /* Number of elements in direct_bv[] */
@@ -262,7 +275,6 @@ struct netfs_io_request {
atomic_t subreq_counter; /* Next subreq->debug_index */
unsigned int nr_group_rel; /* Number of refs to release on ->group */
spinlock_t lock; /* Lock for queuing subreqs */
- unsigned char front_folio_order; /* Order (size) of front folio */
enum netfs_io_origin origin; /* Origin of the request */
bool direct_bv_unpin; /* T if direct_bv[] must be unpinned */
refcount_t ref;
@@ -273,13 +285,18 @@ struct netfs_io_request {
#define NETFS_RREQ_FAILED 3 /* The request failed */
#define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */
#define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */
-#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */
-#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */
-#define NETFS_RREQ_FOLIO_COPY_TO_CACHE 10 /* Copy current folio to cache from read */
-#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */
-#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */
+#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */
+#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */
+#define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */
+#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */
+#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */
+#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */
+#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */
+#define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */
+#ifdef CONFIG_NETFS_PGPRIV2
#define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark
* write to cache on read */
+#endif
const struct netfs_request_ops *netfs_ops;
};
@@ -298,12 +315,12 @@ struct netfs_request_ops {
int (*prepare_read)(struct netfs_io_subrequest *subreq);
void (*issue_read)(struct netfs_io_subrequest *subreq);
bool (*is_still_valid)(struct netfs_io_request *rreq);
- int (*check_write_begin)(struct file *file, loff_t pos, unsigned len,
+ int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len,
struct folio **foliop, void **_fsdata);
void (*done)(struct netfs_io_request *rreq);
/* Modification handling */
- void (*update_i_size)(struct inode *inode, loff_t i_size);
+ void (*update_i_size)(struct inode *inode, uoff_t i_size);
void (*post_modify)(struct inode *inode);
/* Write request handling */
@@ -331,7 +348,7 @@ struct netfs_cache_ops {
/* Read data from the cache */
int (*read)(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
enum netfs_read_from_hole read_hole,
netfs_io_terminated_t term_func,
@@ -339,7 +356,7 @@ struct netfs_cache_ops {
/* Write data to the cache */
int (*write)(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
netfs_io_terminated_t term_func,
void *term_func_priv);
@@ -349,15 +366,14 @@ struct netfs_cache_ops {
/* Expand readahead request */
void (*expand_readahead)(struct netfs_cache_resources *cres,
- unsigned long long *_start,
- unsigned long long *_len,
- unsigned long long i_size);
+ uoff_t *_start,
+ uoff_t *_len,
+ uoff_t i_size);
/* Prepare a read operation, shortening it to a cached/uncached
* boundary as appropriate.
*/
- enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq,
- unsigned long long i_size);
+ int (*prepare_read)(struct netfs_io_subrequest *subreq);
/* Prepare a write subrequest, working out if we're allowed to do it
* and finding out the maximum amount of data to gather before
@@ -370,15 +386,24 @@ struct netfs_cache_ops {
* actually do.
*/
int (*prepare_write)(struct netfs_cache_resources *cres,
- loff_t *_start, size_t *_len, size_t upper_len,
- loff_t i_size, bool no_space_allocated_yet);
+ uoff_t *_start, size_t *_len, size_t upper_len,
+ uoff_t i_size, bool no_space_allocated_yet);
/* Query the occupancy of the cache in a region, returning where the
* next chunk of data starts and how long it is.
*/
int (*query_occupancy)(struct netfs_cache_resources *cres,
- loff_t start, size_t len, size_t granularity,
- loff_t *_data_start, size_t *_data_len);
+ struct fscache_occupancy *occ);
+
+ /* Collect the result of buffered writeback to the cache. This
+ * includes copying a read to the cache. block_type is one of:
+ * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data
+ * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped
+ * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap
+ */
+ void (*collect_write)(struct netfs_io_request *wreq,
+ uoff_t start, size_t len,
+ enum netfs_cache_collect block_type);
};
/* High-level read API. */
@@ -396,6 +421,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from);
ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *iter,
struct netfs_group *netfs_group);
ssize_t netfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from);
+void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from,
+ uoff_t to);
/* Single, monolithic object read/write API. */
void netfs_single_mark_inode_dirty(struct inode *inode);
@@ -409,7 +436,7 @@ struct readahead_control;
void netfs_readahead(struct readahead_control *);
int netfs_read_folio(struct file *, struct folio *);
int netfs_write_begin(struct netfs_inode *, struct file *,
- struct address_space *, loff_t pos, unsigned int len,
+ struct address_space *, uoff_t pos, unsigned int len,
struct folio **, void **fsdata);
int netfs_writepages(struct address_space *mapping,
struct writeback_control *wbc);
@@ -487,10 +514,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode)
* cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit
* archs it makes no difference if preempt is enabled or not.
*/
-static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode)
+static inline uoff_t netfs_read_remote_i_size(const struct inode *inode)
{
const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode);
- unsigned long long remote_i_size;
+ uoff_t remote_i_size;
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
unsigned int seq;
@@ -525,7 +552,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in
* spinning forever.
*/
static inline void netfs_write_remote_i_size(struct inode *inode,
- unsigned long long remote_i_size)
+ uoff_t remote_i_size)
{
struct netfs_inode *ictx = netfs_inode(inode);
@@ -562,10 +589,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode,
* cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit
* archs it makes no difference if preempt is enabled or not.
*/
-static inline unsigned long long netfs_read_zero_point(const struct inode *inode)
+static inline uoff_t netfs_read_zero_point(const struct inode *inode)
{
struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode);
- unsigned long long zero_point;
+ uoff_t zero_point;
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
unsigned int seq;
@@ -600,7 +627,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode
* forever.
*/
static inline void netfs_write_zero_point(struct inode *inode,
- unsigned long long zero_point)
+ uoff_t zero_point)
{
struct netfs_inode *ictx = netfs_inode(inode);
@@ -641,9 +668,9 @@ static inline void netfs_write_zero_point(struct inode *inode,
* archs it makes no difference if preempt is enabled or not.
*/
static inline void netfs_read_sizes(const struct inode *inode,
- unsigned long long *i_size,
- unsigned long long *remote_i_size,
- unsigned long long *zero_point)
+ uoff_t *i_size,
+ uoff_t *remote_i_size,
+ uoff_t *zero_point)
{
const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode);
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
@@ -689,9 +716,9 @@ static inline void netfs_read_sizes(const struct inode *inode,
* forever.
*/
static inline void netfs_write_sizes(struct inode *inode,
- unsigned long long i_size,
- unsigned long long remote_i_size,
- unsigned long long zero_point)
+ uoff_t i_size,
+ uoff_t remote_i_size,
+ uoff_t zero_point)
{
struct netfs_inode *ictx = netfs_inode(inode);
@@ -759,7 +786,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx,
* Inform the netfs lib that a file got resized so that it can adjust its state.
*/
static inline void netfs_resize_file(struct netfs_inode *ictx,
- unsigned long long new_i_size,
+ uoff_t new_i_size,
bool changed_on_server)
{
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
diff --git a/include/linux/netpoll.h b/include/linux/netpoll.h
index 1c6b1eec5efd..ec0821a5b02d 100644
--- a/include/linux/netpoll.h
+++ b/include/linux/netpoll.h
@@ -16,11 +16,6 @@
#include <linux/ip.h>
#include <linux/udp.h>
-union inet_addr {
- __be32 ip;
- struct in6_addr in6;
-};
-
struct netpoll {
struct net_device *dev;
netdevice_tracker dev_tracker;
diff --git a/include/linux/nfs.h b/include/linux/nfs.h
index 0906a0b40c6a..8c2818db43c5 100644
--- a/include/linux/nfs.h
+++ b/include/linux/nfs.h
@@ -11,59 +11,8 @@
#include <linux/cred.h>
#include <linux/sunrpc/auth.h>
#include <linux/sunrpc/msg_prot.h>
-#include <linux/string.h>
-#include <linux/crc32.h>
-#include <uapi/linux/nfs.h>
-
-/* The LOCALIO program is entirely private to Linux and is
- * NOT part of the uapi.
- */
-#define NFS_LOCALIO_PROGRAM 400122
-#define LOCALIOPROC_NULL 0
-#define LOCALIOPROC_UUID_IS_LOCAL 1
-
-/*
- * This is the kernel NFS client file handle representation
- */
-#define NFS_MAXFHSIZE 128
-struct nfs_fh {
- unsigned short size;
- unsigned char data[NFS_MAXFHSIZE];
-};
-
-/*
- * Returns a zero iff the size and data fields match.
- * Checks only "size" bytes in the data field.
- */
-static inline int nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b)
-{
- return a->size != b->size || memcmp(a->data, b->data, a->size) != 0;
-}
-
-static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source)
-{
- target->size = source->size;
- memcpy(target->data, source->data, source->size);
-}
-
-enum nfs3_stable_how {
- NFS_UNSTABLE = 0,
- NFS_DATA_SYNC = 1,
- NFS_FILE_SYNC = 2,
+#include <linux/nfs_fh.h>
- /* used by direct.c to mark verf as invalid */
- NFS_INVALID_STABLE_HOW = -1
-};
+#include <uapi/linux/nfs.h>
-/**
- * nfs_fhandle_hash - calculate the crc32 hash for the filehandle
- * @fh - pointer to filehandle
- *
- * returns a crc32 hash for the filehandle that is compatible with
- * the one displayed by "wireshark".
- */
-static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh)
-{
- return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size);
-}
#endif /* _LINUX_NFS_H */
diff --git a/include/linux/nfs3.h b/include/linux/nfs3.h
index 404b8f724fc9..b6539a75edea 100644
--- a/include/linux/nfs3.h
+++ b/include/linux/nfs3.h
@@ -7,6 +7,49 @@
#include <uapi/linux/nfs3.h>
+/*
+ * NFSv3 error status values.
+ * See RFC 1813 Section 2.5
+ */
+enum {
+ NFS3ERR_PERM = 1,
+ NFS3ERR_NOENT = 2,
+ NFS3ERR_IO = 5,
+ NFS3ERR_NXIO = 6,
+ NFS3ERR_ACCES = 13,
+ NFS3ERR_EXIST = 17,
+ NFS3ERR_XDEV = 18,
+ NFS3ERR_NODEV = 19,
+ NFS3ERR_NOTDIR = 20,
+ NFS3ERR_ISDIR = 21,
+ NFS3ERR_INVAL = 22,
+ NFS3ERR_FBIG = 27,
+ NFS3ERR_NOSPC = 28,
+ NFS3ERR_ROFS = 30,
+ NFS3ERR_MLINK = 31,
+ NFS3ERR_NAMETOOLONG = 63,
+ NFS3ERR_NOTEMPTY = 66,
+ NFS3ERR_DQUOT = 69,
+ NFS3ERR_STALE = 70,
+ NFS3ERR_REMOTE = 71,
+ NFS3ERR_BADHANDLE = 10001,
+ NFS3ERR_NOT_SYNC = 10002,
+ NFS3ERR_BAD_COOKIE = 10003,
+ NFS3ERR_NOTSUPP = 10004,
+ NFS3ERR_TOOSMALL = 10005,
+ NFS3ERR_SERVERFAULT = 10006,
+ NFS3ERR_BADTYPE = 10007,
+ NFS3ERR_JUKEBOX = 10008,
+};
+
+enum nfs3_stable_how {
+ NFS_UNSTABLE = 0,
+ NFS_DATA_SYNC = 1,
+ NFS_FILE_SYNC = 2,
+
+ /* used to mark verf as invalid */
+ NFS_INVALID_STABLE_HOW = -1
+};
/* Number of 32bit words in post_op_attr */
#define NFS3_POST_OP_ATTR_WORDS 22
diff --git a/include/linux/nfs4.h b/include/linux/nfs4.h
index 1a3981c26b23..41b7cdcc674f 100644
--- a/include/linux/nfs4.h
+++ b/include/linux/nfs4.h
@@ -263,6 +263,12 @@ enum why_no_delegation4 { /* new to v4.1 */
WND4_IS_DIR = 8,
};
+enum stable_how4 {
+ UNSTABLE4 = 0,
+ DATA_SYNC4 = 1,
+ FILE_SYNC4 = 2,
+};
+
enum lock_type4 {
NFS4_UNLOCK_LT = 0,
NFS4_READ_LT = 1,
diff --git a/include/linux/nfs_fh.h b/include/linux/nfs_fh.h
new file mode 100644
index 000000000000..49dfc5ec60fe
--- /dev/null
+++ b/include/linux/nfs_fh.h
@@ -0,0 +1,63 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * struct nfs_fh is an NFS version-agnostic data structure that
+ * stores an NFS file handle. It is also commonly used in NFS
+ * related APIs.
+ */
+#ifndef _LINUX_NFS_FH_H
+#define _LINUX_NFS_FH_H
+
+#include <linux/types.h>
+#include <linux/string.h>
+#include <linux/crc32.h>
+
+/*
+ * The largest file handle size today is an NFSv4 file handle,
+ * which can be up to 128 octets long.
+ */
+#define NFS_MAXFHSIZE 128
+struct nfs_fh {
+ unsigned short size;
+ unsigned char data[NFS_MAXFHSIZE];
+};
+
+/**
+ * nfs_compare_fh - Compare two NFS file handles
+ * @a: An NFS file handle to be compared
+ * @b: An NFS file handle to be compared
+ *
+ * Checks only "size" bytes in each data field.
+ *
+ * Return: %false if the two file handles are equal, otherwise %true
+ */
+static inline bool nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b)
+{
+ return a->size != b->size || memcmp(a->data, b->data, a->size) != 0;
+}
+
+/**
+ * nfs_copy_fh - Copy an NFS file handle
+ * @target: Destination file handle
+ * @source: Source file handle
+ *
+ * Copies source->size bytes of file handle data into target.
+ */
+static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source)
+{
+ target->size = source->size;
+ memcpy(target->data, source->data, source->size);
+}
+
+/**
+ * nfs_fhandle_hash - Calculate the crc32 hash for the filehandle
+ * @fh: An NFS file handle to hash
+ *
+ * Return: a crc32 hash for the filehandle that is compatible with
+ * the one displayed by "wireshark"
+ */
+static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh)
+{
+ return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size);
+}
+
+#endif /* _LINUX_NFS_FH_H */
diff --git a/include/linux/nfs_fs.h b/include/linux/nfs_fs.h
index b85a73ae7919..d2c716322c6f 100644
--- a/include/linux/nfs_fs.h
+++ b/include/linux/nfs_fs.h
@@ -437,11 +437,11 @@ extern int nfs_refresh_inode(struct inode *, struct nfs_fattr *);
extern int nfs_post_op_update_inode(struct inode *inode, struct nfs_fattr *fattr);
extern int nfs_post_op_update_inode_force_wcc(struct inode *inode, struct nfs_fattr *fattr);
extern int nfs_post_op_update_inode_force_wcc_locked(struct inode *inode, struct nfs_fattr *fattr);
-extern int nfs_getattr(struct mnt_idmap *, const struct path *,
+extern int nfs_getattr(const struct mnt_idmap *, const struct path *,
struct kstat *, u32, unsigned int);
extern void nfs_access_add_cache(struct inode *, struct nfs_access_entry *, const struct cred *);
extern void nfs_access_set_mask(struct nfs_access_entry *, u32);
-extern int nfs_permission(struct mnt_idmap *, struct inode *, int);
+extern int nfs_permission(const struct mnt_idmap *, struct inode *, int);
extern int nfs_open(struct inode *, struct file *);
extern int nfs_attribute_cache_expired(struct inode *inode);
extern int nfs_revalidate_inode(struct inode *inode, unsigned long flags);
@@ -450,7 +450,7 @@ extern int nfs_clear_invalid_mapping(struct address_space *mapping);
extern bool nfs_mapping_need_revalidate_inode(struct inode *inode);
extern int nfs_revalidate_mapping(struct inode *inode, struct address_space *mapping);
extern int nfs_revalidate_mapping_rcu(struct inode *inode);
-extern int nfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *);
+extern int nfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *);
extern void nfs_setattr_update_inode(struct inode *inode, struct iattr *attr, struct nfs_fattr *);
extern void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr);
extern struct nfs_open_context *get_nfs_open_context(struct nfs_open_context *ctx);
diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h
index 34d294774f8c..416c6f39f31d 100644
--- a/include/linux/nfs_fs_sb.h
+++ b/include/linux/nfs_fs_sb.h
@@ -74,6 +74,8 @@ struct nfs_client {
u64 cl_clientid; /* constant */
nfs4_verifier cl_confirm; /* Clientid verifier */
unsigned long cl_state;
+ /* bumped on each CB_NOTIFY_DEVICEID CHANGE for this client */
+ atomic_t cl_deviceid_change_epoch;
spinlock_t cl_lock;
@@ -101,6 +103,8 @@ struct nfs_client {
/* The flags used for obtaining the clientid during EXCHANGE_ID */
u32 cl_exchange_flags;
struct nfs4_session *cl_session; /* shared session */
+ /* CB_NOTIFY_DEVICEID DELETE suspects, protected by cl_lock */
+ struct list_head cl_deviceid_deletes;
bool cl_preserve_clid;
struct nfs41_server_owner *cl_serverowner;
struct nfs41_server_scope *cl_serverscope;
@@ -248,6 +252,10 @@ struct nfs_server {
that are supported on this
filesystem */
struct pnfs_layoutdriver_type *pnfs_curr_ld; /* Active layout driver */
+ unsigned int lg_reply_sz; /* Learned LAYOUTGET reply
+ buffer size, when the layout
+ driver's default has proved
+ too small */
struct rpc_wait_queue roc_rpcwaitq;
/* the following fields are protected by nfs_client->cl_lock */
diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h
index 4b9a35dbc062..c38e4b380be5 100644
--- a/include/linux/nfs_page.h
+++ b/include/linux/nfs_page.h
@@ -38,6 +38,7 @@ enum {
PG_REMOVE, /* page group sync bit in write path */
PG_CONTENDED1, /* Is someone waiting for a lock? */
PG_CONTENDED2, /* Is someone waiting for a lock? */
+ PG_PINNED, /* page is pinned by GUP */
};
struct nfs_inode;
@@ -58,6 +59,7 @@ struct nfs_page {
struct nfs_page *wb_this_page; /* list of reqs for this page */
struct nfs_page *wb_head; /* head pointer for req list */
unsigned short wb_nio; /* Number of I/O attempts */
+ unsigned int wb_nr_pinned; /* Number of pinned pages */
};
struct nfs_pgio_mirror;
@@ -125,15 +127,17 @@ struct nfs_pageio_descriptor {
extern struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx,
struct page *page,
+ bool pinned,
unsigned int pgbase,
loff_t offset,
unsigned int count);
extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx,
struct folio *folio,
+ bool pinned,
unsigned int offset,
unsigned int count);
-extern void nfs_release_request(struct nfs_page *);
-
+void nfs_release_request(struct nfs_page *req);
+void nfs_release_request_list(struct list_head *head);
extern void nfs_pageio_init(struct nfs_pageio_descriptor *desc,
struct inode *inode,
diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h
index 22265b1ff080..c199ea23e7eb 100644
--- a/include/linux/nfs_ssc.h
+++ b/include/linux/nfs_ssc.h
@@ -2,80 +2,33 @@
/*
* include/linux/nfs_ssc.h
*
+ * NFSv4.2 server-to-server copy, NFS client side APIs
+ *
* Author: Dai Ngo <dai.ngo@oracle.com>
*
* Copyright (c) 2020, Oracle and/or its affiliates.
*/
-#include <linux/nfs_fs.h>
-#include <linux/sunrpc/svc.h>
+#ifndef _LINUX_NFS_SSC_H
+#define _LINUX_NFS_SSC_H
-extern struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl;
+#include <linux/nfs_fh.h>
+#include <linux/nfs4.h>
+
+struct file;
+struct vfsmount;
-/*
- * NFS_V4
- */
struct nfs4_ssc_client_ops {
+ struct module *owner;
struct file *(*sco_open)(struct vfsmount *ss_mnt,
struct nfs_fh *src_fh, nfs4_stateid *stateid);
void (*sco_close)(struct file *filep);
};
-/*
- * NFS_FS
- */
-struct nfs_ssc_client_ops {
- void (*sco_sb_deactive)(struct super_block *sb);
-};
-
-struct nfs_ssc_client_ops_tbl {
- const struct nfs4_ssc_client_ops *ssc_nfs4_ops;
- const struct nfs_ssc_client_ops *ssc_nfs_ops;
-};
-
extern void nfs42_ssc_register_ops(void);
extern void nfs42_ssc_unregister_ops(void);
extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops);
extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops);
-#ifdef CONFIG_NFSD_V4_2_INTER_SSC
-static inline struct file *nfs42_ssc_open(struct vfsmount *ss_mnt,
- struct nfs_fh *src_fh, nfs4_stateid *stateid)
-{
- if (nfs_ssc_client_tbl.ssc_nfs4_ops)
- return (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_open)(ss_mnt, src_fh, stateid);
- return ERR_PTR(-EIO);
-}
-
-static inline void nfs42_ssc_close(struct file *filep)
-{
- if (nfs_ssc_client_tbl.ssc_nfs4_ops)
- (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_close)(filep);
-}
-#endif
-
-struct nfsd4_ssc_umount_item {
- struct list_head nsui_list;
- bool nsui_busy;
- /*
- * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry
- * is removed when refcnt drops to 1 and nsui_expire expires.
- */
- refcount_t nsui_refcnt;
- unsigned long nsui_expire;
- struct vfsmount *nsui_vfsmount;
- char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1];
-};
-
-/*
- * NFS_FS
- */
-extern void nfs_ssc_register(const struct nfs_ssc_client_ops *ops);
-extern void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops);
-
-static inline void nfs_do_sb_deactive(struct super_block *sb)
-{
- if (nfs_ssc_client_tbl.ssc_nfs_ops)
- (*nfs_ssc_client_tbl.ssc_nfs_ops->sco_sb_deactive)(sb);
-}
+#endif /* _LINUX_NFS_SSC_H */
diff --git a/include/linux/nfs_xdr.h b/include/linux/nfs_xdr.h
index 7ed8fdb930d6..c0e29b4dfa62 100644
--- a/include/linux/nfs_xdr.h
+++ b/include/linux/nfs_xdr.h
@@ -1693,6 +1693,7 @@ struct nfs_pgio_header {
struct nfs_client *ds_clp; /* pNFS data server */
u32 ds_commit_idx; /* ds index if ds_clp is set */
u32 pgio_mirror_idx;/* mirror index in pgio layer */
+ struct nfs4_deviceid_node *ds_dev; /* device node ref held across the I/O */
};
struct nfs_mds_commit_info {
@@ -1731,6 +1732,7 @@ struct nfs_commit_data {
struct nfs_open_context *context;
struct pnfs_layout_segment *lseg;
struct nfs_client *ds_clp; /* pNFS data server */
+ struct nfs4_deviceid_node *ds_dev; /* device node ref held across the commit */
int ds_commit_index;
loff_t lwb;
const struct rpc_call_ops *mds_ops;
diff --git a/include/linux/nfsd_ssc.h b/include/linux/nfsd_ssc.h
new file mode 100644
index 000000000000..7001410f01c2
--- /dev/null
+++ b/include/linux/nfsd_ssc.h
@@ -0,0 +1,38 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * include/linux/nfsd_ssc.h
+ *
+ * NFSv4.2 server-to-server copy, NFS server side APIs
+ *
+ * Author: Dai Ngo <dai.ngo@oracle.com>
+ *
+ * Copyright (c) 2020, Oracle and/or its affiliates.
+ */
+
+#ifndef _LINUX_NFSD_SSC_H
+#define _LINUX_NFSD_SSC_H
+
+#include <linux/nfs_fh.h>
+#include <linux/nfs4.h>
+
+struct file;
+struct vfsmount;
+
+#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER)
+struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh,
+ nfs4_stateid *stateid);
+void nfsd42_ssc_close(struct file *filp);
+#else
+static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt,
+ struct nfs_fh *src_fh,
+ nfs4_stateid *stateid)
+{
+ return ERR_PTR(-EIO);
+}
+
+static inline void nfsd42_ssc_close(struct file *filp)
+{
+}
+#endif
+
+#endif /* _LINUX_NFSD_SSC_H */
diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h
index 3d91043254e6..8ce4d978a636 100644
--- a/include/linux/nfslocalio.h
+++ b/include/linux/nfslocalio.h
@@ -13,9 +13,18 @@
#include <linux/uuid.h>
#include <linux/sunrpc/clnt.h>
#include <linux/sunrpc/svcauth.h>
-#include <linux/nfs.h>
+#include <linux/nfs_fh.h>
+
#include <net/net_namespace.h>
+/*
+ * The LOCALIO program is entirely private to Linux and is NOT part of
+ * the uapi.
+ */
+#define NFS_LOCALIO_PROGRAM 400122
+#define LOCALIOPROC_NULL 0
+#define LOCALIOPROC_UUID_IS_LOCAL 1
+
struct nfs_client;
struct nfs_file_localio;
diff --git a/include/linux/ns/ns_common_types.h b/include/linux/ns/ns_common_types.h
index ea45c54e4435..fddc62be3b62 100644
--- a/include/linux/ns/ns_common_types.h
+++ b/include/linux/ns/ns_common_types.h
@@ -116,10 +116,11 @@ struct ns_common {
struct dentry *stashed;
const struct proc_ns_operations *ops;
unsigned int inum;
- union {
- struct ns_tree;
- struct rcu_head ns_rcu;
- };
+ struct ns_tree;
+ struct rcu_head ns_rcu;
+#ifdef CONFIG_SECURITY
+ void *ns_security;
+#endif
};
#define to_ns_common(__ns) \
diff --git a/include/linux/nvme-tcp.h b/include/linux/nvme-tcp.h
index e435250fcb4d..859338da8573 100644
--- a/include/linux/nvme-tcp.h
+++ b/include/linux/nvme-tcp.h
@@ -77,7 +77,7 @@ struct nvme_tcp_hdr {
__le32 plen;
};
-/**
+/*
* struct nvme_tcp_icreq_pdu - nvme tcp initialize connection request pdu
*
* @hdr: pdu generic header
@@ -95,7 +95,7 @@ struct nvme_tcp_icreq_pdu {
__u8 rsvd2[112];
};
-/**
+/*
* struct nvme_tcp_icresp_pdu - nvme tcp initialize connection response pdu
*
* @hdr: pdu common header
@@ -113,12 +113,13 @@ struct nvme_tcp_icresp_pdu {
__u8 rsvd[112];
};
-/**
+/*
* struct nvme_tcp_term_pdu - nvme tcp terminate connection pdu
*
* @hdr: pdu common header
* @fes: fatal error status
- * @fei: fatal error information
+ * @feil: fatal error information (low 16 bits)
+ * @feih: fatal error information (high 16 bits)
*/
struct nvme_tcp_term_pdu {
struct nvme_tcp_hdr hdr;
@@ -128,7 +129,7 @@ struct nvme_tcp_term_pdu {
__u8 rsvd[10];
};
-/**
+/*
* struct nvme_tcp_cmd_pdu - nvme tcp command capsule pdu
*
* @hdr: pdu common header
@@ -139,10 +140,9 @@ struct nvme_tcp_cmd_pdu {
struct nvme_command cmd;
};
-/**
+/*
* struct nvme_tcp_rsp_pdu - nvme tcp response capsule pdu
*
- * @hdr: pdu common header
* @hdr: nvme-tcp generic header
* @cqe: nvme completion queue entry
*/
@@ -151,7 +151,7 @@ struct nvme_tcp_rsp_pdu {
struct nvme_completion cqe;
};
-/**
+/*
* struct nvme_tcp_r2t_pdu - nvme tcp ready-to-transfer pdu
*
* @hdr: pdu common header
@@ -169,7 +169,7 @@ struct nvme_tcp_r2t_pdu {
__u8 rsvd[4];
};
-/**
+/*
* struct nvme_tcp_data_pdu - nvme tcp data pdu
*
* @hdr: pdu common header
diff --git a/include/linux/once_lite.h b/include/linux/once_lite.h
index 236592c4eeb1..5e2b67039aaf 100644
--- a/include/linux/once_lite.h
+++ b/include/linux/once_lite.h
@@ -10,19 +10,21 @@
#define DO_ONCE_LITE(func, ...) \
DO_ONCE_LITE_IF(true, func, ##__VA_ARGS__)
-#define __ONCE_LITE_IF(condition) \
+#define __ONCE_LITE() \
({ \
static bool __section(".data..once") __already_done; \
- bool __ret_cond = !!(condition); \
bool __ret_once = false; \
\
- if (unlikely(__ret_cond) && unlikely(!__already_done)) {\
+ if (unlikely(!__already_done)) { \
__already_done = true; \
__ret_once = true; \
} \
unlikely(__ret_once); \
})
+#define __ONCE_LITE_IF(condition) \
+ (unlikely(condition) && __ONCE_LITE())
+
#define DO_ONCE_LITE_IF(condition, func, ...) \
({ \
bool __ret_do_once = !!(condition); \
diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h
index 7a863572adce..b0ddc652e76c 100644
--- a/include/linux/page-flags.h
+++ b/include/linux/page-flags.h
@@ -44,10 +44,6 @@
* Consequently, PG_reserved for a page mapped into user space can indicate
* the zero page, the vDSO, MMIO pages or device memory.
*
- * The PG_private bitflag is set on pagecache pages if they contain filesystem
- * specific data (which is normally at page->private). It can be used by
- * private allocations for its own usage.
- *
* During initiation of disk I/O, PG_locked is set. This bit is set before I/O
* and cleared when writeback _starts_ or when read _completes_. PG_writeback
* is set before writeback starts and cleared when it finishes.
@@ -105,7 +101,7 @@ enum pageflags {
PG_owner_2, /* Owner use. If pagecache, fs may use */
PG_arch_1,
PG_reserved,
- PG_private, /* If pagecache, has fs-private data */
+ PG_folio, /* Do not use: reserved for folio identification */
PG_private_2, /* If pagecache, has fs aux data */
PG_reclaim, /* To be reclaimed asap */
PG_swapbacked, /* Page is backed by RAM/swap */
@@ -208,14 +204,13 @@ enum pageflags {
static __always_inline bool compound_info_has_mask(void)
{
/*
- * Limit mask usage to HugeTLB vmemmap optimization (HVO) where it
- * makes a difference.
+ * Limit mask usage to HVO where it makes a difference.
*
* The approach with mask would work in the wider set of conditions,
* but it requires validating that struct pages are naturally aligned
* for all orders up to the MAX_FOLIO_ORDER, which can be tricky.
*/
- if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP))
+ if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION))
return false;
return is_power_of_2(sizeof(struct page));
@@ -576,9 +571,19 @@ FOLIO_FLAG(swapbacked, FOLIO_HEAD_PAGE)
/*
* Private page markings that may be used by the filesystem that owns the page
* for its own purposes.
- * - PG_private and PG_private_2 cause release_folio() and co to be invoked
+ * - folio->private and PG_private_2 cause release_folio() and co to be invoked
*/
-PAGEFLAG(Private, private, PF_ANY)
+
+static __always_inline bool folio_test_private(const struct folio *folio)
+{
+ /*
+ * data_race() is added for readers without holding the folio lock.
+ * Only the NULL/non-NULL answer is used and both are valid while
+ * private is being attached or detached, so the race is benign.
+ */
+ return data_race(folio->private);
+}
+
FOLIO_FLAG(private_2, FOLIO_HEAD_PAGE)
/* owner_2 can be set on tail pages for anon memory */
@@ -588,13 +593,12 @@ FOLIO_FLAG(owner_2, FOLIO_HEAD_PAGE)
* Only test-and-set exist for PG_writeback. The unconditional operators are
* risky: they bypass page accounting.
*/
-TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL)
- TESTSCFLAG(Writeback, writeback, PF_NO_TAIL)
+FOLIO_TEST_FLAG(writeback, FOLIO_HEAD_PAGE)
+ FOLIO_TEST_SET_FLAG(writeback, FOLIO_HEAD_PAGE)
FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE)
/* PG_readahead is only used for reads; PG_reclaim is only for writes */
-PAGEFLAG(Reclaim, reclaim, PF_NO_TAIL)
- TESTCLEARFLAG(Reclaim, reclaim, PF_NO_TAIL)
+FOLIO_FLAG(reclaim, FOLIO_HEAD_PAGE)
FOLIO_FLAG(readahead, FOLIO_HEAD_PAGE)
FOLIO_TEST_CLEAR_FLAG(readahead, FOLIO_HEAD_PAGE)
@@ -656,6 +660,7 @@ TESTSCFLAG(HWPoison, hwpoison, PF_ANY)
#define __PG_HWPOISON (1UL << PG_hwpoison)
#else
PAGEFLAG_FALSE(HWPoison, hwpoison)
+TESTSCFLAG_FALSE(HWPoison, hwpoison)
#define __PG_HWPOISON 0
#endif
@@ -1170,7 +1175,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page)
*/
#define PAGE_FLAGS_CHECK_AT_FREE \
(1UL << PG_lru | 1UL << PG_locked | \
- 1UL << PG_private | 1UL << PG_private_2 | \
+ 1UL << PG_private_2 | \
1UL << PG_writeback | 1UL << PG_reserved | \
1UL << PG_active | \
1UL << PG_unevictable | __PG_MLOCKED | LRU_GEN_MASK)
@@ -1194,8 +1199,31 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page)
(0xffUL /* order */ | 1UL << PG_has_hwpoisoned | \
1UL << PG_large_rmappable | 1UL << PG_partially_mapped)
-#define PAGE_FLAGS_PRIVATE \
- (1UL << PG_private | 1UL << PG_private_2)
+/**
+ * folio_has_attached_private - check if the folio has private data attached
+ * @folio: The folio to check.
+ *
+ * Use this in code that may encounter swapcache or hugetlb folios but only
+ * wants to detect attached private data.
+ *
+ * Return: true if the folio has private data attached.
+ */
+static inline bool folio_has_attached_private(const struct folio *folio)
+{
+ /*
+ * Swapcache stores swp_entry_t in folio->swap, a union with
+ * folio->private, and hugetlb stores its own flags in folio->private;
+ * both are excluded.
+ *
+ * NOTE: For swapcache, folio->swap.val PG_swapcache are not set as
+ * a whole, so folio_test_swapcache() is not reliable to exclude
+ * swapcache. Use folio_test_swapbacked() instead, since it remains set
+ * when a folio is added to/removed from swapcache.
+ */
+
+ return folio_test_private(folio) && !folio_test_swapbacked(folio) &&
+ !folio_test_hugetlb(folio);
+}
/**
* folio_has_private - Determine if folio has private stuff
* @folio: The folio to be checked
@@ -1205,7 +1233,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page)
*/
static inline int folio_has_private(const struct folio *folio)
{
- return !!(folio->flags.f & PAGE_FLAGS_PRIVATE);
+ return folio_has_attached_private(folio) || folio_test_private_2(folio);
}
#undef PF_ANY
diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h
index d649b6bbbc87..2baf7a2b29b2 100644
--- a/include/linux/page_counter.h
+++ b/include/linux/page_counter.h
@@ -63,11 +63,12 @@ static inline void page_counter_init(struct page_counter *counter,
counter->track_failcnt = false;
}
-static inline unsigned long page_counter_read(struct page_counter *counter)
+static inline unsigned long page_counter_read(const struct page_counter *counter)
{
return atomic_long_read(&counter->usage);
}
+long page_counter_margin(const struct page_counter *counter);
void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages);
void page_counter_charge(struct page_counter *counter, unsigned long nr_pages);
bool page_counter_try_charge(struct page_counter *counter,
diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h
index 0adfa6605653..73af18a37367 100644
--- a/include/linux/pagemap.h
+++ b/include/linux/pagemap.h
@@ -14,7 +14,6 @@
#include <linux/gfp.h>
#include <linux/bitops.h>
#include <linux/hardirq.h> /* for in_interrupt() */
-#include <linux/hugetlb_inline.h>
struct folio_batch;
@@ -594,7 +593,6 @@ static inline void folio_attach_private(struct folio *folio, void *data)
{
folio_get(folio);
folio->private = data;
- folio_set_private(folio);
}
/**
@@ -629,9 +627,8 @@ static inline void *folio_detach_private(struct folio *folio)
{
void *data = folio_get_private(folio);
- if (!folio_test_private(folio))
+ if (!data)
return NULL;
- folio_clear_private(folio);
folio->private = NULL;
folio_put(folio);
@@ -1128,8 +1125,7 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma,
const pgoff_t pgoff = __linear_anon_page_index(vma, address);
VM_WARN_ON_ONCE(!vma_is_cow_mapping(vma));
- /* Account for MAP_PRIVATE-/dev/zero which is only semi-anonymous. */
- if (vma_is_anonymous(vma) && !vma->vm_file)
+ if (vma_is_anonymous(vma))
VM_WARN_ON_ONCE(pgoff != linear_page_index(vma, address));
return pgoff;
@@ -1416,6 +1412,7 @@ struct readahead_control {
bool dropbehind;
bool _workingset;
unsigned long _pflags;
+ bool _forward;
};
#define DEFINE_READAHEAD(ractl, f, r, m, i) \
@@ -1480,18 +1477,29 @@ void page_cache_async_readahead(struct address_space *mapping,
page_cache_async_ra(&ractl, folio, req_count);
}
+/*
+ * Adjust readahead_control to ensure next folio comes from
+ * [_index, _index + _nr_pages) afterwards and reset _batch_count.
+ */
+static inline void __readahead_advance(struct readahead_control *rac)
+{
+ if (rac->_forward)
+ rac->_index += rac->_batch_count;
+
+ rac->_nr_pages -= rac->_batch_count;
+ rac->_batch_count = 0;
+}
+
static inline struct folio *__readahead_folio(struct readahead_control *ractl)
{
struct folio *folio;
BUG_ON(ractl->_batch_count > ractl->_nr_pages);
- ractl->_nr_pages -= ractl->_batch_count;
- ractl->_index += ractl->_batch_count;
+ __readahead_advance(ractl);
+ ractl->_forward = true;
- if (!ractl->_nr_pages) {
- ractl->_batch_count = 0;
+ if (!ractl->_nr_pages)
return NULL;
- }
folio = xa_load(&ractl->mapping->i_pages, ractl->_index);
VM_BUG_ON_FOLIO(!folio_test_locked(folio), folio);
@@ -1517,6 +1525,39 @@ static inline struct folio *readahead_folio(struct readahead_control *ractl)
return folio;
}
+/**
+ * readahead_folio_last - Get the next folio to read, from the tail.
+ * @ractl: The current readahead request.
+ *
+ * Like readahead_folio(), but walks the range back-to-front. The folio is
+ * returned locked with its refcount dropped; the caller unlocks it once I/O
+ * completes. Compound folios are returned once, at their head index.
+ *
+ * Context: The folio is locked.
+ * Return: A pointer to the next folio, or %NULL when done.
+ */
+static inline struct folio *readahead_folio_last(struct readahead_control *ractl)
+{
+ struct folio *folio;
+
+ /* Drop the previously returned batch from the remaining range. */
+ __readahead_advance(ractl);
+ ractl->_forward = false;
+
+ if (!ractl->_nr_pages)
+ return NULL;
+
+ /* xa_load() follows sibling entries, so a tail index returns the head */
+ folio = xa_load(&ractl->mapping->i_pages,
+ ractl->_index + ractl->_nr_pages - 1);
+ VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio);
+
+ ractl->_batch_count = folio_nr_pages(folio);
+
+ folio_put(folio);
+ return folio;
+}
+
static inline unsigned int __readahead_batch(struct readahead_control *rac,
struct page **array, unsigned int array_sz)
{
@@ -1525,9 +1566,8 @@ static inline unsigned int __readahead_batch(struct readahead_control *rac,
struct folio *folio;
BUG_ON(rac->_batch_count > rac->_nr_pages);
- rac->_nr_pages -= rac->_batch_count;
- rac->_index += rac->_batch_count;
- rac->_batch_count = 0;
+ __readahead_advance(rac);
+ rac->_forward = true;
xas_set(&xas, rac->_index);
rcu_read_lock();
diff --git a/include/linux/panic.h b/include/linux/panic.h
index f1dd417e54b2..17e61b61c45f 100644
--- a/include/linux/panic.h
+++ b/include/linux/panic.h
@@ -13,7 +13,8 @@ __printf(1, 2)
void panic(const char *fmt, ...) __noreturn __cold;
__printf(1, 0)
void vpanic(const char *fmt, va_list args) __noreturn __cold;
-void nmi_panic(struct pt_regs *regs, const char *msg);
+__printf(2, 3)
+void nmi_panic(struct pt_regs *regs, const char *fmt, ...);
void check_panic_on_warn(const char *origin);
extern void oops_enter(void);
extern void oops_exit(void);
@@ -110,4 +111,6 @@ extern void add_taint(unsigned flag, enum lockdep_ok);
extern int test_taint(unsigned flag);
extern unsigned long get_taint(void);
+void arch_do_panic(void);
+
#endif /* _LINUX_PANIC_H */
diff --git a/include/linux/pci-epf.h b/include/linux/pci-epf.h
index 704e1dc8b30a..226f3e59cf60 100644
--- a/include/linux/pci-epf.h
+++ b/include/linux/pci-epf.h
@@ -163,7 +163,8 @@ enum pci_epf_doorbell_type {
* For MSI-backed doorbells this is the MSI message, while for
* "embedded" doorbells this represents an MMIO write that asserts
* an interrupt on the EP side.
- * @virq: IRQ number of this doorbell message
+ * @virq: IRQ number of this doorbell message. Multiple messages may use the
+ * same IRQ; consumers must request each distinct IRQ only once.
* @irq_flags: Required flags for request_irq()/request_threaded_irq().
* Callers may OR-in additional flags (e.g. IRQF_ONESHOT).
* @type: Doorbell type.
diff --git a/include/linux/pci.h b/include/linux/pci.h
index d31a8d107b1e..d50cf9ae142d 100644
--- a/include/linux/pci.h
+++ b/include/linux/pci.h
@@ -43,6 +43,7 @@
#include <uapi/linux/pci.h>
#include <linux/pci_ids.h>
+#include <linux/pci_liveupdate.h>
#define PCI_STATUS_ERROR_BITS (PCI_STATUS_DETECTED_PARITY | \
PCI_STATUS_SIG_SYSTEM_ERROR | \
@@ -598,6 +599,9 @@ struct pci_dev {
u8 tph_mode; /* TPH mode */
u8 tph_req_type; /* TPH requester type */
#endif
+#ifdef CONFIG_PCI_LIVEUPDATE
+ struct pci_liveupdate liveupdate;
+#endif
};
static inline struct pci_dev *pci_physfn(struct pci_dev *dev)
@@ -832,6 +836,9 @@ static inline struct pci_dev *pci_upstream_bridge(struct pci_dev *dev)
return dev->bus->self;
}
+#define for_each_pci_dev_in_path(dev) \
+ for (; dev; dev = pci_upstream_bridge(dev))
+
#ifdef CONFIG_PCI_MSI
static inline bool pci_dev_msi_enabled(struct pci_dev *pci_dev)
{
@@ -1951,9 +1958,10 @@ int pci_disable_link_state(struct pci_dev *pdev, int state);
int pci_disable_link_state_locked(struct pci_dev *pdev, int state);
int pci_enable_link_state(struct pci_dev *pdev, int state);
int pci_enable_link_state_locked(struct pci_dev *pdev, int state);
+int pci_force_enable_link_state(struct pci_dev *pdev, int state);
void pcie_no_aspm(void);
bool pcie_aspm_support_enabled(void);
-bool pcie_aspm_enabled(struct pci_dev *pdev);
+u32 pcie_aspm_enabled(struct pci_dev *pdev);
#else
static inline int pci_disable_link_state(struct pci_dev *pdev, int state)
{ return 0; }
@@ -1963,9 +1971,11 @@ static inline int pci_enable_link_state(struct pci_dev *pdev, int state)
{ return 0; }
static inline int pci_enable_link_state_locked(struct pci_dev *pdev, int state)
{ return 0; }
+static inline int pci_force_enable_link_state(struct pci_dev *pdev, int state)
+{ return 0; }
static inline void pcie_no_aspm(void) { }
static inline bool pcie_aspm_support_enabled(void) { return false; }
-static inline bool pcie_aspm_enabled(struct pci_dev *pdev) { return false; }
+static inline u32 pcie_aspm_enabled(struct pci_dev *pdev) { return 0; }
#endif
#ifdef CONFIG_HOTPLUG_PCI
diff --git a/include/linux/pci_liveupdate.h b/include/linux/pci_liveupdate.h
new file mode 100644
index 000000000000..bb79348769b7
--- /dev/null
+++ b/include/linux/pci_liveupdate.h
@@ -0,0 +1,72 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * PCI Live Update support (Public/Driver API)
+ *
+ * Copyright (c) 2026, Google LLC.
+ * David Matlack <dmatlack@google.com>
+ */
+#ifndef LINUX_PCI_LIVEUPDATE_H
+#define LINUX_PCI_LIVEUPDATE_H
+
+#include <linux/kho/abi/pci.h>
+#include <linux/liveupdate.h>
+#include <linux/spinlock_types.h>
+#include <linux/types.h>
+
+/**
+ * struct pci_liveupdate - PCI Live Update state for a struct pci_dev
+ * @outgoing: State preserved for the next kernel.
+ * @incoming: State preserved by the previous kernel.
+ * @was_incoming: True if this struct pci_dev was incoming-preserved when it was
+ * set up, i.e. it was matched to state preserved by the previous
+ * kernel. Unlike @incoming, this is never cleared, so it stays
+ * true after the device finishes participating in Live Update.
+ * @frozen: True if the outgoing preservation status of this device is frozen
+ * and thus cannot be changed.
+ */
+struct pci_liveupdate {
+ struct pci_dev_ser *outgoing;
+ struct pci_dev_ser *incoming;
+ bool was_incoming;
+ bool frozen;
+};
+
+struct pci_dev;
+
+#ifdef CONFIG_PCI_LIVEUPDATE
+int pci_liveupdate_register_flb(struct liveupdate_file_handler *fh);
+void pci_liveupdate_unregister_flb(struct liveupdate_file_handler *fh);
+int pci_liveupdate_preserve(struct pci_dev *dev);
+void pci_liveupdate_unpreserve(struct pci_dev *dev);
+void pci_liveupdate_finish(struct pci_dev *dev);
+bool pci_liveupdate_is_incoming(struct pci_dev *dev);
+#else
+static inline int pci_liveupdate_register_flb(struct liveupdate_file_handler *fh)
+{
+ return -EOPNOTSUPP;
+}
+
+static inline void pci_liveupdate_unregister_flb(struct liveupdate_file_handler *fh)
+{
+}
+
+static inline int pci_liveupdate_preserve(struct pci_dev *dev)
+{
+ return -EOPNOTSUPP;
+}
+
+static inline void pci_liveupdate_unpreserve(struct pci_dev *dev)
+{
+}
+
+static inline void pci_liveupdate_finish(struct pci_dev *dev)
+{
+}
+
+static inline bool pci_liveupdate_is_incoming(struct pci_dev *dev)
+{
+ return false;
+}
+#endif
+
+#endif /* LINUX_PCI_LIVEUPDATE_H */
diff --git a/include/linux/perf/riscv_pmu.h b/include/linux/perf/riscv_pmu.h
index f82a28040594..ecaa40370830 100644
--- a/include/linux/perf/riscv_pmu.h
+++ b/include/linux/perf/riscv_pmu.h
@@ -55,7 +55,7 @@ struct riscv_pmu {
irqreturn_t (*handle_irq)(int irq_num, void *dev);
- unsigned long cmask;
+ DECLARE_BITMAP(cmask, RISCV_MAX_COUNTERS);
u64 (*ctr_read)(struct perf_event *event);
int (*ctr_get_idx)(struct perf_event *event);
int (*ctr_get_width)(int idx);
diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h
index 5842552294c1..7797ce207555 100644
--- a/include/linux/perf_event.h
+++ b/include/linux/perf_event.h
@@ -306,6 +306,7 @@ struct perf_event_pmu_context;
#define PERF_PMU_CAP_AUX_PAUSE 0x0200
#define PERF_PMU_CAP_AUX_PREFER_LARGE 0x0400
#define PERF_PMU_CAP_MEDIATED_VPMU 0x0800
+#define PERF_PMU_CAP_SIMD_REGS 0x1000
/**
* pmu::scope
@@ -1467,23 +1468,7 @@ static inline u32 perf_sample_data_size(struct perf_sample_data *data,
return size;
}
-/*
- * Clear all bitfields in the perf_branch_entry.
- * The to and from fields are not cleared because they are
- * systematically modified by caller.
- */
-static inline void perf_clear_branch_entry_bitfields(struct perf_branch_entry *br)
-{
- br->mispred = 0;
- br->predicted = 0;
- br->in_tx = 0;
- br->abort = 0;
- br->cycles = 0;
- br->type = 0;
- br->spec = PERF_BR_SPEC_NA;
- br->reserved = 0;
-}
-
+extern u64 perf_update_xregs_size(struct perf_event *event, bool intr);
extern void perf_output_sample(struct perf_output_handle *handle,
struct perf_event_header *header,
struct perf_sample_data *data,
@@ -1534,6 +1519,27 @@ perf_event__output_id_sample(struct perf_event *event,
extern void
perf_log_lost_samples(struct perf_event *event, u64 lost);
+static inline bool event_has_simd_regs(struct perf_event *event)
+{
+ struct perf_event_attr *attr = &event->attr;
+
+ if (!(event->attr.sample_type &
+ (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)))
+ return false;
+
+ return attr->sample_simd_regs_enabled != 0;
+}
+
+static inline bool event_has_extended_regs(struct perf_event *event)
+{
+ struct perf_event_attr *attr = &event->attr;
+
+ return ((attr->sample_type & PERF_SAMPLE_REGS_USER) &&
+ (attr->sample_regs_user & PERF_REG_EXTENDED_MASK)) ||
+ ((attr->sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK));
+}
+
static inline bool event_has_any_exclude_flag(struct perf_event *event)
{
struct perf_event_attr *attr = &event->attr;
diff --git a/include/linux/perf_regs.h b/include/linux/perf_regs.h
index f632c5725f16..09dbc2fc3859 100644
--- a/include/linux/perf_regs.h
+++ b/include/linux/perf_regs.h
@@ -9,6 +9,16 @@ struct perf_regs {
struct pt_regs *regs;
};
+u64 perf_reg_value(struct pt_regs *regs, int idx);
+int perf_reg_validate(u64 mask, bool simd_enabled);
+u64 perf_reg_abi(struct task_struct *task);
+void perf_get_regs_user(struct perf_regs *regs_user,
+ struct pt_regs *regs);
+int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
+ u16 pred_qwords, u32 pred_mask);
+u64 perf_simd_reg_value(struct pt_regs *regs, int idx,
+ u16 qwords_idx, bool pred);
+
#ifdef CONFIG_HAVE_PERF_REGS
#include <asm/perf_regs.h>
@@ -16,35 +26,9 @@ struct perf_regs {
#define PERF_REG_EXTENDED_MASK 0
#endif
-u64 perf_reg_value(struct pt_regs *regs, int idx);
-int perf_reg_validate(u64 mask);
-u64 perf_reg_abi(struct task_struct *task);
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs);
#else
#define PERF_REG_EXTENDED_MASK 0
-static inline u64 perf_reg_value(struct pt_regs *regs, int idx)
-{
- return 0;
-}
-
-static inline int perf_reg_validate(u64 mask)
-{
- return mask ? -ENOSYS : 0;
-}
-
-static inline u64 perf_reg_abi(struct task_struct *task)
-{
- return PERF_SAMPLE_REGS_ABI_NONE;
-}
-
-static inline void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
#endif /* CONFIG_HAVE_PERF_REGS */
#endif /* _LINUX_PERF_REGS_H */
diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h
index 8c093c119e5a..780fe849ff8b 100644
--- a/include/linux/pgtable.h
+++ b/include/linux/pgtable.h
@@ -490,35 +490,35 @@ static inline int pudp_set_access_flags(struct vm_area_struct *vma,
#endif
#ifndef ptep_get
-static inline pte_t ptep_get(pte_t *ptep)
+static inline pte_t ptep_get(const pte_t *ptep)
{
return READ_ONCE(*ptep);
}
#endif
#ifndef pmdp_get
-static inline pmd_t pmdp_get(pmd_t *pmdp)
+static inline pmd_t pmdp_get(const pmd_t *pmdp)
{
return READ_ONCE(*pmdp);
}
#endif
#ifndef pudp_get
-static inline pud_t pudp_get(pud_t *pudp)
+static inline pud_t pudp_get(const pud_t *pudp)
{
return READ_ONCE(*pudp);
}
#endif
#ifndef p4dp_get
-static inline p4d_t p4dp_get(p4d_t *p4dp)
+static inline p4d_t p4dp_get(const p4d_t *p4dp)
{
return READ_ONCE(*p4dp);
}
#endif
#ifndef pgdp_get
-static inline pgd_t pgdp_get(pgd_t *pgdp)
+static inline pgd_t pgdp_get(const pgd_t *pgdp)
{
return READ_ONCE(*pgdp);
}
@@ -2313,6 +2313,20 @@ static inline const char *pgtable_level_to_str(enum pgtable_level level)
}
}
+void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size);
+
+#define ptval_to_str(buf, val) \
+ do { \
+ auto __val = (val); \
+ \
+ ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \
+ } while (0)
+
+#if defined(__SIZEOF_INT128__)
+#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */
+#else
+#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */
+#endif
#endif /* !__ASSEMBLER__ */
#if !defined(MAX_POSSIBLE_PHYSMEM_BITS) && !defined(CONFIG_64BIT)
diff --git a/include/linux/phy.h b/include/linux/phy.h
index 5f8d65868e0f..7c5098a0dd6c 100644
--- a/include/linux/phy.h
+++ b/include/linux/phy.h
@@ -376,6 +376,24 @@ struct mii_bus {
int regnum, u16 val);
/** @reset: Perform a reset of the bus */
int (*reset)(struct mii_bus *bus);
+ /**
+ * @notify_phy_attach: Perform post-attach handling for MDIO bus
+ * drivers. Optional and independent of @notify_phy_detach. Called
+ * in phy_attach_direct() right before phy_resume(). Runs in process
+ * context, may sleep and may be called with RTNL held. Must not
+ * acquire or rely on RTNL. Returns 0 on success or negative errno
+ * on failure. Must unwind its own state on error as attachment is
+ * aborted.
+ */
+ int (*notify_phy_attach)(struct phy_device *phydev);
+ /**
+ * @notify_phy_detach: Perform pre-detach handling for MDIO bus
+ * drivers. Optional and independent of @notify_phy_attach. Called
+ * in phy_detach() right after phy_suspend(). Runs in process context,
+ * may sleep and may be called with RTNL held. Must not acquire or
+ * rely on RTNL.
+ */
+ void (*notify_phy_detach)(struct phy_device *phydev);
/** @stats: Statistic counters per device on the bus */
struct mdio_bus_stats stats[PHY_MAX_ADDR];
@@ -1697,7 +1715,7 @@ static inline bool phy_can_wakeup(struct phy_device *phydev)
* phy_may_wakeup() - indicate whether PHY has wakeup enabled
* @phydev: The phy_device struct
*
- * Returns: true/false depending on the PHY driver's device_set_wakeup_enabled()
+ * Returns: true/false depending on the PHY driver's device_set_wakeup_enable()
* setting if using the driver model, otherwise the legacy determination.
*/
bool phy_may_wakeup(struct phy_device *phydev);
@@ -2422,10 +2440,10 @@ int phy_get_mac_termination(struct phy_device *phydev, struct device *dev,
void phy_resolve_pause(unsigned long *local_adv, unsigned long *partner_adv,
bool *tx_pause, bool *rx_pause);
-int phy_register_fixup_for_id(const char *bus_id,
- int (*run)(struct phy_device *));
-int phy_register_fixup_for_uid(u32 phy_uid, u32 phy_uid_mask,
- int (*run)(struct phy_device *));
+void __init phy_register_fixup_for_id(const char *bus_id,
+ int (*run)(struct phy_device *));
+void __init phy_register_fixup_for_uid(u32 phy_uid, u32 phy_uid_mask,
+ int (*run)(struct phy_device *));
int phy_eee_tx_clock_stop_capable(struct phy_device *phydev);
int phy_eee_rx_clock_stop(struct phy_device *phydev, bool clk_stop_enable);
diff --git a/include/linux/phy/phy.h b/include/linux/phy/phy.h
index ea47975e288a..14b924a88411 100644
--- a/include/linux/phy/phy.h
+++ b/include/linux/phy/phy.h
@@ -284,6 +284,8 @@ struct phy *devm_of_phy_optional_get(struct device *dev, struct device_node *np,
const char *con_id);
struct phy *devm_of_phy_get_by_index(struct device *dev, struct device_node *np,
int index);
+struct phy *phy_get_by_of_node(struct device_node *np);
+struct phy *devm_phy_get_by_of_node(struct device *dev, struct device_node *np);
void of_phy_put(struct phy *phy);
void phy_put(struct device *dev, struct phy *phy);
void devm_phy_put(struct device *dev, struct phy *phy);
@@ -493,6 +495,17 @@ static inline struct phy *devm_of_phy_get_by_index(struct device *dev,
return ERR_PTR(-ENOSYS);
}
+static inline struct phy *phy_get_by_of_node(struct device_node *np)
+{
+ return ERR_PTR(-ENOSYS);
+}
+
+static inline struct phy *devm_phy_get_by_of_node(struct device *dev,
+ struct device_node *np)
+{
+ return ERR_PTR(-ENOSYS);
+}
+
static inline void of_phy_put(struct phy *phy)
{
}
diff --git a/include/linux/phylink.h b/include/linux/phylink.h
index 1dda5c7ed5f1..3a88a69882a6 100644
--- a/include/linux/phylink.h
+++ b/include/linux/phylink.h
@@ -843,4 +843,6 @@ void phylink_replay_link_begin(struct phylink *pl);
void phylink_replay_link_end(struct phylink *pl);
+void phylink_update_mac_pause_capabilities(struct phylink *pl, unsigned long mac_pause);
+
#endif
diff --git a/include/linux/platform_data/microchip-ksz.h b/include/linux/platform_data/microchip-ksz.h
index 028781ad4059..411d5e164383 100644
--- a/include/linux/platform_data/microchip-ksz.h
+++ b/include/linux/platform_data/microchip-ksz.h
@@ -31,9 +31,11 @@ enum ksz_chip_id {
KSZ88X3_CHIP_ID = 0x8830,
KSZ8864_CHIP_ID = 0x8864,
KSZ8895_CHIP_ID = 0x8895,
+ KSZ8995XA_CHIP_ID = 0x8995,
KSZ9477_CHIP_ID = 0x00947700,
KSZ9896_CHIP_ID = 0x00989600,
KSZ9897_CHIP_ID = 0x00989700,
+ KSZ9897S_CHIP_ID = 0x00989701,
KSZ9893_CHIP_ID = 0x00989300,
KSZ9563_CHIP_ID = 0x00956300,
KSZ8567_CHIP_ID = 0x00856700,
diff --git a/include/linux/platform_data/mipi-i3c-hci.h b/include/linux/platform_data/mipi-i3c-hci.h
index ab7395f455f9..6f61d8a02378 100644
--- a/include/linux/platform_data/mipi-i3c-hci.h
+++ b/include/linux/platform_data/mipi-i3c-hci.h
@@ -3,13 +3,17 @@
#define INCLUDE_PLATFORM_DATA_MIPI_I3C_HCI_H
#include <linux/compiler_types.h>
+#include <linux/types.h>
/**
* struct mipi_i3c_hci_platform_data - Platform-dependent data for mipi_i3c_hci
* @base_regs: Register set base address (to support multi-bus instances)
+ * @instance: Zero-based instance number of the Bus Controller as defined by the
+ * DisCo specification I3C Target Address (_ADR) Encoding
*/
struct mipi_i3c_hci_platform_data {
void __iomem *base_regs;
+ u8 instance;
};
#endif
diff --git a/include/linux/platform_data/x86/asus-wmi.h b/include/linux/platform_data/x86/asus-wmi.h
index b5ed8c83ace1..be4d2873ffc9 100644
--- a/include/linux/platform_data/x86/asus-wmi.h
+++ b/include/linux/platform_data/x86/asus-wmi.h
@@ -54,6 +54,7 @@
#define ASUS_WMI_DEVID_LED5 0x00020015
#define ASUS_WMI_DEVID_LED6 0x00020016
#define ASUS_WMI_DEVID_MICMUTE_LED 0x00040017
+#define ASUS_WMI_DEVID_MUTE_LED 0x0004001C
/* Disable Camera LED */
#define ASUS_WMI_DEVID_CAMERA_LED_NEG 0x00060078 /* 0 = on (unused) */
@@ -140,8 +141,9 @@
#define ASUS_WMI_DEVID_APU_MEM 0x000600C1
-#define ASUS_WMI_DEVID_DGPU_BASE_TGP 0x00120099
+#define ASUS_WMI_DEVID_DGPU_POWER_STATE 0x00120097
#define ASUS_WMI_DEVID_DGPU_SET_TGP 0x00120098
+#define ASUS_WMI_DEVID_DGPU_BASE_TGP 0x00120099
/* gpu mux switch, 0 = dGPU, 1 = Optimus */
#define ASUS_WMI_DEVID_GPU_MUX 0x00090016
diff --git a/include/linux/platform_data/x86/int3472.h b/include/linux/platform_data/x86/int3472.h
index a73841dfae27..b1040e36deb8 100644
--- a/include/linux/platform_data/x86/int3472.h
+++ b/include/linux/platform_data/x86/int3472.h
@@ -25,6 +25,8 @@
#define INT3472_GPIO_TYPE_RESET 0x00
#define INT3472_GPIO_TYPE_POWERDOWN 0x01
#define INT3472_GPIO_TYPE_STROBE 0x02
+#define INT3472_GPIO_TYPE_POWER0 0x07
+#define INT3472_GPIO_TYPE_POWER1 0x08
#define INT3472_GPIO_TYPE_POWER_ENABLE 0x0b
#define INT3472_GPIO_TYPE_CLK_ENABLE 0x0c
#define INT3472_GPIO_TYPE_PRIVACY_LED 0x0d
diff --git a/include/linux/platform_device.h b/include/linux/platform_device.h
index 3d5bbcbae730..3bd0f1f0d870 100644
--- a/include/linux/platform_device.h
+++ b/include/linux/platform_device.h
@@ -344,6 +344,19 @@ static inline void platform_set_drvdata(struct platform_device *pdev,
#define builtin_platform_driver(__platform_driver) \
builtin_driver(__platform_driver, platform_driver_register)
+/*
+ * subsys_platform_driver() - Helper macro for drivers that don't do anything
+ * special in module init/exit but need to register earlier, at
+ * subsys_initcall level, when built in. This eliminates a lot of
+ * boilerplate. Each driver may only use this macro once, and calling it
+ * replaces module_init() and module_exit(). This is meant to be a parallel
+ * of module_platform_driver() above, but with the init call promoted to
+ * subsys_initcall() so built-in providers are available earlier during boot.
+ */
+#define subsys_platform_driver(__platform_driver) \
+ subsys_driver(__platform_driver, platform_driver_register, \
+ platform_driver_unregister)
+
/* module_platform_driver_probe() - Helper macro for drivers that don't do
* anything special in module init/exit. This eliminates a lot of
* boilerplate. Each module may only use this macro once, and
@@ -376,6 +389,23 @@ static int __init __platform_driver##_init(void) \
} \
device_initcall(__platform_driver##_init); \
+/*
+ * subsys_platform_driver_probe() - Helper macro for drivers that don't do
+ * anything special in device init and have no exit, but need to register
+ * earlier, at subsys_initcall level. This eliminates some boilerplate. Each
+ * driver may only use this macro once, and using it replaces subsys_initcall.
+ * This is meant to be a parallel of builtin_platform_driver_probe above, but
+ * with the init call promoted to subsys_initcall so the provider is available
+ * earlier during boot.
+ */
+#define subsys_platform_driver_probe(__platform_driver, __platform_probe) \
+static int __init __platform_driver##_init(void) \
+{ \
+ return platform_driver_probe(&(__platform_driver), \
+ __platform_probe); \
+} \
+subsys_initcall(__platform_driver##_init) \
+
#define platform_create_bundle(driver, probe, res, n_res, data, size) \
__platform_create_bundle(driver, probe, res, n_res, data, size, THIS_MODULE, KBUILD_MODNAME)
extern struct platform_device *__platform_create_bundle(
diff --git a/include/linux/pm.h b/include/linux/pm.h
index afcaaa37a812..ef3f1310e749 100644
--- a/include/linux/pm.h
+++ b/include/linux/pm.h
@@ -663,6 +663,99 @@ struct pm_subsys_data {
#define DPM_FLAG_SMART_SUSPEND BIT(2)
#define DPM_FLAG_MAY_SKIP_RESUME BIT(3)
+/**
+ * struct dev_pm_info - Device power management information.
+ *
+ * @power_state: Legacy power state (mostly unused in modern kernels).
+ * @can_wakeup: Device is capable of generating wakeup signals.
+ * @async_suspend: Device can be suspended and resumed asynchronously.
+ * @in_dpm_list: Device is on the dpm_list.
+ * @is_prepared: Device's ->prepare() callback has run successfully.
+ * @is_suspended: Device is suspended during a system sleep transition.
+ * @is_noirq_suspended: Device's noirq suspend callback has run successfully.
+ * @is_late_suspended: Device's late suspend callback has run successfully.
+ * @no_pm: Device does not participate in power management transitions.
+ * @early_init: Device was initialized before standard PM initialization.
+ * @direct_complete: Device can skip suspend/resume callbacks and remain
+ * runtime-suspended during system sleep.
+ * @driver_flags: Driver flags (e.g. %DPM_FLAG_SMART_SUSPEND) set at probe time.
+ * @lock: Spinlock used for synchronizing PM state transitions and runtime PM
+ * operations.
+ * @entry: List head for device power management lists.
+ * @completion: Completion for synchronization during asynchronous system
+ * suspend/resume.
+ * @wakeup: Wakeup source object associated with the device.
+ * @work_in_progress: Asynchronous PM operation in progress.
+ * @wakeup_path: Device is in the wakeup path or can wake the system up.
+ * @syscore: Device participates in syscore power management operations.
+ * @no_pm_callbacks: Device has no PM callbacks; handled by parent or subsystem.
+ * @smart_suspend: Driver requested smart-suspend behavior.
+ * @must_resume: Device must be resumed during system resume.
+ * @may_skip_resume: Set by subsystems to indicate driver resume callbacks may
+ * be skipped.
+ * @out_band_wakeup: Out-of-band wakeup is supported.
+ * @strict_midlayer: Middle layer code does not want callbacks invoked via
+ * pm_runtime_force_suspend() / pm_runtime_force_resume().
+ * @should_wakeup: Wakeup flag when system sleep is not enabled.
+ * @suspend_timer: High-resolution timer used for scheduling delayed runtime
+ * suspend and autosuspend requests.
+ * @timer_expires: Timer expiration time in nanoseconds monotonic time
+ * (runtime PM).
+ * @work: Work structure used for queuing up requests into pm_wq (runtime PM).
+ * @wait_queue: Wait queue used if any helper functions need to wait for another
+ * state change to complete (runtime PM).
+ * @wakeirq: Dedicated wakeup interrupt for the device.
+ * @usage_count: Device runtime PM usage counter.
+ * @child_count: Count of active children of the device (runtime PM).
+ * @disable_depth: Disable counter for runtime PM (runtime PM is enabled when
+ * this is 0; initial value is 1).
+ * @idle_notification: Set if ->runtime_idle() is being executed.
+ * @request_pending: Set if a work item is queued into pm_wq (runtime PM).
+ * @deferred_resume: Set if ->runtime_resume() should run as soon as
+ * ->runtime_suspend() completes.
+ * @needs_force_resume: Indicates the device was forced into suspend by
+ * pm_runtime_force_suspend() and must be resumed by
+ * pm_runtime_force_resume().
+ * @runtime_auto: User space has allowed the driver to power manage the device
+ * at runtime via sysfs control attribute; also can be set by
+ * pm_runtime_allow() or pm_runtime_forbid().
+ * @ignore_children: If set, the value of child_count is ignored for runtime
+ * suspend and idle decisions.
+ * @no_callbacks: Indicates the device does not use runtime PM callbacks.
+ * @irq_safe: Indicates runtime PM callbacks will be invoked with the spinlock
+ * held and interrupts disabled.
+ * @use_autosuspend: Indicates the device driver supports delayed runtime
+ * autosuspend.
+ * @timer_autosuspends: Indicates the runtime PM core should attempt an
+ * autosuspend rather than a normal suspend when the timer expires.
+ * @memalloc_noio: Indicates memory allocation during runtime PM transitions
+ * must avoid I/O (GFP_NOIO).
+ * @links_count: Number of device links that require runtime PM coordination.
+ * @request: Type of pending runtime PM request (valid if request_pending is
+ * set).
+ * @runtime_status: Runtime PM status of the device.
+ * @last_status: Last status captured before disabling runtime PM, or
+ * %RPM_BLOCKED / %RPM_INVALID.
+ * @runtime_error: Fatal error code returned by a failing callback, blocking
+ * helpers until cleared.
+ * @autosuspend_delay: Delay time in milliseconds to be used for runtime
+ * autosuspend.
+ * @last_busy: Timestamp in nanoseconds when pm_runtime_mark_last_busy() was
+ * last called. Used in calculating inactivity periods for autosuspend.
+ * @active_time: Accumulated time in nanoseconds spent in %RPM_ACTIVE state.
+ * @suspended_time: Accumulated time in nanoseconds spent in %RPM_SUSPENDED
+ * state.
+ * @accounting_timestamp: Timestamp in nanoseconds of the last runtime PM state
+ * accounting update.
+ * @subsys_data: Subsystem-specific power management data.
+ * @set_latency_tolerance: Callback for setting latency tolerance.
+ * @qos: Per-device PM Quality of Service (QoS) constraints.
+ * @detach_power_off: Indicates device should be detached from PM domain on
+ * power off.
+ *
+ * Device power management information stored in the "power" member of struct
+ * device.
+ */
struct dev_pm_info {
pm_message_t power_state;
bool can_wakeup:1;
diff --git a/include/linux/pm_domain.h b/include/linux/pm_domain.h
index f925614aebdb..14e0e346c610 100644
--- a/include/linux/pm_domain.h
+++ b/include/linux/pm_domain.h
@@ -121,6 +121,14 @@ struct dev_pm_domain_list {
* powered-off until the ->sync_state() callback is
* invoked. This flag informs genpd to allow a
* power-off without waiting for ->sync_state().
+ *
+ * GENPD_FLAG_POWER_UNKNOWN: Use this flag to inform genpd that its initial
+ * status for the PM domain is set to powered off,
+ * which may not correctly reflect the state of the
+ * HW, as it's unknown. If the PM domain becomes
+ * powered on during boot, genpd will prevent it
+ * from being powered off until the ->sync_state
+ * callback is invoked for it.
*/
#define GENPD_FLAG_PM_CLK (1U << 0)
#define GENPD_FLAG_IRQ_SAFE (1U << 1)
@@ -133,6 +141,7 @@ struct dev_pm_domain_list {
#define GENPD_FLAG_DEV_NAME_FW (1U << 8)
#define GENPD_FLAG_NO_SYNC_STATE (1U << 9)
#define GENPD_FLAG_NO_STAY_ON (1U << 10)
+#define GENPD_FLAG_POWER_UNKNOWN (1U << 11)
enum gpd_status {
GENPD_STATE_ON = 0, /* PM domain is on */
diff --git a/include/linux/pm_qos.h b/include/linux/pm_qos.h
index 6cea4455f867..439a9e779d81 100644
--- a/include/linux/pm_qos.h
+++ b/include/linux/pm_qos.h
@@ -37,6 +37,8 @@ enum pm_qos_flags_status {
#define PM_QOS_LATENCY_TOLERANCE_NO_CONSTRAINT (-1)
#define PM_QOS_FLAG_NO_POWER_OFF (1 << 0)
+/* latency value applies to system-wide suspend/s2idle */
+#define PM_QOS_FLAG_LATENCY_SYS (2 << 0)
enum pm_qos_type {
PM_QOS_UNITIALIZED,
@@ -217,6 +219,12 @@ static inline s32 dev_pm_qos_raw_resume_latency(struct device *dev)
PM_QOS_RESUME_LATENCY_NO_CONSTRAINT :
pm_qos_read_value(&dev->power.qos->resume_latency);
}
+
+static inline s32 dev_pm_qos_raw_flags(struct device *dev)
+{
+ return IS_ERR_OR_NULL(dev->power.qos) ?
+ 0 : READ_ONCE(dev->power.qos->flags.effective_flags);
+}
#else
static inline enum pm_qos_flags_status __dev_pm_qos_flags(struct device *dev,
s32 mask)
@@ -298,6 +306,7 @@ static inline s32 dev_pm_qos_raw_resume_latency(struct device *dev)
{
return PM_QOS_RESUME_LATENCY_NO_CONSTRAINT;
}
+static inline s32 dev_pm_qos_raw_flags(struct device *dev) { return 0; }
#endif
static inline int freq_qos_request_active(struct freq_qos_request *req)
diff --git a/include/linux/pm_runtime.h b/include/linux/pm_runtime.h
index 64921b10ac74..322e3b17f987 100644
--- a/include/linux/pm_runtime.h
+++ b/include/linux/pm_runtime.h
@@ -137,13 +137,14 @@ static inline void pm_runtime_put_noidle(struct device *dev)
* pm_runtime_suspended - Check whether or not a device is runtime-suspended.
* @dev: Target device.
*
- * Return %true if runtime PM is enabled for @dev and its runtime PM status is
- * %RPM_SUSPENDED, or %false otherwise.
- *
* Note that the return value of this function can only be trusted if it is
* called under the runtime PM lock of @dev or under conditions in which
* runtime PM cannot be either disabled or enabled for @dev and its runtime PM
* status cannot change.
+ *
+ * Return:
+ * * %true: @dev has runtime PM enabled and its status is %RPM_SUSPENDED.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_suspended(struct device *dev)
{
@@ -155,13 +156,14 @@ static inline bool pm_runtime_suspended(struct device *dev)
* pm_runtime_active - Check whether or not a device is runtime-active.
* @dev: Target device.
*
- * Return %true if runtime PM is disabled for @dev or its runtime PM status is
- * %RPM_ACTIVE, or %false otherwise.
- *
* Note that the return value of this function can only be trusted if it is
* called under the runtime PM lock of @dev or under conditions in which
* runtime PM cannot be either disabled or enabled for @dev and its runtime PM
* status cannot change.
+ *
+ * Return:
+ * * %true: Runtime PM is disabled for @dev or its status is %RPM_ACTIVE.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_active(struct device *dev)
{
@@ -173,12 +175,13 @@ static inline bool pm_runtime_active(struct device *dev)
* pm_runtime_status_suspended - Check if runtime PM status is "suspended".
* @dev: Target device.
*
- * Return %true if the runtime PM status of @dev is %RPM_SUSPENDED, or %false
- * otherwise, regardless of whether or not runtime PM has been enabled for @dev.
- *
* Note that the return value of this function can only be trusted if it is
* called under the runtime PM lock of @dev or under conditions in which the
* runtime PM status of @dev cannot change.
+ *
+ * Return:
+ * * %true: Runtime PM status of @dev is %RPM_SUSPENDED.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_status_suspended(struct device *dev)
{
@@ -189,11 +192,13 @@ static inline bool pm_runtime_status_suspended(struct device *dev)
* pm_runtime_enabled - Check if runtime PM is enabled.
* @dev: Target device.
*
- * Return %true if runtime PM is enabled for @dev or %false otherwise.
- *
* Note that the return value of this function can only be trusted if it is
* called under the runtime PM lock of @dev or under conditions in which
* runtime PM cannot be either disabled or enabled for @dev.
+ *
+ * Return:
+ * * %true: Runtime PM is enabled for @dev.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_enabled(struct device *dev)
{
@@ -205,6 +210,10 @@ static inline bool pm_runtime_enabled(struct device *dev)
* @dev: Target device.
*
* Do not call this function outside system suspend/resume code paths.
+ *
+ * Return:
+ * * %true: Runtime PM enabling is blocked for @dev.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_blocked(struct device *dev)
{
@@ -215,8 +224,9 @@ static inline bool pm_runtime_blocked(struct device *dev)
* pm_runtime_has_no_callbacks - Check if runtime PM callbacks may be present.
* @dev: Target device.
*
- * Return %true if @dev is a special device without runtime PM callbacks or
- * %false otherwise.
+ * Return:
+ * * %true: @dev is marked as having no runtime PM callbacks.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_has_no_callbacks(struct device *dev)
{
@@ -239,9 +249,11 @@ static inline void pm_runtime_mark_last_busy(struct device *dev)
* pm_runtime_is_irq_safe - Check if runtime PM can work in interrupt context.
* @dev: Target device.
*
- * Return %true if @dev has been marked as an "IRQ-safe" device (with respect
- * to runtime PM), in which case its runtime PM callabcks can be expected to
- * work correctly when invoked from interrupt handlers.
+ * Return:
+ * * %true: @dev has been marked as an "IRQ-safe" device, in which case its
+ * runtime PM callbacks can be expected to work correctly from interrupt
+ * handlers.
+ * * %false: Otherwise.
*/
static inline bool pm_runtime_is_irq_safe(struct device *dev)
{
@@ -340,25 +352,25 @@ static inline int pm_runtime_force_resume(struct device *dev) { return -ENXIO; }
#endif /* CONFIG_PM_SLEEP */
/**
- * pm_runtime_idle - Conditionally set up autosuspend of a device or suspend it.
+ * pm_runtime_idle - Conditionally initiate autosuspend of a device or suspend it.
* @dev: Target device.
*
* Invoke the "idle check" callback of @dev and, depending on its return value,
- * set up autosuspend of @dev or suspend it (depending on whether or not
+ * initiate autosuspend of @dev or suspend it (depending on whether or not
* autosuspend has been enabled for it).
*
* Return:
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change
- * ongoing or device not in %RPM_ACTIVE state.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -EINPROGRESS: Suspend already in progress.
- * * -ENOSYS: CONFIG_PM not enabled.
- * Other values and conditions for the above values are possible as returned by
- * Runtime PM idle and suspend callbacks.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change
+ * ongoing or device not in %RPM_ACTIVE state.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-EINPROGRESS: Suspend already in progress.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
+ * * Other values and conditions for the above values are possible as returned
+ * by Runtime PM idle and suspend callbacks.
*/
static inline int pm_runtime_idle(struct device *dev)
{
@@ -370,17 +382,17 @@ static inline int pm_runtime_idle(struct device *dev)
* @dev: Target device.
*
* Return:
- * * 1: Success; device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change
- * ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -ENOSYS: CONFIG_PM not enabled.
- * Other values and conditions for the above values are possible as returned by
- * Runtime PM suspend callbacks.
+ * * %1: Success; device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change
+ * ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
+ * * Other values and conditions for the above values are possible as returned
+ * by Runtime PM suspend callbacks.
*/
static inline int pm_runtime_suspend(struct device *dev)
{
@@ -388,26 +400,26 @@ static inline int pm_runtime_suspend(struct device *dev)
}
/**
- * pm_runtime_autosuspend - Update the last access time and set up autosuspend
+ * pm_runtime_autosuspend - Update the last access time and initiate autosuspend
* of a device.
* @dev: Target device.
*
- * First update the last access time, then set up autosuspend of @dev or suspend
- * it (depending on whether or not autosuspend is enabled for it) without
- * engaging its "idle check" callback.
+ * First update the last access time, then initiate autosuspend of @dev or
+ * suspend it (depending on whether or not autosuspend is enabled for it)
+ * without engaging its "idle check" callback.
*
* Return:
- * * 1: Success; device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change
- * ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -ENOSYS: CONFIG_PM not enabled.
- * Other values and conditions for the above values are possible as returned by
- * Runtime PM suspend callbacks.
+ * * %1: Success; device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change
+ * ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
+ * * Other values and conditions for the above values are possible as returned
+ * by Runtime PM suspend callbacks.
*/
static inline int pm_runtime_autosuspend(struct device *dev)
{
@@ -418,6 +430,11 @@ static inline int pm_runtime_autosuspend(struct device *dev)
/**
* pm_runtime_resume - Resume a device synchronously.
* @dev: Target device.
+ *
+ * Return:
+ * * %1: Success; @dev is already %RPM_ACTIVE.
+ * * %0: Success.
+ * * Error code on failure.
*/
static inline int pm_runtime_resume(struct device *dev)
{
@@ -425,22 +442,22 @@ static inline int pm_runtime_resume(struct device *dev)
}
/**
- * pm_request_idle - Queue up "idle check" execution for a device.
+ * pm_request_idle - Request an asynchronous idle check for a device.
* @dev: Target device.
*
- * Queue up a work item to run an equivalent of pm_runtime_idle() for @dev
- * asynchronously.
+ * Asynchronously request the PM core to evaluate whether @dev can be idled
+ * or suspended, invoking its ->runtime_idle() callback if provided.
*
* Return:
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change
- * ongoing or device not in %RPM_ACTIVE state.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -EINPROGRESS: Suspend already in progress.
- * * -ENOSYS: CONFIG_PM not enabled.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change
+ * ongoing or device not in %RPM_ACTIVE state.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-EINPROGRESS: Suspend already in progress.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
*/
static inline int pm_request_idle(struct device *dev)
{
@@ -448,8 +465,16 @@ static inline int pm_request_idle(struct device *dev)
}
/**
- * pm_request_resume - Queue up runtime-resume of a device.
+ * pm_request_resume - Request an asynchronous runtime resume for a device.
* @dev: Target device.
+ *
+ * Asynchronously request the PM core to resume @dev to %RPM_ACTIVE state
+ * without modifying its usage counter.
+ *
+ * Return:
+ * * %1: Success; @dev is already %RPM_ACTIVE.
+ * * %0: Success.
+ * * Error code on failure.
*/
static inline int pm_request_resume(struct device *dev)
{
@@ -457,24 +482,23 @@ static inline int pm_request_resume(struct device *dev)
}
/**
- * pm_request_autosuspend - Update the last access time and queue up autosuspend
- * of a device.
+ * pm_request_autosuspend - Update access time and request delayed suspension.
* @dev: Target device.
*
- * Update the last access time of a device and queue up a work item to run an
- * equivalent pm_runtime_autosuspend() for @dev asynchronously.
+ * Update the last access time of @dev and asynchronously request the PM core
+ * to suspend it after the autosuspend delay has elapsed.
*
* Return:
- * * 1: Success; device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change
- * ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -EINPROGRESS: Suspend already in progress.
- * * -ENOSYS: CONFIG_PM not enabled.
+ * * %1: Success; device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change
+ * ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-EINPROGRESS: Suspend already in progress.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
*/
static inline int pm_request_autosuspend(struct device *dev)
{
@@ -483,11 +507,16 @@ static inline int pm_request_autosuspend(struct device *dev)
}
/**
- * pm_runtime_get - Bump up usage counter and queue up resume of a device.
+ * pm_runtime_get - Increment usage counter and request asynchronous resume.
* @dev: Target device.
*
- * Bump up the runtime PM usage counter of @dev and queue up a work item to
- * carry out runtime-resume of it.
+ * Increment the runtime PM usage counter of @dev and, if the device is
+ * currently suspended, asynchronously request the PM core to resume it.
+ *
+ * Return:
+ * * %1: Success; @dev is already %RPM_ACTIVE.
+ * * %0: Success; runtime-resume was queued.
+ * * Error code on failure.
*/
static inline int pm_runtime_get(struct device *dev)
{
@@ -501,12 +530,15 @@ static inline int pm_runtime_get(struct device *dev)
* Bump up the runtime PM usage counter of @dev and carry out runtime-resume of
* it synchronously.
*
- * The possible return values of this function are the same as for
- * pm_runtime_resume() and the runtime PM usage counter of @dev remains
- * incremented in all cases, even if it returns an error code.
- * Consider using pm_runtime_resume_and_get() instead of it, especially
- * if its return value is checked by the caller, as this is likely to result
- * in cleaner code.
+ * Note that the runtime PM usage counter of @dev remains incremented in all
+ * cases, even if it returns an error code. Consider using
+ * pm_runtime_resume_and_get() instead, especially if the return value is
+ * checked by the caller, as this is likely to result in cleaner code.
+ *
+ * Return:
+ * * %1: Success; @dev is already %RPM_ACTIVE.
+ * * %0: Success.
+ * * Error code on failure.
*/
static inline int pm_runtime_get_sync(struct device *dev)
{
@@ -531,8 +563,11 @@ static inline int pm_runtime_get_active(struct device *dev, int rpmflags)
* @dev: Target device.
*
* Resume @dev synchronously and if that is successful, increment its runtime
- * PM usage counter. Return 0 if the runtime PM usage counter of @dev has been
- * incremented or a negative error code otherwise.
+ * PM usage counter.
+ *
+ * Return:
+ * * %0: Success; @dev is active and its usage counter has been incremented.
+ * * Negative error code on failure; usage counter is unchanged.
*/
static inline int pm_runtime_resume_and_get(struct device *dev)
{
@@ -540,11 +575,12 @@ static inline int pm_runtime_resume_and_get(struct device *dev)
}
/**
- * pm_runtime_put - Drop device usage counter and queue up "idle check" if 0.
+ * pm_runtime_put - Drop device usage counter and request asynchronous idle check.
* @dev: Target device.
*
- * Decrement the runtime PM usage counter of @dev and if it turns out to be
- * equal to 0, queue up a work item for @dev like in pm_request_idle().
+ * Decrement the runtime PM usage counter of @dev. If the counter reaches zero
+ * and the device has no active child dependencies, asynchronously request the
+ * PM core to idle or suspend the device.
*/
static inline void pm_runtime_put(struct device *dev)
{
@@ -559,16 +595,16 @@ static inline void pm_runtime_put(struct device *dev)
* equal to 0, queue up a work item for @dev like in pm_request_autosuspend().
*
* Return:
- * * 1: Success. Usage counter dropped to zero, but device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
- * change ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -EINPROGRESS: Suspend already in progress.
- * * -ENOSYS: CONFIG_PM not enabled.
+ * * %1: Success. Usage counter dropped to zero, but device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
+ * change ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-EINPROGRESS: Suspend already in progress.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
*/
static inline int __pm_runtime_put_autosuspend(struct device *dev)
{
@@ -576,25 +612,25 @@ static inline int __pm_runtime_put_autosuspend(struct device *dev)
}
/**
- * pm_runtime_put_autosuspend - Update the last access time of a device, drop
- * its usage counter and queue autosuspend if the usage counter becomes 0.
+ * pm_runtime_put_autosuspend - Update the last access time, drop usage counter
+ * and request autosuspend.
* @dev: Target device.
*
- * Update the last access time of @dev, decrement runtime PM usage counter of
- * @dev and if it turns out to be equal to 0, queue up a work item for @dev like
- * in pm_request_autosuspend().
+ * Update the last access time of @dev and decrement its runtime PM usage
+ * counter. If the counter drops to zero, asynchronously request the PM core to
+ * suspend the device once its autosuspend delay has elapsed.
*
* Return:
- * * 1: Success. Usage counter dropped to zero, but device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
- * change ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -EINPROGRESS: Suspend already in progress.
- * * -ENOSYS: CONFIG_PM not enabled.
+ * * %1: Success. Usage counter dropped to zero, but device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
+ * change ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-EINPROGRESS: Suspend already in progress.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
*/
static inline int pm_runtime_put_autosuspend(struct device *dev)
{
@@ -653,26 +689,28 @@ DEFINE_GUARD_COND(pm_runtime_active_auto, _try_enabled,
* pm_runtime_put_sync - Drop device usage counter and run "idle check" if 0.
* @dev: Target device.
*
- * Decrement the runtime PM usage counter of @dev and if it turns out to be
- * equal to 0, invoke the "idle check" callback of @dev and, depending on its
- * return value, set up autosuspend of @dev or suspend it (depending on whether
- * or not autosuspend has been enabled for it).
+ * Decrement the runtime PM usage counter of @dev. If the counter drops to zero,
+ * synchronously evaluate and trigger idle/suspend handling.
+ *
+ * Note that this does not update the last access time, but it does respect
+ * existing autosuspend timers. If @dev uses autosuspend, consider using
+ * pm_runtime_put_sync_autosuspend() or pm_runtime_put_sync_suspend() instead.
*
* The runtime PM usage counter of @dev remains decremented in all cases, even
* if it returns an error code.
*
* Return:
- * * 1: Success. Usage counter dropped to zero, but device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
- * change ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -ENOSYS: CONFIG_PM not enabled.
- * Other values and conditions for the above values are possible as returned by
- * Runtime PM suspend callbacks.
+ * * %1: Success. Usage counter dropped to zero, but device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
+ * change ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
+ * * Other values and conditions for the above values are possible as returned
+ * by Runtime PM suspend callbacks.
*/
static inline int pm_runtime_put_sync(struct device *dev)
{
@@ -683,24 +721,28 @@ static inline int pm_runtime_put_sync(struct device *dev)
* pm_runtime_put_sync_suspend - Drop device usage counter and suspend if 0.
* @dev: Target device.
*
- * Decrement the runtime PM usage counter of @dev and if it turns out to be
- * equal to 0, carry out runtime-suspend of @dev synchronously.
+ * Decrement the runtime PM usage counter of @dev. If the counter drops to zero,
+ * suspend the device synchronously.
+ *
+ * This API differs from pm_runtime_put_sync() and
+ * pm_runtime_put_sync_autosuspend() in that it ignores any outstanding
+ * autosuspend delays.
*
* The runtime PM usage counter of @dev remains decremented in all cases, even
* if it returns an error code.
*
* Return:
- * * 1: Success. Usage counter dropped to zero, but device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
- * change ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -ENOSYS: CONFIG_PM not enabled.
- * Other values and conditions for the above values are possible as returned by
- * Runtime PM suspend callbacks.
+ * * %1: Success. Usage counter dropped to zero, but device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
+ * change ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
+ * * Other values and conditions for the above values are possible as returned
+ * by Runtime PM suspend callbacks.
*/
static inline int pm_runtime_put_sync_suspend(struct device *dev)
{
@@ -712,27 +754,28 @@ static inline int pm_runtime_put_sync_suspend(struct device *dev)
* drop device usage counter and autosuspend if 0.
* @dev: Target device.
*
- * Update the last access time of @dev, decrement the runtime PM usage counter
- * of @dev and if it turns out to be equal to 0, set up autosuspend of @dev or
- * suspend it synchronously (depending on whether or not autosuspend has been
- * enabled for it).
+ * Update the last access time of @dev and decrement its runtime PM usage
+ * counter. If the counter drops to zero, synchronously suspend the device (or
+ * schedule autosuspend if the delay has not elapsed).
+ *
+ * Prefer this API over pm_runtime_put_sync() for devices that use autosuspend.
*
* The runtime PM usage counter of @dev remains decremented in all cases, even
* if it returns an error code.
*
* Return:
- * * 1: Success. Usage counter dropped to zero, but device was already suspended.
- * * 0: Success.
- * * -EINVAL: Runtime PM error.
- * * -EACCES: Runtime PM disabled.
- * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
- * change ongoing.
- * * -EBUSY: Runtime PM child_count non-zero.
- * * -EPERM: Device PM QoS resume latency 0.
- * * -EINPROGRESS: Suspend already in progress.
- * * -ENOSYS: CONFIG_PM not enabled.
- * Other values and conditions for the above values are possible as returned by
- * Runtime PM suspend callbacks.
+ * * %1: Success. Usage counter dropped to zero, but device was already suspended.
+ * * %0: Success.
+ * * %-EINVAL: Runtime PM error.
+ * * %-EACCES: Runtime PM disabled.
+ * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status
+ * change ongoing.
+ * * %-EBUSY: Runtime PM child_count non-zero.
+ * * %-EPERM: Device PM QoS resume latency 0.
+ * * %-EINPROGRESS: Suspend already in progress.
+ * * %-ENOSYS: %CONFIG_PM not enabled.
+ * * Other values and conditions for the above values are possible as returned
+ * by Runtime PM suspend callbacks.
*/
static inline int pm_runtime_put_sync_autosuspend(struct device *dev)
{
@@ -741,13 +784,22 @@ static inline int pm_runtime_put_sync_autosuspend(struct device *dev)
}
/**
- * pm_runtime_set_active - Set runtime PM status to "active".
+ * pm_runtime_set_active - Set runtime PM status to "active" and clear errors.
* @dev: Target device.
*
- * Set the runtime PM status of @dev to %RPM_ACTIVE and ensure that dependencies
- * of it will be taken into account.
+ * Set the runtime PM status of @dev to %RPM_ACTIVE and ensure that its
+ * dependencies will be taken into account. Also clear the device's error
+ * status (@dev->power.runtime_error).
+ *
+ * It is only valid to call this function if runtime PM is disabled or if
+ * @dev->power.runtime_error is set.
*
- * It is not valid to call this function for devices with runtime PM enabled.
+ * This will fail if suppliers cannot be resumed, or if the parent is not in
+ * the correct state.
+ *
+ * Return:
+ * * %0: Success.
+ * * Error code on failure.
*/
static inline int pm_runtime_set_active(struct device *dev)
{
@@ -755,13 +807,19 @@ static inline int pm_runtime_set_active(struct device *dev)
}
/**
- * pm_runtime_set_suspended - Set runtime PM status to "suspended".
+ * pm_runtime_set_suspended - Set runtime PM status to "suspended" and clear errors.
* @dev: Target device.
*
- * Set the runtime PM status of @dev to %RPM_SUSPENDED and ensure that
- * dependencies of it will be taken into account.
+ * Set the runtime PM status of @dev to %RPM_SUSPENDED and ensure that its
+ * dependencies will be taken into account. Also clear the device's error
+ * status (@dev->power.runtime_error).
*
- * It is not valid to call this function for devices with runtime PM enabled.
+ * It is only valid to call this function if runtime PM is disabled or if
+ * @dev->power.runtime_error is set.
+ *
+ * Return:
+ * * %0: Success.
+ * * Error code on failure.
*/
static inline int pm_runtime_set_suspended(struct device *dev)
{
@@ -777,9 +835,8 @@ static inline int pm_runtime_set_suspended(struct device *dev)
*
* If the counter is zero when this function runs and there is a pending runtime
* resume request for @dev, it will be resumed. If the counter is still zero at
- * that point, all of the pending runtime PM requests for @dev will be canceled
- * and all runtime PM operations in progress involving it will be waited for to
- * complete.
+ * that point, this function cancels all pending runtime PM requests for @dev
+ * and waits for its runtime PM operations to complete (if any).
*
* For each invocation of this function for @dev, there must be a matching
* pm_runtime_enable() call, so that runtime PM is eventually enabled for it
diff --git a/include/linux/posix-timers.h b/include/linux/posix-timers.h
index 9a1a0c61361c..00767acbc111 100644
--- a/include/linux/posix-timers.h
+++ b/include/linux/posix-timers.h
@@ -66,38 +66,6 @@ struct cpu_timer {
struct task_struct __rcu *handling;
};
-static inline bool cpu_timer_enqueue(struct timerqueue_head *head,
- struct cpu_timer *ctmr)
-{
- ctmr->head = head;
- return timerqueue_add(head, &ctmr->node);
-}
-
-static inline bool cpu_timer_queued(struct cpu_timer *ctmr)
-{
- return !!ctmr->head;
-}
-
-static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr)
-{
- if (cpu_timer_queued(ctmr)) {
- timerqueue_del(ctmr->head, &ctmr->node);
- ctmr->head = NULL;
- return true;
- }
- return false;
-}
-
-static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr)
-{
- return ctmr->node.expires;
-}
-
-static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp)
-{
- ctmr->node.expires = exp;
-}
-
static inline void posix_cputimers_init(struct posix_cputimers *pct)
{
memset(pct, 0, sizeof(*pct));
@@ -224,14 +192,15 @@ struct k_itimer {
} ____cacheline_aligned_in_smp;
void run_posix_cpu_timers(void);
-void posix_cpu_timers_exit(struct task_struct *task);
-void posix_cpu_timers_exit_group(struct task_struct *task);
void set_process_cpu_timer(struct task_struct *task, unsigned int clock_idx,
u64 *newval, u64 *oldval);
int update_rlimit_cpu(struct task_struct *task, unsigned long rlim_new);
#ifdef CONFIG_POSIX_TIMERS
+void posixtimer_exec(void);
+void posixtimer_exit(bool group_dead);
+
static inline void posixtimer_putref(struct k_itimer *tmr)
{
if (rcuref_put(&tmr->rcuref))
@@ -259,6 +228,8 @@ static inline bool posixtimer_valid(const struct k_itimer *timer)
return !(val & 0x1UL);
}
#else /* CONFIG_POSIX_TIMERS */
+static inline void posixtimer_exec(void) { }
+static inline void posixtimer_exit(bool group_dead) { }
static inline void posixtimer_sigqueue_getref(struct sigqueue *q) { }
static inline void posixtimer_sigqueue_putref(struct sigqueue *q) { }
#endif /* !CONFIG_POSIX_TIMERS */
diff --git a/include/linux/posix_acl.h b/include/linux/posix_acl.h
index 62d497763e25..caf500bed993 100644
--- a/include/linux/posix_acl.h
+++ b/include/linux/posix_acl.h
@@ -74,20 +74,20 @@ extern int __posix_acl_create(struct posix_acl **, gfp_t, umode_t *);
extern int __posix_acl_chmod(struct posix_acl **, gfp_t, umode_t);
extern struct posix_acl *get_posix_acl(struct inode *, int);
-int set_posix_acl(struct mnt_idmap *, struct dentry *, int,
+int set_posix_acl(const struct mnt_idmap *, struct dentry *, int,
struct posix_acl *);
struct posix_acl *get_cached_acl_rcu(struct inode *inode, int type);
struct posix_acl *posix_acl_clone(const struct posix_acl *acl, gfp_t flags);
#ifdef CONFIG_FS_POSIX_ACL
-int posix_acl_chmod(struct mnt_idmap *, struct dentry *, umode_t);
+int posix_acl_chmod(const struct mnt_idmap *, struct dentry *, umode_t);
extern int posix_acl_create(struct inode *, umode_t *, struct posix_acl **,
struct posix_acl **);
-int posix_acl_update_mode(struct mnt_idmap *, struct inode *, umode_t *,
+int posix_acl_update_mode(const struct mnt_idmap *, struct inode *, umode_t *,
struct posix_acl **);
-int simple_set_acl(struct mnt_idmap *, struct dentry *,
+int simple_set_acl(const struct mnt_idmap *, struct dentry *,
struct posix_acl *, int);
extern int simple_acl_create(struct inode *, struct inode *);
@@ -96,7 +96,7 @@ void set_cached_acl(struct inode *inode, int type, struct posix_acl *acl);
void forget_cached_acl(struct inode *inode, int type);
void forget_all_cached_acls(struct inode *inode);
int posix_acl_valid(struct user_namespace *, const struct posix_acl *);
-int posix_acl_permission(struct mnt_idmap *, struct inode *,
+int posix_acl_permission(const struct mnt_idmap *, struct inode *,
const struct posix_acl *, int);
static inline void cache_no_acl(struct inode *inode)
@@ -105,16 +105,16 @@ static inline void cache_no_acl(struct inode *inode)
inode->i_default_acl = NULL;
}
-int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry,
+int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry,
const char *acl_name, struct posix_acl *kacl);
-struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap,
+struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name);
-int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry,
+int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry,
const char *acl_name);
int posix_acl_listxattr(struct inode *inode, char **buffer,
ssize_t *remaining_size);
#else
-static inline int posix_acl_chmod(struct mnt_idmap *idmap,
+static inline int posix_acl_chmod(const struct mnt_idmap *idmap,
struct dentry *dentry, umode_t mode)
{
return 0;
@@ -141,21 +141,21 @@ static inline void forget_all_cached_acls(struct inode *inode)
{
}
-static inline int vfs_set_acl(struct mnt_idmap *idmap,
+static inline int vfs_set_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name,
struct posix_acl *acl)
{
return -EOPNOTSUPP;
}
-static inline struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap,
+static inline struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{
return ERR_PTR(-EOPNOTSUPP);
}
-static inline int vfs_remove_acl(struct mnt_idmap *idmap,
+static inline int vfs_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
{
return -EOPNOTSUPP;
diff --git a/include/linux/power/bq27xxx_battery.h b/include/linux/power/bq27xxx_battery.h
index d56e1276aafe..0b833c4bb583 100644
--- a/include/linux/power/bq27xxx_battery.h
+++ b/include/linux/power/bq27xxx_battery.h
@@ -33,6 +33,7 @@ enum bq27xxx_chip {
BQ27441,
BQ27621,
BQ27Z561,
+ BQ27Z746,
BQ28Z610,
BQ34Z100,
BQ78Z100,
diff --git a/include/linux/power_supply.h b/include/linux/power_supply.h
index e749d2189335..bee659427166 100644
--- a/include/linux/power_supply.h
+++ b/include/linux/power_supply.h
@@ -104,6 +104,14 @@ enum {
POWER_SUPPLY_SCOPE_DEVICE,
};
+enum {
+ POWER_SUPPLY_LOAD_SWITCH_UNKNOWN = 0,
+ POWER_SUPPLY_LOAD_SWITCH_ON,
+ POWER_SUPPLY_LOAD_SWITCH_OFF,
+ POWER_SUPPLY_LOAD_SWITCH_STANDBY,
+ POWER_SUPPLY_LOAD_SWITCH_SHIP,
+};
+
enum power_supply_property {
/* Properties of type `int' */
POWER_SUPPLY_PROP_STATUS = 0,
@@ -182,6 +190,7 @@ enum power_supply_property {
POWER_SUPPLY_PROP_MANUFACTURE_DAY,
POWER_SUPPLY_PROP_INTERNAL_RESISTANCE,
POWER_SUPPLY_PROP_STATE_OF_HEALTH,
+ POWER_SUPPLY_PROP_LOAD_SWITCH,
/* Properties of type `const char *' */
POWER_SUPPLY_PROP_MODEL_NAME,
POWER_SUPPLY_PROP_MANUFACTURER,
@@ -259,9 +268,9 @@ struct power_supply_config {
struct power_supply_desc {
const char *name;
enum power_supply_type type;
- u8 charge_behaviours;
u32 charge_types;
u32 usb_types;
+ u32 load_switches;
const enum power_supply_property *properties;
size_t num_properties;
@@ -295,14 +304,17 @@ struct power_supply_desc {
*/
int (*init)(struct power_supply *psy);
+ /* For APM emulation, think legacy userspace. */
+ int use_for_apm;
+
/*
* Set if thermal zone should not be created for this power supply.
* For example for virtual supplies forwarding calls to actual
* sensors or other supplies.
*/
bool no_thermal;
- /* For APM emulation, think legacy userspace. */
- int use_for_apm;
+
+ u8 charge_behaviours;
};
struct power_supply_ext {
@@ -414,9 +426,9 @@ struct power_supply_vbat_ri_table {
* @charge_voltage_max_uv: maintenance charging voltage that is usually a bit
* lower than the constant_charge_voltage_max_uv. We can apply this settings
* charge_current_max_ua until we get back up to this voltage.
- * @safety_timer_minutes: maintenance charging safety timer, with an expiry
- * time in minutes. We will only use maintenance charging in this setting
- * for a certain amount of time, then we will first move to the next
+ * @charge_safety_timer_minutes: maintenance charging safety timer, with an
+ * expiry time in minutes. We will only use maintenance charging in this
+ * setting for a certain amount of time, then we will first move to the next
* maintenance charge current and voltage pair in respective array and wait
* for the next safety timer timeout, or, if we reached the last maintencance
* charging setting, disable charging until we reach
diff --git a/include/linux/preempt.h b/include/linux/preempt.h
index 8299657f0f86..06ff44a4b8b6 100644
--- a/include/linux/preempt.h
+++ b/include/linux/preempt.h
@@ -168,10 +168,6 @@ static __always_inline unsigned char interrupt_context_level(void)
#define in_softirq() (softirq_count())
#define in_interrupt() (irq_count())
-#define hardirq_disable_count() ((preempt_count() & HARDIRQ_DISABLE_MASK) >> HARDIRQ_DISABLE_SHIFT)
-#define hardirq_disable_enter() __preempt_count_add_return(HARDIRQ_DISABLE_OFFSET)
-#define hardirq_disable_exit() __preempt_count_sub_return(HARDIRQ_DISABLE_OFFSET)
-
/*
* The preempt_count offset after preempt_disable();
*/
@@ -502,21 +498,11 @@ DEFINE_LOCK_GUARD_0(preempt_notrace, preempt_disable_notrace(), preempt_enable_n
#ifdef CONFIG_PREEMPT_DYNAMIC
-extern bool preempt_model_none(void);
-extern bool preempt_model_voluntary(void);
extern bool preempt_model_full(void);
extern bool preempt_model_lazy(void);
#else
-static inline bool preempt_model_none(void)
-{
- return IS_ENABLED(CONFIG_PREEMPT_NONE);
-}
-static inline bool preempt_model_voluntary(void)
-{
- return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY);
-}
static inline bool preempt_model_full(void)
{
return IS_ENABLED(CONFIG_PREEMPT);
@@ -529,6 +515,16 @@ static inline bool preempt_model_lazy(void)
#endif
+static inline bool preempt_model_none(void)
+{
+ return IS_ENABLED(CONFIG_PREEMPT_NONE);
+}
+
+static inline bool preempt_model_voluntary(void)
+{
+ return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY);
+}
+
static inline bool preempt_model_rt(void)
{
return IS_ENABLED(CONFIG_PREEMPT_RT);
diff --git a/include/linux/property.h b/include/linux/property.h
index 907c790a3f01..9143fe4f5c05 100644
--- a/include/linux/property.h
+++ b/include/linux/property.h
@@ -547,6 +547,11 @@ unsigned int fwnode_graph_get_endpoint_count(const struct fwnode_handle *fwnode,
for (child = fwnode_graph_get_next_endpoint(fwnode, NULL); child; \
child = fwnode_graph_get_next_endpoint(fwnode, child))
+#define fwnode_graph_for_each_endpoint_scoped(fwnode, child) \
+ for (struct fwnode_handle *child __free(fwnode_handle) = \
+ fwnode_graph_get_next_endpoint(fwnode, NULL); \
+ child; child = fwnode_graph_get_next_endpoint(fwnode, child))
+
int fwnode_graph_parse_endpoint(const struct fwnode_handle *fwnode,
struct fwnode_endpoint *endpoint);
diff --git a/include/linux/psp-sev.h b/include/linux/psp-sev.h
index 03a79786df1d..fab62228f981 100644
--- a/include/linux/psp-sev.h
+++ b/include/linux/psp-sev.h
@@ -49,7 +49,7 @@
#define SEV_FW_BLOB_MAX_SIZE 0x4000 /* 16KB */
-/**
+/*
* SEV platform state
*/
enum sev_state {
@@ -60,7 +60,7 @@ enum sev_state {
SEV_STATE_MAX
};
-/**
+/*
* SEV platform and guest management commands
*/
enum sev_cmd {
@@ -157,6 +157,7 @@ enum sev_cmd {
* struct sev_data_init - INIT command parameters
*
* @flags: processing flags
+ * @reserved: reserved
* @tmr_address: system physical address used for SEV-ES
* @tmr_len: len of tmr_address
*/
@@ -174,6 +175,7 @@ struct sev_data_init {
* @flags: processing flags
* @tmr_address: system physical address used for SEV-ES
* @tmr_len: len of tmr_address
+ * @reserved: reserved
* @nv_address: system physical address used for PSP NV storage
* @nv_len: len of nv_address
*/
@@ -201,12 +203,13 @@ struct sev_data_pek_csr {
} __packed;
/**
- * struct sev_data_cert_import - PEK_CERT_IMPORT command parameters
+ * struct sev_data_pek_cert_import - PEK_CERT_IMPORT command parameters
*
- * @pek_address: PEK certificate chain
- * @pek_len: len of PEK certificate
- * @oca_address: OCA certificate chain
- * @oca_len: len of OCA certificate
+ * @pek_cert_address: PEK certificate chain
+ * @pek_cert_len: len of PEK certificate
+ * @reserved: reserved
+ * @oca_cert_address: OCA certificate chain
+ * @oca_cert_len: len of OCA certificate
*/
struct sev_data_pek_cert_import {
u64 pek_cert_address; /* In */
@@ -240,8 +243,9 @@ struct sev_data_get_id {
/**
* struct sev_data_pdh_cert_export - PDH_CERT_EXPORT command parameters
*
- * @pdh_address: PDH certificate address
- * @pdh_len: len of PDH certificate
+ * @pdh_cert_address: PDH certificate address
+ * @pdh_cert_len: len of PDH certificate
+ * @reserved: reserved
* @cert_chain_address: PDH certificate chain
* @cert_chain_len: len of PDH certificate chain
*/
@@ -304,6 +308,7 @@ struct sev_data_guest_status {
* @policy: guest launch policy
* @dh_cert_address: physical address of DH certificate blob
* @dh_cert_len: len of DH certificate blob
+ * @reserved: reserved
* @session_address: physical address of session parameters
* @session_len: len of session parameters
*/
@@ -321,6 +326,7 @@ struct sev_data_launch_start {
* struct sev_data_launch_update_data - LAUNCH_UPDATE_DATA command parameter
*
* @handle: handle of the VM to update
+ * @reserved: reserved
* @len: len of memory to be encrypted
* @address: physical address of memory region to encrypt
*/
@@ -335,6 +341,7 @@ struct sev_data_launch_update_data {
* struct sev_data_launch_update_vmsa - LAUNCH_UPDATE_VMSA command
*
* @handle: handle of the VM
+ * @reserved: reserved
* @address: physical address of memory region to encrypt
* @len: len of memory region to encrypt
*/
@@ -349,6 +356,7 @@ struct sev_data_launch_update_vmsa {
* struct sev_data_launch_measure - LAUNCH_MEASURE command parameters
*
* @handle: handle of the VM to process
+ * @reserved: reserved
* @address: physical address containing the measurement blob
* @len: len of measurement blob
*/
@@ -363,10 +371,13 @@ struct sev_data_launch_measure {
* struct sev_data_launch_secret - LAUNCH_SECRET command parameters
*
* @handle: handle of the VM to process
+ * @reserved1: reserved
* @hdr_address: physical address containing the packet header
* @hdr_len: len of packet header
+ * @reserved2: reserved
* @guest_address: system physical address of guest memory region
* @guest_len: len of guest_paddr
+ * @reserved3: reserved
* @trans_address: physical address of transport memory buffer
* @trans_len: len of transport memory buffer
*/
@@ -399,10 +410,13 @@ struct sev_data_launch_finish {
* @policy: policy information for the VM
* @pdh_cert_address: physical address containing PDH certificate
* @pdh_cert_len: len of PDH certificate
+ * @reserved1: reserved
* @plat_certs_address: physical address containing platform certificate
* @plat_certs_len: len of platform certificate
+ * @reserved2: reserved
* @amd_certs_address: physical address containing AMD certificate
* @amd_certs_len: len of AMD certificate
+ * @reserved3: reserved
* @session_address: physical address containing Session data
* @session_len: len of session data
*/
@@ -423,13 +437,16 @@ struct sev_data_send_start {
} __packed;
/**
- * struct sev_data_send_update - SEND_UPDATE_DATA command
+ * struct sev_data_send_update_data - SEND_UPDATE_DATA command
*
* @handle: handle of the VM to process
+ * @reserved1: reserved
* @hdr_address: physical address containing packet header
* @hdr_len: len of packet header
+ * @reserved2: reserved
* @guest_address: physical address of guest memory region to send
* @guest_len: len of guest memory region to send
+ * @reserved3: reserved
* @trans_address: physical address of host memory region
* @trans_len: len of host memory region
*/
@@ -447,13 +464,15 @@ struct sev_data_send_update_data {
} __packed;
/**
- * struct sev_data_send_update - SEND_UPDATE_VMSA command
+ * struct sev_data_send_update_vmsa - SEND_UPDATE_VMSA command
*
* @handle: handle of the VM to process
* @hdr_address: physical address containing packet header
* @hdr_len: len of packet header
+ * @reserved2: reserved
* @guest_address: physical address of guest memory region to send
* @guest_len: len of guest memory region to send
+ * @reserved3: reserved
* @trans_address: physical address of host memory region
* @trans_len: len of host memory region
*/
@@ -491,8 +510,10 @@ struct sev_data_send_cancel {
* struct sev_data_receive_start - RECEIVE_START command parameters
*
* @handle: handle of the VM to perform receive operation
+ * @policy: policy information for the VM
* @pdh_cert_address: system physical address containing PDH certificate blob
* @pdh_cert_len: len of PDH certificate blob
+ * @reserved1: reserved
* @session_address: system physical address containing session blob
* @session_len: len of session blob
*/
@@ -510,10 +531,13 @@ struct sev_data_receive_start {
* struct sev_data_receive_update_data - RECEIVE_UPDATE_DATA command parameters
*
* @handle: handle of the VM to update
+ * @reserved1: reserved
* @hdr_address: physical address containing packet header blob
* @hdr_len: len of packet header
+ * @reserved2: reserved
* @guest_address: system physical address of guest memory region
* @guest_len: len of guest memory region
+ * @reserved3: reserved
* @trans_address: system physical address of transport buffer
* @trans_len: len of transport buffer
*/
@@ -534,10 +558,13 @@ struct sev_data_receive_update_data {
* struct sev_data_receive_update_vmsa - RECEIVE_UPDATE_VMSA command parameters
*
* @handle: handle of the VM to update
+ * @reserved1: reserved
* @hdr_address: physical address containing packet header blob
* @hdr_len: len of packet header
+ * @reserved2: reserved
* @guest_address: system physical address of guest memory region
* @guest_len: len of guest memory region
+ * @reserved3: reserved
* @trans_address: system physical address of transport buffer
* @trans_len: len of transport buffer
*/
@@ -567,6 +594,7 @@ struct sev_data_receive_finish {
* struct sev_data_dbg - DBG_ENCRYPT/DBG_DECRYPT command parameters
*
* @handle: handle of the VM to perform debug operation
+ * @reserved: reserved
* @src_addr: source address of data to operate on
* @dst_addr: destination address of data to operate on
* @len: len of data to operate on
@@ -583,6 +611,7 @@ struct sev_data_dbg {
* struct sev_data_attestation_report - SEV_ATTESTATION_REPORT command parameters
*
* @handle: handle of the VM
+ * @reserved: reserved
* @mnonce: a random nonce that will be included in the report.
* @address: physical address where the report will be copied.
* @len: length of the physical buffer.
@@ -781,9 +810,14 @@ struct sev_data_snp_guest_request {
*
* @init_rmp: indicate that the RMP should be initialized.
* @list_paddr_en: indicate that list_paddr is valid
+ * @rapl_dis: whether RAPL is disabled
+ * @ciphertext_hiding_en: whether ciphertext hiding is enabled
+ * @tio_en: Indicates that SNP_INIT_EX initialized the RMP for SEV-TIO
* @rsvd: reserved
* @rsvd1: reserved
* @list_paddr: system physical address of range list
+ * @max_snp_asid: When non-zero, enable ciphertext hiding and specify the
+ * maximum ASID that can be used for an SEV-SNP guest.
* @rsvd2: reserved
*/
struct sev_data_snp_init_ex {
@@ -841,7 +875,7 @@ struct sev_data_snp_shutdown_ex {
} __packed;
/**
- * struct sev_platform_init_args
+ * struct sev_platform_init_args - parameters for sev_platform_init()
*
* @error: SEV firmware error code
* @probe: True if this is being called as part of CCP module probe, which
@@ -879,12 +913,12 @@ struct sev_data_snp_feature_info {
} __packed;
/**
- * struct feature_info - FEATURE_INFO structure
+ * struct snp_feature_info - FEATURE_INFO structure
*
* @eax: output of SNP_FEATURE_INFO command
* @ebx: output of SNP_FEATURE_INFO command
* @ecx: output of SNP_FEATURE_INFO command
- * #edx: output of SNP_FEATURE_INFO command
+ * @edx: output of SNP_FEATURE_INFO command
*/
struct snp_feature_info {
u32 eax;
@@ -954,7 +988,7 @@ struct sev_data_snp_verify_mitigation_dst {
} __packed;
/**
- * struct sev_snp_tcb_version_genoa_milan
+ * struct sev_snp_tcb_version_genoa_milan - v1 SVN payload
*
* @boot_loader: SVN of PSP bootloader
* @tee: SVN of PSP operating system
@@ -971,7 +1005,7 @@ struct sev_snp_tcb_version_genoa_milan {
};
/**
- * struct sev_snp_tcb_version_turin
+ * struct sev_snp_tcb_version_turin - v2 SVN payload
*
* @fmc: SVN of FMC firmware
* @boot_loader: SVN of PSP bootloader
@@ -1037,9 +1071,9 @@ int sev_platform_status(struct sev_user_data_status *status, int *error);
* behalf of userspace. The caller must pass a valid SEV file descriptor
* so that we know that it has access to SEV device.
*
- * @filep - SEV device file pointer
- * @cmd - command to issue
- * @data - command buffer
+ * @filep: SEV device file pointer
+ * @id: command to issue
+ * @data: command buffer
* @error: SEV command return code
*
* Returns:
@@ -1056,8 +1090,8 @@ int sev_issue_cmd_external_user(struct file *filep, unsigned int id,
/**
* sev_guest_deactivate - perform SEV DEACTIVATE command
*
- * @deactivate: sev_data_deactivate structure to be processed
- * @sev_ret: sev command return code
+ * @data: sev_data_deactivate structure to be processed
+ * @error: sev command return code
*
* Returns:
* 0 if the sev successfully processed the command
@@ -1071,8 +1105,8 @@ int sev_guest_deactivate(struct sev_data_deactivate *data, int *error);
/**
* sev_guest_activate - perform SEV ACTIVATE command
*
- * @activate: sev_data_activate structure to be processed
- * @sev_ret: sev command return code
+ * @data: sev_data_activate structure to be processed
+ * @error: sev command return code
*
* Returns:
* 0 if the sev successfully processed the command
@@ -1086,7 +1120,7 @@ int sev_guest_activate(struct sev_data_activate *data, int *error);
/**
* sev_guest_df_flush - perform SEV DF_FLUSH command
*
- * @sev_ret: sev command return code
+ * @error: sev command return code
*
* Returns:
* 0 if the sev successfully processed the command
@@ -1100,8 +1134,8 @@ int sev_guest_df_flush(int *error);
/**
* sev_guest_decommission - perform SEV DECOMMISSION command
*
- * @decommission: sev_data_decommission structure to be processed
- * @sev_ret: sev command return code
+ * @data: sev_data_decommission structure to be processed
+ * @error: sev command return code
*
* Returns:
* 0 if the sev successfully processed the command
diff --git a/include/linux/ptdump.h b/include/linux/ptdump.h
index 240bd3bff18d..af18d1459b2f 100644
--- a/include/linux/ptdump.h
+++ b/include/linux/ptdump.h
@@ -31,9 +31,9 @@ bool ptdump_walk_pgd_level_core(struct seq_file *m,
void ptdump_walk_pgd(struct ptdump_state *st, struct mm_struct *mm, pgd_t *pgd);
bool ptdump_check_wx(void);
-static inline void debug_checkwx(void)
+static inline void pgtable_checkwx(void)
{
- if (IS_ENABLED(CONFIG_DEBUG_WX))
+ if (IS_ENABLED(CONFIG_CHECK_WX))
ptdump_check_wx();
}
diff --git a/include/linux/quotaops.h b/include/linux/quotaops.h
index f9c0f9d7c9d9..0c64ca674e77 100644
--- a/include/linux/quotaops.h
+++ b/include/linux/quotaops.h
@@ -20,7 +20,7 @@ static inline struct quota_info *sb_dqopt(struct super_block *sb)
}
/* i_rwsem must being held */
-static inline bool is_quota_modification(struct mnt_idmap *idmap,
+static inline bool is_quota_modification(const struct mnt_idmap *idmap,
struct inode *inode, struct iattr *ia)
{
return ((ia->ia_valid & ATTR_SIZE) ||
@@ -109,7 +109,7 @@ int dquot_set_dqblk(struct super_block *sb, struct kqid id,
struct qc_dqblk *di);
int __dquot_transfer(struct inode *inode, struct dquot **transfer_to);
-int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode,
+int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode,
struct iattr *iattr);
static inline struct mem_dqinfo *sb_dqinfo(struct super_block *sb, int type)
@@ -229,7 +229,7 @@ static inline void dquot_free_inode(struct inode *inode)
{
}
-static inline int dquot_transfer(struct mnt_idmap *idmap,
+static inline int dquot_transfer(const struct mnt_idmap *idmap,
struct inode *inode, struct iattr *iattr)
{
return 0;
diff --git a/include/linux/rbtree_augmented.h b/include/linux/rbtree_augmented.h
index 6dbc5a1bf6a8..d2fa1c41bfd2 100644
--- a/include/linux/rbtree_augmented.h
+++ b/include/linux/rbtree_augmented.h
@@ -87,18 +87,18 @@ rb_add_augmented_cached(struct rb_node *node, struct rb_root_cached *tree,
}
/*
- * Template for declaring augmented rbtree callbacks (generic case)
+ * Template for declaring augmented rbtree callbacks (generic multi fields)
*
* RBSTATIC: 'static' or empty
* RBNAME: name of the rb_augment_callbacks structure
* RBSTRUCT: struct type of the tree nodes
* RBFIELD: name of struct rb_node field within RBSTRUCT
- * RBAUGMENTED: name of field within RBSTRUCT holding data for subtree
- * RBCOMPUTE: name of function that recomputes the RBAUGMENTED data
+ * RBCOPY: name of function that copies the RBAUGMENTED datas
+ * RBCOMPUTE: name of function that recomputes the RBAUGMENTED datas
*/
-#define RB_DECLARE_CALLBACKS(RBSTATIC, RBNAME, \
- RBSTRUCT, RBFIELD, RBAUGMENTED, RBCOMPUTE) \
+#define RB_DECLARE_CALLBACKS_MULTI(RBSTATIC, RBNAME, \
+ RBSTRUCT, RBFIELD, RBCOPY, RBCOMPUTE) \
static inline void \
RBNAME ## _propagate(struct rb_node *rb, struct rb_node *stop) \
{ \
@@ -114,14 +114,14 @@ RBNAME ## _copy(struct rb_node *rb_old, struct rb_node *rb_new) \
{ \
RBSTRUCT *old = rb_entry(rb_old, RBSTRUCT, RBFIELD); \
RBSTRUCT *new = rb_entry(rb_new, RBSTRUCT, RBFIELD); \
- new->RBAUGMENTED = old->RBAUGMENTED; \
+ RBCOPY(new, old); \
} \
static void \
RBNAME ## _rotate(struct rb_node *rb_old, struct rb_node *rb_new) \
{ \
RBSTRUCT *old = rb_entry(rb_old, RBSTRUCT, RBFIELD); \
RBSTRUCT *new = rb_entry(rb_new, RBSTRUCT, RBFIELD); \
- new->RBAUGMENTED = old->RBAUGMENTED; \
+ RBCOPY(new, old); \
RBCOMPUTE(old, false); \
} \
RBSTATIC const struct rb_augment_callbacks RBNAME = { \
@@ -131,6 +131,27 @@ RBSTATIC const struct rb_augment_callbacks RBNAME = { \
};
/*
+ * Template for declaring augmented rbtree callbacks (generic single field)
+ *
+ * RBSTATIC: 'static' or empty
+ * RBNAME: name of the rb_augment_callbacks structure
+ * RBSTRUCT: struct type of the tree nodes
+ * RBFIELD: name of struct rb_node field within RBSTRUCT
+ * RBAUGMENTED: name of field within RBSTRUCT holding data for subtree
+ * RBCOMPUTE: name of function that recomputes the RBAUGMENTED data
+ */
+
+#define RB_DECLARE_CALLBACKS(RBSTATIC, RBNAME, \
+ RBSTRUCT, RBFIELD, RBAUGMENTED, RBCOMPUTE) \
+static inline void \
+RBNAME ## _copy_single(RBSTRUCT *new, RBSTRUCT *old) \
+{ \
+ new->RBAUGMENTED = old->RBAUGMENTED; \
+} \
+RB_DECLARE_CALLBACKS_MULTI(RBSTATIC, RBNAME, \
+ RBSTRUCT, RBFIELD, RBNAME ## _copy_single, RBCOMPUTE)
+
+/*
* Template for declaring augmented rbtree callbacks,
* computing RBAUGMENTED scalar as max(RBCOMPUTE(node)) for all subtree nodes.
*
diff --git a/include/linux/rcupdate.h b/include/linux/rcupdate.h
index 44c07a66edff..3f74ae6d6e1f 100644
--- a/include/linux/rcupdate.h
+++ b/include/linux/rcupdate.h
@@ -488,7 +488,7 @@ static __always_inline bool lockdep_assert_rcu_helper(bool c, const struct __ctx
context_unsafe( \
typeof(*p) *local = (typeof(*p) *__force)(p); \
rcu_check_sparse(p, __rcu); \
- ((typeof(*p) __force __kernel *)(local)) \
+ ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)) \
)
/**
* unrcu_pointer - mark a pointer as not being RCU protected
@@ -503,7 +503,7 @@ context_unsafe( \
({ \
typeof(*p) *local = (typeof(*p) *__force)READ_ONCE(p); \
rcu_check_sparse(p, space); \
- ((typeof(*p) __force __kernel *)(local)); \
+ ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \
}) )
#define __rcu_dereference_check(p, local, c, space) \
({ \
@@ -511,19 +511,19 @@ context_unsafe( \
typeof(*p) *local = (typeof(*p) *__force)READ_ONCE(p); \
RCU_LOCKDEP_WARN(!(c), "suspicious rcu_dereference_check() usage"); \
rcu_check_sparse(p, space); \
- ((typeof(*p) __force __kernel *)(local)); \
+ ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \
})
#define __rcu_dereference_protected(p, local, c, space) \
({ \
RCU_LOCKDEP_WARN(!(c), "suspicious rcu_dereference_protected() usage"); \
rcu_check_sparse(p, space); \
- ((typeof(*p) __force __kernel *)(p)); \
+ ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(p)); \
})
#define __rcu_dereference_raw(p, local) \
({ \
/* Dependency order vs. p above. */ \
typeof(p) local = READ_ONCE(p); \
- ((typeof(*p) __force __kernel *)(local)); \
+ ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \
})
#define rcu_dereference_raw(p) __rcu_dereference_raw(p, __UNIQUE_ID(rcu))
diff --git a/include/linux/rcuref.h b/include/linux/rcuref.h
index 2fb2af6d9824..01fee161b67c 100644
--- a/include/linux/rcuref.h
+++ b/include/linux/rcuref.h
@@ -34,7 +34,7 @@ static inline void rcuref_init(rcuref_t *ref, unsigned int cnt)
* indicate that it is safe to schedule the object, protected by this reference
* counter, for deconstruction.
* If you want to know if the reference counter has been marked DEAD (as
- * signaled by rcuref_put()) please use rcuread_is_dead().
+ * signaled by rcuref_put()) please use rcuref_is_dead().
*/
static inline unsigned int rcuref_read(rcuref_t *ref)
{
diff --git a/include/linux/regulator/driver.h b/include/linux/regulator/driver.h
index cc6ce709ec86..55cc519103ed 100644
--- a/include/linux/regulator/driver.h
+++ b/include/linux/regulator/driver.h
@@ -362,7 +362,7 @@ enum regulator_type {
* @off_on_delay: guard time (in uS), before re-enabling a regulator
*
* @poll_enabled_time: The polling interval (in uS) to use while checking that
- * the regulator was actually enabled. Max upto enable_time.
+ * the regulator was actually enabled. Max up to enable_time.
*
* @of_map_mode: Maps a hardware mode defined in a DeviceTree to a standard mode
*/
diff --git a/include/linux/regulator/pca9450.h b/include/linux/regulator/pca9450.h
index 0df8b3c48082..bf94df5fafe3 100644
--- a/include/linux/regulator/pca9450.h
+++ b/include/linux/regulator/pca9450.h
@@ -210,9 +210,11 @@ enum {
#define LDO5L_EN_MASK 0xC0
#define LDO5LOUT_MASK 0x0F
-#define LDO5H_EN_MASK 0xC0
#define LDO5HOUT_MASK 0x0F
+/* LDO ENMODE value: ON in RUN, OFF while PMIC_STBY_REQ is asserted */
+#define LDO_ENMODE_ONREQ_STBYREQ 0x80
+
/* PCA9450_REG_IRQ bits */
#define IRQ_PWRON 0x80
#define IRQ_WDOGB 0x40
diff --git a/include/linux/resctrl.h b/include/linux/resctrl.h
index dd09c2ce9a0f..10dfdca7f4bf 100644
--- a/include/linux/resctrl.h
+++ b/include/linux/resctrl.h
@@ -505,6 +505,25 @@ bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r);
*/
int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable);
+/**
+ * resctrl_arch_preconvert_bw() - Prepare bandwidth control value for arch use.
+ * @r: Resource whose schema was written.
+ * @val: Bandwidth control value written to the schemata file by userspace.
+ *
+ * Convert the user provided bandwidth control value to an appropriate form for
+ * consumption by the hardware driver for resource @r. Converted value is stored
+ * in rdt_ctrl_domain::staged_config[] for later consumption by
+ * resctrl_arch_update_domains(). Is not called when MBA software controller is
+ * enabled.
+ *
+ * Architectures for which this pre-conversion hook is not useful should supply
+ * an implementation of this function that just returns @val unmodified.
+ *
+ * Return:
+ * The converted value.
+ */
+u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val);
+
/*
* Update the ctrl_val and apply this config right now.
* Must be called on one of the domain's CPUs.
diff --git a/include/linux/rhashtable-types.h b/include/linux/rhashtable-types.h
index 57c11ec9dc64..3576b8f08aff 100644
--- a/include/linux/rhashtable-types.h
+++ b/include/linux/rhashtable-types.h
@@ -70,6 +70,12 @@ struct rhashtable_params {
rht_obj_cmpfn_t obj_cmpfn;
};
+struct rhashtable_lockdep_keys {
+ struct lock_class_key lock_key;
+ struct lock_class_key mutex_key;
+ struct lock_class_key bucket_key;
+};
+
/**
* struct rhashtable - Hash table handle
* @tbl: Bucket table
@@ -97,6 +103,9 @@ struct rhashtable {
#ifdef CONFIG_MEM_ALLOC_PROFILING
struct alloc_tag *alloc_tag;
#endif
+#ifdef CONFIG_LOCKDEP
+ struct lock_class_key *lockdep_key;
+#endif
};
/**
@@ -138,24 +147,77 @@ struct rhashtable_iter {
int __rhashtable_init_noprof(struct rhashtable *ht,
const struct rhashtable_params *params,
- struct lock_class_key *key);
+ struct rhashtable_lockdep_keys *keys);
#define rhashtable_init_noprof(ht, params) \
({ \
- static struct lock_class_key __key; \
+ static struct rhashtable_lockdep_keys __keys; \
\
- __rhashtable_init_noprof(ht, params, &__key); \
+ __rhashtable_init_noprof(ht, params, &__keys); \
})
+
+/**
+ * rhashtable_init - initialize a new hash table
+ * @ht: hash table to be initialized
+ * @params: configuration parameters
+ *
+ * Initializes a new hash table based on the provided configuration
+ * parameters. A table can be configured either with a variable or
+ * fixed length key:
+ *
+ * Configuration Example 1: Fixed length keys
+ * struct test_obj {
+ * int key;
+ * void * my_member;
+ * struct rhash_head node;
+ * };
+ *
+ * struct rhashtable_params params = {
+ * .head_offset = offsetof(struct test_obj, node),
+ * .key_offset = offsetof(struct test_obj, key),
+ * .key_len = sizeof(int),
+ * .hashfn = jhash,
+ * };
+ *
+ * Configuration Example 2: Variable length keys
+ * struct test_obj {
+ * [...]
+ * struct rhash_head node;
+ * };
+ *
+ * u32 my_hash_fn(const void *data, u32 len, u32 seed)
+ * {
+ * struct test_obj *obj = data;
+ *
+ * return [... hash ...];
+ * }
+ *
+ * struct rhashtable_params params = {
+ * .head_offset = offsetof(struct test_obj, node),
+ * .hashfn = jhash,
+ * .obj_hashfn = my_hash_fn,
+ * };
+ */
#define rhashtable_init(...) alloc_hooks(rhashtable_init_noprof(__VA_ARGS__))
int __rhltable_init_noprof(struct rhltable *hlt,
const struct rhashtable_params *params,
- struct lock_class_key *key);
+ struct rhashtable_lockdep_keys *keys);
#define rhltable_init_noprof(hlt, params) \
({ \
- static struct lock_class_key __key; \
+ static struct rhashtable_lockdep_keys __keys; \
\
- __rhltable_init_noprof(hlt, params, &__key); \
+ __rhltable_init_noprof(hlt, params, &__keys); \
})
+
+/**
+ * rhltable_init - initialize a new hash list table
+ * @hlt: hash list table to be initialized
+ * @params: configuration parameters
+ *
+ * Initializes a new hash list table.
+ *
+ * See documentation for rhashtable_init.
+ */
#define rhltable_init(...) alloc_hooks(rhltable_init_noprof(__VA_ARGS__))
#endif /* _LINUX_RHASHTABLE_TYPES_H */
diff --git a/include/linux/rhashtable.h b/include/linux/rhashtable.h
index 57a2a29bef0e..ec853c1b9af3 100644
--- a/include/linux/rhashtable.h
+++ b/include/linux/rhashtable.h
@@ -319,18 +319,6 @@ static inline struct rhash_lock_head __rcu **rht_bucket_insert(
* When we write to a bucket without unlocking, we use rht_assign_locked().
*/
-static inline unsigned long rht_lock(struct bucket_table *tbl,
- struct rhash_lock_head __rcu **bkt)
- __acquires(__bitlock(0, bkt))
-{
- unsigned long flags;
-
- local_irq_save(flags);
- bit_spin_lock(0, (unsigned long *)bkt);
- lock_map_acquire(&tbl->dep_map);
- return flags;
-}
-
static inline unsigned long rht_lock_nested(struct bucket_table *tbl,
struct rhash_lock_head __rcu **bucket,
unsigned int subclass)
@@ -344,6 +332,13 @@ static inline unsigned long rht_lock_nested(struct bucket_table *tbl,
return flags;
}
+static inline unsigned long rht_lock(struct bucket_table *tbl,
+ struct rhash_lock_head __rcu **bkt)
+ __acquires(__bitlock(0, bkt))
+{
+ return rht_lock_nested(tbl, bkt, 0);
+}
+
static inline void rht_unlock(struct bucket_table *tbl,
struct rhash_lock_head __rcu **bkt,
unsigned long flags)
diff --git a/include/linux/ring_buffer.h b/include/linux/ring_buffer.h
index 0670742b2d60..eac3e9080c3c 100644
--- a/include/linux/ring_buffer.h
+++ b/include/linux/ring_buffer.h
@@ -3,8 +3,9 @@
#define _LINUX_RING_BUFFER_H
#include <linux/mm.h>
-#include <linux/seq_file.h>
#include <linux/poll.h>
+#include <linux/ring_buffer_types.h>
+#include <linux/seq_file.h>
#include <uapi/linux/trace_mmap.h>
@@ -218,14 +219,15 @@ bool ring_buffer_time_stamp_abs(struct trace_buffer *buffer);
size_t ring_buffer_nr_dirty_pages(struct trace_buffer *buffer, int cpu);
struct buffer_data_read_page;
-struct buffer_data_read_page *
-ring_buffer_alloc_read_page(struct trace_buffer *buffer, int cpu);
+int ring_buffer_alloc_read_page(struct trace_buffer *buffer, int cpu,
+ struct buffer_data_read_page **rpage);
void ring_buffer_free_read_page(struct trace_buffer *buffer, int cpu,
struct buffer_data_read_page *page);
int ring_buffer_read_page(struct trace_buffer *buffer,
struct buffer_data_read_page *data_page,
size_t len, int cpu, int full);
void *ring_buffer_read_page_data(struct buffer_data_read_page *page);
+unsigned int ring_buffer_read_page_size(struct buffer_data_read_page *rpage);
struct trace_seq;
@@ -278,11 +280,25 @@ static inline struct ring_buffer_desc *__first_ring_buffer_desc(struct trace_buf
return (struct ring_buffer_desc *)(&desc->__data[0]);
}
+/*
+ * Returns the number of pages for a ring_buffer_desc. The caller must ensure it
+ * does not overflow ring_buffer_desc::nr_page_va.
+ */
+static inline unsigned long __calc_nr_pages_ring_buffer_desc(size_t size)
+{
+ /* Takes into account the reader page */
+ return max(DIV_ROUND_UP(size, PAGE_SIZE - BUF_PAGE_HDR_SIZE), 2UL) + 1;
+}
+
static inline size_t trace_buffer_desc_size(size_t buffer_size, unsigned int nr_cpus)
{
- unsigned int nr_pages = max(DIV_ROUND_UP(buffer_size, PAGE_SIZE), 2UL) + 1;
+ unsigned long nr_pages = __calc_nr_pages_ring_buffer_desc(buffer_size);
struct ring_buffer_desc *rbdesc;
+ /* Capped by ring_buffer_desc::nr_page_va */
+ if (nr_pages > UINT_MAX)
+ return SIZE_MAX;
+
return size_add(offsetof(struct trace_buffer_desc, __data),
size_mul(nr_cpus, struct_size(rbdesc, page_va, nr_pages)));
}
diff --git a/include/linux/rmap.h b/include/linux/rmap.h
index 0b332770abee..74cca0e3c726 100644
--- a/include/linux/rmap.h
+++ b/include/linux/rmap.h
@@ -888,7 +888,7 @@ struct page_vma_mapped_walk {
static inline void page_vma_mapped_walk_done(struct page_vma_mapped_walk *pvmw)
{
/* HugeTLB pte is set to the relevant page table entry without pte_mapped. */
- if (pvmw->pte && !is_vm_hugetlb_page(pvmw->vma))
+ if (pvmw->pte && !vma_is_hugetlb(pvmw->vma))
pte_unmap(pvmw->pte);
if (pvmw->ptl)
spin_unlock(pvmw->ptl);
diff --git a/include/linux/rolling_buffer.h b/include/linux/rolling_buffer.h
index 9e5dad29669c..a97f7cfaacaa 100644
--- a/include/linux/rolling_buffer.h
+++ b/include/linux/rolling_buffer.h
@@ -45,9 +45,9 @@ struct rolling_buffer_snapshot {
int rolling_buffer_init(struct rolling_buffer *roll, unsigned int rreq_id,
unsigned int direction, gfp_t gfp);
int rolling_buffer_make_space(struct rolling_buffer *roll, gfp_t gfp);
-ssize_t rolling_buffer_load_from_ra(struct rolling_buffer *roll,
- struct readahead_control *ractl,
- struct folio_batch *put_batch);
+ssize_t rolling_buffer_bulk_load_from_ra(struct rolling_buffer *roll,
+ struct readahead_control *ractl,
+ unsigned int rreq_id, gfp_t gfp);
ssize_t rolling_buffer_append(struct rolling_buffer *roll, struct folio *folio,
unsigned int flags, gfp_t gfp);
struct folio_queue *rolling_buffer_delete_spent(struct rolling_buffer *roll);
diff --git a/include/linux/rtnetlink.h b/include/linux/rtnetlink.h
index 95729339e7a5..a54ec40d095c 100644
--- a/include/linux/rtnetlink.h
+++ b/include/linux/rtnetlink.h
@@ -186,7 +186,12 @@ void net_dec_ingress_queue(void);
#ifdef CONFIG_NET_EGRESS
void net_inc_egress_queue(void);
void net_dec_egress_queue(void);
-void netdev_xmit_skip_txqueue(bool skip);
+bool netdev_xmit_skip_txqueue(bool skip);
+#else
+static inline bool netdev_xmit_skip_txqueue(bool skip)
+{
+ return false;
+}
#endif
void rtnetlink_init(void);
diff --git a/include/linux/sched.h b/include/linux/sched.h
index 8b3d47a325cc..d828f0fb1896 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -554,6 +554,7 @@ struct sched_statistics {
u64 nr_failed_migrations_running;
u64 nr_failed_migrations_hot;
u64 nr_forced_migrations;
+ u64 nr_migrations_cpu_non_preferred;
u64 nr_wakeups;
u64 nr_wakeups_sync;
@@ -1172,8 +1173,12 @@ struct task_struct {
/* Objective and real subjective task credentials (COW): */
const struct cred __rcu *real_cred;
- /* Effective (overridable) subjective task credentials (COW): */
- const struct cred __rcu *cred;
+ /*
+ * Effective (overridable) subjective task credentials (COW).
+ * Only accessible for the current task and during task creation/freeing.
+ * This pointer is not managed by RCU!
+ */
+ const struct cred *cred;
#ifdef CONFIG_KEYS
/* Cached requested key. */
@@ -1433,6 +1438,7 @@ struct task_struct {
#ifdef CONFIG_SCHED_CACHE
struct callback_head cache_work;
+ struct sched_cache_group __rcu *sched_cache_grp;
int preferred_llc;
/* 1: task was enqueued to its preferred LLC, 0 otherwise */
int pref_llc_queued;
@@ -1787,7 +1793,7 @@ static inline bool is_lazy_mmu_mode_active(void)
}
#endif
-extern struct pid *cad_pid;
+extern struct pid __rcu *cad_pid;
/*
* Per process flags
@@ -1808,15 +1814,15 @@ extern struct pid *cad_pid;
#define PF_USED_MATH 0x00002000 /* If unset the fpu must be initialized before use */
#define PF_USER_WORKER 0x00004000 /* Kernel thread cloned from userspace thread */
#define PF_NOFREEZE 0x00008000 /* This thread should not be frozen */
-#define PF_KCOMPACTD 0x00010000 /* I am kcompactd */
-#define PF_KSWAPD 0x00020000 /* I am kswapd */
+#define PF__HOLE__00010000 0x00010000
+#define PF__HOLE__00020000 0x00020000
#define PF_MEMALLOC_NOFS 0x00040000 /* All allocations inherit GFP_NOFS. See memalloc_nfs_save() */
#define PF_MEMALLOC_NOIO 0x00080000 /* All allocations inherit GFP_NOIO. See memalloc_noio_save() */
#define PF_LOCAL_THROTTLE 0x00100000 /* Throttle writes only against the bdi I write to,
* I am cleaning dirty pages from some other bdi. */
#define PF_KTHREAD 0x00200000 /* I am a kernel thread */
#define PF_RANDOMIZE 0x00400000 /* Randomize virtual address space */
-#define PF__HOLE__00800000 0x00800000
+#define PF_NO_NOTIFY_SIGNAL 0x00800000 /* see no_notify_signal_save() */
#define PF__HOLE__01000000 0x01000000
#define PF__HOLE__02000000 0x02000000
#define PF_NO_SETAFFINITY 0x04000000 /* Userland is not allowed to meddle with cpus_mask */
@@ -2134,44 +2140,19 @@ static inline void set_need_resched_current(void)
* value indicates whether a reschedule was done in fact.
* cond_resched_lock() will drop the spinlock before scheduling,
*/
-#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
+#if !defined(CONFIG_PREEMPTION)
extern int __cond_resched(void);
-#if defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-
-DECLARE_STATIC_CALL(cond_resched, __cond_resched);
-
-static __always_inline int _cond_resched(void)
-{
- return static_call_mod(cond_resched)();
-}
-
-#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-
-extern int dynamic_cond_resched(void);
-
-static __always_inline int _cond_resched(void)
-{
- return dynamic_cond_resched();
-}
-
-#else /* !CONFIG_PREEMPTION */
-
static inline int _cond_resched(void)
{
return __cond_resched();
}
-
-#endif /* PREEMPT_DYNAMIC && CONFIG_HAVE_PREEMPT_DYNAMIC_CALL */
-
-#else /* CONFIG_PREEMPTION && !CONFIG_PREEMPT_DYNAMIC */
-
+#else
static inline int _cond_resched(void)
{
return 0;
}
-
-#endif /* !CONFIG_PREEMPTION || CONFIG_PREEMPT_DYNAMIC */
+#endif
#define cond_resched() ({ \
__might_resched(__FILE__, __LINE__, 0); \
@@ -2405,7 +2386,7 @@ struct sched_cache_time {
unsigned long epoch;
};
-struct sched_cache_stat {
+struct sched_cache_group {
struct sched_cache_time __percpu *pcpu_sched;
raw_spinlock_t lock;
unsigned long epoch;
@@ -2413,11 +2394,26 @@ struct sched_cache_stat {
unsigned long next_scan;
unsigned long footprint;
int cpu;
+ refcount_t refcnt;
+ struct rcu_head rcu;
} ____cacheline_aligned_in_smp;
+struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp);
+struct sched_cache_group *task_cache_group_get(struct task_struct *p);
+
+void sched_cache_fork(struct task_struct *p);
+void sched_cache_fork_cleanup(struct task_struct *p);
+void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm);
+void sched_cache_exit_mm(struct task_struct *p);
+
#else
-struct sched_cache_stat { };
+struct sched_cache_group { };
+
+static inline void sched_cache_fork(struct task_struct *p) { }
+static inline void sched_cache_fork_cleanup(struct task_struct *p) { }
+static inline void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm) { }
+static inline void sched_cache_exit_mm(struct task_struct *p) { }
#endif
diff --git a/include/linux/sched/cputime.h b/include/linux/sched/cputime.h
index e90efaf6d26e..694126411dfe 100644
--- a/include/linux/sched/cputime.h
+++ b/include/linux/sched/cputime.h
@@ -182,9 +182,9 @@ extern unsigned long long
task_sched_runtime(struct task_struct *task);
#ifdef CONFIG_PARAVIRT
-struct static_key;
-extern struct static_key paravirt_steal_enabled;
-extern struct static_key paravirt_steal_rq_enabled;
+#include <linux/jump_label.h>
+DECLARE_STATIC_KEY_FALSE(paravirt_steal_enabled);
+DECLARE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled);
#ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN
u64 dummy_steal_clock(int cpu);
diff --git a/include/linux/sched/ext.h b/include/linux/sched/ext.h
index 582d7cd4a983..23f9e178bc5a 100644
--- a/include/linux/sched/ext.h
+++ b/include/linux/sched/ext.h
@@ -91,7 +91,7 @@ struct scx_dispatch_q {
struct rhash_head hash_node;
struct llist_node free_node;
struct scx_sched *sched;
- struct scx_dsq_pcpu __percpu *pcpu;
+ struct scx_dsq_pcpu __percpu *pcpu_user;
struct rcu_head rcu;
};
@@ -104,6 +104,7 @@ enum scx_ent_flags {
SCX_TASK_SUB_INIT = 1 << 4, /* task being initialized for a sub sched */
SCX_TASK_IMMED = 1 << 5, /* task is on local DSQ with %SCX_ENQ_IMMED */
SCX_TASK_PROTECTED = 1 << 6, /* slice and DSQ head position protected */
+ SCX_TASK_RUN_TRACKED = 1 << 7, /* task is in an ops.running()/stopping() session */
/*
* Bits 8 to 10 are used to carry task state:
@@ -136,6 +137,7 @@ enum scx_ent_flags {
* IMMED reenqueued due to failed ENQ_IMMED
* PREEMPTED preempted while running
* CAP sub-sched cap miss, see p->scx.reenq_reason_*
+ * PROXY proxy state prevented a remote DSQ transfer
*/
SCX_TASK_REENQ_REASON_SHIFT = 12,
SCX_TASK_REENQ_REASON_BITS = 3,
@@ -146,6 +148,7 @@ enum scx_ent_flags {
SCX_TASK_REENQ_IMMED = 2 << SCX_TASK_REENQ_REASON_SHIFT,
SCX_TASK_REENQ_PREEMPTED = 3 << SCX_TASK_REENQ_REASON_SHIFT,
SCX_TASK_REENQ_CAP = 4 << SCX_TASK_REENQ_REASON_SHIFT,
+ SCX_TASK_REENQ_PROXY = 5 << SCX_TASK_REENQ_REASON_SHIFT,
/* iteration cursor, not a task */
SCX_TASK_CURSOR = 1 << 31,
@@ -197,9 +200,19 @@ struct sched_ext_entity {
u64 ddsp_slice;
u64 ddsp_vtime;
struct scx_dsq_list_node dsq_list; /* dispatch order */
- struct rb_node dsq_priq; /* p->scx.dsq_vtime order */
u32 dsq_seq;
u32 dsq_flags; /* protected by DSQ lock */
+
+ /*
+ * Used to order tasks when dispatching to the vtime-ordered priority
+ * queue of a dsq. This is usually set through
+ * scx_bpf_dsq_insert_vtime() but can also be modified directly by the
+ * BPF scheduler. Modifying it while a task is queued on a dsq may
+ * mangle the ordering and is not recommended. Kept next to @dsq_priq
+ * as rbtree insertion reads both on every visited node.
+ */
+ u64 dsq_vtime;
+ struct rb_node dsq_priq; /* p->scx.dsq_vtime order */
u32 flags; /* protected by rq lock */
u32 weight;
u32 reenq_cnt; /* reenqueues since last run */
@@ -225,7 +238,7 @@ struct sched_ext_entity {
u64 tid;
struct rhash_head tid_hash_node; /* see SCX_OPS_TID_TO_TASK */
- /* BPF scheduler modifiable fields */
+ /* BPF scheduler modifiable fields, along with @dsq_vtime above */
/*
* Runtime budget in nsecs - how long the task may hold its cpu. Owned
@@ -241,15 +254,6 @@ struct sched_ext_entity {
u64 slice;
/*
- * Used to order tasks when dispatching to the vtime-ordered priority
- * queue of a dsq. This is usually set through
- * scx_bpf_dsq_insert_vtime() but can also be modified directly by the
- * BPF scheduler. Modifying it while a task is queued on a dsq may
- * mangle the ordering and is not recommended.
- */
- u64 dsq_vtime;
-
- /*
* Out-of-band slice request from scx_bpf_task_set_slice() when the
* caller does not hold the rq lock, applied under the rq lock at the
* next slice consideration. One atomic64 packs the pending flag, the
@@ -278,6 +282,14 @@ struct sched_ext_entity {
*/
bool disallow; /* reject switching into SCX */
+ /*
+ * If set, depletion of this task's slice at the scheduler tick requests
+ * lazy instead of immediate rescheduling. Initialized from
+ * %SCX_OPS_LAZY_RESCHED immediately before ops.enable() and may be
+ * modified afterwards with scx_bpf_task_set_lazy_resched().
+ */
+ bool lazy_resched;
+
/* cold fields */
#ifdef CONFIG_EXT_GROUP_SCHED
struct cgroup *cgrp_moving_from;
@@ -323,7 +335,7 @@ struct scx_task_group {
u64 bw_period_us;
u64 bw_quota_us;
u64 bw_burst_us;
- bool idle;
+ bool sched_idle;
#endif
};
diff --git a/include/linux/sched/rt.h b/include/linux/sched/rt.h
index 4e3338103654..922935cc3383 100644
--- a/include/linux/sched/rt.h
+++ b/include/linux/sched/rt.h
@@ -52,8 +52,10 @@ static inline bool rt_or_dl_task_policy(struct task_struct *tsk)
#ifdef CONFIG_RT_MUTEXES
extern void rt_mutex_pre_schedule(void);
+extern void rt_mutex_futex_pre_schedule(void);
extern void rt_mutex_schedule(void);
extern void rt_mutex_post_schedule(void);
+extern void rt_mutex_futex_post_schedule(void);
/*
* Must hold either p->pi_lock or task_rq(p)->lock.
diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h
index 584ae88b435e..d9419dc902f6 100644
--- a/include/linux/sched/signal.h
+++ b/include/linux/sched/signal.h
@@ -2,6 +2,7 @@
#ifndef _LINUX_SCHED_SIGNAL_H
#define _LINUX_SCHED_SIGNAL_H
+#include <linux/cleanup.h>
#include <linux/rculist.h>
#include <linux/signal.h>
#include <linux/sched.h>
@@ -79,9 +80,9 @@ struct core_thread {
};
struct core_state {
- atomic_t nr_threads;
- struct core_thread dumper;
- struct completion startup;
+ /* Threads the dumper still waits for. */
+ atomic_t threads_remaining;
+ struct core_thread *tasks;
};
/*
@@ -384,14 +385,36 @@ static inline int task_sigpending(struct task_struct *p)
return unlikely(test_tsk_thread_flag(p,TIF_SIGPENDING));
}
+/* Prevent TIF_NOTIFY_SIGNAL from interrupting this task. */
+static inline unsigned int no_notify_signal_save(void)
+{
+ unsigned int flags = current->flags;
+
+ current->flags |= PF_NO_NOTIFY_SIGNAL;
+ return flags;
+}
+
+/* Restore the previous PF_NO_NOTIFY_SIGNAL state. */
+static inline void no_notify_signal_restore(unsigned int flags)
+{
+ current_restore_flags(flags, PF_NO_NOTIFY_SIGNAL);
+}
+
+DEFINE_LOCK_GUARD_0(no_notify_signal,
+ _T->flags = no_notify_signal_save(),
+ no_notify_signal_restore(_T->flags),
+ unsigned int flags)
+
static inline int signal_pending(struct task_struct *p)
{
/*
* TIF_NOTIFY_SIGNAL isn't really a signal, but it requires the same
* behavior in terms of ensuring that we break out of wait loops
- * so that notify signal callbacks can be processed.
+ * so that notify signal callbacks can be processed. Not for a task
+ * that asked not to be interrupted by it, see no_notify_signal_save().
*/
- if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)))
+ if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)) &&
+ likely(!(READ_ONCE(p->flags) & PF_NO_NOTIFY_SIGNAL)))
return 1;
return task_sigpending(p);
}
@@ -562,10 +585,7 @@ static inline sigset_t *sigmask_to_save(void)
return res;
}
-static inline int kill_cad_pid(int sig, int priv)
-{
- return kill_pid(cad_pid, sig, priv);
-}
+int kill_cad_pid(int sig, int priv);
/* These can be the second arg to send_sig_info/send_group_sig_info. */
#define SEND_SIG_NOINFO ((struct kernel_siginfo *) 0)
diff --git a/include/linux/sched/task.h b/include/linux/sched/task.h
index e0c1ca8c6a18..90ed5bc3c7af 100644
--- a/include/linux/sched/task.h
+++ b/include/linux/sched/task.h
@@ -94,7 +94,6 @@ static inline void exit_thread(struct task_struct *tsk)
extern __noreturn void do_group_exit(int);
extern void exit_files(struct task_struct *);
-extern void exit_itimers(struct task_struct *);
extern pid_t kernel_clone(struct kernel_clone_args *kargs);
struct task_struct *copy_process(struct pid *pid, int trace, int node,
diff --git a/include/linux/sched/topology.h b/include/linux/sched/topology.h
index b5d9d7c2b8ad..f96812d71c51 100644
--- a/include/linux/sched/topology.h
+++ b/include/linux/sched/topology.h
@@ -281,9 +281,9 @@ static inline int task_node(const struct task_struct *p)
}
#ifdef CONFIG_SCHED_CACHE
-extern void sched_update_llc_bytes(unsigned int cpu);
+extern void sched_update_llc_bytes(const struct cpumask *cpus);
#else
-static inline void sched_update_llc_bytes(unsigned int cpu) { }
+static inline void sched_update_llc_bytes(const struct cpumask *cpus) { }
#endif
#endif /* _LINUX_SCHED_TOPOLOGY_H */
diff --git a/include/linux/sched/user.h b/include/linux/sched/user.h
index 4cc52698e214..8d7e5521f7cd 100644
--- a/include/linux/sched/user.h
+++ b/include/linux/sched/user.h
@@ -25,7 +25,8 @@ struct user_struct {
#if defined(CONFIG_PERF_EVENTS) || defined(CONFIG_BPF_SYSCALL) || \
defined(CONFIG_NET) || defined(CONFIG_IO_URING) || \
- defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD)
+ defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD) || \
+ defined(CONFIG_SECRETMEM)
atomic_long_t locked_vm;
#endif
#ifdef CONFIG_WATCH_QUEUE
diff --git a/include/linux/scmi_protocol.h b/include/linux/scmi_protocol.h
index 5ab73b1ab9aa..3f7cc35eb861 100644
--- a/include/linux/scmi_protocol.h
+++ b/include/linux/scmi_protocol.h
@@ -9,6 +9,7 @@
#define _LINUX_SCMI_PROTOCOL_H
#include <linux/bitfield.h>
+#include <linux/device-id/scmi.h>
#include <linux/device.h>
#include <linux/notifier.h>
#include <linux/types.h>
@@ -528,22 +529,25 @@ struct scmi_sensor_proto_ops {
u32 sensor_id, u32 sensor_config);
};
+struct scmi_reset_domain_info {
+ char name[SCMI_MAX_STR_SIZE];
+ u32 latency_us;
+};
+
/**
* struct scmi_reset_proto_ops - represents the various operations provided
* by SCMI Reset Protocol
*
* @num_domains_get: get the count of reset domains provided by SCMI
- * @name_get: gets the name of a reset domain
- * @latency_get: gets the reset latency for the specified reset domain
+ * @info_get: gets the information of the specified reset domain
* @reset: resets the specified reset domain
* @assert: explicitly assert reset signal of the specified reset domain
* @deassert: explicitly deassert reset signal of the specified reset domain
*/
struct scmi_reset_proto_ops {
int (*num_domains_get)(const struct scmi_protocol_handle *ph);
- const char *(*name_get)(const struct scmi_protocol_handle *ph,
- u32 domain);
- int (*latency_get)(const struct scmi_protocol_handle *ph, u32 domain);
+ const struct scmi_reset_domain_info __must_check *(*info_get)
+ (const struct scmi_protocol_handle *ph, u32 domain);
int (*reset)(const struct scmi_protocol_handle *ph, u32 domain);
int (*assert)(const struct scmi_protocol_handle *ph, u32 domain);
int (*deassert)(const struct scmi_protocol_handle *ph, u32 domain);
@@ -930,6 +934,7 @@ enum scmi_std_protocol {
SCMI_PROTOCOL_VOLTAGE = 0x17,
SCMI_PROTOCOL_POWERCAP = 0x18,
SCMI_PROTOCOL_PINCTRL = 0x19,
+ SCMI_PROTOCOL_TELEMETRY = 0x1B,
};
enum scmi_system_events {
@@ -951,11 +956,6 @@ struct scmi_device {
#define to_scmi_dev(d) container_of_const(d, struct scmi_device, dev)
-struct scmi_device_id {
- u8 protocol_id;
- const char *name;
-};
-
struct scmi_driver {
const char *name;
int (*probe)(struct scmi_device *sdev);
diff --git a/include/linux/security.h b/include/linux/security.h
index 153e9043058f..8b8d6b8b802f 100644
--- a/include/linux/security.h
+++ b/include/linux/security.h
@@ -67,6 +67,7 @@ enum fs_value_type;
struct watch;
struct watch_notification;
struct lsm_ctx;
+struct nsset;
/* Default (no) options for the capable function */
#define CAP_OPT_NONE 0x0
@@ -80,6 +81,7 @@ struct lsm_ctx;
struct ctl_table;
struct audit_krule;
+struct ns_common;
struct user_namespace;
struct timezone;
@@ -185,11 +187,11 @@ extern int cap_capset(struct cred *new, const struct cred *old,
extern int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file);
int cap_inode_setxattr(struct dentry *dentry, const char *name,
const void *value, size_t size, int flags);
-int cap_inode_removexattr(struct mnt_idmap *idmap,
+int cap_inode_removexattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name);
int cap_inode_need_killpriv(struct dentry *dentry);
-int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry);
-int cap_inode_getsecurity(struct mnt_idmap *idmap,
+int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry);
+int cap_inode_getsecurity(const struct mnt_idmap *idmap,
struct inode *inode, const char *name, void **buffer,
bool alloc);
extern int cap_mmap_addr(unsigned long addr);
@@ -338,6 +340,7 @@ int security_binder_transfer_file(const struct cred *from,
const struct cred *to, const struct file *file);
int security_ptrace_access_check(struct task_struct *child, unsigned int mode);
int security_ptrace_traceme(struct task_struct *parent);
+int security_mem_foll_force(const struct cred *subject, bool opened_by_owner);
int security_capget(const struct task_struct *target,
kernel_cap_t *effective,
kernel_cap_t *inheritable,
@@ -404,49 +407,53 @@ int security_inode_init_security(struct inode *inode, struct inode *dir,
int security_inode_init_security_anon(struct inode *inode,
const struct qstr *name,
const struct inode *context_inode);
-int security_inode_create(struct inode *dir, struct dentry *dentry, umode_t mode);
-void security_inode_post_create_tmpfile(struct mnt_idmap *idmap,
+int security_inode_create(const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, umode_t mode);
+void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap,
struct inode *inode);
-int security_inode_link(struct dentry *old_dentry, struct inode *dir,
- struct dentry *new_dentry);
+int security_inode_link(const struct mnt_idmap *idmap, struct dentry *old_dentry,
+ struct inode *dir, struct dentry *new_dentry);
int security_inode_unlink(struct inode *dir, struct dentry *dentry);
-int security_inode_symlink(struct inode *dir, struct dentry *dentry,
- const char *old_name);
-int security_inode_mkdir(struct inode *dir, struct dentry *dentry, umode_t mode);
+int security_inode_symlink(const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, const char *old_name);
+int security_inode_mkdir(const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, umode_t mode);
int security_inode_rmdir(struct inode *dir, struct dentry *dentry);
-int security_inode_mknod(struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev);
+int security_inode_mknod(const struct mnt_idmap *idmap, struct inode *dir,
+ struct dentry *dentry, umode_t mode, dev_t dev);
int security_inode_rename(struct inode *old_dir, struct dentry *old_dentry,
struct inode *new_dir, struct dentry *new_dentry,
unsigned int flags);
int security_inode_readlink(struct dentry *dentry);
int security_inode_follow_link(struct dentry *dentry, struct inode *inode,
bool rcu);
-int security_inode_permission(struct inode *inode, int mask);
-int security_inode_setattr(struct mnt_idmap *idmap,
+int security_inode_permission(const struct mnt_idmap *idmap, struct inode *inode,
+ int mask);
+int security_inode_setattr(const struct mnt_idmap *idmap,
struct dentry *dentry, struct iattr *attr);
-void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
+void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry,
int ia_valid);
int security_inode_getattr(const struct path *path);
-int security_inode_setxattr(struct mnt_idmap *idmap,
+int security_inode_setxattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name,
const void *value, size_t size, int flags);
-int security_inode_set_acl(struct mnt_idmap *idmap,
+int security_inode_set_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name,
struct posix_acl *kacl);
void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name,
struct posix_acl *kacl);
-int security_inode_get_acl(struct mnt_idmap *idmap,
+int security_inode_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name);
-int security_inode_remove_acl(struct mnt_idmap *idmap,
+int security_inode_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name);
-void security_inode_post_remove_acl(struct mnt_idmap *idmap,
+void security_inode_post_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name);
void security_inode_post_setxattr(struct dentry *dentry, const char *name,
const void *value, size_t size, int flags);
int security_inode_getxattr(struct dentry *dentry, const char *name);
int security_inode_listxattr(struct dentry *dentry);
-int security_inode_removexattr(struct mnt_idmap *idmap,
+int security_inode_removexattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name);
void security_inode_post_removexattr(struct dentry *dentry, const char *name);
int security_inode_file_setattr(struct dentry *dentry,
@@ -454,8 +461,8 @@ int security_inode_file_setattr(struct dentry *dentry,
int security_inode_file_getattr(struct dentry *dentry,
struct file_kattr *fa);
int security_inode_need_killpriv(struct dentry *dentry);
-int security_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry);
-int security_inode_getsecurity(struct mnt_idmap *idmap,
+int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry);
+int security_inode_getsecurity(const struct mnt_idmap *idmap,
struct inode *inode, const char *name,
void **buffer, bool alloc);
int security_inode_setsecurity(struct inode *inode, const char *name, const void *value, size_t size, int flags);
@@ -468,7 +475,7 @@ int security_inode_setintegrity(const struct inode *inode,
size_t size);
int security_kernfs_init_security(struct kernfs_node *kn_dir,
struct kernfs_node *kn);
-int security_file_permission(struct file *file, int mask);
+int security_file_permission(const struct file *file, int mask);
int security_file_alloc(struct file *file);
void security_file_release(struct file *file);
void security_file_free(struct file *file);
@@ -540,6 +547,9 @@ int security_task_prctl(int option, unsigned long arg2, unsigned long arg3,
unsigned long arg4, unsigned long arg5);
void security_task_to_inode(struct task_struct *p, struct inode *inode);
int security_create_user_ns(const struct cred *cred);
+int security_namespace_init(struct ns_common *ns);
+void security_namespace_free(struct ns_common *ns);
+int security_namespace_install(const struct nsset *nsset, struct ns_common *ns);
int security_ipc_permission(struct kern_ipc_perm *ipcp, short flag);
void security_ipc_getlsmprop(struct kern_ipc_perm *ipcp, struct lsm_prop *prop);
int security_msg_msg_alloc(struct msg_msg *msg);
@@ -676,6 +686,12 @@ static inline int security_ptrace_traceme(struct task_struct *parent)
return cap_ptrace_traceme(parent);
}
+static inline int security_mem_foll_force(const struct cred *subject,
+ bool opened_by_owner)
+{
+ return 0;
+}
+
static inline int security_capget(const struct task_struct *target,
kernel_cap_t *effective,
kernel_cap_t *inheritable,
@@ -902,20 +918,22 @@ static inline int security_inode_init_security_anon(struct inode *inode,
return 0;
}
-static inline int security_inode_create(struct inode *dir,
- struct dentry *dentry,
- umode_t mode)
+static inline int security_inode_create(const struct mnt_idmap *idmap,
+ struct inode *dir,
+ struct dentry *dentry,
+ umode_t mode)
{
return 0;
}
static inline void
-security_inode_post_create_tmpfile(struct mnt_idmap *idmap, struct inode *inode)
+security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode)
{ }
-static inline int security_inode_link(struct dentry *old_dentry,
- struct inode *dir,
- struct dentry *new_dentry)
+static inline int security_inode_link(const struct mnt_idmap *idmap,
+ struct dentry *old_dentry,
+ struct inode *dir,
+ struct dentry *new_dentry)
{
return 0;
}
@@ -926,16 +944,18 @@ static inline int security_inode_unlink(struct inode *dir,
return 0;
}
-static inline int security_inode_symlink(struct inode *dir,
- struct dentry *dentry,
- const char *old_name)
+static inline int security_inode_symlink(const struct mnt_idmap *idmap,
+ struct inode *dir,
+ struct dentry *dentry,
+ const char *old_name)
{
return 0;
}
-static inline int security_inode_mkdir(struct inode *dir,
- struct dentry *dentry,
- int mode)
+static inline int security_inode_mkdir(const struct mnt_idmap *idmap,
+ struct inode *dir,
+ struct dentry *dentry,
+ int mode)
{
return 0;
}
@@ -946,9 +966,10 @@ static inline int security_inode_rmdir(struct inode *dir,
return 0;
}
-static inline int security_inode_mknod(struct inode *dir,
- struct dentry *dentry,
- int mode, dev_t dev)
+static inline int security_inode_mknod(const struct mnt_idmap *idmap,
+ struct inode *dir,
+ struct dentry *dentry,
+ int mode, dev_t dev)
{
return 0;
}
@@ -974,12 +995,13 @@ static inline int security_inode_follow_link(struct dentry *dentry,
return 0;
}
-static inline int security_inode_permission(struct inode *inode, int mask)
+static inline int security_inode_permission(const struct mnt_idmap *idmap,
+ struct inode *inode, int mask)
{
return 0;
}
-static inline int security_inode_setattr(struct mnt_idmap *idmap,
+static inline int security_inode_setattr(const struct mnt_idmap *idmap,
struct dentry *dentry,
struct iattr *attr)
{
@@ -987,7 +1009,7 @@ static inline int security_inode_setattr(struct mnt_idmap *idmap,
}
static inline void
-security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
+security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry,
int ia_valid)
{ }
@@ -996,14 +1018,14 @@ static inline int security_inode_getattr(const struct path *path)
return 0;
}
-static inline int security_inode_setxattr(struct mnt_idmap *idmap,
+static inline int security_inode_setxattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name, const void *value,
size_t size, int flags)
{
return cap_inode_setxattr(dentry, name, value, size, flags);
}
-static inline int security_inode_set_acl(struct mnt_idmap *idmap,
+static inline int security_inode_set_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name,
struct posix_acl *kacl)
@@ -1016,21 +1038,21 @@ static inline void security_inode_post_set_acl(struct dentry *dentry,
struct posix_acl *kacl)
{ }
-static inline int security_inode_get_acl(struct mnt_idmap *idmap,
+static inline int security_inode_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{
return 0;
}
-static inline int security_inode_remove_acl(struct mnt_idmap *idmap,
+static inline int security_inode_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{
return 0;
}
-static inline void security_inode_post_remove_acl(struct mnt_idmap *idmap,
+static inline void security_inode_post_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{ }
@@ -1050,7 +1072,7 @@ static inline int security_inode_listxattr(struct dentry *dentry)
return 0;
}
-static inline int security_inode_removexattr(struct mnt_idmap *idmap,
+static inline int security_inode_removexattr(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *name)
{
@@ -1078,13 +1100,13 @@ static inline int security_inode_need_killpriv(struct dentry *dentry)
return cap_inode_need_killpriv(dentry);
}
-static inline int security_inode_killpriv(struct mnt_idmap *idmap,
+static inline int security_inode_killpriv(const struct mnt_idmap *idmap,
struct dentry *dentry)
{
return cap_inode_killpriv(idmap, dentry);
}
-static inline int security_inode_getsecurity(struct mnt_idmap *idmap,
+static inline int security_inode_getsecurity(const struct mnt_idmap *idmap,
struct inode *inode,
const char *name, void **buffer,
bool alloc)
@@ -1132,7 +1154,7 @@ static inline int security_inode_copy_up_xattr(struct dentry *src, const char *n
return -EOPNOTSUPP;
}
-static inline int security_file_permission(struct file *file, int mask)
+static inline int security_file_permission(const struct file *file, int mask)
{
return 0;
}
@@ -1431,6 +1453,21 @@ static inline int security_create_user_ns(const struct cred *cred)
return 0;
}
+static inline int security_namespace_init(struct ns_common *ns)
+{
+ return 0;
+}
+
+static inline void security_namespace_free(struct ns_common *ns)
+{
+}
+
+static inline int security_namespace_install(const struct nsset *nsset,
+ struct ns_common *ns)
+{
+ return 0;
+}
+
static inline int security_ipc_permission(struct kern_ipc_perm *ipcp,
short flag)
{
@@ -2085,7 +2122,7 @@ int security_path_mkdir(const struct path *dir, struct dentry *dentry, umode_t m
int security_path_rmdir(const struct path *dir, struct dentry *dentry);
int security_path_mknod(const struct path *dir, struct dentry *dentry, umode_t mode,
unsigned int dev);
-void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry);
+void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry);
int security_path_truncate(const struct path *path);
int security_path_symlink(const struct path *dir, struct dentry *dentry,
const char *old_name);
@@ -2120,7 +2157,7 @@ static inline int security_path_mknod(const struct path *dir, struct dentry *den
return 0;
}
-static inline void security_path_post_mknod(struct mnt_idmap *idmap,
+static inline void security_path_post_mknod(const struct mnt_idmap *idmap,
struct dentry *dentry)
{ }
diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h
index 3030d9245f5a..27c32fbabfec 100644
--- a/include/linux/set_memory.h
+++ b/include/linux/set_memory.h
@@ -5,16 +5,92 @@
#ifndef _LINUX_SET_MEMORY_H_
#define _LINUX_SET_MEMORY_H_
+/**
+ * DOC: Kernel page table permissions
+ *
+ * The set_memory() and set_direct_map() APIs update permissions of existing
+ * kernel mappings.
+ *
+ * The set_memory() functions operate on a range of kernel virtual addresses,
+ * the set_direct_map() functions operate on the direct map.
+ *
+ * The updates are not atomic: when a call fails, an arbitrary prefix of the
+ * range may have been updated already and there is no automatic rollback.
+ * A caller must restore the required permissions before reusing or freeing
+ * the memory.
+ *
+ * When an architecture does not implement these APIs they succeed without
+ * doing anything, so a return value of 0 does not mean that the permissions
+ * were actually changed.
+ *
+ * Callers that depend on the permissions being applied must ensure that the
+ * architecture supports the required operation for the target addresses. The
+ * Kconfig symbols alone do not guarantee this.
+ *
+ * See Documentation/mm/kernel-page-tables.rst for the details and for the
+ * differences between the architecture implementations.
+ */
+
#ifdef CONFIG_ARCH_HAS_SET_MEMORY
#include <asm/set_memory.h>
#else
+/**
+ * set_memory_ro - make a kernel mapping read-only
+ * @addr: page aligned start of the kernel virtual address range
+ * @numpages: number of pages in the range
+ *
+ * Flushes the TLB for the range.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
static inline int __must_check set_memory_ro(unsigned long addr, int numpages) { return 0; }
+
+/**
+ * set_memory_rw - make a kernel mapping writable
+ * @addr: page aligned start of the kernel virtual address range
+ * @numpages: number of pages in the range
+ *
+ * Flushes the TLB for the range.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
static inline int __must_check set_memory_rw(unsigned long addr, int numpages) { return 0; }
+
+/**
+ * set_memory_x - make a kernel mapping executable
+ * @addr: page aligned start of the kernel virtual address range
+ * @numpages: number of pages in the range
+ *
+ * Flushes the TLB for the range.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
static inline int __must_check set_memory_x(unsigned long addr, int numpages) { return 0; }
+
+/**
+ * set_memory_nx - make a kernel mapping non-executable
+ * @addr: page aligned start of the kernel virtual address range
+ * @numpages: number of pages in the range
+ *
+ * Flushes the TLB for the range.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
static inline int __must_check set_memory_nx(unsigned long addr, int numpages) { return 0; }
#endif
#ifndef set_memory_rox
+/**
+ * set_memory_rox - make a kernel mapping read-only and executable
+ * @addr: page aligned start of the kernel virtual address range
+ * @numpages: number of pages in the range
+ *
+ * A failure may leave the range read-only but not executable.
+ *
+ * Flushes the TLB for the range.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
static inline int set_memory_rox(unsigned long addr, int numpages)
{
int ret = set_memory_ro(addr, numpages);
@@ -25,17 +101,35 @@ static inline int set_memory_rox(unsigned long addr, int numpages)
#endif
#ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP
-static inline int set_direct_map_invalid_noflush(struct page *page)
-{
- return 0;
-}
-static inline int set_direct_map_default_noflush(struct page *page)
+/**
+ * set_direct_map_invalid_noflush - remove pages from the direct map
+ * @page: first page to update
+ * @nr: number of pages to update
+ *
+ * Makes the direct mapping of @nr pages starting at @page not present.
+ * The caller is responsible for any required TLB flushing.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
+static inline int set_direct_map_invalid_noflush(struct page *page,
+ unsigned int nr)
{
return 0;
}
-static inline int set_direct_map_valid_noflush(struct page *page,
- unsigned nr, bool valid)
+/**
+ * set_direct_map_default_noflush - restore the direct map of pages
+ * @page: first page to update
+ * @nr: number of pages to update
+ *
+ * Restores the default kernel permissions of the direct mapping of @nr
+ * pages starting at @page.
+ * The caller is responsible for any required TLB flushing.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
+static inline int set_direct_map_default_noflush(struct page *page,
+ unsigned int nr)
{
return 0;
}
@@ -50,6 +144,18 @@ static inline bool kernel_page_present(struct page *page)
* boot time. Let them overrive this query.
*/
#ifndef can_set_direct_map
+/**
+ * can_set_direct_map - check if the direct map can be modified
+ *
+ * Available with CONFIG_ARCH_HAS_SET_DIRECT_MAP. Architectures may override
+ * this to report whether direct map updates are enabled at runtime.
+ * Even though the generic implementation returns true this does not guarantee
+ * that every address can be updated.
+ *
+ * See Documentation/mm/kernel-page-tables.rst for the details
+ *
+ * Return: true unless the architecture reports direct map updates disabled.
+ */
static inline bool can_set_direct_map(void)
{
return true;
diff --git a/include/linux/shmem_fs.h b/include/linux/shmem_fs.h
index 5663dff53186..a7c7a96a7cbf 100644
--- a/include/linux/shmem_fs.h
+++ b/include/linux/shmem_fs.h
@@ -11,6 +11,7 @@
#include <linux/fs_parser.h>
#include <linux/userfaultfd_k.h>
#include <linux/bits.h>
+#include <linux/list_lru.h>
/* inode in-kernel data */
@@ -54,6 +55,11 @@ struct shmem_inode_info {
struct dquot __rcu *i_dquot[MAXQUOTAS];
#endif
struct inode vfs_inode;
+
+#ifdef CONFIG_TRANSPARENT_HUGEPAGE
+ struct obj_cgroup *shrinklist_objcg;
+ int shrinklist_nid;
+#endif
};
#define SHMEM_FL_USER_VISIBLE (FS_FL_USER_VISIBLE | FS_CASEFOLD_FL)
@@ -83,9 +89,9 @@ struct shmem_sb_info {
ino_t next_ino; /* The next per-sb inode number to use */
ino_t __percpu *ino_batch; /* The next per-cpu inode number to use */
struct mempolicy *mpol; /* default memory policy for mappings */
- spinlock_t shrinklist_lock; /* Protects shrinklist */
- struct list_head shrinklist; /* List of shinkable inodes */
- unsigned long shrinklist_len; /* Length of shrinklist */
+#ifdef CONFIG_TRANSPARENT_HUGEPAGE
+ struct list_lru shrinklist; /* List of shrinkable inodes */
+#endif
struct shmem_quota_limits qlimits; /* Default quota limits */
struct simple_xattr_cache xa_cache;
};
diff --git a/include/linux/skbuff.h b/include/linux/skbuff.h
index 671c13494566..27ec1e38c828 100644
--- a/include/linux/skbuff.h
+++ b/include/linux/skbuff.h
@@ -1020,7 +1020,7 @@ struct sk_buff {
#ifdef CONFIG_NET_REDIRECT
__u8 from_ingress:1;
#endif
-#ifdef CONFIG_NETFILTER_SKIP_EGRESS
+#ifdef CONFIG_NET_EGRESS
__u8 nf_skip_egress:1;
#endif
#ifdef CONFIG_SKB_DECRYPTED
@@ -1834,22 +1834,6 @@ static inline void skb_zcopy_set(struct sk_buff *skb, struct ubuf_info *uarg,
}
}
-static inline void skb_zcopy_set_nouarg(struct sk_buff *skb, void *val)
-{
- skb_shinfo(skb)->destructor_arg = (void *)((uintptr_t) val | 0x1UL);
- skb_shinfo(skb)->flags |= SKBFL_ZEROCOPY_FRAG;
-}
-
-static inline bool skb_zcopy_is_nouarg(struct sk_buff *skb)
-{
- return (uintptr_t) skb_shinfo(skb)->destructor_arg & 0x1UL;
-}
-
-static inline void *skb_zcopy_get_nouarg(struct sk_buff *skb)
-{
- return (void *)((uintptr_t) skb_shinfo(skb)->destructor_arg & ~0x1UL);
-}
-
static inline void net_zcopy_put(struct ubuf_info *uarg)
{
if (uarg)
@@ -1872,8 +1856,7 @@ static inline void skb_zcopy_clear(struct sk_buff *skb, bool zerocopy_success)
struct ubuf_info *uarg = skb_zcopy(skb);
if (uarg) {
- if (!skb_zcopy_is_nouarg(skb))
- uarg->ops->complete(skb, uarg, zerocopy_success);
+ uarg->ops->complete(skb, uarg, zerocopy_success);
skb_shinfo(skb)->flags &= ~SKBFL_ALL_ZEROCOPY;
}
@@ -3082,6 +3065,11 @@ static inline bool skb_transport_header_was_set(const struct sk_buff *skb)
return skb->transport_header != (typeof(skb->transport_header))~0U;
}
+static inline void skb_unset_transport_header(struct sk_buff *skb)
+{
+ skb->transport_header = (typeof(skb->transport_header))~0U;
+}
+
static inline unsigned char *skb_transport_header(const struct sk_buff *skb)
{
DEBUG_NET_WARN_ON_ONCE(!skb_transport_header_was_set(skb));
@@ -3828,9 +3816,12 @@ static inline dma_addr_t __skb_frag_dma_map(struct device *dev,
size_t offset, size_t size,
enum dma_data_direction dir)
{
+ dma_addr_t addr;
+
if (skb_frag_is_net_iov(frag)) {
- return netmem_to_net_iov(frag->netmem)->desc.dma_addr +
- offset + frag->offset;
+ addr = netmem_dma_addr_decode(
+ netmem_get_dma_addr(frag->netmem));
+ return addr + offset + frag->offset;
}
return dma_map_page(dev, skb_frag_page(frag),
skb_frag_off(frag) + offset, size, dir);
@@ -4367,7 +4358,10 @@ skb_header_pointer_careful(const struct sk_buff *skb, int offset,
static inline void * __must_check
skb_pointer_if_linear(const struct sk_buff *skb, int offset, int len)
{
- if (likely(skb_headlen(skb) - offset >= len))
+ unsigned int uoffset = (unsigned int)offset;
+
+ if (likely(uoffset <= skb_headlen(skb) &&
+ (unsigned int)len <= skb_headlen(skb) - uoffset))
return skb->data + offset;
return NULL;
}
diff --git a/include/linux/soc/qcom/apr.h b/include/linux/soc/qcom/apr.h
index 909e84f84e0c..2e4231092088 100644
--- a/include/linux/soc/qcom/apr.h
+++ b/include/linux/soc/qcom/apr.h
@@ -5,7 +5,6 @@
#include <linux/spinlock.h>
#include <linux/device.h>
-#include <linux/device-id/apr.h>
#include <dt-bindings/soc/qcom,apr.h>
#include <dt-bindings/soc/qcom,gpr.h>
@@ -135,6 +134,8 @@ struct pkt_router_svc {
typedef struct pkt_router_svc gpr_port_t;
+#define APR_NAME_SIZE 32
+
struct apr_device {
struct device dev;
uint16_t svc_id;
@@ -158,7 +159,6 @@ struct apr_driver {
const struct apr_resp_pkt *d);
gpr_port_cb gpr_callback;
struct device_driver driver;
- const struct apr_device_id *id_table;
};
typedef struct apr_driver gpr_driver_t;
diff --git a/include/linux/soc/qcom/geni-se.h b/include/linux/soc/qcom/geni-se.h
index 29a53bbc0dd4..5f18d281e6a4 100644
--- a/include/linux/soc/qcom/geni-se.h
+++ b/include/linux/soc/qcom/geni-se.h
@@ -347,17 +347,12 @@ struct geni_se {
#define QUP_SE_VERSION_2_5 0x20050000
/*
- * Define bandwidth thresholds that cause the underlying Core 2X interconnect
- * clock to run at the named frequency. These baseline values are recommended
- * by the hardware team, and are not dynamically scaled with GENI bandwidth
- * beyond basic on/off.
+ * QUP Core 2X clock votes used by GENI clients through the "qup-core" ICC
+ * path. Values are in Bps and must be converted with Bps_to_icc() before
+ * setting avg_bw.
*/
-#define CORE_2X_19_2_MHZ 960
-#define CORE_2X_50_MHZ 2500
-#define CORE_2X_100_MHZ 5000
-#define CORE_2X_150_MHZ 7500
-#define CORE_2X_200_MHZ 10000
-#define CORE_2X_236_MHZ 16383
+#define CORE_2X_19_2_MHZ 9600000
+#define CORE_2X_50_MHZ 25000000
#define GENI_DEFAULT_BW Bps_to_icc(1000)
diff --git a/include/linux/soc/qcom/llcc-qcom.h b/include/linux/soc/qcom/llcc-qcom.h
index f3ed63e475ab..713cb0221b14 100644
--- a/include/linux/soc/qcom/llcc-qcom.h
+++ b/include/linux/soc/qcom/llcc-qcom.h
@@ -74,6 +74,7 @@
#define LLCC_CAMSRTIP 73
#define LLCC_CAMRTRF 74
#define LLCC_CAMSRTRF 75
+#define LLCC_GPU_LITTLE 80
#define LLCC_OOBM_NS 81
#define LLCC_OOBM_S 82
#define LLCC_VIDEO_APV 83
@@ -85,7 +86,12 @@
#define LLCC_CAM_OFE_STROV 93
#define LLCC_CPUSS_HEU 94
#define LLCC_PCIE_TCU 97
+#define LLCC_GPUHTW_LITTLE 98
#define LLCC_MDM_PNG_FIXED 100
+#define LLCC_PPE_RXDESC 101
+#define LLCC_PPE_RXFILL 102
+#define LLCC_WLAN_5G 103
+#define LLCC_WLAN_6G 104
/**
* struct llcc_slice_desc - Cache slice descriptor
diff --git a/include/linux/soc/qcom/tc9563.h b/include/linux/soc/qcom/tc9563.h
new file mode 100644
index 000000000000..086f37a40d80
--- /dev/null
+++ b/include/linux/soc/qcom/tc9563.h
@@ -0,0 +1,19 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Copyright (c) 2026 Qualcomm Innovation Center, Inc. All rights reserved.
+ * Author: Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>
+ */
+
+#ifndef __QCOM_TC9563_H
+#define __QCOM_TC9563_H
+
+#define TC9563_GPIO_DEV_NAME "tc9563-gpio"
+
+#define TC9563_GPIO_IN0_OFFSET 0x801200
+#define TC9563_GPIO_EN0_OFFSET 0x801208
+#define TC9563_GPIO_OUT0_OFFSET 0x801210
+
+#define TC9563_GPIO_CONFIG TC9563_GPIO_EN0_OFFSET
+#define TC9563_RESET_GPIO TC9563_GPIO_OUT0_OFFSET
+
+#endif /* __QCOM_TC9563_H */
diff --git a/include/linux/soc/qcom/ubwc.h b/include/linux/soc/qcom/ubwc.h
index f3a70360b177..b09ae05cbeae 100644
--- a/include/linux/soc/qcom/ubwc.h
+++ b/include/linux/soc/qcom/ubwc.h
@@ -36,6 +36,7 @@ struct qcom_ubwc_cfg_data {
#define UBWC_4_3 0x40030000
#define UBWC_5_0 0x50000000
#define UBWC_6_0 0x60000000
+#define UBWC_7_0 0x70000000
#if IS_ENABLED(CONFIG_QCOM_UBWC_CONFIG)
const struct qcom_ubwc_cfg_data *qcom_ubwc_config_get_data(void);
@@ -108,6 +109,8 @@ static inline u32 qcom_ubwc_swizzle(const struct qcom_ubwc_cfg_data *cfg)
static inline u32 qcom_ubwc_version_tag(const struct qcom_ubwc_cfg_data *cfg)
{
+ if (cfg->ubwc_enc_version >= UBWC_7_0)
+ return 6;
if (cfg->ubwc_enc_version >= UBWC_6_0)
return 5;
if (cfg->ubwc_enc_version >= UBWC_5_0)
diff --git a/include/linux/soc/ti/k3-ringacc.h b/include/linux/soc/ti/k3-ringacc.h
index 39b022b92598..7b7fc237c08f 100644
--- a/include/linux/soc/ti/k3-ringacc.h
+++ b/include/linux/soc/ti/k3-ringacc.h
@@ -57,13 +57,13 @@ struct k3_ringacc;
struct k3_ring;
/**
- * enum k3_ring_cfg - RA ring configuration structure
+ * struct k3_ring_cfg - RA ring configuration structure
*
* @size: Ring size, number of elements
* @elm_size: Ring element size
* @mode: Ring operational mode
* @flags: Ring configuration flags. Possible values:
- * @K3_RINGACC_RING_SHARED: when set allows to request the same ring
+ * %K3_RINGACC_RING_SHARED: when set allows to request the same ring
* few times. It's usable when the same ring is used as Free Host PD ring
* for different flows, for example.
* Note: Locking should be done by consumer if required
@@ -88,11 +88,10 @@ struct k3_ring_cfg {
/**
* of_k3_ringacc_get_by_phandle - find a RA by phandle property
* @np: device node
- * @propname: property name containing phandle on RA node
+ * @property: property name containing phandle on RA node
*
- * Returns pointer on the RA - struct k3_ringacc
- * or -ENODEV if not found,
- * or -EPROBE_DEFER if not yet registered
+ * Return: Pointer to the RA, or ERR_PTR(-ENODEV) if the phandle cannot
+ * be resolved, or ERR_PTR(-EPROBE_DEFER) if the RA is not yet registered.
*/
struct k3_ringacc *of_k3_ringacc_get_by_phandle(struct device_node *np,
const char *property);
@@ -103,10 +102,8 @@ struct k3_ringacc *of_k3_ringacc_get_by_phandle(struct device_node *np,
* k3_ringacc_request_ring - request ring from ringacc
* @ringacc: pointer on ringacc
* @id: ring id or K3_RINGACC_RING_ID_ANY for any general purpose ring
- * @flags:
- * @K3_RINGACC_RING_USE_PROXY: if set - proxy will be allocated and
- * used to access ring memory. Sopported only for rings in
- * Message/Credentials/Queue mode.
+ * @flags: Set %K3_RINGACC_RING_USE_PROXY to allocate a proxy for
+ * accessing ring memory in Message mode.
*
* Returns pointer on the Ring - struct k3_ring
* or NULL in case of failure.
@@ -126,8 +123,9 @@ int k3_ringacc_request_rings_pair(struct k3_ringacc *ringacc,
*/
void k3_ringacc_ring_reset(struct k3_ring *ring);
/**
- * k3_ringacc_ring_reset - ring reset for DMA rings
+ * k3_ringacc_ring_reset_dma - ring reset for DMA rings
* @ring: pointer on Ring
+ * @occ: occupancy used by the DMA reset quirk, or zero to read it from hardware
*
* Resets ring internal state ((hw)occ, (hw)idx). Should be used for rings
* which are read by K3 UDMA, like TX or Free Host PD rings.
@@ -217,8 +215,8 @@ int k3_ringacc_ring_push(struct k3_ring *ring, void *elem);
* @ring: pointer on ring
* @elem: pointer on ring element buffer
*
- * Push one ring element from the ring head. Size of the ring element is
- * determined by ring configuration &struct k3_ring_cfg elm_size..
+ * Pop one ring element from the ring head. Size of the ring element is
+ * determined by ring configuration &struct k3_ring_cfg elm_size.
*
* Returns 0 on success, errno otherwise.
*/
@@ -242,7 +240,7 @@ int k3_ringacc_ring_push_head(struct k3_ring *ring, void *elem);
* @ring: pointer on ring
* @elem: pointer on ring element buffer
*
- * Push one ring element from the ring tail. Size of the ring element is
+ * Pop one ring element from the ring tail. Size of the ring element is
* determined by ring configuration &struct k3_ring_cfg elm_size.
*
* Returns 0 on success, errno otherwise.
@@ -256,7 +254,10 @@ u32 k3_ringacc_get_tisci_dev_id(struct k3_ring *ring);
struct ti_sci_handle;
/**
- * struct struct k3_ringacc_init_data - Initialization data for DMA rings
+ * struct k3_ringacc_init_data - Initialization data for DMA rings
+ * @tisci: TI SCI firmware handle
+ * @tisci_dev_id: TI SCI device ID of the DMA controller
+ * @num_rings: Total number of available rings
*/
struct k3_ringacc_init_data {
const struct ti_sci_handle *tisci;
diff --git a/include/linux/spi/spi.h b/include/linux/spi/spi.h
index 88d17fce02dc..fb4baa0d3398 100644
--- a/include/linux/spi/spi.h
+++ b/include/linux/spi/spi.h
@@ -169,6 +169,12 @@ extern void spi_transfer_cs_change_delay_exec(struct spi_message *msg,
* @cs_inactive: delay to be introduced by the controller after CS is
* deasserted. If @cs_change_delay is used from @spi_transfer, then the
* two delays will be added up.
+ * @rx_sample_delay_ns: Delay in nanoseconds by which the controller should
+ * postpone sampling the incoming data, relative to the sampling point it
+ * uses by default. Describes the board rather than the device, namely the
+ * flight time of the clock and data signals between controller and
+ * device, and comes from the "rx-sample-delay-ns" property. Zero when the
+ * property is absent.
* @chip_select: Array of physical chipselect, spi->chipselect[i] gives
* the corresponding physical CS for logical CS i.
* @num_chipselect: Number of physical chipselects used.
@@ -235,6 +241,9 @@ struct spi_device {
struct spi_delay cs_hold;
struct spi_delay cs_inactive;
+ /* Additional delay before the incoming data is sampled, in ns */
+ u32 rx_sample_delay_ns;
+
u8 chip_select[SPI_DEVICE_CS_CNT_MAX];
u8 num_chipselect;
diff --git a/include/linux/splice.h b/include/linux/splice.h
index 9dec4861d09f..0e6c955dc6ff 100644
--- a/include/linux/splice.h
+++ b/include/linux/splice.h
@@ -79,8 +79,8 @@ ssize_t add_to_pipe(struct pipe_inode_info *pipe, struct pipe_buffer *buf);
ssize_t vfs_splice_read(struct file *in, loff_t *ppos,
struct pipe_inode_info *pipe, size_t len,
unsigned int flags);
-ssize_t splice_direct_to_actor(struct file *file, struct splice_desc *sd,
- splice_direct_actor *actor);
+ssize_t vfs_splice_to_actor(struct file *file, loff_t pos, size_t count,
+ splice_direct_actor *actor, void *private);
ssize_t do_splice(struct file *in, loff_t *off_in, struct file *out,
loff_t *off_out, size_t len, unsigned int flags);
ssize_t do_splice_direct(struct file *in, loff_t *ppos, struct file *out,
diff --git a/include/linux/srcu.h b/include/linux/srcu.h
index 7d9bc06df98d..1a8a465a5650 100644
--- a/include/linux/srcu.h
+++ b/include/linux/srcu.h
@@ -36,6 +36,8 @@ static inline int __init_srcu_struct(struct srcu_struct *ssp, const char *name,
int __init_srcu_struct_fast(struct srcu_struct *ssp, const char *name, struct lock_class_key *key);
int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name,
struct lock_class_key *key);
+int __init_srcu_struct_atomic(struct srcu_struct *ssp, const char *name,
+ struct lock_class_key *key);
#endif // #ifndef CONFIG_TINY_SRCU
#define init_srcu_struct_fast(ssp) \
@@ -52,6 +54,13 @@ int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name,
__init_srcu_struct_fast_updown((ssp), #ssp, &__srcu_key); \
})
+#define init_srcu_struct_atomic(ssp) \
+({ \
+ static struct lock_class_key __srcu_key; \
+ \
+ __init_srcu_struct_atomic((ssp), #ssp, &__srcu_key); \
+})
+
#define __SRCU_DEP_MAP_INIT(srcu_name) .dep_map = { .name = #srcu_name },
#else /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */
@@ -64,6 +73,7 @@ static inline int __init_srcu_struct(struct srcu_struct *ssp, const char *name,
#ifndef CONFIG_TINY_SRCU
int init_srcu_struct_fast(struct srcu_struct *ssp);
int init_srcu_struct_fast_updown(struct srcu_struct *ssp);
+int init_srcu_struct_atomic(struct srcu_struct *ssp);
#endif // #ifndef CONFIG_TINY_SRCU
#define __SRCU_DEP_MAP_INIT(srcu_name)
@@ -77,14 +87,18 @@ int init_srcu_struct_fast_updown(struct srcu_struct *ssp);
})
/* Values for SRCU Tree srcu_data ->srcu_reader_flavor, but also used by rcutorture. */
-#define SRCU_READ_FLAVOR_NORMAL 0x1 // srcu_read_lock().
-#define SRCU_READ_FLAVOR_NMI 0x2 // srcu_read_lock_nmisafe().
-// 0x4 // SRCU-lite is no longer with us.
-#define SRCU_READ_FLAVOR_FAST 0x4 // srcu_read_lock_fast(), also NMI-safe.
-#define SRCU_READ_FLAVOR_FAST_UPDOWN 0x8 // srcu_read_lock_fast_updown().
+#define SRCU_READ_FLAVOR_NORMAL 0x01 // srcu_read_lock().
+#define SRCU_READ_FLAVOR_NMI 0x02 // srcu_read_lock_nmisafe().
+// 0x04 // SRCU-lite is no longer with us.
+#define SRCU_READ_FLAVOR_FAST 0x04 // srcu_read_lock_fast(), also NMI-safe.
+#define SRCU_READ_FLAVOR_FAST_UPDOWN 0x08 // srcu_read_lock_fast_updown().
+#define SRCU_READ_FLAVOR_ATOMIC 0x10 // srcu_read_lock_atomic().
#define SRCU_READ_FLAVOR_ALL (SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_NMI | \
- SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN)
+ SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN | \
+ SRCU_READ_FLAVOR_ATOMIC)
// All of the above.
+#define SRCU_READ_FLAVOR_PREDEF (SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_ATOMIC)
+ // Flavors special DEFINE_SRCU() flavors.
#define SRCU_READ_FLAVOR_SLOWGP (SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN)
// Flavors requiring synchronize_rcu()
// instead of smp_mb().
@@ -102,6 +116,7 @@ void call_srcu(struct srcu_struct *ssp, struct rcu_head *head,
void (*func)(struct rcu_head *head));
void cleanup_srcu_struct(struct srcu_struct *ssp);
void synchronize_srcu(struct srcu_struct *ssp);
+void synchronize_srcu_atomic(struct srcu_struct *ssp);
#define SRCU_GET_STATE_COMPLETED 0x1
@@ -307,6 +322,45 @@ static inline int srcu_read_lock(struct srcu_struct *ssp)
}
/**
+ * srcu_read_lock_atomic - register a new reader promising an atomic section
+ * @ssp: srcu_struct in which to register the new reader.
+ *
+ * As srcu_read_lock(), but the caller promises that the read-side
+ * critical section never sleeps and never blocks on anything which
+ * may itself depend on memory allocation to make progress. Preemption
+ * is disabled for the duration, which both enforces that promise (any
+ * sleepable call in the section will splat on every configuration)
+ * and bounds the section so that the update side may spin rather
+ * than sleep when waiting for readers: see synchronize_srcu_atomic().
+ *
+ * The lock and matching srcu_read_unlock_atomic() must be invoked on
+ * the same CPU, from the same context; passing the return value to
+ * another task is not permitted for this flavor.
+ */
+static inline int srcu_read_lock_atomic(struct srcu_struct *ssp)
+ __acquires_shared(ssp)
+{
+ int retval;
+
+ preempt_disable();
+ /*
+ * Arm might_sleep() to catch even a *potentially* sleeping call
+ * in the section, not just an actual schedule: the atomic-domain
+ * promise must hold on every path, contended or not. In hardirq,
+ * softirq, or NMI the annotation would land on the interrupted
+ * task, and can also result in data races against that task's
+ * own non_block_start()/non_block_end() invocations; it is also
+ * redundant there, so skip it.
+ */
+ if (in_task())
+ non_block_start();
+ srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC);
+ retval = __srcu_read_lock(ssp);
+ srcu_lock_acquire(&ssp->dep_map);
+ return retval;
+}
+
+/**
* srcu_read_lock_fast - register a new reader for an SRCU-protected structure.
* @ssp: srcu_struct in which to register the new reader.
*
@@ -499,6 +553,23 @@ static inline void srcu_read_unlock(struct srcu_struct *ssp, int idx)
}
/**
+ * srcu_read_unlock_atomic - unregister an atomic-section reader
+ * @ssp: srcu_struct from which to unregister the old reader.
+ * @idx: return value from corresponding srcu_read_lock_atomic().
+ */
+static inline void srcu_read_unlock_atomic(struct srcu_struct *ssp, int idx)
+ __releases_shared(ssp)
+{
+ WARN_ON_ONCE(idx & ~0x1);
+ srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC);
+ srcu_lock_release(&ssp->dep_map);
+ __srcu_read_unlock(ssp, idx);
+ if (in_task())
+ non_block_end();
+ preempt_enable();
+}
+
+/**
* srcu_read_unlock_fast - unregister a old reader from an SRCU-protected structure.
* @ssp: srcu_struct in which to unregister the old reader.
* @scp: return value from corresponding srcu_read_lock_fast().
diff --git a/include/linux/srcutiny.h b/include/linux/srcutiny.h
index fbcf13bc12d1..47a368f945e3 100644
--- a/include/linux/srcutiny.h
+++ b/include/linux/srcutiny.h
@@ -12,12 +12,14 @@
#define _LINUX_SRCU_TINY_H
#include <linux/irq_work_types.h>
+#include <linux/llist.h>
#include <linux/swait.h>
struct srcu_struct {
short srcu_lock_nesting[2]; /* srcu_read_lock() nesting depth. */
u8 srcu_gp_running; /* GP workqueue running? */
u8 srcu_gp_waiting; /* GP waiting for readers? */
+ u8 srcu_atomic_gp_flag; /* Serialize atomic GP work.*/
unsigned long srcu_idx; /* Current reader array element in bit 0x2. */
unsigned long srcu_idx_max; /* Furthest future srcu_idx request. */
struct swait_queue_head srcu_wq;
@@ -26,6 +28,8 @@ struct srcu_struct {
struct rcu_head **srcu_cb_tail; /* Pending callbacks: Tail. */
struct work_struct srcu_work; /* For driving grace periods. */
struct irq_work srcu_irq_work; /* Defer schedule_work() to irq work. */
+ struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */
+ struct irq_work defer_iw; /* Re-issues defer_cbs later. */
#ifdef CONFIG_DEBUG_LOCK_ALLOC
struct lockdep_map dep_map;
#endif /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */
@@ -33,6 +37,7 @@ struct srcu_struct {
void srcu_drive_gp(struct work_struct *wp);
void srcu_tiny_irq_work(struct irq_work *irq_work);
+void srcu_defer_drain(struct irq_work *irq_work);
#define __SRCU_STRUCT_INIT(name, __ignored, ___ignored, ____ignored) \
{ \
@@ -40,6 +45,9 @@ void srcu_tiny_irq_work(struct irq_work *irq_work);
.srcu_cb_tail = &name.srcu_cb_head, \
.srcu_work = __WORK_INITIALIZER(name.srcu_work, srcu_drive_gp), \
.srcu_irq_work = { .func = srcu_tiny_irq_work }, \
+ .defer_cbs = LLIST_HEAD_INIT(name.defer_cbs), \
+ .defer_iw = { .node = { .u_flags = IRQ_WORK_HARD_IRQ }, \
+ .func = srcu_defer_drain }, \
__SRCU_DEP_MAP_INIT(name) \
}
@@ -57,15 +65,20 @@ void srcu_tiny_irq_work(struct irq_work *irq_work);
#define DEFINE_SRCU_FAST_UPDOWN(name) DEFINE_SRCU(name)
#define DEFINE_STATIC_SRCU_FAST_UPDOWN(name) \
static struct srcu_struct name = __SRCU_STRUCT_INIT(name, name, name, name)
+#define DEFINE_SRCU_ATOMIC(name) DEFINE_SRCU(name)
+#define DEFINE_STATIC_SRCU_ATOMIC(name) \
+ static struct srcu_struct name = __SRCU_STRUCT_INIT(name, name, name, name)
// Dummy structure for srcu_notifier_head.
struct srcu_usage { };
#define __SRCU_USAGE_INIT(name) { }
#define __init_srcu_struct_fast __init_srcu_struct
#define __init_srcu_struct_fast_updown __init_srcu_struct
+#define __init_srcu_struct_atomic __init_srcu_struct
#ifndef CONFIG_DEBUG_LOCK_ALLOC
#define init_srcu_struct_fast init_srcu_struct
#define init_srcu_struct_fast_updown init_srcu_struct
+#define init_srcu_struct_atomic init_srcu_struct
#endif // #ifndef CONFIG_DEBUG_LOCK_ALLOC
void synchronize_srcu(struct srcu_struct *ssp);
@@ -131,10 +144,7 @@ static inline void synchronize_srcu_expedited(struct srcu_struct *ssp)
synchronize_srcu(ssp);
}
-static inline void srcu_barrier(struct srcu_struct *ssp)
-{
- synchronize_srcu(ssp);
-}
+void srcu_barrier(struct srcu_struct *ssp);
static inline void srcu_expedite_current(struct srcu_struct *ssp) { }
#define srcu_check_read_flavor(ssp, read_flavor) do { } while (0)
diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h
index 75e54e4f963f..93b76e543894 100644
--- a/include/linux/srcutree.h
+++ b/include/linux/srcutree.h
@@ -13,6 +13,8 @@
#include <linux/rcu_node_tree.h>
#include <linux/completion.h>
+#include <linux/irq_work_types.h>
+#include <linux/llist.h>
struct srcu_node;
struct srcu_struct;
@@ -41,6 +43,8 @@ struct srcu_data {
bool srcu_cblist_invoking; /* Invoking these CBs? */
struct timer_list delay_work; /* Delay for CB invoking */
struct work_struct work; /* Context for CB invoking. */
+ struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */
+ struct llist_node defer_link; /* Links onto the per-CPU deferral drain list */
struct rcu_head srcu_barrier_head; /* For srcu_barrier() use. */
struct rcu_head srcu_ec_head; /* For srcu_expedite_current() use. */
int srcu_ec_state; /* State for srcu_expedite_current(). */
@@ -76,6 +80,7 @@ struct srcu_usage {
struct mutex srcu_cb_mutex; /* Serialize CB preparation. */
raw_spinlock_t __private lock; /* Protect counters and size state. */
struct mutex srcu_gp_mutex; /* Serialize GP work. */
+ atomic_t srcu_atomic_gp_flag; /* Serialize atomic GP work. */
unsigned long srcu_gp_seq; /* Grace-period seq #. */
unsigned long srcu_gp_seq_needed; /* Latest gp_seq needed. */
unsigned long srcu_gp_seq_needed_exp; /* Furthest future exp GP. */
@@ -209,6 +214,12 @@ struct srcu_struct {
* instead of smp_mb(), and given that the first (for example)
* srcu_read_lock_fast() might race with the first synchronize_srcu(),
* this different must be specified at initialization time.
+ *
+ * If you use any of the DEFINE_SRCU() functions within a module, the
+ * module-entry code will invoke init_srcu_struct() and the module-exit
+ * code will invoke cleanup_srcu_struct(). This means that if your module
+ * passes the resulting srcu_struct structure to call_srcu(), you will
+ * need to also pass this structure to srcu_barrier() prior to module exit.
*/
#ifdef MODULE
# define __DEFINE_SRCU(name, fast, is_static) \
@@ -225,14 +236,16 @@ struct srcu_struct {
is_static struct srcu_struct name = \
__SRCU_STRUCT_INIT(name, name##_srcu_usage, name##_srcu_data, fast)
#endif
-#define DEFINE_SRCU(name) __DEFINE_SRCU(name, 0, /* not static */)
+#define DEFINE_SRCU(name) __DEFINE_SRCU(name, 0, /* !static */)
#define DEFINE_STATIC_SRCU(name) __DEFINE_SRCU(name, 0, static)
-#define DEFINE_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, /* not static */)
+#define DEFINE_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, /* !static */)
#define DEFINE_STATIC_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, static)
#define DEFINE_SRCU_FAST_UPDOWN(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST_UPDOWN, \
- /* not static */)
+ /* !static */)
#define DEFINE_STATIC_SRCU_FAST_UPDOWN(name) \
__DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST_UPDOWN, static)
+#define DEFINE_SRCU_ATOMIC(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_ATOMIC, /* !static */)
+#define DEFINE_STATIC_SRCU_ATOMIC(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_ATOMIC, static)
int __srcu_read_lock(struct srcu_struct *ssp) __acquires_shared(ssp);
void synchronize_srcu_expedited(struct srcu_struct *ssp);
diff --git a/include/linux/stmmac.h b/include/linux/stmmac.h
index 4430b967abde..00be2df63d22 100644
--- a/include/linux/stmmac.h
+++ b/include/linux/stmmac.h
@@ -242,6 +242,8 @@ struct plat_stmmacenet_data {
* that phylink uses.
*/
phy_interface_t phy_interface;
+ bool has_internal_tx_delay;
+ bool has_internal_rx_delay;
struct stmmac_mdio_bus_data *mdio_bus_data;
struct device_node *phy_node;
struct device_node *mdio_node;
diff --git a/include/linux/string.h b/include/linux/string.h
index 5702daca4326..6cb5cdd01158 100644
--- a/include/linux/string.h
+++ b/include/linux/string.h
@@ -278,6 +278,19 @@ static inline void memcpy_flushcache(void *dst, const void *src, size_t cnt)
}
#endif
+#ifndef memcpy_nontemporal
+/*
+ * memcpy_nontemporal() requests a non-temporal copy when the
+ * architecture has a suitable backend. Architectures without a
+ * specialized backend fall back to memcpy(). Keep this as a
+ * function-like macro so the compiler can still see the original
+ * memcpy() call site and preserve the usual FORTIFY coverage when
+ * object sizes remain visible there, while keeping the API void.
+ */
+#define memcpy_nontemporal(dst, src, len) \
+ ((void)memcpy(dst, src, len))
+#endif
+
void *memchr_inv(const void *s, int c, size_t n);
char *strreplace(char *str, char old, char new);
diff --git a/include/linux/string_choices.h b/include/linux/string_choices.h
index ee84087d4b26..a2eba454004f 100644
--- a/include/linux/string_choices.h
+++ b/include/linux/string_choices.h
@@ -65,6 +65,12 @@ static inline const char *str_read_write(bool v)
}
#define str_write_read(v) str_read_write(!(v))
+static inline const char *str_supported_unsupported(bool v)
+{
+ return v ? "supported" : "unsupported";
+}
+#define str_unsupported_supported(v) str_supported_unsupported(!(v))
+
static inline const char *str_true_false(bool v)
{
return v ? "true" : "false";
diff --git a/include/linux/sunrpc/svc_xprt.h b/include/linux/sunrpc/svc_xprt.h
index da2a2531e110..2af222f3ea2c 100644
--- a/include/linux/sunrpc/svc_xprt.h
+++ b/include/linux/sunrpc/svc_xprt.h
@@ -37,6 +37,9 @@ struct svc_xprt_class {
struct list_head xcl_list;
u32 xcl_max_payload;
int xcl_ident;
+ u32 xcl_flags;
+/* Set only on classes whose xpo_has_wspace() reads xpt_reserved */
+#define SVC_XPRT_FLAG_WSPACE_RESERVE BIT(0)
};
/*
@@ -59,7 +62,7 @@ struct svc_xprt {
unsigned long xpt_flags;
struct svc_serv *xpt_server; /* service for transport */
- atomic_t xpt_reserved; /* space on outq that is rsvd */
+ atomic_t xpt_reserved; /* outq space rsvd, UDP only */
atomic_t xpt_nr_rqsts; /* Number of requests */
struct mutex xpt_mutex; /* to serialize sending data */
spinlock_t xpt_lock; /* protects sk_deferred
diff --git a/include/linux/swap.h b/include/linux/swap.h
index 5658a1634b85..cb434cccd653 100644
--- a/include/linux/swap.h
+++ b/include/linux/swap.h
@@ -25,12 +25,6 @@
#define SWAP_FLAGS_VALID (SWAP_FLAG_PRIO_MASK | SWAP_FLAG_PREFER | \
SWAP_FLAG_DISCARD | SWAP_FLAG_DISCARD_ONCE | \
SWAP_FLAG_DISCARD_PAGES)
-
-static inline int current_is_kswapd(void)
-{
- return current->flags & PF_KSWAPD;
-}
-
/*
* MAX_SWAPFILES defines the maximum number of swaptypes: things which can
* be swapped to. The swap type and the offset into that swap type are
@@ -246,7 +240,7 @@ struct swap_info_struct {
struct plist_node list; /* entry in swap_active_head */
signed char type; /* strange name for an index */
unsigned int max; /* size of this swap device */
- struct swap_cluster_info *cluster_info; /* cluster info. Only for SSD */
+ struct swap_cluster_info *cluster_info; /* array, one entry per cluster */
struct list_head free_clusters; /* free clusters list */
struct list_head full_clusters; /* full clusters list */
struct list_head nonfull_clusters[SWAP_NR_ORDERS];
@@ -278,15 +272,39 @@ struct swap_info_struct {
const struct swap_ops *ops;
};
-static inline swp_entry_t page_swap_entry(struct page *page)
+/**
+ * folio_swap_entry - Return the swap entry at a page index within a folio.
+ * @folio: The folio.
+ * @idx: The index of the page within the folio.
+ *
+ * A folio in the swap cache occupies folio_nr_pages() contiguous swap
+ * entries starting at folio->swap. The caller must ensure the folio is
+ * in the swap cache and that @idx is within the folio.
+ */
+static inline
+swp_entry_t folio_swap_entry(const struct folio *folio, unsigned long idx)
{
- struct folio *folio = page_folio(page);
swp_entry_t entry = folio->swap;
- entry.val += folio_page_idx(folio, page);
+ VM_WARN_ON_ONCE_FOLIO(idx >= folio_nr_pages(folio), folio);
+ entry.val += idx;
return entry;
}
+/**
+ * folio_page_swap_entry - Return the swap entry of a page in a folio.
+ * @folio: The folio containing @page.
+ * @page: A page within @folio.
+ *
+ * The caller must ensure the folio is in the swap cache and that @page
+ * is part of @folio.
+ */
+static inline swp_entry_t folio_page_swap_entry(const struct folio *folio,
+ const struct page *page)
+{
+ return folio_swap_entry(folio, folio_page_idx(folio, page));
+}
+
/* linux/mm/page_alloc.c */
extern unsigned long totalreserve_pages;
@@ -318,6 +336,9 @@ static inline bool lru_cache_disabled(void)
extern unsigned long shrink_all_memory(unsigned long nr_pages);
long remove_mapping(struct address_space *mapping, struct folio *folio);
+long remove_mapping_set_shadow(struct address_space *mapping,
+ struct folio *folio,
+ struct mem_cgroup *target_memcg);
#if defined(CONFIG_SYSFS) && defined(CONFIG_NUMA)
extern int reclaim_register_node(struct node *node);
@@ -339,6 +360,7 @@ void check_move_unevictable_folios(struct folio_batch *fbatch);
extern void __meminit kswapd_run(int nid);
extern void __meminit kswapd_stop(int nid);
+bool current_is_kswapd(void);
#ifdef CONFIG_SWAP
int add_swap_extent(struct swap_info_struct *sis, unsigned long start_page,
@@ -402,6 +424,8 @@ void swap_put_entries_direct(swp_entry_t entry, int nr);
*/
bool folio_free_swap(struct folio *folio);
+void swap_writeback_dropbehind_folio(struct folio *folio);
+
/* Allocate / free (hibernation) exclusive entries */
swp_entry_t swap_alloc_hibernation_slot(int type);
void swap_free_hibernation_slot(swp_entry_t entry);
@@ -412,6 +436,7 @@ static inline void put_swap_device(struct swap_info_struct *si)
}
#else /* CONFIG_SWAP */
+static inline void swap_writeback_dropbehind_folio(struct folio *folio) {}
static inline struct swap_info_struct *get_swap_device(swp_entry_t entry)
{
return NULL;
@@ -508,8 +533,9 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p
__mem_cgroup_uncharge_swap(id, nr_pages);
}
-extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg);
-extern bool mem_cgroup_swap_full(struct folio *folio);
+long mem_cgroup_get_folio_swap_margin(const struct folio *folio);
+long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg);
+bool mem_cgroup_swap_full(const struct folio *folio);
#else
static inline int mem_cgroup_try_charge_swap(struct folio *folio)
{
@@ -521,12 +547,17 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id,
{
}
-static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg)
+static inline long mem_cgroup_get_folio_swap_margin(const struct folio *folio)
+{
+ return PAGE_COUNTER_MAX;
+}
+
+static inline long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg)
{
return get_nr_swap_pages();
}
-static inline bool mem_cgroup_swap_full(struct folio *folio)
+static inline bool mem_cgroup_swap_full(const struct folio *folio)
{
return vm_swap_full();
}
diff --git a/include/linux/swapops.h b/include/linux/swapops.h
index e7d0d529f3e0..603f9e3f909c 100644
--- a/include/linux/swapops.h
+++ b/include/linux/swapops.h
@@ -261,11 +261,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page)
return swp_entry(SWP_HWPOISON, page_to_pfn(page));
}
-static inline int is_hwpoison_entry(swp_entry_t entry)
-{
- return swp_type(entry) == SWP_HWPOISON;
-}
-
#else
static inline swp_entry_t make_hwpoison_entry(struct page *page)
@@ -273,10 +268,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page)
return swp_entry(0, 0);
}
-static inline int is_hwpoison_entry(swp_entry_t swp)
-{
- return 0;
-}
#endif
typedef unsigned long pte_marker;
diff --git a/include/linux/thermal.h b/include/linux/thermal.h
index 083b4f533933..306ad17aed89 100644
--- a/include/linux/thermal.h
+++ b/include/linux/thermal.h
@@ -293,8 +293,9 @@ struct device *thermal_zone_device(struct thermal_zone_device *tzd);
void thermal_zone_device_update(struct thermal_zone_device *,
enum thermal_notify_event);
-struct thermal_cooling_device *thermal_cooling_device_register(const char *,
- void *, const struct thermal_cooling_device_ops *);
+struct thermal_cooling_device *thermal_cooling_device_create(
+ struct device *parent, const char *type, void *devdata,
+ const struct thermal_cooling_device_ops *ops);
struct thermal_cooling_device *
devm_thermal_cooling_device_register(struct device *dev, const char *type, void *devdata,
@@ -340,9 +341,9 @@ static inline void thermal_zone_device_update(struct thermal_zone_device *tz,
enum thermal_notify_event event)
{ }
-static inline struct thermal_cooling_device *
-thermal_cooling_device_register(const char *type, void *devdata,
- const struct thermal_cooling_device_ops *ops)
+static inline struct thermal_cooling_device *thermal_cooling_device_create(
+ struct device *parent, const char *type, void *devdata,
+ const struct thermal_cooling_device_ops *ops)
{ return ERR_PTR(-ENODEV); }
static inline struct thermal_cooling_device *
@@ -391,4 +392,12 @@ static inline void thermal_pm_prepare(void) {}
static inline void thermal_pm_complete(void) {}
#endif /* CONFIG_THERMAL */
+static inline struct thermal_cooling_device *thermal_cooling_device_register(
+ const char *type, void *devdata,
+ const struct thermal_cooling_device_ops *ops)
+{
+ return thermal_cooling_device_create(NULL, type, devdata, ops);
+}
+
+
#endif /* __THERMAL_H__ */
diff --git a/include/linux/thunderbolt.h b/include/linux/thunderbolt.h
index d48623fda79b..b62dfa52b149 100644
--- a/include/linux/thunderbolt.h
+++ b/include/linux/thunderbolt.h
@@ -22,6 +22,7 @@ struct device;
#include <linux/device.h>
#include <linux/idr.h>
#include <linux/list.h>
+#include <linux/lockdep.h>
#include <linux/mutex.h>
#include <linux/device-id/tb.h>
#include <linux/pci.h>
@@ -565,6 +566,7 @@ struct tb_nhi {
* @interval_nsec: Interval counter if interrupt throttling is to be
* used with this ring (in ns)
* @wait: Used to signal that the ring may be empty now
+ * @lock_key: Lock validator class key per-ring
*/
struct tb_ring {
spinlock_t lock;
@@ -590,6 +592,7 @@ struct tb_ring {
void *poll_data;
unsigned int interval_nsec;
wait_queue_head_t wait;
+ struct lock_class_key lock_key;
};
/* Leave ring interrupt enabled on suspend */
diff --git a/include/linux/torture.h b/include/linux/torture.h
index c9b47d138302..1b6d5be641f8 100644
--- a/include/linux/torture.h
+++ b/include/linux/torture.h
@@ -98,6 +98,7 @@ void torture_shutdown_absorb(const char *title);
int torture_shutdown_init(int ssecs, void (*cleanup)(void));
/* Task stuttering, which forces load/no-load transitions. */
+bool stutter_will_wait(void);
bool stutter_wait(const char *title);
int torture_stutter_init(int s, int sgap);
@@ -131,8 +132,11 @@ void _torture_stop_kthread(char *m, struct task_struct **tp);
#endif
void torture_sched_set_normal(struct task_struct *t, int nice);
-#if IS_ENABLED(CONFIG_RCU_TORTURE_TEST) || IS_MODULE(CONFIG_RCU_TORTURE_TEST) || IS_ENABLED(CONFIG_LOCK_TORTURE_TEST) || IS_MODULE(CONFIG_LOCK_TORTURE_TEST)
+#if IS_ENABLED(CONFIG_RCU_TORTURE_TEST) || IS_ENABLED(CONFIG_LOCK_TORTURE_TEST) || IS_ENABLED(CONFIG_HAZPTR_TORTURE_TEST)
long torture_sched_setaffinity(pid_t pid, const struct cpumask *in_mask, bool dowarn);
#endif
+/* Atomic per-CPU counters. */
+s64 torture_sum_pcpu_atomic_long(atomic_long_t __percpu *pcp);
+
#endif /* __LINUX_TORTURE_H */
diff --git a/include/linux/trace_remote_event.h b/include/linux/trace_remote_event.h
index c8ae1e1f5e72..e4cc2d4497bc 100644
--- a/include/linux/trace_remote_event.h
+++ b/include/linux/trace_remote_event.h
@@ -3,6 +3,8 @@
#ifndef _LINUX_TRACE_REMOTE_EVENTS_H
#define _LINUX_TRACE_REMOTE_EVENTS_H
+#include <linux/types.h>
+
struct trace_remote;
struct trace_event_fields;
struct trace_seq;
diff --git a/include/linux/tty.h b/include/linux/tty.h
index 0a46e4054dec..833bb7b97ceb 100644
--- a/include/linux/tty.h
+++ b/include/linux/tty.h
@@ -168,6 +168,7 @@ struct tty_operations;
* @write_wait: concurrent writers are waiting in this queue until they are
* allowed to write
* @read_wait: readers wait for data in this queue
+ * @break_wait: wait queue for timed breaks
* @hangup_work: normally a work to perform a hangup (do_tty_hangup()); while
* freeing the tty, (re)used to release_one_tty()
* @disc_data: pointer to @ldisc's private data (e.g. to &struct n_tty_data)
@@ -230,6 +231,7 @@ struct tty_struct {
struct fasync_struct *fasync;
wait_queue_head_t write_wait;
wait_queue_head_t read_wait;
+ wait_queue_head_t break_wait;
struct work_struct hangup_work;
void *disc_data;
void *driver_data;
diff --git a/include/linux/tty_driver.h b/include/linux/tty_driver.h
index 1f2896e56e77..62e459c5d3ae 100644
--- a/include/linux/tty_driver.h
+++ b/include/linux/tty_driver.h
@@ -73,6 +73,11 @@ struct serial_struct;
* @TTY_DRIVER_NO_WORKQUEUE:
* Do not create workqueue when tty_register_driver(). Whenever set, flip
* buffer workqueue can be set by tty_port_link_wq() for every port.
+ *
+ * @TTY_DRIVER_RESET_SAVED_TERMIOS
+ * Reset any saved termios settings on device registration when reusing a
+ * minor number. Must only be set by drivers that guarantee that the minor
+ * number is no longer in use.
*/
enum tty_driver_flag {
TTY_DRIVER_INSTALLED = BIT(0),
@@ -84,6 +89,7 @@ enum tty_driver_flag {
TTY_DRIVER_DYNAMIC_ALLOC = BIT(6),
TTY_DRIVER_UNNUMBERED_NODE = BIT(7),
TTY_DRIVER_NO_WORKQUEUE = BIT(8),
+ TTY_DRIVER_RESET_SAVED_TERMIOS = BIT(9),
};
enum tty_driver_type {
diff --git a/include/linux/tty_port.h b/include/linux/tty_port.h
index 23cad403bb8f..8d22c59c6f15 100644
--- a/include/linux/tty_port.h
+++ b/include/linux/tty_port.h
@@ -245,7 +245,8 @@ bool tty_port_carrier_raised(struct tty_port *port);
void tty_port_raise_dtr_rts(struct tty_port *port);
void tty_port_lower_dtr_rts(struct tty_port *port);
void tty_port_hangup(struct tty_port *port);
-void __tty_port_tty_hangup(struct tty_port *port, bool check_clocal, bool async);
+void tty_port_tty_hangup(struct tty_port *port, bool check_clocal);
+void tty_port_tty_vhangup(struct tty_port *port);
void tty_port_tty_wakeup(struct tty_port *port);
int tty_port_block_til_ready(struct tty_port *port, struct tty_struct *tty,
struct file *filp);
@@ -264,25 +265,6 @@ static inline int tty_port_users(struct tty_port *port)
return port->count + port->blocked_open;
}
-/**
- * tty_port_tty_hangup - helper to hang up a tty asynchronously
- * @port: tty port
- * @check_clocal: hang only ttys with %CLOCAL unset?
- */
-static inline void tty_port_tty_hangup(struct tty_port *port, bool check_clocal)
-{
- __tty_port_tty_hangup(port, check_clocal, true);
-}
-
-/**
- * tty_port_tty_vhangup - helper to hang up a tty synchronously
- * @port: tty port
- */
-static inline void tty_port_tty_vhangup(struct tty_port *port)
-{
- __tty_port_tty_hangup(port, false, false);
-}
-
#ifdef CONFIG_TTY
void tty_kref_put(struct tty_struct *tty);
__DEFINE_CLASS_IS_CONDITIONAL(tty_port_tty, true);
diff --git a/include/linux/uidgid.h b/include/linux/uidgid.h
index 2dc767e08f54..02403629b49f 100644
--- a/include/linux/uidgid.h
+++ b/include/linux/uidgid.h
@@ -130,9 +130,9 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid)
return from_kgid(ns, gid) != (gid_t) -1;
}
-u32 map_id_down(struct uid_gid_map *map, u32 id);
-u32 map_id_up(struct uid_gid_map *map, u32 id);
-u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count);
+u32 map_id_down(const struct uid_gid_map *map, u32 id);
+u32 map_id_up(const struct uid_gid_map *map, u32 id);
+u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count);
#else
@@ -182,17 +182,17 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid)
return gid_valid(gid);
}
-static inline u32 map_id_down(struct uid_gid_map *map, u32 id)
+static inline u32 map_id_down(const struct uid_gid_map *map, u32 id)
{
return id;
}
-static inline u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count)
+static inline u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count)
{
return id;
}
-static inline u32 map_id_up(struct uid_gid_map *map, u32 id)
+static inline u32 map_id_up(const struct uid_gid_map *map, u32 id)
{
return id;
}
diff --git a/include/linux/usb/serial.h b/include/linux/usb/serial.h
index 534e6650e2aa..ab0a32a90629 100644
--- a/include/linux/usb/serial.h
+++ b/include/linux/usb/serial.h
@@ -381,6 +381,7 @@ void usb_serial_handle_dcd_change(struct usb_serial_port *usb_port,
int usb_serial_bus_register(struct usb_serial_driver *device);
void usb_serial_bus_deregister(struct usb_serial_driver *device);
+void usb_serial_bus_remove_new_id(struct usb_serial_driver *driver);
extern const struct bus_type usb_serial_bus_type;
extern struct tty_driver *usb_serial_tty_driver;
diff --git a/include/linux/user_namespace.h b/include/linux/user_namespace.h
index e38d9e60569f..91232053775d 100644
--- a/include/linux/user_namespace.h
+++ b/include/linux/user_namespace.h
@@ -29,8 +29,8 @@ struct uid_gid_map { /* 64 bytes -- 1 cache line */
u32 nr_extents;
};
struct {
- struct uid_gid_extent *forward;
- struct uid_gid_extent *reverse;
+ struct uid_gid_extent *forward __counted_by_ptr(nr_extents);
+ struct uid_gid_extent *reverse __counted_by_ptr(nr_extents);
};
};
};
@@ -207,6 +207,13 @@ extern bool in_userns(const struct user_namespace *ancestor,
const struct user_namespace *child);
extern bool current_in_userns(const struct user_namespace *target_ns);
struct ns_common *ns_get_owner(struct ns_common *ns);
+
+#if IS_ENABLED(CONFIG_KUNIT)
+extern int uid_gid_map_insert_extent(struct uid_gid_map *map,
+ struct uid_gid_extent *extent);
+extern int uid_gid_map_sort(struct uid_gid_map *map);
+#endif /* CONFIG_KUNIT */
+
#else
static inline struct user_namespace *get_user_ns(struct user_namespace *ns)
diff --git a/include/linux/userfaultfd_k.h b/include/linux/userfaultfd_k.h
index a4351cffc60c..a14b8a9ffb7b 100644
--- a/include/linux/userfaultfd_k.h
+++ b/include/linux/userfaultfd_k.h
@@ -18,7 +18,6 @@
#include <linux/swap.h>
#include <linux/leafops.h>
#include <asm-generic/pgtable_uffd.h>
-#include <linux/hugetlb_inline.h>
/* The set of all possible UFFD-related VM flags. */
#define __VM_UFFD_FLAGS (VM_UFFD_MISSING | VM_UFFD_MINOR | \
diff --git a/include/linux/vga_switcheroo.h b/include/linux/vga_switcheroo.h
index 7e6ac0114d55..d7eee65664ab 100644
--- a/include/linux/vga_switcheroo.h
+++ b/include/linux/vga_switcheroo.h
@@ -31,8 +31,12 @@
#ifndef _LINUX_VGA_SWITCHEROO_H_
#define _LINUX_VGA_SWITCHEROO_H_
-#include <linux/fb.h>
+#include <linux/errno.h>
+#include <linux/types.h>
+struct device;
+struct dev_pm_domain;
+struct fb_info;
struct pci_dev;
/**
@@ -127,23 +131,28 @@ struct vga_switcheroo_handler {
* @set_gpu_state: do the equivalent of suspend/resume for the card.
* Mandatory. This should not cut power to the discrete GPU,
* which is the job of the handler
- * @reprobe: poll outputs.
- * Optional. This gets called after waking the GPU and switching
- * the outputs to it
* @can_switch: check if the device is in a position to switch now.
* Mandatory. The client should return false if a user space process
* has one of its device files open
+ * @pre_switch: prepare switch
+ * Optional. This gets called before switching the outputs to the
+ * GPU. Allows drivers to prepare for the switch.
+ * @post_switch: completes switch
+ * Optional. This gets called after waking the GPU and switching
+ * the outputs to it. Allows drivers to poll the switched outputs.
* @gpu_bound: notify the client id to audio client when the GPU is bound.
*
* Client callbacks. A client can be either a GPU or an audio device on a GPU.
- * The @set_gpu_state and @can_switch methods are mandatory, @reprobe may be
- * set to NULL. For audio clients, the @reprobe member is bogus.
- * OTOH, @gpu_bound is only for audio clients, and not used for GPU clients.
+ * The @set_gpu_state and @can_switch methods are mandatory, @pre_switch and
+ * @post_switch may be set to NULL. For audio clients, the @pre_switch and
+ * @post_switch members are bogus. OTOH, @gpu_bound is only for audio clients,
+ * and not used for GPU clients.
*/
struct vga_switcheroo_client_ops {
void (*set_gpu_state)(struct pci_dev *dev, enum vga_switcheroo_state);
- void (*reprobe)(struct pci_dev *dev);
bool (*can_switch)(struct pci_dev *dev);
+ void (*pre_switch)(struct pci_dev *dev);
+ void (*post_switch)(struct pci_dev *dev);
void (*gpu_bound)(struct pci_dev *dev, enum vga_switcheroo_client_id);
};
@@ -156,9 +165,6 @@ int vga_switcheroo_register_audio_client(struct pci_dev *pdev,
const struct vga_switcheroo_client_ops *ops,
struct pci_dev *vga_dev);
-void vga_switcheroo_client_fb_set(struct pci_dev *dev,
- struct fb_info *info);
-
int vga_switcheroo_register_handler(const struct vga_switcheroo_handler *handler,
enum vga_switcheroo_handler_flags_t handler_flags);
void vga_switcheroo_unregister_handler(void);
@@ -178,7 +184,6 @@ void vga_switcheroo_fini_domain_pm_ops(struct device *dev);
static inline void vga_switcheroo_unregister_client(struct pci_dev *dev) {}
static inline int vga_switcheroo_register_client(struct pci_dev *dev,
const struct vga_switcheroo_client_ops *ops, bool driver_power_control) { return 0; }
-static inline void vga_switcheroo_client_fb_set(struct pci_dev *dev, struct fb_info *info) {}
static inline int vga_switcheroo_register_handler(const struct vga_switcheroo_handler *handler,
enum vga_switcheroo_handler_flags_t handler_flags) { return 0; }
static inline int vga_switcheroo_register_audio_client(struct pci_dev *pdev,
diff --git a/include/linux/vgaarb.h b/include/linux/vgaarb.h
index 97129a1bbb7d..71a364669eaf 100644
--- a/include/linux/vgaarb.h
+++ b/include/linux/vgaarb.h
@@ -33,7 +33,8 @@ struct pci_dev *vga_default_device(void);
void vga_set_default_device(struct pci_dev *pdev);
int vga_remove_vgacon(struct pci_dev *pdev);
int vga_client_register(struct pci_dev *pdev,
- unsigned int (*set_decode)(struct pci_dev *pdev, bool state));
+ unsigned int (*set_decode)(void *data, bool state),
+ void *data);
#else /* CONFIG_VGA_ARB */
static inline void vga_set_legacy_decoding(struct pci_dev *pdev,
unsigned int decodes)
@@ -59,7 +60,8 @@ static inline int vga_remove_vgacon(struct pci_dev *pdev)
return 0;
}
static inline int vga_client_register(struct pci_dev *pdev,
- unsigned int (*set_decode)(struct pci_dev *pdev, bool state))
+ unsigned int (*set_decode)(void *data, bool state),
+ void *data)
{
return 0;
}
@@ -97,7 +99,7 @@ static inline int vga_get_uninterruptible(struct pci_dev *pdev,
static inline void vga_client_unregister(struct pci_dev *pdev)
{
- vga_client_register(pdev, NULL);
+ vga_client_register(pdev, NULL, NULL);
}
#endif /* LINUX_VGA_H */
diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h
index aed121d729b0..034a693777ca 100644
--- a/include/linux/vmalloc.h
+++ b/include/linux/vmalloc.h
@@ -3,6 +3,7 @@
#define _LINUX_VMALLOC_H
#include <linux/alloc_tag.h>
+#include <linux/cleanup.h>
#include <linux/sched.h>
#include <linux/spinlock.h>
#include <linux/init.h>
@@ -38,6 +39,7 @@ struct iov_iter; /* in uio.h */
#define VM_DEFER_KMEMLEAK 0
#endif
#define VM_SPARSE 0x00001000 /* sparse vm_area. not all pages are present. */
+#define VM_REQUIRE_HUGE_VMAP 0x00002000 /* huge page mapping or nothing */
/* bits [20..32] reserved for arch specific ioremap internals */
@@ -214,6 +216,8 @@ void *__must_check vrealloc_node_align_noprof(const void *p, size_t size,
extern void vfree(const void *addr);
extern void vfree_atomic(const void *addr);
+DEFINE_FREE(vfree, void *, if (!IS_ERR_OR_NULL(_T)) vfree(_T))
+
extern void *vmap(struct page **pages, unsigned int count,
unsigned long flags, pgprot_t prot);
void *vmap_pfn(unsigned long *pfns, unsigned int count, pgprot_t prot);
diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h
new file mode 100644
index 000000000000..fa9e9abd6656
--- /dev/null
+++ b/include/linux/vmemmap-optimization.h
@@ -0,0 +1,115 @@
+/* SPDX-License-Identifier: GPL-2.0-or-later */
+/*
+ * vmemmap-optimization.h
+ *
+ * Generic vmemmap optimization declarations.
+ *
+ * Author: Muchun Song <songmuchun@bytedance.com>
+ */
+#ifndef _LINUX_VMEMMAP_OPTIMIZATION_H
+#define _LINUX_VMEMMAP_OPTIMIZATION_H
+
+#include <linux/align.h>
+#include <linux/log2.h>
+#include <linux/mmdebug.h>
+#include <linux/mmzone.h>
+
+/*
+ * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to
+ * be naturally aligned with regard to the folio size.
+ *
+ * HVO which is only active if the size of struct page is a power of 2.
+ */
+#define MAX_FOLIO_VMEMMAP_ALIGN \
+ (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \
+ is_power_of_2(sizeof(struct page)) ? \
+ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0)
+
+/* The number of retained vmemmap pages with HVO enabled. */
+#define VMEMMAP_OPTIMIZATION_PAGES 1
+#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \
+ (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page))
+#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1)
+
+#ifdef CONFIG_VMEMMAP_OPTIMIZATION
+static inline unsigned int section_compound_order(const struct mem_section *section)
+{
+ return section->compound_page_order;
+}
+
+static inline void section_set_compound_order(struct mem_section *section,
+ unsigned int order)
+{
+ VM_WARN_ON(section_compound_order(section) && order &&
+ section_compound_order(section) != order);
+ section->compound_page_order = order;
+}
+
+static inline void section_set_compound_order_range(unsigned long pfn,
+ unsigned long nr_pages, unsigned int order)
+{
+ unsigned long section_nr = pfn_to_section_nr(pfn);
+
+ if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION))
+ return;
+
+ for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++)
+ section_set_compound_order(__nr_to_section(section_nr + i), order);
+}
+
+static inline unsigned int pfn_to_section_compound_order(unsigned long pfn)
+{
+ return section_compound_order(__pfn_to_section(pfn));
+}
+
+struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone);
+#else
+static inline unsigned int section_compound_order(const struct mem_section *section)
+{
+ return 0;
+}
+
+static inline void section_set_compound_order(struct mem_section *section,
+ unsigned int order)
+{
+}
+
+static inline void section_set_compound_order_range(unsigned long pfn,
+ unsigned long nr_pages, unsigned int order)
+{
+}
+
+static inline unsigned int pfn_to_section_compound_order(unsigned long pfn)
+{
+ return 0;
+}
+
+static inline struct page *vmemmap_shared_tail_page(unsigned int order,
+ struct zone *zone)
+{
+ return NULL;
+}
+#endif /* CONFIG_VMEMMAP_OPTIMIZATION */
+
+static inline bool vmemmap_optimizable_pfn(unsigned long pfn)
+{
+ const unsigned int order = pfn_to_section_compound_order(pfn);
+ const unsigned long nr_pages = 1UL << order;
+
+ if (!is_power_of_2(sizeof(struct page)))
+ return false;
+
+ return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES;
+}
+
+static inline bool vmemmap_optimizable_order(unsigned int order)
+{
+ if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION))
+ return false;
+
+ if (!is_power_of_2(sizeof(struct page)))
+ return false;
+
+ return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER;
+}
+#endif /* _LINUX_VMEMMAP_OPTIMIZATION_H */
diff --git a/include/linux/wait.h b/include/linux/wait.h
index 7e215330199c..c2af98b0074d 100644
--- a/include/linux/wait.h
+++ b/include/linux/wait.h
@@ -556,7 +556,7 @@ do { \
} \
\
__ret = ___wait_event(wq_head, condition, state, 0, 0, \
- if (!__t.task) { \
+ if (!hrtimer_sleeper_task_get(&__t)) { \
__ret = -ETIME; \
break; \
} \
diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h
index 553d7b23e3ad..af077ed4caf6 100644
--- a/include/linux/wait_bit.h
+++ b/include/linux/wait_bit.h
@@ -433,6 +433,32 @@ do { \
})
/**
+ * wait_var_event_state - wait for a variable to be updated and notified
+ * @var: the address of variable being waited on
+ * @condition: the condition to wait for
+ * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc.
+ *
+ * Wait for a @condition to be true, only re-checking when a wake up is
+ * received for the given @var (an arbitrary kernel address which need
+ * not be directly related to the given condition, but usually is).
+ *
+ * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal
+ * arrived which @state allows to interrupt.
+ *
+ * The condition should normally use smp_load_acquire() or a similarly
+ * ordered access to ensure that any changes to memory made before the
+ * condition became true will be visible after the wait completes.
+ */
+#define wait_var_event_state(var, condition, state) \
+({ \
+ int __ret = 0; \
+ might_sleep(); \
+ if (!(condition)) \
+ __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \
+ __ret; \
+})
+
+/**
* wait_var_event_any_lock - wait for a variable to be updated under a lock
* @var: the address of the variable being waited on
* @condition: condition to wait for
diff --git a/include/linux/xattr.h b/include/linux/xattr.h
index 54ac3cbc133f..4cc4257de084 100644
--- a/include/linux/xattr.h
+++ b/include/linux/xattr.h
@@ -47,7 +47,7 @@ struct xattr_handler {
struct inode *inode, const char *name, void *buffer,
size_t size);
int (*set)(const struct xattr_handler *,
- struct mnt_idmap *idmap, struct dentry *dentry,
+ const struct mnt_idmap *idmap, struct dentry *dentry,
struct inode *inode, const char *name, const void *buffer,
size_t size, int flags);
};
@@ -77,25 +77,25 @@ struct xattr {
};
ssize_t __vfs_getxattr(struct dentry *, struct inode *, const char *, void *, size_t);
-ssize_t vfs_getxattr(struct mnt_idmap *, struct dentry *, const char *,
+ssize_t vfs_getxattr(const struct mnt_idmap *, struct dentry *, const char *,
void *, size_t);
ssize_t vfs_listxattr(struct dentry *d, char *list, size_t size);
-int __vfs_setxattr(struct mnt_idmap *, struct dentry *, struct inode *,
+int __vfs_setxattr(const struct mnt_idmap *, struct dentry *, struct inode *,
const char *, const void *, size_t, int);
-int __vfs_setxattr_noperm(struct mnt_idmap *, struct dentry *,
+int __vfs_setxattr_noperm(const struct mnt_idmap *, struct dentry *,
const char *, const void *, size_t, int);
-int __vfs_setxattr_locked(struct mnt_idmap *, struct dentry *,
+int __vfs_setxattr_locked(const struct mnt_idmap *, struct dentry *,
const char *, const void *, size_t, int,
struct delegated_inode *);
-int vfs_setxattr(struct mnt_idmap *, struct dentry *, const char *,
+int vfs_setxattr(const struct mnt_idmap *, struct dentry *, const char *,
const void *, size_t, int);
-int __vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *);
-int __vfs_removexattr_locked(struct mnt_idmap *, struct dentry *,
+int __vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *);
+int __vfs_removexattr_locked(const struct mnt_idmap *, struct dentry *,
const char *, struct delegated_inode *);
-int vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *);
+int vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *);
ssize_t generic_listxattr(struct dentry *dentry, char *buffer, size_t buffer_size);
-int vfs_getxattr_alloc(struct mnt_idmap *idmap,
+int vfs_getxattr_alloc(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name,
char **xattr_value, size_t size, gfp_t flags);
diff --git a/include/linux/zsmalloc.h b/include/linux/zsmalloc.h
index 478410c880b1..5b7298a5026b 100644
--- a/include/linux/zsmalloc.h
+++ b/include/linux/zsmalloc.h
@@ -40,10 +40,6 @@ unsigned int zs_lookup_class_index(struct zs_pool *pool, unsigned int size);
void zs_pool_stats(struct zs_pool *pool, struct zs_pool_stats *stats);
-void *zs_obj_read_begin(struct zs_pool *pool, unsigned long handle,
- size_t mem_len, void *local_copy);
-void zs_obj_read_end(struct zs_pool *pool, unsigned long handle,
- size_t mem_len, void *handle_mem);
void zs_obj_read_sg_begin(struct zs_pool *pool, unsigned long handle,
struct scatterlist *sg, size_t mem_len);
void zs_obj_read_sg_end(struct zs_pool *pool, unsigned long handle);
diff --git a/include/linux/zswap.h b/include/linux/zswap.h
index 30c193a1207e..df6cafbe95dc 100644
--- a/include/linux/zswap.h
+++ b/include/linux/zswap.h
@@ -27,7 +27,7 @@ struct zswap_lruvec_state {
unsigned long zswap_total_pages(void);
bool zswap_store(struct folio *folio);
int zswap_load(struct folio *folio);
-void zswap_invalidate(swp_entry_t swp);
+void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries);
int zswap_swapon(int type, unsigned long nr_pages);
void zswap_swapoff(int type);
void zswap_memcg_offline_cleanup(struct mem_cgroup *memcg);
@@ -49,7 +49,11 @@ static inline int zswap_load(struct folio *folio)
return -ENOENT;
}
-static inline void zswap_invalidate(swp_entry_t swp) {}
+static inline void zswap_invalidate(int type, pgoff_t offset,
+ unsigned long nr_entries)
+{
+}
+
static inline int zswap_swapon(int type, unsigned long nr_pages)
{
return 0;