diff options
Diffstat (limited to 'include/linux')
284 files changed, 6831 insertions, 2655 deletions
diff --git a/include/linux/acpi.h b/include/linux/acpi.h index ddacac812094..aef0a01f4cfa 100644 --- a/include/linux/acpi.h +++ b/include/linux/acpi.h @@ -25,17 +25,19 @@ struct irq_domain_ops; #define _LINUX #endif #include <acpi/acpi.h> +#include <acpi/acpi_bus.h> #include <acpi/acpi_numa.h> #ifdef CONFIG_ACPI +DEFINE_FREE(acpi_object_free, union acpi_object *, if (_T) ACPI_FREE(_T)); + #include <linux/list.h> #include <linux/dynamic_debug.h> #include <linux/module.h> #include <linux/mutex.h> #include <linux/fw_table.h> -#include <acpi/acpi_bus.h> #include <acpi/acpi_drivers.h> #include <acpi/acpi_io.h> #include <asm/acpi.h> @@ -434,7 +436,7 @@ extern acpi_handle ec_get_handle(void); extern bool acpi_is_pnp_device(struct acpi_device *); -#if defined(CONFIG_ACPI_WMI) || defined(CONFIG_ACPI_WMI_MODULE) +#if IS_ENABLED(CONFIG_ACPI_WMI) typedef void (*wmi_notify_handler) (union acpi_object *data, void *context); @@ -1305,34 +1307,34 @@ void __acpi_handle_debug(struct _ddebug *descriptor, acpi_handle handle, const c #endif /* - * acpi_handle_<level>: Print message with ACPI prefix and object path + * acpi_handle_<level> - Print a message with ACPI prefix and object path * - * These interfaces acquire the global namespace mutex to obtain an object - * path. In interrupt context, it shows the object path as <n/a>. + * In thread context, the global namespace mutex is acquired to obtain the + * object path. In interrupt context, the object path is shown as <n/a>. */ #define acpi_handle_emerg(handle, fmt, ...) \ - acpi_handle_printk(KERN_EMERG, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_EMERG, handle, dev_fmt(fmt), ##__VA_ARGS__) #define acpi_handle_alert(handle, fmt, ...) \ - acpi_handle_printk(KERN_ALERT, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_ALERT, handle, dev_fmt(fmt), ##__VA_ARGS__) #define acpi_handle_crit(handle, fmt, ...) \ - acpi_handle_printk(KERN_CRIT, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_CRIT, handle, dev_fmt(fmt), ##__VA_ARGS__) #define acpi_handle_err(handle, fmt, ...) \ - acpi_handle_printk(KERN_ERR, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_ERR, handle, dev_fmt(fmt), ##__VA_ARGS__) #define acpi_handle_warn(handle, fmt, ...) \ - acpi_handle_printk(KERN_WARNING, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_WARNING, handle, dev_fmt(fmt), ##__VA_ARGS__) #define acpi_handle_notice(handle, fmt, ...) \ - acpi_handle_printk(KERN_NOTICE, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_NOTICE, handle, dev_fmt(fmt), ##__VA_ARGS__) #define acpi_handle_info(handle, fmt, ...) \ - acpi_handle_printk(KERN_INFO, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_INFO, handle, dev_fmt(fmt), ##__VA_ARGS__) #if defined(DEBUG) #define acpi_handle_debug(handle, fmt, ...) \ - acpi_handle_printk(KERN_DEBUG, handle, fmt, ##__VA_ARGS__) + acpi_handle_printk(KERN_DEBUG, handle, dev_fmt(fmt), ##__VA_ARGS__) #else #if defined(CONFIG_DYNAMIC_DEBUG) #define acpi_handle_debug(handle, fmt, ...) \ _dynamic_func_call(fmt, __acpi_handle_debug, \ - handle, pr_fmt(fmt), ##__VA_ARGS__) + handle, dev_fmt(fmt), ##__VA_ARGS__) #else #define acpi_handle_debug(handle, fmt, ...) \ ({ \ diff --git a/include/linux/amd-iommu.h b/include/linux/amd-iommu.h index edcee9f5335a..104817e8d897 100644 --- a/include/linux/amd-iommu.h +++ b/include/linux/amd-iommu.h @@ -11,11 +11,11 @@ #include <linux/types.h> struct amd_iommu; +struct pci_dev; #ifdef CONFIG_AMD_IOMMU struct task_struct; -struct pci_dev; extern void amd_iommu_detect(void); @@ -76,4 +76,15 @@ static inline int amd_iommu_snp_disable(void) { return 0; } static inline bool amd_iommu_sev_tio_supported(void) { return false; } #endif +#ifdef CONFIG_AMD_IOMMU +int amd_iommu_enable_perfopt(struct pci_dev *pdev); +void amd_iommu_disable_perfopt(struct pci_dev *pdev); +#else +static inline int amd_iommu_enable_perfopt(struct pci_dev *pdev) +{ + return 0; +} +static inline void amd_iommu_disable_perfopt(struct pci_dev *pdev) { } +#endif + #endif /* _ASM_X86_AMD_IOMMU_H */ diff --git a/include/linux/apple-gmux.h b/include/linux/apple-gmux.h index 206d97ffda79..d7939c3a08fd 100644 --- a/include/linux/apple-gmux.h +++ b/include/linux/apple-gmux.h @@ -109,7 +109,7 @@ static inline bool apple_gmux_detect(struct pnp_dev *pnp_dev, enum apple_gmux_ty if (!adev) return false; - dev = get_device(acpi_get_first_physical_node(adev)); + dev = acpi_bus_get_primary_device(adev); acpi_dev_put(adev); if (!dev) return false; diff --git a/include/linux/arm-rsi-cmds.h b/include/linux/arm-rsi-cmds.h new file mode 100644 index 000000000000..3f7a6a833993 --- /dev/null +++ b/include/linux/arm-rsi-cmds.h @@ -0,0 +1,239 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (C) 2023 ARM Ltd. + */ + +#ifndef __LINUX_ARM_RSI_CMDS_H_ +#define __LINUX_ARM_RSI_CMDS_H_ + +#include <linux/arm-smccc-rsi.h> +#include <linux/jump_label.h> +#include <linux/string.h> +#include <asm/memory.h> + +#ifdef CONFIG_ARM_RMM_RSI +DECLARE_STATIC_KEY_FALSE(rsi_present); + +void __init arm64_rsi_init(void); + +bool arm64_rsi_is_protected(phys_addr_t base, size_t size); + +static inline bool is_realm_world(void) +{ + return static_branch_unlikely(&rsi_present); +} +#else +static inline void arm64_rsi_init(void) { } + +static inline bool arm64_rsi_is_protected(phys_addr_t base, size_t size) +{ + return false; +} + +static inline bool is_realm_world(void) { return false; } +#endif + +#define RSI_GRANULE_SHIFT 12 +#define RSI_GRANULE_SIZE (_AC(1, UL) << RSI_GRANULE_SHIFT) + +enum ripas { + RSI_RIPAS_EMPTY = 0, + RSI_RIPAS_RAM = 1, + RSI_RIPAS_DESTROYED = 2, + RSI_RIPAS_DEV = 3, +}; + +static inline unsigned long rsi_request_version(unsigned long req, + unsigned long *out_lower, + unsigned long *out_higher) +{ + struct arm_smccc_res res; + + arm_smccc_smc(SMC_RSI_ABI_VERSION, req, 0, 0, 0, 0, 0, 0, &res); + + if (out_lower) + *out_lower = res.a1; + if (out_higher) + *out_higher = res.a2; + + return res.a0; +} + +static inline unsigned long rsi_get_realm_config(struct realm_config *cfg) +{ + struct arm_smccc_res res; + + arm_smccc_smc(SMC_RSI_REALM_CONFIG, virt_to_phys(cfg), + 0, 0, 0, 0, 0, 0, &res); + return res.a0; +} + +static inline unsigned long rsi_ipa_state_get(phys_addr_t start, + phys_addr_t end, + enum ripas *state, + phys_addr_t *top) +{ + struct arm_smccc_res res; + + arm_smccc_smc(SMC_RSI_IPA_STATE_GET, + start, end, 0, 0, 0, 0, 0, + &res); + + if (res.a0 == RSI_SUCCESS) { + if (top) + *top = res.a1; + if (state) + *state = res.a2; + } + + return res.a0; +} + +static inline long rsi_set_addr_range_state(phys_addr_t start, + phys_addr_t end, + enum ripas state, + unsigned long flags, + phys_addr_t *top) +{ + struct arm_smccc_res res; + + arm_smccc_smc(SMC_RSI_IPA_STATE_SET, start, end, state, + flags, 0, 0, 0, &res); + + if (top) + *top = res.a1; + + if (res.a2 != RSI_ACCEPT) + return -EPERM; + + return res.a0; +} + +static inline int rsi_set_memory_range(phys_addr_t start, phys_addr_t end, + enum ripas state, unsigned long flags) +{ + unsigned long ret; + phys_addr_t top; + + while (start != end) { + ret = rsi_set_addr_range_state(start, end, state, flags, &top); + if (ret || top < start || top > end) + return -EINVAL; + start = top; + } + + return 0; +} + +/* + * Convert the specified range to RAM. Do not use this if you rely on the + * contents of a page that may already be in RAM state. + */ +static inline int rsi_set_memory_range_protected(phys_addr_t start, + phys_addr_t end) +{ + return rsi_set_memory_range(start, end, RSI_RIPAS_RAM, + RSI_CHANGE_DESTROYED); +} + +/* + * Convert the specified range to RAM. Do not convert any pages that may have + * been DESTROYED, without our permission. + */ +static inline int rsi_set_memory_range_protected_safe(phys_addr_t start, + phys_addr_t end) +{ + return rsi_set_memory_range(start, end, RSI_RIPAS_RAM, + RSI_NO_CHANGE_DESTROYED); +} + +static inline int rsi_set_memory_range_shared(phys_addr_t start, + phys_addr_t end) +{ + return rsi_set_memory_range(start, end, RSI_RIPAS_EMPTY, + RSI_CHANGE_DESTROYED); +} + +#define RSI_ATTEST_CHALLENGE_MIN_SIZE 32 +#define RSI_ATTEST_CHALLENGE_MAX_SIZE 64 + +struct rsi_attestation_token_init_args { + unsigned long fid; + u8 challenge[RSI_ATTEST_CHALLENGE_MAX_SIZE]; +}; + +/** + * rsi_attestation_token_init - Initialise the operation to retrieve an + * attestation token. + * + * @challenge: The challenge data to be used in the attestation token + * generation. + * @size: Size of the challenge data in bytes. + * + * Initialises the attestation token generation and returns an upper bound + * on the attestation token size that can be used to allocate an adequate + * buffer. The caller is expected to subsequently call + * rsi_attestation_token_continue() to retrieve the attestation token data on + * the same CPU. + * + * Returns: + * On success, returns the upper limit of the attestation report size. + * Otherwise, -EINVAL + */ +static inline long +rsi_attestation_token_init(const u8 *challenge, unsigned long size) +{ + union { + struct arm_smccc_1_2_regs regs; + struct rsi_attestation_token_init_args init; + } args = { 0 }; + + if (!challenge || size < RSI_ATTEST_CHALLENGE_MIN_SIZE || + size > RSI_ATTEST_CHALLENGE_MAX_SIZE) + return -EINVAL; + + args.init.fid = SMC_RSI_ATTESTATION_TOKEN_INIT; + memcpy(args.init.challenge, challenge, size); + arm_smccc_1_2_smc(&args.regs, &args.regs); + + if (args.regs.a0 == RSI_SUCCESS) + return args.regs.a1; + + return -EINVAL; +} + +/** + * rsi_attestation_token_continue - Continue the operation to retrieve an + * attestation token. + * + * @granule: {I}PA of the Granule to which the token will be written. + * @offset: Offset within Granule to start of buffer in bytes. + * @size: The size of the buffer. + * @len: The number of bytes written to the buffer. + * + * Retrieves up to a RSI_GRANULE_SIZE worth of token data per call. The caller + * is expected to call rsi_attestation_token_init() before calling this + * function to retrieve the attestation token. + * + * Return: + * * %RSI_SUCCESS - Attestation token retrieved successfully. + * * %RSI_INCOMPLETE - Token generation is not complete. + * * %RSI_ERROR_INPUT - A parameter was not valid. + * * %RSI_ERROR_STATE - Attestation not in progress. + */ +static inline unsigned long rsi_attestation_token_continue(phys_addr_t granule, + unsigned long offset, + unsigned long size, + unsigned long *len) +{ + struct arm_smccc_res res; + + arm_smccc_1_1_invoke(SMC_RSI_ATTESTATION_TOKEN_CONTINUE, + granule, offset, size, 0, &res); + + if (len) + *len = res.a1; + return res.a0; +} + +#endif /* __LINUX_ARM_RSI_CMDS_H_ */ diff --git a/include/linux/arm-smccc-bus.h b/include/linux/arm-smccc-bus.h new file mode 100644 index 000000000000..05c0c850510e --- /dev/null +++ b/include/linux/arm-smccc-bus.h @@ -0,0 +1,48 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (C) 2026 Arm Limited + */ +#ifndef __LINUX_ARM_SMCCC_BUS_H +#define __LINUX_ARM_SMCCC_BUS_H + +#include <linux/device.h> +#include <linux/device-id/arm_smccc.h> +#include <linux/module.h> + +struct arm_smccc_device { + struct device dev; + u32 func_id; +}; + +#define to_arm_smccc_device(d) container_of_const(d, struct arm_smccc_device, dev) + +struct arm_smccc_driver { + struct device_driver driver; + + const char *name; + int (*probe)(struct arm_smccc_device *sdev); + void (*remove)(struct arm_smccc_device *sdev); + const struct arm_smccc_device_id *id_table; +}; + +#define to_arm_smccc_driver(d) \ + container_of_const(d, struct arm_smccc_driver, driver) + +int arm_smccc_driver_register(struct arm_smccc_driver *driver, + struct module *owner, const char *mod_name); +void arm_smccc_driver_unregister(struct arm_smccc_driver *driver); +struct arm_smccc_device *arm_smccc_device_register(const char *name, u32 func_id); +void arm_smccc_device_unregister(struct arm_smccc_device *smcc_dev); + +#define arm_smccc_register(driver) \ + arm_smccc_driver_register(driver, THIS_MODULE, KBUILD_MODNAME) +#define arm_smccc_unregister(driver) \ + arm_smccc_driver_unregister(driver) + +#define module_arm_smccc_driver(__arm_smccc_driver) \ + module_driver(__arm_smccc_driver, arm_smccc_register, \ + arm_smccc_unregister) + +extern const struct bus_type arm_smccc_bus_type; + +#endif /* __LINUX_ARM_SMCCC_BUS_H */ diff --git a/include/linux/arm-smccc-rsi.h b/include/linux/arm-smccc-rsi.h new file mode 100644 index 000000000000..fddb77986f70 --- /dev/null +++ b/include/linux/arm-smccc-rsi.h @@ -0,0 +1,193 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (C) 2023 ARM Ltd. + */ + +#ifndef __LINUX_ARM_SMCCC_RSI_H_ +#define __LINUX_ARM_SMCCC_RSI_H_ + +#include <linux/arm-smccc.h> + +/* + * This file describes the Realm Services Interface (RSI) Application Binary + * Interface (ABI) for SMC calls made from within the Realm to the RMM and + * serviced by the RMM. + */ + +/* + * The major version number of the RSI implementation. This is increased when + * the binary format or semantics of the SMC calls change. + */ +#define RSI_ABI_VERSION_MAJOR UL(1) + +/* + * The minor version number of the RSI implementation. This is increased when + * a bug is fixed, or a feature is added without breaking binary compatibility. + */ +#define RSI_ABI_VERSION_MINOR UL(0) + +#define RSI_ABI_VERSION ((RSI_ABI_VERSION_MAJOR << 16) | \ + RSI_ABI_VERSION_MINOR) + +#define RSI_ABI_VERSION_GET_MAJOR(_version) ((_version) >> 16) +#define RSI_ABI_VERSION_GET_MINOR(_version) ((_version) & 0xFFFF) + +#define RSI_SUCCESS UL(0) +#define RSI_ERROR_INPUT UL(1) +#define RSI_ERROR_STATE UL(2) +#define RSI_INCOMPLETE UL(3) +#define RSI_ERROR_UNKNOWN UL(4) + +#define SMC_RSI_FID(n) ARM_SMCCC_CALL_VAL(ARM_SMCCC_FAST_CALL, \ + ARM_SMCCC_SMC_64, \ + ARM_SMCCC_OWNER_STANDARD, \ + n) + +/* + * Returns RSI version. + * + * arg1 == Requested interface revision + * ret0 == Status / error + * ret1 == Lower implemented interface revision + * ret2 == Higher implemented interface revision + */ +#define SMC_RSI_ABI_VERSION SMC_RSI_FID(0x190) + +/* + * Read feature register. + * + * arg1 == Feature register index + * ret0 == Status / error + * ret1 == Feature register value + */ +#define SMC_RSI_FEATURES SMC_RSI_FID(0x191) + +/* + * Read measurement for the current Realm. + * + * arg1 == Index, which measurements slot to read + * ret0 == Status / error + * ret1 == Measurement value, bytes: 0 - 7 + * ret2 == Measurement value, bytes: 8 - 15 + * ret3 == Measurement value, bytes: 16 - 23 + * ret4 == Measurement value, bytes: 24 - 31 + * ret5 == Measurement value, bytes: 32 - 39 + * ret6 == Measurement value, bytes: 40 - 47 + * ret7 == Measurement value, bytes: 48 - 55 + * ret8 == Measurement value, bytes: 56 - 63 + */ +#define SMC_RSI_MEASUREMENT_READ SMC_RSI_FID(0x192) + +/* + * Extend Realm Extensible Measurement (REM) value. + * + * arg1 == Index, which measurements slot to extend + * arg2 == Size of realm measurement in bytes, max 64 bytes + * arg3 == Measurement value, bytes: 0 - 7 + * arg4 == Measurement value, bytes: 8 - 15 + * arg5 == Measurement value, bytes: 16 - 23 + * arg6 == Measurement value, bytes: 24 - 31 + * arg7 == Measurement value, bytes: 32 - 39 + * arg8 == Measurement value, bytes: 40 - 47 + * arg9 == Measurement value, bytes: 48 - 55 + * arg10 == Measurement value, bytes: 56 - 63 + * ret0 == Status / error + */ +#define SMC_RSI_MEASUREMENT_EXTEND SMC_RSI_FID(0x193) + +/* + * Initialize the operation to retrieve an attestation token. + * + * arg1 == Challenge value, bytes: 0 - 7 + * arg2 == Challenge value, bytes: 8 - 15 + * arg3 == Challenge value, bytes: 16 - 23 + * arg4 == Challenge value, bytes: 24 - 31 + * arg5 == Challenge value, bytes: 32 - 39 + * arg6 == Challenge value, bytes: 40 - 47 + * arg7 == Challenge value, bytes: 48 - 55 + * arg8 == Challenge value, bytes: 56 - 63 + * ret0 == Status / error + * ret1 == Upper bound of token size in bytes + */ +#define SMC_RSI_ATTESTATION_TOKEN_INIT SMC_RSI_FID(0x194) + +/* + * Continue the operation to retrieve an attestation token. + * + * arg1 == The IPA of token buffer + * arg2 == Offset within the granule of the token buffer + * arg3 == Size of the granule buffer + * ret0 == Status / error + * ret1 == Length of token bytes copied to the granule buffer + */ +#define SMC_RSI_ATTESTATION_TOKEN_CONTINUE SMC_RSI_FID(0x195) + +#ifndef __ASSEMBLER__ + +struct realm_config { + union { + struct { + unsigned long ipa_bits; /* Width of IPA in bits */ + unsigned long hash_algo; /* Hash algorithm */ + }; + u8 pad[0x200]; + }; + union { + u8 rpv[64]; /* Realm Personalization Value */ + u8 pad2[0xe00]; + }; + /* + * The RMM requires the configuration structure to be aligned to a 4k + * boundary, ensure this happens by aligning this structure. + */ +} __aligned(0x1000); + +#endif /* __ASSEMBLER__ */ + +/* + * Read configuration for the current Realm. + * + * arg1 == struct realm_config addr + * ret0 == Status / error + */ +#define SMC_RSI_REALM_CONFIG SMC_RSI_FID(0x196) + +/* + * Request RIPAS of a target IPA range to be changed to a specified value. + * + * arg1 == Base IPA address of target region + * arg2 == Top of the region + * arg3 == RIPAS value + * arg4 == flags + * ret0 == Status / error + * ret1 == Top of modified IPA range + * ret2 == Whether the Host accepted or rejected the request + */ +#define SMC_RSI_IPA_STATE_SET SMC_RSI_FID(0x197) + +#define RSI_NO_CHANGE_DESTROYED UL(0) +#define RSI_CHANGE_DESTROYED UL(1) + +#define RSI_ACCEPT UL(0) +#define RSI_REJECT UL(1) + +/* + * Get RIPAS of a target IPA range. + * + * arg1 == Base IPA of target region + * arg2 == End of target IPA region + * ret0 == Status / error + * ret1 == Top of IPA region which has the reported RIPAS value + * ret2 == RIPAS value + */ +#define SMC_RSI_IPA_STATE_GET SMC_RSI_FID(0x198) + +/* + * Make a Host call. + * + * arg1 == IPA of host call structure + * ret0 == Status / error + */ +#define SMC_RSI_HOST_CALL SMC_RSI_FID(0x199) + +#endif /* __LINUX_ARM_SMCCC_RSI_H_ */ diff --git a/include/linux/audit.h b/include/linux/audit.h index 45abb3722d30..18b44a1c5397 100644 --- a/include/linux/audit.h +++ b/include/linux/audit.h @@ -547,7 +547,7 @@ static inline void audit_openat2_how(struct open_how *how) static inline void audit_log_kern_module(const char *name) { - if (!audit_dummy_context()) + if (unlikely(!audit_dummy_context())) __audit_log_kern_module(name); } @@ -563,7 +563,7 @@ static inline void audit_tk_injoffset(struct timespec64 offset) if (offset.tv_sec == 0 && offset.tv_nsec == 0) return; - if (!audit_dummy_context()) + if (unlikely(!audit_dummy_context())) __audit_tk_injoffset(offset); } @@ -586,7 +586,7 @@ static inline void audit_ntp_set_new(struct audit_ntp_data *ad, static inline void audit_ntp_log(const struct audit_ntp_data *ad) { - if (!audit_dummy_context()) + if (unlikely(!audit_dummy_context())) __audit_ntp_log(ad); } diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h index f686a37f7a0a..2e87faf9a8c2 100644 --- a/include/linux/binfmts.h +++ b/include/linux/binfmts.h @@ -128,7 +128,8 @@ struct linux_binfmt { struct module *module; int (*load_binary)(struct linux_binprm *); #ifdef CONFIG_COREDUMP - int (*core_dump)(struct coredump_params *cprm); + /* Returns true if the whole coredump was written. */ + bool (*core_dump)(struct coredump_params *cprm); unsigned long min_coredump; /* minimal dump size */ #endif } __randomize_layout; diff --git a/include/linux/bio-integrity.h b/include/linux/bio-integrity.h index 0ea2a8bf7efb..a954c97be0b3 100644 --- a/include/linux/bio-integrity.h +++ b/include/linux/bio-integrity.h @@ -151,7 +151,6 @@ void bio_integrity_setup_default(struct bio *bio); unsigned int fs_bio_integrity_alloc(struct bio *bio); void fs_bio_integrity_free(struct bio *bio); void fs_bio_integrity_generate(struct bio *bio); -int fs_bio_integrity_verify(struct bio *bio, sector_t sector, - unsigned int size); +int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter); #endif /* _LINUX_BIO_INTEGRITY_H */ diff --git a/include/linux/bio.h b/include/linux/bio.h index bb3235497e67..75f8086351c0 100644 --- a/include/linux/bio.h +++ b/include/linux/bio.h @@ -256,6 +256,12 @@ static inline struct folio *bio_first_folio_all(struct bio *bio) return page_folio(bio_first_page_all(bio)); } +static inline struct bio_vec *bio_last_bvec_all(struct bio *bio) +{ + WARN_ON_ONCE(bio_flagged(bio, BIO_CLONED)); + return &bio->bi_io_vec[bio->bi_vcnt - 1]; +} + /** * struct folio_iter - State for iterating all folios in a bio. * @folio: The current folio we're iterating. NULL after the last folio. @@ -479,6 +485,7 @@ static inline void bio_init_inline(struct bio *bio, struct block_device *bdev, extern void bio_uninit(struct bio *); void bio_reset(struct bio *bio, struct block_device *bdev, blk_opf_t opf); void bio_reuse(struct bio *bio, blk_opf_t opf); +void bio_prepare_reissue(struct bio *bio, struct block_device *bdev); void bio_chain(struct bio *, struct bio *); void bio_await(struct bio *bio, void *priv, void (*submit)(struct bio *bio, void *priv)); @@ -516,16 +523,18 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data, size_t len, enum req_op op); int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, - unsigned mem_align_mask, unsigned len_align_mask); + unsigned maxlen, unsigned mem_align_mask, + unsigned len_align_mask); bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter); void __bio_release_pages(struct bio *bio, bool mark_dirty); extern void bio_set_pages_dirty(struct bio *bio); extern void bio_check_pages_dirty(struct bio *bio); -int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen, - size_t minsize); -void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty); +int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize); +void bio_free_folios(struct bio *bio); +int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, + size_t maxlen, size_t minsize); extern void bio_copy_data(struct bio *dst, struct bio *src); extern void bio_free_pages(struct bio *bio); diff --git a/include/linux/bitmap.h b/include/linux/bitmap.h index 7df1573a409c..adafbcf2016b 100644 --- a/include/linux/bitmap.h +++ b/include/linux/bitmap.h @@ -52,6 +52,7 @@ struct device; * bitmap_complement(dst, src, nbits) *dst = ~(*src) * bitmap_equal(src1, src2, nbits) Are *src1 and *src2 equal? * bitmap_intersects(src1, src2, nbits) Do *src1 and *src2 overlap? + * bitmap_intersects_and(src1, src2, src3, nbits) Do *src1, *src2 and *src3 overlap? * bitmap_subset(src1, src2, nbits) Is *src1 a subset of *src2? * bitmap_empty(src, nbits) Are all bits zero in *src? * bitmap_full(src, nbits) Are all bits set in *src? @@ -181,6 +182,9 @@ void __bitmap_replace(unsigned long *dst, const unsigned long *mask, unsigned int nbits); bool __bitmap_intersects(const unsigned long *bitmap1, const unsigned long *bitmap2, unsigned int nbits); +bool __bitmap_intersects_and(const unsigned long *bitmap1, + const unsigned long *bitmap2, + const unsigned long *bitmap3, unsigned int nbits); bool __bitmap_subset(const unsigned long *bitmap1, const unsigned long *bitmap2, unsigned int nbits); unsigned int __bitmap_weight(const unsigned long *bitmap, unsigned int nbits); @@ -446,6 +450,16 @@ bool bitmap_intersects(const unsigned long *src1, const unsigned long *src2, uns } static __always_inline +bool bitmap_intersects_and(const unsigned long *src1, const unsigned long *src2, + const unsigned long *src3, unsigned int nbits) +{ + if (small_const_nbits(nbits)) + return ((*src1 & *src2 & *src3) & BITMAP_LAST_WORD_MASK(nbits)) != 0; + else + return __bitmap_intersects_and(src1, src2, src3, nbits); +} + +static __always_inline bool bitmap_subset(const unsigned long *src1, const unsigned long *src2, unsigned int nbits) { if (small_const_nbits(nbits)) diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h index 4f7905c3412b..d003a9d2d1f6 100644 --- a/include/linux/blkdev.h +++ b/include/linux/blkdev.h @@ -191,14 +191,17 @@ struct gendisk { #ifdef CONFIG_BLK_DEV_ZONED /* * Zoned block device information. Reads of this information must be - * protected with blk_queue_enter() / blk_queue_exit(). Modifying this - * information is only allowed while no requests are being processed. - * See also blk_mq_freeze_queue() and blk_mq_unfreeze_queue(). + * protected with blk_queue_enter() / blk_queue_exit() or by holding a + * lock on zone_revalidate_mutex. blk_revalidate_disk_zones() may modify + * this information while no requests are being processed (disk queue + * frozen with blk_mq_freeze_queue()) and while holding a lock on + * zone_revalidate_mutex. */ + struct mutex zone_revalidate_mutex; unsigned int nr_zones; unsigned int zone_capacity; unsigned int last_zone_capacity; - u8 __rcu *zones_cond; + u8 __rcu *zones_state; unsigned int zone_wplugs_hash_bits; atomic_t nr_zone_wplugs; spinlock_t zone_wplugs_hash_lock; @@ -1816,9 +1819,11 @@ static inline int bio_split_rw_at(struct bio *bio, */ static inline unsigned int max_integrity_io_size(struct queue_limits *lim) { - return min_t(unsigned int, lim->max_segment_size, - (BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size) << - lim->integrity.interval_exp); + u64 max_intervals; + + max_intervals = BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size; + return min_t(u64, lim->max_segment_size, + max_intervals << lim->integrity.interval_exp); } #define DEFINE_IO_COMP_BATCH(name) struct io_comp_batch name = { } diff --git a/include/linux/bnge/hsi.h b/include/linux/bnge/hsi.h index 1f7bd96415a5..d4fe662d60b1 100644 --- a/include/linux/bnge/hsi.h +++ b/include/linux/bnge/hsi.h @@ -8607,6 +8607,7 @@ struct hwrm_cfa_l2_filter_alloc_input { #define CFA_L2_FILTER_ALLOC_REQ_ENABLES_MIRROR_VNIC_ID 0x10000UL #define CFA_L2_FILTER_ALLOC_REQ_ENABLES_NUM_VLANS 0x20000UL #define CFA_L2_FILTER_ALLOC_REQ_ENABLES_T_NUM_VLANS 0x40000UL + #define CFA_L2_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX 0x80000UL u8 l2_addr[6]; u8 num_vlans; u8 t_num_vlans; @@ -8663,7 +8664,8 @@ struct hwrm_cfa_l2_filter_alloc_input { #define CFA_L2_FILTER_ALLOC_REQ_PRI_HINT_MIN 0x4UL #define CFA_L2_FILTER_ALLOC_REQ_PRI_HINT_LAST CFA_L2_FILTER_ALLOC_REQ_PRI_HINT_MIN u8 unused_5; - __le32 unused_6; + __le16 rfs_ring_tbl_idx; + __le16 unused_6; __le64 l2_filter_id_hint; }; @@ -8900,6 +8902,95 @@ struct hwrm_cfa_tunnel_filter_free_output { u8 valid; }; +/* hwrm_cfa_ntuple_filter_alloc_input (size:1024b/128B) */ +struct hwrm_cfa_ntuple_filter_alloc_input { + __le16 req_type; + __le16 cmpl_ring; + __le16 seq_id; + __le16 target_id; + __le64 resp_addr; + __le32 flags; + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_LOOPBACK 0x1UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DROP 0x2UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_METER 0x4UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DEST_FID 0x8UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_ARP_REPLY 0x10UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DEST_RFS_RING_IDX 0x20UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_NO_L2_CONTEXT 0x40UL + __le32 enables; + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_L2_FILTER_ID 0x1UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_ETHERTYPE 0x2UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_TUNNEL_TYPE 0x4UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_MACADDR 0x8UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_IPADDR_TYPE 0x10UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_IPADDR 0x20UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_IPADDR_MASK 0x40UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_IPADDR 0x80UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_IPADDR_MASK 0x100UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_IP_PROTOCOL 0x200UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_PORT 0x400UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_PORT_MASK 0x800UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_PORT 0x1000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_PORT_MASK 0x2000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_PRI_HINT 0x4000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_NTUPLE_FILTER_ID 0x8000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_ID 0x10000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_MIRROR_VNIC_ID 0x20000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_MACADDR 0x40000UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX 0x80000UL + __le64 l2_filter_id; + u8 src_macaddr[6]; + __be16 ethertype; + u8 ip_addr_type; + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_UNKNOWN 0x0UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV4 0x4UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV6 0x6UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV6 + u8 ip_protocol; + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_UNKNOWN 0x0UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_TCP 0x6UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_UDP 0x11UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_ICMP 0x1UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_ICMPV6 0x3aUL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_RSVD 0xffUL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_IP_PROTOCOL_RSVD + __le16 dst_id; + __le16 rfs_ring_tbl_idx; + u8 tunnel_type; + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_NONTUNNEL 0x0UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN 0x1UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_NVGRE 0x2UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_L2GRE 0x3UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_IPIP 0x4UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_GENEVE 0x5UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_MPLS 0x6UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_STT 0x7UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_IPGRE 0x8UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN_V4 0x9UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_IPGRE_V1 0xaUL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_L2_ETYPE 0xbUL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN_GPE_V6 0xcUL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_VXLAN_GPE 0x10UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_ANYTUNNEL 0xffUL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_ANYTUNNEL + u8 pri_hint; + #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_NO_PREFER 0x0UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_ABOVE 0x1UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_BELOW 0x2UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_HIGHEST 0x3UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_LOWEST 0x4UL + #define CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_LAST CFA_NTUPLE_FILTER_ALLOC_REQ_PRI_HINT_LOWEST + __be32 src_ipaddr[4]; + __be32 src_ipaddr_mask[4]; + __be32 dst_ipaddr[4]; + __be32 dst_ipaddr_mask[4]; + __be16 src_port; + __be16 src_port_mask; + __be16 dst_port; + __be16 dst_port_mask; + __le64 ntuple_filter_id_hint; +}; + /* hwrm_cfa_ntuple_filter_alloc_output (size:192b/24B) */ struct hwrm_cfa_ntuple_filter_alloc_output { __le16 error_code; @@ -9014,6 +9105,49 @@ struct hwrm_cfa_ntuple_filter_cfg_output { u8 valid; }; +/* hwrm_cfa_adv_flow_mgnt_qcaps_input (size:256b/32B) */ +struct hwrm_cfa_adv_flow_mgnt_qcaps_input { + __le16 req_type; + __le16 cmpl_ring; + __le16 seq_id; + __le16 target_id; + __le64 resp_addr; + __le32 unused_0[4]; +}; + +/* hwrm_cfa_adv_flow_mgnt_qcaps_output (size:128b/16B) */ +struct hwrm_cfa_adv_flow_mgnt_qcaps_output { + __le16 error_code; + __le16 req_type; + __le16 seq_id; + __le16 resp_len; + __le32 flags; + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_HND_16BIT_SUPPORTED 0x1UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_HND_64BIT_SUPPORTED 0x2UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_BATCH_DELETE_SUPPORTED 0x4UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_RESET_ALL_SUPPORTED 0x8UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_DEST_FUNC_SUPPORTED 0x10UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_TX_EEM_FLOW_SUPPORTED 0x20UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RX_EEM_FLOW_SUPPORTED 0x40UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_FLOW_COUNTER_ALLOC_SUPPORTED 0x80UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RFS_RING_TBL_IDX_SUPPORTED 0x100UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_UNTAGGED_VLAN_SUPPORTED 0x200UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_XDP_SUPPORTED 0x400UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_L2_HEADER_SOURCE_FIELDS_SUPPORTED 0x800UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_RX_ARP_SUPPORTED 0x1000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RFS_RING_TBL_IDX_V2_SUPPORTED 0x2000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_RX_ETHERTYPE_IP_SUPPORTED 0x4000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_TRUFLOW_CAPABLE 0x8000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_L2_FILTER_TRAFFIC_TYPE_L2_ROCE_SUPPORTED 0x10000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_LAG_SUPPORTED 0x20000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_NO_L2CTX_SUPPORTED 0x40000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NIC_FLOW_STATS_SUPPORTED 0x80000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_NTUPLE_FLOW_RX_EXT_IP_PROTO_SUPPORTED 0x100000UL + #define CFA_ADV_FLOW_MGNT_QCAPS_RESP_FLAGS_RFS_RING_TBL_IDX_V3_SUPPORTED 0x200000UL + u8 unused_0[3]; + u8 valid; +}; + /* hwrm_tunnel_dst_port_alloc_input (size:192b/24B) */ struct hwrm_tunnel_dst_port_alloc_input { __le16 req_type; diff --git a/include/linux/bootconfig.h b/include/linux/bootconfig.h index deda507500da..fdc15b6f4d1b 100644 --- a/include/linux/bootconfig.h +++ b/include/linux/bootconfig.h @@ -291,12 +291,7 @@ int __init xbc_init(const char *buf, size_t size, const char **emsg, int *epos); int __init xbc_get_info(int *node_size, size_t *data_size); /* XBC cleanup data structures */ -void __init _xbc_exit(bool early); - -static __always_inline void xbc_exit(void) -{ - _xbc_exit(false); -} +void __init xbc_exit(void); /* XBC embedded bootconfig data in kernel */ #ifdef CONFIG_BOOT_CONFIG_EMBED diff --git a/include/linux/bpf-cgroup-defs.h b/include/linux/bpf-cgroup-defs.h index c9e6b26abab6..0147b8bec973 100644 --- a/include/linux/bpf-cgroup-defs.h +++ b/include/linux/bpf-cgroup-defs.h @@ -47,6 +47,7 @@ enum cgroup_bpf_attach_type { CGROUP_INET6_GETSOCKNAME, CGROUP_UNIX_GETSOCKNAME, CGROUP_INET_SOCK_RELEASE, + CGROUP_TCP_SOCK_OPS, CGROUP_LSM_START, CGROUP_LSM_END = CGROUP_LSM_START + CGROUP_LSM_NUM - 1, MAX_CGROUP_BPF_ATTACH_TYPE diff --git a/include/linux/bpf-cgroup.h b/include/linux/bpf-cgroup.h index 4d0cc65976a1..8a75a6cd7309 100644 --- a/include/linux/bpf-cgroup.h +++ b/include/linux/bpf-cgroup.h @@ -100,6 +100,8 @@ struct bpf_cgroup_storage { struct bpf_cgroup_link { struct bpf_link link; struct cgroup *cgroup; + struct bpf_map *map; + wait_queue_head_t wait_hup; }; struct bpf_prog_list { @@ -110,6 +112,18 @@ struct bpf_prog_list { u32 flags; }; +#define bpf_cgroup_struct_ops_foreach(var, item, cgrp, atype) \ + for (item = rcu_dereference((cgrp)->bpf.effective[atype])->items;\ + ((var) = READ_ONCE(item->kdata)); \ + item++) + +static inline bool cgroup_bpf_is_struct_ops_atype(enum cgroup_bpf_attach_type atype) +{ + return atype == CGROUP_TCP_SOCK_OPS; +} +void cgroup_bpf_struct_ops_register(int atype, u32 type_id, void *cfi_stubs, bool mult_trace); +int cgroup_bpf_struct_ops_attach(struct bpf_map *map, const union bpf_attr *attr); + void __init cgroup_bpf_lifetime_notifier_init(void); int __cgroup_bpf_run_filter_skb(struct sock *sk, @@ -479,6 +493,20 @@ static inline int bpf_percpu_cgroup_storage_update(struct bpf_map *map, return 0; } +static inline bool cgroup_bpf_is_struct_ops_atype(int atype) +{ + return false; +} +static inline void cgroup_bpf_struct_ops_register(int atype, u32 type_id, void *cfi_stubs, + bool mult_trace) +{ +} +static inline int cgroup_bpf_struct_ops_attach(struct bpf_map *map, + const union bpf_attr *attr) +{ + return -EOPNOTSUPP; +} + #define cgroup_bpf_enabled(atype) (0) #define BPF_CGROUP_RUN_SA_PROG_LOCK(sk, uaddr, uaddrlen, atype, t_ctx) ({ 0; }) #define BPF_CGROUP_RUN_SA_PROG(sk, uaddr, uaddrlen, atype) ({ 0; }) diff --git a/include/linux/bpf.h b/include/linux/bpf.h index ffa5626411ac..d23a188a6f40 100644 --- a/include/linux/bpf.h +++ b/include/linux/bpf.h @@ -17,6 +17,7 @@ #include <linux/numa.h> #include <linux/mm_types.h> #include <linux/wait.h> +#include <linux/irq_work_types.h> #include <linux/refcount.h> #include <linux/mutex.h> #include <linux/module.h> @@ -90,6 +91,7 @@ struct bpf_map_ops { struct bpf_map *(*map_alloc)(union bpf_attr *attr); void (*map_release)(struct bpf_map *map, struct file *map_file); void (*map_free)(struct bpf_map *map); + void (*map_free_pre_rcu)(struct bpf_map *map); int (*map_get_next_key)(struct bpf_map *map, void *key, void *next_key); void (*map_release_uref)(struct bpf_map *map); void *(*map_lookup_elem_sys_only)(struct bpf_map *map, void *key); @@ -214,6 +216,7 @@ enum btf_field_type { BPF_UPTR = (1 << 11), BPF_RES_SPIN_LOCK = (1 << 12), BPF_TASK_WORK = (1 << 13), + BPF_RCU_HEAD = (1 << 14), }; enum bpf_cgroup_storage_type { @@ -268,6 +271,7 @@ struct btf_record { int wq_off; int refcount_off; int task_work_off; + int rcu_head_off; struct btf_field fields[]; }; @@ -341,8 +345,18 @@ struct bpf_map { s64 __percpu *elem_count; u64 cookie; /* write-once */ char *excl_prog_sha; + /* + * Which programs use the map, see bpf_map_claim(): 0 - none so far, + * aux of the program - only that one, the same with BPF_MAP_USER_PATCHED + * set - only that one and it stored the addresses of its functions into + * the map, BPF_MAP_USER_MANY - more than one. + */ + unsigned long user; }; +#define BPF_MAP_USER_MANY 1UL +#define BPF_MAP_USER_PATCHED 1UL + static inline const char *btf_field_type_name(enum btf_field_type type) { switch (type) { @@ -373,6 +387,8 @@ static inline const char *btf_field_type_name(enum btf_field_type type) return "bpf_refcount"; case BPF_TASK_WORK: return "bpf_task_work"; + case BPF_RCU_HEAD: + return "bpf_rcu_head"; default: WARN_ON_ONCE(1); return "unknown"; @@ -413,6 +429,8 @@ static inline u32 btf_field_type_size(enum btf_field_type type) return sizeof(struct bpf_refcount); case BPF_TASK_WORK: return sizeof(struct bpf_task_work); + case BPF_RCU_HEAD: + return sizeof(struct bpf_rcu_head); default: WARN_ON_ONCE(1); return 0; @@ -447,6 +465,8 @@ static inline u32 btf_field_type_align(enum btf_field_type type) return __alignof__(struct bpf_refcount); case BPF_TASK_WORK: return __alignof__(struct bpf_task_work); + case BPF_RCU_HEAD: + return __alignof__(struct bpf_rcu_head); default: WARN_ON_ONCE(1); return 0; @@ -479,6 +499,7 @@ static inline void bpf_obj_init_field(const struct btf_field *field, void *addr) case BPF_KPTR_PERCPU: case BPF_UPTR: case BPF_TASK_WORK: + case BPF_RCU_HEAD: break; default: WARN_ON_ONCE(1); @@ -572,7 +593,7 @@ static inline void bpf_obj_memcpy(struct btf_record *rec, if (long_memcpy) bpf_long_memcpy(dst, src, size); else - memcpy(dst, src, size); + data_race(memcpy(dst, src, size)); return; } @@ -580,10 +601,10 @@ static inline void bpf_obj_memcpy(struct btf_record *rec, u32 next_off = rec->fields[i].offset; u32 sz = next_off - curr_off; - memcpy(dst + curr_off, src + curr_off, sz); + data_race(memcpy(dst + curr_off, src + curr_off, sz)); curr_off += rec->fields[i].size + sz; } - memcpy(dst + curr_off, src + curr_off, size - curr_off); + data_race(memcpy(dst + curr_off, src + curr_off, size - curr_off)); } static inline void copy_map_value(struct bpf_map *map, void *dst, void *src) @@ -780,7 +801,10 @@ enum bpf_type_flag { */ PTR_UNTRUSTED = BIT(6 + BPF_BASE_TYPE_BITS), - /* MEM can be uninitialized. */ + /* + * MEM can be uninitialized. Generic memory outputs need not be fully + * initialized by the callee. + */ MEM_UNINIT = BIT(7 + BPF_BASE_TYPE_BITS), /* DYNPTR points to memory local to the bpf program. */ @@ -874,7 +898,7 @@ enum bpf_type_flag { /* function argument constraints */ enum bpf_arg_type { - ARG_DONTCARE = 0, /* unused argument in helper function */ + ARG_UNUSED = 0, /* unused argument; terminates argument iteration */ /* the following constraints used to prototype * bpf_map_lookup/update/delete_elem() functions @@ -894,6 +918,7 @@ enum bpf_arg_type { ARG_PTR_TO_CTX, /* pointer to context */ ARG_ANYTHING, /* any (initialized) argument is ok */ + ARG_SCALAR, /* scalar argument */ ARG_PTR_TO_SPIN_LOCK, /* pointer to bpf_spin_lock */ ARG_PTR_TO_SOCK_COMMON, /* pointer to sock_common */ ARG_PTR_TO_SOCKET, /* pointer to bpf_sock (fullsock) */ @@ -908,6 +933,24 @@ enum bpf_arg_type { ARG_PTR_TO_TIMER, /* pointer to bpf_timer */ ARG_KPTR_XCHG_DEST, /* pointer to destination that kptrs are bpf_kptr_xchg'd into */ ARG_PTR_TO_DYNPTR, /* pointer to bpf_dynptr. See bpf_type_flag for dynptr type */ + + ARG_CONST_SCALAR, /* scalar known at verification time */ + ARG_CONST_MEM_SIZE, /* ARG_MEM_SIZE that must be constant */ + ARG_PTR_TO_ALLOC_BTF_ID, /* pointer to an allocated object */ + ARG_PTR_TO_REFCOUNTED_KPTR, /* pointer to a refcounted local kptr */ + ARG_PTR_TO_ITER, /* pointer to an iterator */ + ARG_PTR_TO_LIST_HEAD, /* pointer to bpf_list_head */ + ARG_PTR_TO_LIST_NODE, /* pointer to bpf_list_node */ + ARG_PTR_TO_RB_ROOT, /* pointer to bpf_rb_root */ + ARG_PTR_TO_RB_NODE, /* pointer to bpf_rb_node */ + ARG_PTR_TO_WORKQUEUE, /* pointer to bpf_wq */ + ARG_PTR_TO_TASK_WORK, /* pointer to bpf_task_work */ + ARG_PTR_TO_RCU_HEAD, /* pointer to bpf_rcu_head */ + ARG_PTR_TO_IRQ_FLAG, /* pointer to saved IRQ flags on the stack */ + ARG_PTR_TO_RES_SPIN_LOCK, /* pointer to bpf_res_spin_lock */ + ARG_PTR_TO_CTX_OUT, /* hook output argument passed through from ctx */ + ARG_PTR_TO_PROG_AUX, /* pointer to the caller's bpf_prog_aux */ + ARG_IGNORE, /* argument the verifier does not check at all */ __BPF_ARG_TYPE_MAX, /* Extended arg_types. */ @@ -976,6 +1019,13 @@ static_assert(__BPF_RET_TYPE_MAX <= BPF_BASE_TYPE_LIMIT); */ #define MAX_BPF_FUNC_REG_ARGS 5 +/* A by-value argument takes two eightbytes at most, so the maximum number of + * argument slots of any function is 2 * MAX_BPF_FUNC_ARGS. A local array may + * need that size for processing, although eventually the maximum slots will + * be capped at MAX_BPF_FUNC_ARGS. + */ +#define MAX_BPF_FUNC_ARG_SLOTS (2 * MAX_BPF_FUNC_ARGS) + /* eBPF function prototype used by verifier to allow BPF_CALLs from eBPF programs * to in-kernel helper functions and for adjusting imm32 field in BPF_CALL * instructions after verifying @@ -1004,13 +1054,13 @@ struct bpf_func_proto { }; union { struct { - u32 *arg1_btf_id; - u32 *arg2_btf_id; - u32 *arg3_btf_id; - u32 *arg4_btf_id; - u32 *arg5_btf_id; + const u32 *arg1_btf_id; + const u32 *arg2_btf_id; + const u32 *arg3_btf_id; + const u32 *arg4_btf_id; + const u32 *arg5_btf_id; }; - u32 *arg_btf_id[MAX_BPF_FUNC_ARGS]; + const u32 *arg_btf_id[MAX_BPF_FUNC_ARGS]; struct { size_t arg1_size; size_t arg2_size; @@ -1113,6 +1163,7 @@ struct bpf_insn_access_aux { u32 ref_id; }; }; + u32 mem_size; struct bpf_verifier_log *log; /* for verbose logs */ bool is_retval; /* is accessing function return value ? */ }; @@ -1193,6 +1244,9 @@ struct bpf_prog_offload { u32 jited_len; }; +/* The argument is aligned to 16 bytes. */ +#define BTF_FMODEL_ALIGN16_ARG BIT(0) + /* The argument is signed. */ #define BTF_FMODEL_SIGNED_ARG BIT(1) @@ -1210,6 +1264,11 @@ struct btf_func_model { u8 arg_flags[MAX_BPF_FUNC_ARGS]; }; +static inline u32 btf_func_model_arg_slots(const struct btf_func_model *m, u32 arg) +{ + return (m->arg_size[arg] + sizeof(u64) - 1) / sizeof(u64); +} + /* Restore arguments before returning from trampoline to let original function * continue executing. This flag is used for fentry progs when there are no * fexit progs. @@ -1257,11 +1316,15 @@ struct btf_func_model { #define BPF_TRAMP_F_INDIRECT BIT(8) /* Each call __bpf_prog_enter + call bpf_func + call __bpf_prog_exit is ~50 - * bytes on x86. + * bytes on x86. The trampoline image has to fit in PAGE_SIZE. */ enum { -#if defined(__s390x__) +#if defined(__s390x__) || defined(__powerpc64__) BPF_MAX_TRAMP_LINKS = 27, +#elif defined(__x86_64__) + BPF_MAX_TRAMP_LINKS = 36, +#elif defined(__aarch64__) + BPF_MAX_TRAMP_LINKS = 37, #else BPF_MAX_TRAMP_LINKS = 38, #endif @@ -1314,6 +1377,7 @@ int arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *image, void *i void *arch_alloc_bpf_trampoline(unsigned int size); void arch_free_bpf_trampoline(void *image, unsigned int size); int __must_check arch_protect_bpf_trampoline(void *image, unsigned int size); +int arch_bpf_trampoline_skip(void *nop, void *target); int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *func_addr); @@ -1362,19 +1426,48 @@ enum bpf_tramp_prog_type { BPF_TRAMP_FSESSION, }; +/* + * Each prog call in a trampoline image is preceded by a nop. When the prog is + * detached, the nop is patched to a jump to target, right after the call, so + * that tasks still running in the image skip the prog. + */ +struct bpf_tramp_skip { + struct bpf_prog *prog; + void *nop; + void *target; +}; + struct bpf_tramp_image { void *image; int size; struct bpf_ksym ksym; struct percpu_ref pcref; - void *ip_after_call; - void *ip_epilogue; + bool call_orig; + /* entry in tr->images, the image holds a reference on tr */ + struct bpf_trampoline *tr; + struct list_head list; + int nr_skips; + struct bpf_tramp_skip *skips; union { struct rcu_head rcu; struct work_struct work; }; }; +static inline void bpf_tramp_image_add_skip(struct bpf_tramp_image *im, struct bpf_prog *prog, + void *nop, void *target) +{ + struct bpf_tramp_skip *skip; + + /* struct_ops trampolines and arch_bpf_trampoline_size() have no image */ + if (!im || !im->skips) + return; + skip = &im->skips[im->nr_skips++]; + skip->prog = prog; + skip->nop = nop; + skip->target = target; +} + struct bpf_trampoline { /* hlist for trampoline_key_table */ struct hlist_node hlist_key; @@ -1401,6 +1494,8 @@ struct bpf_trampoline { int progs_cnt[BPF_TRAMP_MAX]; /* Executable image of trampoline */ struct bpf_tramp_image *cur_image; + /* Images not freed yet, cur_image and older ones still in use */ + struct list_head images; /* Used as temporary old image storage for multi_attach */ struct { struct bpf_tramp_image *old_image; @@ -1650,8 +1745,9 @@ static inline void bpf_trampoline_set_flags(struct bpf_trampoline *tr, u32 flags struct bpf_func_info_aux { u16 linkage; bool unreliable; - bool called : 1; - bool verified : 1; + /* Indexed by in_sleepable. */ + bool called[2]; + bool verified[2]; }; enum bpf_jit_poke_reason { @@ -1683,6 +1779,7 @@ struct bpf_ctx_arg_aux { struct btf *btf; u32 btf_id; u32 ref_id; + u32 mem_size; bool refcounted; }; @@ -1711,12 +1808,18 @@ enum { }; struct bpf_stream { - atomic_t capacity; + refcount_t refcnt; + atomic_t capacity; /* bytes reserved against the stream limit */ + atomic_t readable; /* published bytes available to readers */ struct llist_head log; /* list of in-flight stream elements in LIFO order */ struct mutex lock; /* lock protecting backlog_{head,tail} */ struct llist_node *backlog_head; /* list of in-flight stream elements in FIFO order */ struct llist_node *backlog_tail; /* tail of the list above */ + wait_queue_head_t waitq; + struct irq_work notify_work; + bool notify_used; /* notify_work was queued at least once */ + bool dead; }; struct bpf_stream_stage { @@ -1735,6 +1838,7 @@ enum bpf_sig_keyring { BPF_SIG_KEYRING_SECONDARY, BPF_SIG_KEYRING_PLATFORM, BPF_SIG_KEYRING_USER, + BPF_SIG_KEYRING_BPF, }; struct bpf_prog_aux { @@ -1768,12 +1872,12 @@ struct bpf_prog_aux { bool offload_requested; /* Program is bound and offloaded to the netdev. */ bool attach_btf_trace; /* true if attaching to BTF-enabled raw tp */ bool attach_tracing_prog; /* true if tracing another tracing program */ + bool tramp_linked; /* true if it was ever linked to a trampoline */ bool func_proto_unreliable; bool tail_call_reachable; bool xdp_has_frags; bool exception_cb; bool exception_boundary; - bool is_extended; /* true if extended by freplace program */ bool jits_use_priv_stack; bool priv_stack_requested; bool changes_pkt_data; @@ -1785,7 +1889,7 @@ struct bpf_prog_aux { u8 verdict; } sig; u64 prog_array_member_cnt; /* counts how many times as member of prog_array */ - struct mutex ext_mutex; /* mutex for is_extended and prog_array_member_cnt */ + struct mutex ext_mutex; /* mutex for freplace_link_cnt and prog_array_member_cnt */ struct bpf_arena *arena; void (*recursion_detected)(struct bpf_prog *prog); /* callback if recursion is detected */ /* BTF_KIND_FUNC_PROTO for valid attach_btf_id */ @@ -1817,6 +1921,7 @@ struct bpf_prog_aux { char name[BPF_OBJ_NAME_LEN]; u64 (*bpf_exception_cb)(u64 cookie, u64 sp, u64 bp, u64, u64); u16 stack_arg_sp_adjust; + u16 freplace_link_cnt; /* counts freplace links extending this prog */ #ifdef CONFIG_SECURITY void *security; #endif @@ -1854,7 +1959,7 @@ struct bpf_prog_aux { struct work_struct work; struct rcu_head rcu; }; - struct bpf_stream stream[2]; + struct bpf_stream *stream[2]; struct mutex st_ops_assoc_mutex; struct bpf_map __rcu *st_ops_assoc; }; @@ -2098,6 +2203,18 @@ struct btf_member; * unloaded while in use. * @name: The name of the struct bpf_struct_ops object. * @func_models: Func models + * @cgroup_atype: A value in enum cgroup_bpf_attach_type for cgroup attachment. + * 0 means the struct_ops type does not support cgroup attachment. + * If cgroup_atype is non-zero, the @reg and @unreg must be NULL + * because the attachment/detachment will be handled by the bpf core. + * @free_after_tasks_rcu_gp: Set to true if it needs the bpf core to wait for + * a tasks_rcu gp before freeing the struct_ops map + * and its progs. It is unnecessary if the @unreg + * has waited for the correct rcu gp or the @unreg + * has ensured all struct_ops prog has finished running. + * @free_after_mult_rcu_gp: Same as @free_after_tasks_rcu_gp but waiting for + * both tasks_trace_rcu and regular rcu grace period. + * It is usually needed if the struct_ops has sleepable prog. */ struct bpf_struct_ops { const struct bpf_verifier_ops *verifier_ops; @@ -2116,6 +2233,9 @@ struct bpf_struct_ops { struct module *owner; const char *name; struct btf_func_model func_models[BPF_STRUCT_OPS_MAX_NR_MEMBERS]; + int cgroup_atype; + bool free_after_tasks_rcu_gp; + bool free_after_mult_rcu_gp; }; /* Every member of a struct_ops type has an instance even a member is not @@ -2250,6 +2370,12 @@ u32 bpf_struct_ops_id(const void *kdata); int bpf_struct_ops_for_each_prog(const void *kdata, int (*cb)(struct bpf_prog *prog, void *data), void *data); +void *bpf_struct_ops_map_kdata(struct bpf_map *map); +void *bpf_struct_ops_map_cfi_stubs(struct bpf_map *map); +bool bpf_struct_ops_valid_to_reg(struct bpf_map *map); +int bpf_struct_ops_link_update_check(struct bpf_map *new_map, struct bpf_map *old_map, + struct bpf_map *expected_old_map); +int bpf_struct_ops_map_cgroup_atype(struct bpf_map *map); #ifdef CONFIG_NET /* Define it here to avoid the use of forward declaration */ @@ -2314,6 +2440,32 @@ static inline void bpf_map_struct_ops_info_fill(struct bpf_map_info *info, struc static inline void bpf_struct_ops_desc_release(struct bpf_struct_ops_desc *st_ops_desc) { } +static inline u32 bpf_struct_ops_id(void *kdata) +{ + return 0; +} +static inline void *bpf_struct_ops_map_kdata(struct bpf_map *map) +{ + return NULL; +} +static inline int bpf_struct_ops_map_cgroup_atype(struct bpf_map *map) +{ + return 0; +} +static inline void *bpf_struct_ops_map_cfi_stubs(struct bpf_map *map) +{ + return NULL; +} +static inline bool bpf_struct_ops_valid_to_reg(struct bpf_map *map) +{ + return false; +} +static inline int bpf_struct_ops_link_update_check(struct bpf_map *new_map, + struct bpf_map *old_map, + struct bpf_map *expected_old_map) +{ + return -EOPNOTSUPP; +} #endif @@ -2489,7 +2641,10 @@ u64 bpf_event_output(struct bpf_map *map, u64 flags, void *meta, u64 meta_size, * since other cpus are walking the array of pointers in parallel. */ struct bpf_prog_array_item { - struct bpf_prog *prog; + union { + struct bpf_prog *prog; + void *kdata; + }; union { struct bpf_cgroup_storage *cgroup_storage[MAX_BPF_CGROUP_STORAGE_TYPE]; u64 bpf_cookie; @@ -2531,6 +2686,7 @@ int bpf_prog_array_copy(struct bpf_prog_array *old_array, struct bpf_prog *include_prog, u64 bpf_cookie, struct bpf_prog_array **new_array); +struct bpf_prog *bpf_prog_dummy(void); struct bpf_run_ctx {}; @@ -2549,6 +2705,7 @@ struct bpf_trace_run_ctx { struct bpf_tramp_run_ctx { struct bpf_run_ctx run_ctx; u64 bpf_cookie; + int retval; struct bpf_run_ctx *saved_run_ctx; }; @@ -3819,6 +3976,8 @@ struct bpf_key { #if defined(CONFIG_KEYS) && defined(CONFIG_BPF_SYSCALL) struct bpf_key *bpf_lookup_user_key(s32 serial, u64 flags); struct bpf_key *bpf_lookup_system_key(u64 id); +struct bpf_key *bpf_lookup_keyring(void); +bool bpf_keyring_enforced(void); void bpf_key_put(struct bpf_key *bkey); int bpf_verify_pkcs7_signature(const struct bpf_dynptr *data_p, const struct bpf_dynptr *sig_p, @@ -3839,6 +3998,16 @@ static inline struct bpf_key *bpf_lookup_system_key(u64 id) return NULL; } +static inline struct bpf_key *bpf_lookup_keyring(void) +{ + return NULL; +} + +static inline bool bpf_keyring_enforced(void) +{ + return false; +} + static inline void bpf_key_put(struct bpf_key *bkey) { } @@ -4099,9 +4268,10 @@ void bpf_bprintf_cleanup(struct bpf_bprintf_data *data); int bpf_try_get_buffers(struct bpf_bprintf_buffers **bufs); void bpf_put_buffers(void); -void bpf_prog_stream_init(struct bpf_prog *prog); +int bpf_prog_stream_init(struct bpf_prog *prog, gfp_t gfp_extra_flags); void bpf_prog_stream_free(struct bpf_prog *prog); -int bpf_prog_stream_read(struct bpf_prog *prog, enum bpf_stream_id stream_id, void __user *buf, int len); +int bpf_prog_stream_read(struct bpf_prog *prog, enum bpf_stream_id stream_id, void __user *buf, u32 len); +int bpf_prog_stream_new_fd(struct bpf_prog *prog, enum bpf_stream_id stream_id, u32 flags); void bpf_stream_stage_init(struct bpf_stream_stage *ss); void bpf_stream_stage_free(struct bpf_stream_stage *ss); __printf(2, 3) @@ -4164,7 +4334,7 @@ struct bpf_prog *bpf_prog_find_from_stack(void); int bpf_insn_array_init(struct bpf_map *map, const struct bpf_prog *prog); int bpf_insn_array_ready(struct bpf_map *map); void bpf_insn_array_release(struct bpf_map *map); -void bpf_insn_array_adjust(struct bpf_map *map, u32 off, u32 len); +void bpf_insn_array_adjust(struct bpf_map *map, u32 first, u32 len); void bpf_insn_array_adjust_after_remove(struct bpf_map *map, u32 off, u32 len); #ifdef CONFIG_BPF_SYSCALL @@ -4209,7 +4379,7 @@ static inline int bpf_map_check_op_flags(struct bpf_map *map, u64 flags, u64 all return -EINVAL; cpu = flags >> 32; - if ((flags & BPF_F_CPU) && cpu >= num_possible_cpus()) + if ((flags & BPF_F_CPU) && (cpu >= nr_cpu_ids || !cpu_possible(cpu))) return -ERANGE; } diff --git a/include/linux/bpf_crypto.h b/include/linux/bpf_crypto.h deleted file mode 100644 index a41e71d4e2d9..000000000000 --- a/include/linux/bpf_crypto.h +++ /dev/null @@ -1,24 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-only */ -/* Copyright (c) 2024 Meta Platforms, Inc. and affiliates. */ -#ifndef _BPF_CRYPTO_H -#define _BPF_CRYPTO_H - -struct bpf_crypto_type { - void *(*alloc_tfm)(const char *algo); - void (*free_tfm)(void *tfm); - int (*has_algo)(const char *algo); - int (*setkey)(void *tfm, const u8 *key, unsigned int keylen); - int (*setauthsize)(void *tfm, unsigned int authsize); - int (*encrypt)(void *tfm, const u8 *src, u8 *dst, unsigned int len, u8 *iv); - int (*decrypt)(void *tfm, const u8 *src, u8 *dst, unsigned int len, u8 *iv); - unsigned int (*ivsize)(void *tfm); - unsigned int (*statesize)(void *tfm); - u32 (*get_flags)(void *tfm); - struct module *owner; - char name[14]; -}; - -int bpf_crypto_register_type(const struct bpf_crypto_type *type); -int bpf_crypto_unregister_type(const struct bpf_crypto_type *type); - -#endif /* _BPF_CRYPTO_H */ diff --git a/include/linux/bpf_lsm.h b/include/linux/bpf_lsm.h index dda272d78f01..eb4a38eef87e 100644 --- a/include/linux/bpf_lsm.h +++ b/include/linux/bpf_lsm.h @@ -21,6 +21,13 @@ extern bool bpf_lsm_initialized __ro_after_init; #include <linux/lsm_hook_defs.h> #undef LSM_HOOK +/* + * Number of xattr slots the BPF LSM reserves in the array handed to + * security_inode_init_security(), i.e. the maximum number of labels + * a policy may attach to an inode while it is being created. + */ +#define BPF_LSM_INODE_INIT_XATTRS 2 + struct bpf_storage_blob { struct bpf_local_storage __rcu *storage; }; @@ -28,7 +35,7 @@ struct bpf_storage_blob { extern struct lsm_blob_sizes bpf_lsm_blob_sizes; int bpf_lsm_verify_prog(struct bpf_verifier_log *vlog, - const struct bpf_prog *prog); + struct bpf_prog *prog); bool bpf_lsm_is_sleepable_hook(u32 btf_id); bool bpf_lsm_is_trusted(const struct bpf_prog *prog); @@ -71,7 +78,7 @@ static inline bool bpf_lsm_is_trusted(const struct bpf_prog *prog) } static inline int bpf_lsm_verify_prog(struct bpf_verifier_log *vlog, - const struct bpf_prog *prog) + struct bpf_prog *prog) { return -EOPNOTSUPP; } diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h index 5fad59fdab0d..811342e3c041 100644 --- a/include/linux/bpf_verifier.h +++ b/include/linux/bpf_verifier.h @@ -19,11 +19,12 @@ * that converting umax_value to int cannot overflow. */ #define BPF_MAX_VAR_SIZ (1 << 29) -/* size of tmp_str_buf in bpf_verifier. - * we need at least 306 bytes to fit full stack mask representation - * (in the "-8,-16,...,-512" form) +/* + * size of tmp_str_buf in bpf_verifier. + * we need at least 1399 bytes to fit full stack mask representation + * (in the "-8,-16,...,-2048" form) */ -#define TMP_STR_BUF_LEN 320 +#define TMP_STR_BUF_LEN 1408 /* Patch buffer size */ #define INSN_BUF_SIZE 32 @@ -45,18 +46,20 @@ struct bpf_reg_state { union { /* valid when type == PTR_TO_PACKET */ int range; + /* + * Valid when type == PTR_TO_STACK. Inside the callee two registers + * can be both PTR_TO_STACK like R1=fp-8 and R2=fp-8, but one of them + * points to this function stack while another to the caller's stack. + * To differentiate them 'frameno' is used which is an index in + * bpf_verifier_state->frame[] array pointing to bpf_func_state. + */ + u8 frameno; - /* valid when type == CONST_PTR_TO_MAP | PTR_TO_MAP_VALUE | - * PTR_TO_MAP_VALUE_OR_NULL + /* + * For CONST_PTR_TO_MAP, PTR_TO_MAP_KEY, PTR_TO_MAP_VALUE and + * PTR_TO_INSN. */ - struct { - struct bpf_map *map_ptr; - /* To distinguish map lookups from outer map - * the map_uid is non-zero for registers - * pointing to inner maps. - */ - u32 map_uid; - }; + struct bpf_map *map_ptr; /* for PTR_TO_BTF_ID */ struct { @@ -155,13 +158,12 @@ struct bpf_reg_state { * gets parent_id set to the dynptr's id. */ u32 parent_id; - /* Inside the callee two registers can be both PTR_TO_STACK like - * R1=fp-8 and R2=fp-8, but one of them points to this function stack - * while another to the caller's stack. To differentiate them 'frameno' - * is used which is an index in bpf_verifier_state->frame[] array - * pointing to bpf_func_state. + /* + * Distinguishes inner-map lookups and their keys and values. Zero for + * other registers. Kept outside the metadata union for ID remapping + * during state comparisons. */ - u32 frameno; + u32 map_uid; /* if (!precise && SCALAR_VALUE) min/max/tnum don't affect safety */ bool precise; }; @@ -242,55 +244,18 @@ enum bpf_stack_slot_type { #define BPF_REG_SIZE 8 /* size of eBPF register in bytes */ +/* + * Largest number of BPF_REG_SIZE stack slots a single frame can have, sized + * for the largest stack budget any JIT supports. A frame may use any part of + * its program's budget; check_max_stack_depth() enforces the budget on the + * combined depth of frames sharing the kernel stack and on each frame using + * a private stack. + */ +#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK_JIT / BPF_REG_SIZE) + /* 4-byte stack slot granularity for liveness analysis */ #define BPF_HALF_REG_SIZE 4 #define STACK_SLOT_SZ 4 -#define STACK_SLOTS (MAX_BPF_STACK / BPF_HALF_REG_SIZE) /* 128 */ - -typedef struct { - u64 v[2]; -} spis_t; - -#define SPIS_ZERO ((spis_t){}) -#define SPIS_ALL ((spis_t){{ U64_MAX, U64_MAX }}) - -static inline bool spis_is_zero(spis_t s) -{ - return s.v[0] == 0 && s.v[1] == 0; -} - -static inline bool spis_equal(spis_t a, spis_t b) -{ - return a.v[0] == b.v[0] && a.v[1] == b.v[1]; -} - -static inline spis_t spis_or(spis_t a, spis_t b) -{ - return (spis_t){{ a.v[0] | b.v[0], a.v[1] | b.v[1] }}; -} - -static inline spis_t spis_and(spis_t a, spis_t b) -{ - return (spis_t){{ a.v[0] & b.v[0], a.v[1] & b.v[1] }}; -} - -static inline spis_t spis_not(spis_t s) -{ - return (spis_t){{ ~s.v[0], ~s.v[1] }}; -} - -static inline bool spis_test_bit(spis_t s, u32 slot) -{ - return s.v[slot / 64] & BIT_ULL(slot % 64); -} - -static inline void spis_or_range(spis_t *mask, u32 lo, u32 hi) -{ - u32 w; - - for (w = lo; w <= hi && w < STACK_SLOTS; w++) - mask->v[w / 64] |= BIT_ULL(w % 64); -} #define BPF_REGMASK_ARGS ((1 << BPF_REG_1) | (1 << BPF_REG_2) | \ (1 << BPF_REG_3) | (1 << BPF_REG_4) | \ @@ -420,30 +385,36 @@ enum { INSN_F_STACK_ARG_ACCESS = BIT(3), }; +/* Registers linked to one jump condition that a history entry can record */ +#define BPF_LINKED_REGS_MAX 5 + struct bpf_jmp_history_entry { /* insn idx can't be bigger than 1 million */ u32 idx : 20; u32 frame : 4; /* stack access frame number */ - u32 spi : 6; /* stack slot index (0..63) */ - u32 : 2; - u32 prev_idx : 20; /* special INSN_F_xxx flags */ u32 flags : 4; - u32 : 8; + u32 : 4; + u32 prev_idx : 20; + u32 spi : 12; /* stack slot index */ /* - * additional registers that need precision tracking when this - * jump is backtracked, vector of five 11-bit records + * Scalar registers and spilled scalars linked to the condition of + * this jump, which need precision tracking together when the jump is + * backtracked. Each is packed as 4 bits of frame number, one bit + * telling a register from a stack slot and 11 bits of register or + * slot index, see linked_regs_pack(). */ - u64 linked_regs; + u16 linked_regs[BPF_LINKED_REGS_MAX]; + u8 linked_regs_cnt; }; static_assert(MAX_CALL_FRAMES <= (1 << 4)); -static_assert(MAX_BPF_STACK / 8 <= (1 << 6)); +static_assert(MAX_BPF_STACK_SLOTS <= (1 << 12)); /* Maximum number of bpf_reg_state objects that can exist at once */ #define MAX_STACK_ARG_SLOTS (MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS) -#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK / BPF_REG_SIZE + \ - MAX_STACK_ARG_SLOTS) * MAX_CALL_FRAMES) +#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK_SLOTS + MAX_STACK_ARG_SLOTS) * \ + MAX_CALL_FRAMES) struct bpf_verifier_state { /* call stack tracking */ struct bpf_func_state *frame[MAX_CALL_FRAMES]; @@ -529,12 +500,31 @@ struct bpf_verifier_state { u32 may_goto_depth; }; +/* Number of BPF_REG_SIZE stack slots tracked for the frame so far. */ +static inline u32 bpf_stack_nr_slots(const struct bpf_func_state *frame) +{ + return frame->allocated_stack / BPF_REG_SIZE; +} + +/* + * Stack slot @spi of @frame, covering bytes [fp - (spi + 1) * 8, fp - spi * 8). + * The caller must ensure spi < bpf_stack_nr_slots(frame), see grow_stack_state(). + */ +static inline struct bpf_stack_state *bpf_stack_slot(const struct bpf_func_state *frame, u32 spi) +{ + return &frame->stack[spi]; +} + static inline struct bpf_reg_state * bpf_get_spilled_reg(int slot, struct bpf_func_state *frame, u32 mask) { - if (slot < frame->allocated_stack / BPF_REG_SIZE && - (1 << frame->stack[slot].slot_type[BPF_REG_SIZE - 1]) & mask) - return &frame->stack[slot].spilled_ptr; + struct bpf_stack_state *ss; + + if (slot >= bpf_stack_nr_slots(frame)) + return NULL; + ss = bpf_stack_slot(frame, slot); + if ((1 << ss->slot_type[BPF_REG_SIZE - 1]) & mask) + return &ss->spilled_ptr; return NULL; } @@ -550,7 +540,7 @@ bpf_get_spilled_stack_arg(int slot, struct bpf_func_state *frame) /* Iterate over 'frame', setting 'reg' to either NULL or a spilled register. */ #define bpf_for_each_spilled_reg(iter, frame, reg, mask) \ for (iter = 0, reg = bpf_get_spilled_reg(iter, frame, mask); \ - iter < frame->allocated_stack / BPF_REG_SIZE; \ + iter < bpf_stack_nr_slots(frame); \ iter++, reg = bpf_get_spilled_reg(iter, frame, mask)) /* Iterate over 'frame', setting 'reg' to either NULL or a spilled stack arg. */ @@ -575,7 +565,7 @@ bpf_get_spilled_stack_arg(int slot, struct bpf_func_state *frame) bpf_for_each_spilled_reg(___j, __state, __reg, __mask) { \ if (!__reg) \ continue; \ - __stack = &__state->stack[___j]; \ + __stack = bpf_stack_slot(__state, ___j); \ (void)(__expr); \ } \ __stack = NULL; \ @@ -670,7 +660,6 @@ struct bpf_insn_aux_data { /* remember the offset of node field within type to rewrite */ u64 insert_off; }; - struct bpf_iarray *jt; /* jump table for gotox or bpf_tailcall call instruction */ struct btf_struct_meta *kptr_struct_meta; u64 map_key_state; /* constant (32 bit) key tracking for maps */ int ctx_field_size; /* the ctx field size for load insn, maybe 0 */ @@ -679,6 +668,7 @@ struct bpf_insn_aux_data { bool nospec_result; /* result is unsafe under speculation, nospec must follow */ bool zext_dst; /* this insn zero extends dst reg */ bool needs_zext; /* alu op needs to clear upper bits */ + bool prevent_zext; /* alu op cannot be zext (already used with 64-bit scalars) */ bool non_sleepable; /* helper/kfunc may be called from non-sleepable context */ bool is_iter_next; /* bpf_iter_<type>_next() kfunc call */ bool call_with_percpu_alloc_ptr; /* {this,per}_cpu_ptr() with prog percpu alloc */ @@ -706,6 +696,9 @@ struct bpf_insn_aux_data { */ u32 calls_callback:1; u32 indirect_target:1; /* if it is an indirect jump target */ + u32 non_stack_access:1; /* instruction can access non-stack memory */ + /* true if some jump or call instruction targets this instruction */ + u32 jump_target:1; /* * CFG strongly connected component this instruction belongs to, * zero if it is a singleton SCC. @@ -730,7 +723,11 @@ struct bpf_insn_aux_data { #define MAX_USED_MAPS 64 /* max number of maps accessed by one eBPF program */ #define MAX_USED_BTFS 64 /* max number of BTFs accessed by one BPF program */ -#define BPF_VERIFIER_TMP_LOG_SIZE 1024 +/* + * Longest line the verifier log can carry: a full stack mask of + * MAX_BPF_STACK_SLOTS slots, see TMP_STR_BUF_LEN, plus its prefix. + */ +#define BPF_VERIFIER_TMP_LOG_SIZE 2048 struct bpf_verifier_log { /* Logical start and end positions of a "log window" of the verifier log. @@ -783,6 +780,21 @@ int bpf_log_attr_finalize(struct bpf_log_attr *attr, struct bpf_verifier_log *lo #define BPF_MAX_SUBPROGS 256 +/* + * A pointer to a static subprog in the value of a frozen read-only array map: + * a 64-bit value that is the offset in bytes of the first instruction of + * the subprog in the program. + */ +struct bpf_func_ptr { + struct bpf_map *map; + u32 map_off; /* offset of the pointer in the value of the map */ + u32 orig_off; /* what the map has: the first instruction of the subprog */ + u32 xlated_off; /* the same after instructions were patched and removed */ +}; + +/* the subprog that a bpf_func_ptr pointed to was removed as dead code */ +#define BPF_FUNC_PTR_DELETED ((u32)-1) + struct bpf_subprog_arg_info { enum bpf_arg_type arg_type; union { @@ -802,7 +814,12 @@ struct bpf_subprog_info { u32 start; /* insn idx of function entry point */ u32 linfo_idx; /* The idx to the main_prog->aux->linfo */ u32 postorder_start; /* The idx to the env->cfg.insn_postorder */ - u32 exit_idx; /* Index of one of the BPF_EXIT instructions in this subprogram */ + /* + * Index of one of the BPF_EXIT instructions in this subprogram, or + * U32_MAX when it has none. + */ + u32 exit_idx; + struct bpf_iarray *jt; /* jump table shared by all gotox of this subprogram */ u16 stack_depth; /* max. stack depth used by this function */ u16 stack_extra; u32 insns_total; @@ -819,11 +836,13 @@ struct bpf_subprog_info { bool is_async_cb: 1; bool is_exception_cb: 1; bool args_cached: 1; + /* true if the return value is passed in the R0:R2 register pair */ + bool ret_reg_pair: 1; /* true if bpf_fastcall stack region is used by functions that can't be inlined */ bool keep_fastcall_stack: 1; bool changes_pkt_data: 1; bool might_sleep: 1; - u8 arg_cnt:4; + u8 arg_slot_cnt:4; enum priv_stack_mode priv_stack_mode; struct bpf_subprog_arg_info args[MAX_BPF_FUNC_ARGS]; @@ -833,8 +852,8 @@ struct bpf_subprog_info { static inline u16 bpf_in_stack_arg_cnt(const struct bpf_subprog_info *sub) { - if (sub->arg_cnt > MAX_BPF_FUNC_REG_ARGS) - return sub->arg_cnt - MAX_BPF_FUNC_REG_ARGS; + if (sub->arg_slot_cnt > MAX_BPF_FUNC_REG_ARGS) + return sub->arg_slot_cnt - MAX_BPF_FUNC_REG_ARGS; return 0; } @@ -845,7 +864,7 @@ struct backtrack_state { struct bpf_verifier_env *env; u32 frame; u32 reg_masks[MAX_CALL_FRAMES]; - u64 stack_masks[MAX_CALL_FRAMES]; + unsigned long stack_masks[MAX_CALL_FRAMES][BITS_TO_LONGS(MAX_BPF_STACK_SLOTS)]; u8 stack_arg_masks[MAX_CALL_FRAMES]; }; @@ -957,9 +976,24 @@ struct bpf_verifier_env { const struct bpf_line_info *prev_linfo; struct bpf_verifier_log log; struct bpf_diag *diag; + struct bpf_func_proto bpf_subprog_scratch; struct bpf_subprog_info subprog_info[BPF_MAX_SUBPROGS + 2]; /* max + 2 for the fake and exception subprogs */ /* subprog indices sorted in topological order: leaves first, callers last */ int subprog_topo_order[BPF_MAX_SUBPROGS + 2]; + /* + * Pointers to static subprogs found in frozen read-only maps of the + * program, see resolve_func_ptrs(). Sorted by map and map_off. + */ + struct bpf_func_ptr *func_ptrs; + u32 func_ptr_cnt; + bool has_callx; + /* + * Call graph edges created by callx instructions. A bitmap of + * subprog_cnt * subprog_cnt bits, where bit (caller * subprog_cnt + callee) + * is set when the main verification pass sees 'caller' calling 'callee' + * via callx. Allocated when the first such edge is recorded. + */ + unsigned long *callx_edges; union { struct bpf_idmap idmap_scratch; struct bpf_idset idset_scratch; @@ -975,11 +1009,13 @@ struct bpf_verifier_env { int cur_stack; /* current position in the insn_postorder vector */ int cur_postorder; + u32 gotox_edges; + bool subprog_jts_ready; } cfg; struct backtrack_state bt; struct bpf_jmp_history_entry *cur_hist_ent; - /* Per-callsite copy of parent's converged at_stack_in for cross-frame fills. */ - struct arg_track **callsite_at_stack; + /* Per-callsite snapshot of the parent's tracked spill slots for cross-frame fills. */ + struct spill_snapshot **callsite_at_stack; u32 pass_cnt; /* number of times do_check() was called */ u32 subprog_cnt; /* number of instructions analyzed by the verifier */ @@ -988,6 +1024,8 @@ struct bpf_verifier_env { u32 prev_jmps_processed, jmps_processed; /* maximum combined stack depth */ u32 max_stack_depth; + /* stack budget of the program, see bpf_prog_stack_limit() */ + u32 stack_limit; /* total verification time */ u64 verification_time; /* maximum number of verifier states kept in 'branching' instructions */ @@ -1023,7 +1061,7 @@ struct bpf_verifier_env { */ u32 scratched_regs; /* Same as scratched_regs but for stack slots */ - u64 scratched_stack_slots; + DECLARE_BITMAP(scratched_stack_slots, MAX_BPF_STACK_SLOTS); u64 prev_log_pos, prev_insn_print_pos; /* buffer used to temporary hold constants as scalar registers */ struct bpf_reg_state fake_reg[1]; @@ -1055,8 +1093,13 @@ static inline struct bpf_subprog_info *subprog_info(struct bpf_verifier_env *env return &env->subprog_info[subprog]; } +static inline bool bpf_ret_reg_pair(struct bpf_verifier_env *env, int subprog) +{ + return subprog_info(env, subprog)->ret_reg_pair; +} + struct bpf_call_summary { - u8 num_params; + u8 arg_slot_cnt; bool is_void; bool fastcall; }; @@ -1079,6 +1122,12 @@ static inline bool bpf_pseudo_kfunc_call(const struct bpf_insn *insn) insn->src_reg == BPF_PSEUDO_KFUNC_CALL; } +/* callx: indirect call of a bpf subprog whose address is in insn->dst_reg */ +static inline bool bpf_is_callx(const struct bpf_insn *insn) +{ + return insn->code == (BPF_JMP | BPF_CALL | BPF_X); +} + __printf(2, 0) void bpf_verifier_vlog(struct bpf_verifier_log *log, const char *fmt, va_list args); __printf(2, 3) void bpf_verifier_log_write(struct bpf_verifier_env *env, @@ -1142,6 +1191,16 @@ static inline void mark_jmp_point(struct bpf_verifier_env *env, int idx) env->insn_aux_data[idx].jmp_point = true; } +static inline void mark_jump_target(struct bpf_verifier_env *env, int idx) +{ + env->insn_aux_data[idx].jump_target = true; +} + +static inline bool bpf_is_jump_target(struct bpf_verifier_env *env, int insn_idx) +{ + return env->insn_aux_data[insn_idx].jump_target; +} + static inline struct bpf_func_state *cur_func(struct bpf_verifier_env *env) { struct bpf_verifier_state *cur = env->cur_state; @@ -1185,6 +1244,8 @@ static inline void bpf_trampoline_unpack_key(u64 key, u32 *obj_id, u32 *btf_id) int bpf_prepare_btf_info(struct bpf_verifier_env *env, const union bpf_attr *attr, bpfptr_t uattr); +int bpf_check_core_relo(struct bpf_verifier_env *env, + const union bpf_attr *attr, bpfptr_t uattr); int bpf_check_btf_info(struct bpf_verifier_env *env, const union bpf_attr *attr, bpfptr_t uattr); @@ -1207,7 +1268,8 @@ struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx); void bpf_free_verifier_state(struct bpf_verifier_state *state, bool free_self); void bpf_free_backedges(struct bpf_scc_visit *visit); int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state *cur, - int insn_flags, int spi, int frame, u64 linked_regs); + int insn_flags, int spi, int frame, const u16 *linked_regs, + u8 linked_regs_cnt); void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist); void bpf_mark_reg_not_init(const struct bpf_verifier_env *env, struct bpf_reg_state *reg); @@ -1224,11 +1286,36 @@ static inline int bpf_get_spi(s32 off) return (-off - 1) / BPF_REG_SIZE; } +/* + * Stack a program may use in total: combined over the frames of a call + * chain on the kernel stack, or per frame on a private stack. Any single + * frame may reach that deep. Only a JIT that lays out such frames may go + * beyond MAX_BPF_STACK, the interpreter's frame size, and only one whose + * tail calls let the target set up its own frame: without subprogram + * tail calls, do_misc_fixups() gives every program with tail calls a + * MAX_BPF_STACK frame, which a deeper frame would overrun. + */ +static inline u32 bpf_prog_stack_limit(const struct bpf_prog *prog) +{ + /* an offloaded program never runs on the host JIT, whatever it supports */ + if (prog->jit_requested && !bpf_prog_is_offloaded(prog->aux) && + bpf_jit_supports_large_stack() && bpf_jit_supports_subprog_tailcalls()) + return MAX_BPF_STACK_JIT; + return MAX_BPF_STACK; +} + +/* + * Return the function state a stack pointer register refers to. frameno + * shares storage with other pointer metadata, so return NULL for any + * other register type instead of indexing frame[] with aliased bytes. + */ static inline struct bpf_func_state *bpf_func(struct bpf_verifier_env *env, const struct bpf_reg_state *reg) { struct bpf_verifier_state *cur = env->cur_state; + if (reg->type != PTR_TO_STACK) + return NULL; return cur->frame[reg->frameno]; } @@ -1267,12 +1354,7 @@ static inline void bpf_bt_set_frame_reg(struct backtrack_state *bt, u32 frame, u static inline void bpf_bt_set_frame_slot(struct backtrack_state *bt, u32 frame, u32 slot) { - bt->stack_masks[frame] |= 1ull << slot; -} - -static inline void bpf_bt_set_frame_slot_mask(struct backtrack_state *bt, u32 frame, u64 mask) -{ - bt->stack_masks[frame] |= mask; + __set_bit(slot, bt->stack_masks[frame]); } static inline void bt_set_frame_stack_arg_slot(struct backtrack_state *bt, u32 frame, u32 slot) @@ -1287,10 +1369,17 @@ static inline bool bt_is_frame_reg_set(struct backtrack_state *bt, u32 frame, u3 static inline bool bt_is_frame_slot_set(struct backtrack_state *bt, u32 frame, u32 slot) { - return bt->stack_masks[frame] & (1ull << slot); + return test_bit(slot, bt->stack_masks[frame]); } bool bpf_map_is_rdonly(const struct bpf_map *map); +struct bpf_func_ptr *bpf_map_func_ptrs(struct bpf_verifier_env *env, + const struct bpf_map *map, u32 *cnt); +struct bpf_func_ptr *bpf_map_range_func_ptrs(struct bpf_verifier_env *env, + const struct bpf_map *map, + u64 off, u64 size, u32 *cnt); +void bpf_adjust_func_ptrs(struct bpf_verifier_env *env, u32 off, u32 len); +void bpf_adjust_func_ptrs_after_remove(struct bpf_verifier_env *env, u32 off, u32 len); int bpf_map_direct_read(struct bpf_map *map, int off, int size, u64 *val, bool is_ldsx); @@ -1369,7 +1458,9 @@ static inline bool bpf_type_has_unsafe_modifiers(u32 type) static inline bool type_is_ptr_alloc_obj(u32 type) { - return base_type(type) == PTR_TO_BTF_ID && type_flag(type) & MEM_ALLOC; + return base_type(type) == PTR_TO_BTF_ID && + type_flag(type) & MEM_ALLOC && + !(type_flag(type) & PTR_UNTRUSTED); } static inline bool type_is_non_owning_ref(u32 type) @@ -1416,7 +1507,7 @@ static inline void mark_reg_scratched(struct bpf_verifier_env *env, u32 regno) static inline void mark_stack_slot_scratched(struct bpf_verifier_env *env, u32 spi) { - env->scratched_stack_slots |= 1ULL << spi; + __set_bit(spi, env->scratched_stack_slots); } static inline bool reg_scratched(const struct bpf_verifier_env *env, u32 regno) @@ -1424,27 +1515,28 @@ static inline bool reg_scratched(const struct bpf_verifier_env *env, u32 regno) return (env->scratched_regs >> regno) & 1; } -static inline bool stack_slot_scratched(const struct bpf_verifier_env *env, u64 regno) +static inline bool stack_slot_scratched(const struct bpf_verifier_env *env, u32 spi) { - return (env->scratched_stack_slots >> regno) & 1; + return test_bit(spi, env->scratched_stack_slots); } static inline bool verifier_state_scratched(const struct bpf_verifier_env *env) { - return env->scratched_regs || env->scratched_stack_slots; + return env->scratched_regs || + !bitmap_empty(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS); } static inline void mark_verifier_state_clean(struct bpf_verifier_env *env) { env->scratched_regs = 0U; - env->scratched_stack_slots = 0ULL; + bitmap_zero(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS); } /* Used for printing the entire verifier state. */ static inline void mark_verifier_state_scratched(struct bpf_verifier_env *env) { env->scratched_regs = ~0U; - env->scratched_stack_slots = ~0ULL; + bitmap_fill(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS); } static inline bool bpf_stack_narrow_access_ok(int off, int fill_size, int spill_size) @@ -1479,14 +1571,25 @@ struct bpf_subprog_info *bpf_find_containing_subprog(struct bpf_verifier_env *en const char *bpf_subprog_name(const struct bpf_verifier_env *env, int subprog); int bpf_jmp_offset(struct bpf_insn *insn); struct bpf_iarray *bpf_insn_successors(struct bpf_verifier_env *env, u32 idx); -void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, u64 stack_mask); +void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, const unsigned long *stack_mask); bool bpf_subprog_is_global(const struct bpf_verifier_env *env, int subprog); +/* Kinds of member a by-value struct or union may be composed of. */ +enum btf_member_kind { + BTF_MEMBER_SCALAR = BIT(0), /* an int or an enum */ + BTF_MEMBER_ARENA_PTR = BIT(1), /* a pointer carrying the "arena" type tag */ +}; + +bool btf_struct_is_composed_of(struct bpf_verifier_env *env, const struct btf *btf, + const struct btf_type *t, u32 member_kinds); +u32 btf_func_arg_align(const struct btf *btf, const struct btf_type *t); + int bpf_find_subprog(struct bpf_verifier_env *env, int off); bool bpf_is_throw_kfunc(struct bpf_insn *insn); int bpf_compute_const_regs(struct bpf_verifier_env *env); int bpf_prune_dead_branches(struct bpf_verifier_env *env); int bpf_check_cfg(struct bpf_verifier_env *env); +void bpf_free_subprog_jts(struct bpf_verifier_env *env); int bpf_compute_postorder(struct bpf_verifier_env *env); int bpf_compute_scc(struct bpf_verifier_env *env); @@ -1515,12 +1618,15 @@ struct ref_obj_desc { }; /* - * A memory argument a call fills in. The verifier allows the stack to be uninitialized if - * the range is a known constant. Stack slots are marked as STACK_MISC by check_mem_access(). + * Generic MEM_UNINIT arguments, indexed by ABI slot. var_size_mask excludes + * variable-sized buffers from raw mode without losing the output annotation. + * size records constant ranges to mark initialized after checking all arguments, + * only when the caller is allowed to read uninitialized stack memory. */ struct arg_raw_mem_desc { - u8 regno; - int size; + u16 mask; + u16 var_size_mask; + int size[MAX_BPF_FUNC_ARGS]; }; /* Size of PTR_TO_MEM returned, taken from a constant allocation-size argument */ @@ -1540,6 +1646,7 @@ struct bpf_call_arg_meta { struct btf *btf; u32 func_id; const struct bpf_func_proto *fn; + const struct btf_type *func_proto; u8 release_regno; u32 ret_btf_id; u32 subprogno; @@ -1547,11 +1654,14 @@ struct bpf_call_arg_meta { struct bpf_dynptr_desc dynptr; struct ref_obj_desc ref_obj; struct ret_mem_desc ret_mem; + struct arg_raw_mem_desc arg_raw_mem; + + /* Only set by subprog */ + bool subprog_may_change_pkt; /* Only set by kfunc */ bool r0_rdonly; u32 kfunc_flags; - const struct btf_type *func_proto; const char *func_name; struct arg_constant_desc arg_constant; @@ -1560,7 +1670,7 @@ struct bpf_call_arg_meta { * verification logic * bpf_obj_drop/bpf_percpu_obj_drop * Record the local kptr type to be drop'd - * bpf_refcount_acquire (via KF_ARG_PTR_TO_REFCOUNTED_KPTR arg type) + * bpf_refcount_acquire (via ARG_PTR_TO_REFCOUNTED_KPTR arg type) * Record the local kptr type to be refcount_incr'd and use * arg_owning_ref to determine whether refcount_acquire should be * fallible @@ -1568,7 +1678,6 @@ struct bpf_call_arg_meta { struct btf *arg_btf; u32 arg_btf_id; bool arg_owning_ref; - bool arg_prog; struct { struct btf_field *field; @@ -1586,7 +1695,6 @@ struct bpf_call_arg_meta { s64 const_map_key; struct btf *ret_btf; struct btf_field *kptr_field; - struct arg_raw_mem_desc arg_raw_mem; }; int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id, @@ -1655,6 +1763,21 @@ static inline u64 bpf_map_key_immediate(const struct bpf_insn_aux_data *aux) return aux->map_key_state & ~(BPF_MAP_KEY_SEEN | BPF_MAP_KEY_POISON); } +static inline bool bpf_is_mem_insn(struct bpf_insn *insn) +{ + if (BPF_CLASS(insn->code) != BPF_ST && + BPF_CLASS(insn->code) != BPF_STX && + BPF_CLASS(insn->code) != BPF_LDX) + return false; + + if (insn->code == (BPF_ST | BPF_NOSPEC)) + return false; + + return (BPF_MODE(insn->code) == BPF_MEM || + BPF_MODE(insn->code) == BPF_MEMSX || + BPF_MODE(insn->code) == BPF_ATOMIC); +} + #define MAX_PACKET_OFF 0xffff #define CALLER_SAVED_REGS 6 @@ -1688,7 +1811,6 @@ struct bpf_kfunc_desc_tab { }; /* Functions exported from verifier.c, used by fixups.c */ -void bpf_clear_insn_aux_data(struct bpf_verifier_env *env, int start, int len); void bpf_mark_subprog_exc_cb(struct bpf_verifier_env *env, int subprog); bool bpf_allow_tail_call_in_subprogs(struct bpf_verifier_env *env); bool bpf_verifier_inlines_helper_call(struct bpf_verifier_env *env, s32 imm); diff --git a/include/linux/btf.h b/include/linux/btf.h index 89d5a5c4f117..4b63bb91550a 100644 --- a/include/linux/btf.h +++ b/include/linux/btf.h @@ -80,6 +80,7 @@ #define KF_ARENA_ARG2 (1 << 15) /* kfunc takes an arena pointer as its second argument */ #define KF_IMPLICIT_ARGS (1 << 16) /* kfunc has implicit arguments supplied by the verifier */ #define KF_SPINLOCK_SAFE (1 << 17) /* kfunc is allowed inside bpf_spin_lock-ed region */ +#define KF_PERFMON (1 << 18) /* kfunc requires CAP_PERFMON */ /* * Tag marking a kernel function as a kfunc. This is meant to minimize the @@ -235,6 +236,7 @@ struct btf_record *btf_parse_fields(const struct btf *btf, const struct btf_type u32 field_mask, u32 value_size); int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec); bool btf_type_is_void(const struct btf_type *t); +bool btf_type_is_arena_ptr(const struct btf *btf, const struct btf_type *t); s32 btf_find_by_name_kind(const struct btf *btf, const char *name, u8 kind); s32 bpf_find_btf_id(const char *name, u32 kind, struct btf **btf_p); struct btf *btf_get_module_btf(const struct module *module); @@ -260,6 +262,11 @@ const char *btf_type_str(const struct btf_type *t); i < btf_type_vlen(datasec_type); \ i++, member++) +#define for_each_loc(i, locsec_type, member) \ + for (i = 0, member = btf_type_loc_secinfo(locsec_type); \ + i < btf_type_vlen(locsec_type); \ + i++, member++) + static inline bool btf_type_is_ptr(const struct btf_type *t) { return BTF_INFO_KIND(t->info) == BTF_KIND_PTR; @@ -328,6 +335,21 @@ static inline u64 btf_enum64_value(const struct btf_enum64 *e) return ((u64)e->val_hi32 << 32) | e->val_lo32; } +static inline struct btf_loc_param *btf_loc_param(const struct btf_type *t) +{ + return (struct btf_loc_param *)(t + 1); +} + +static inline __u32 *btf_loc_proto_params(const struct btf_type *t) +{ + return (__u32 *)(t + 1); +} + +static inline struct btf_loc *btf_type_loc_secinfo(const struct btf_type *t) +{ + return (struct btf_loc *)(t + 1); +} + static inline bool btf_is_composite(const struct btf_type *t) { u16 kind = btf_kind(t); @@ -558,7 +580,7 @@ struct btf_field_desc { /* member struct size, or zero, if no members */ int m_sz; /* repeated per-member offsets */ - int m_off_cnt, m_offs[1]; + int m_off_cnt, m_offs[2]; }; struct btf_field_iter { diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index fd2c7115c054..f8782b719026 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -59,10 +59,7 @@ struct address_space; struct buffer_head { unsigned long b_state; /* buffer state bitmap (see above) */ struct buffer_head *b_this_page;/* circular list of page's buffers */ - union { - struct page *b_page; /* the page this bh is mapped to */ - struct folio *b_folio; /* the folio this bh is mapped to */ - }; + struct folio *b_folio; /* the folio this bh is mapped to */ sector_t b_blocknr; /* start block number */ size_t b_size; /* size of mapping */ @@ -172,15 +169,39 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh) static inline unsigned long bh_offset(const struct buffer_head *bh) { - return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); + return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1); } -/* If we *know* page->private refers to buffer_heads */ -#define page_buffers(page) \ - ({ \ - BUG_ON(!PagePrivate(page)); \ - ((struct buffer_head *)page_private(page)); \ - }) +/** + * kmap_local_bh - Map the data of a buffer. + * @bh: The buffer. + * + * Buffers usually live in the page cache, but a few are built over memory + * which is not. Those carry no folio and b_data is already a kernel address + * which is always mapped, so there is nothing to do for them. Pair with + * kunmap_local_bh(). + * + * Return: A pointer to the buffer's data. + */ +static inline void *kmap_local_bh(const struct buffer_head *bh) +{ + if (!bh->b_folio) + return bh->b_data; + return kmap_local_folio(bh->b_folio, bh_offset(bh)); +} + +/** + * kunmap_local_bh - Unmap the data of a buffer. + * @bh: The buffer. + * @addr: The address returned by kmap_local_bh(). + */ +static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr) +{ + if (bh->b_folio) + kunmap_local(addr); +} + +/* If we *know* folio->private refers to buffer_heads */ #define folio_buffers(folio) folio_get_private(folio) void buffer_check_dirty_writeback(struct folio *folio, @@ -197,7 +218,6 @@ void folio_set_bh(struct buffer_head *bh, struct folio *folio, unsigned long offset); struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size, gfp_t gfp); -struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size); struct buffer_head *create_empty_buffers(struct folio *folio, unsigned long blocksize, unsigned long b_state); void end_buffer_read_sync(struct buffer_head *bh, int uptodate); @@ -338,20 +358,58 @@ static inline void bforget(struct buffer_head *bh) __bforget(bh); } -static inline struct buffer_head * -sb_bread(struct super_block *sb, sector_t block) +/** + * sb_bread - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers + * to it. The memory is allocated from the movable area so that it can + * be migrated. The returned buffer head has its refcount increased. + * The caller should call brelse() when it has finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, __GFP_MOVABLE); } -static inline struct buffer_head * -sb_bread_unmovable(struct super_block *sb, sector_t block) +/** + * sb_bread_unmovable - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers to it. + * The memory is allocated from the unmovable area so that pointers into + * it remain valid after compaction runs. The returned buffer head has + * its refcount increased. The caller should call brelse() when it has + * finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread_unmovable(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, 0); } -static inline void -sb_breadahead(struct super_block *sb, sector_t block) +/** + * sb_breadahead - Start readahead. + * @sb: Superblock identifying the block device. + * @block: The block to read. + * + * Read this block. The I/O will be flagged as being readahead rather + * than immediate read, but (unlike the page cache), surrounding blocks + * will not be read. + * + * Context: May sleep in order to allocate memory. + */ +static inline +void sb_breadahead(struct super_block *sb, sector_t block) { __breadahead(sb->s_bdev, block, sb->s_blocksize); } diff --git a/include/linux/cache.h b/include/linux/cache.h index e69768f50d53..b6e857b985bc 100644 --- a/include/linux/cache.h +++ b/include/linux/cache.h @@ -89,6 +89,7 @@ */ #ifndef INTERNODE_CACHE_SHIFT #define INTERNODE_CACHE_SHIFT L1_CACHE_SHIFT +#define INTERNODE_CACHE_BYTES (1 << INTERNODE_CACHE_SHIFT) #endif #if !defined(____cacheline_internodealigned_in_smp) diff --git a/include/linux/cacheinfo.h b/include/linux/cacheinfo.h index fc879ac4cc4f..56fa646df0d1 100644 --- a/include/linux/cacheinfo.h +++ b/include/linux/cacheinfo.h @@ -82,6 +82,7 @@ struct cpu_cacheinfo { }; struct cpu_cacheinfo *get_cpu_cacheinfo(unsigned int cpu); +int get_cpu_cacheinfo_id(int cpu, int level); int early_cache_level(unsigned int cpu); int init_cache_level(unsigned int cpu); int init_of_cache_level(unsigned int cpu); @@ -137,17 +138,6 @@ static inline struct cacheinfo *get_cpu_cacheinfo_level(int cpu, int level) return NULL; } -/* - * Get the id of the cache associated with @cpu at level @level. - * cpuhp lock must be held. - */ -static inline int get_cpu_cacheinfo_id(int cpu, int level) -{ - struct cacheinfo *ci = get_cpu_cacheinfo_level(cpu, level); - - return ci ? ci->id : -1; -} - #if defined(CONFIG_ARM64) || defined(CONFIG_ARM) #define use_arch_cache_info() (true) #else diff --git a/include/linux/capability.h b/include/linux/capability.h index 37db92b3d6f8..622137f66f09 100644 --- a/include/linux/capability.h +++ b/include/linux/capability.h @@ -145,6 +145,7 @@ extern bool has_capability_noaudit(struct task_struct *t, int cap); extern bool has_ns_capability_noaudit(struct task_struct *t, struct user_namespace *ns, int cap); extern bool capable(int cap); +bool capable_noaudit(int cap); extern bool ns_capable(struct user_namespace *ns, int cap); extern bool ns_capable_noaudit(struct user_namespace *ns, int cap); extern bool ns_capable_setid(struct user_namespace *ns, int cap); @@ -167,6 +168,10 @@ static inline bool capable(int cap) { return true; } +static inline bool capable_noaudit(int cap) +{ + return true; +} static inline bool ns_capable(struct user_namespace *ns, int cap) { return true; @@ -181,9 +186,9 @@ static inline bool ns_capable_setid(struct user_namespace *ns, int cap) } #endif /* CONFIG_MULTIUSER */ bool privileged_wrt_inode_uidgid(struct user_namespace *ns, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct inode *inode); -bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap, +bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap, const struct inode *inode, int cap); extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap); extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns); @@ -210,11 +215,11 @@ static inline bool checkpoint_restore_ns_capable_noaudit(struct user_namespace * } /* audit system wants to get cap info from files as well */ -int get_vfs_caps_from_disk(struct mnt_idmap *idmap, +int get_vfs_caps_from_disk(const struct mnt_idmap *idmap, const struct dentry *dentry, struct cpu_vfs_cap_data *cpu_caps); -int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry, +int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry, const void **ivalue, size_t size); #endif /* !_LINUX_CAPABILITY_H */ diff --git a/include/linux/ccp.h b/include/linux/ccp.h index 868924dec5a1..e6c599243f11 100644 --- a/include/linux/ccp.h +++ b/include/linux/ccp.h @@ -26,7 +26,7 @@ struct ccp_cmd; /** * ccp_present - check if a CCP device is present * - * Returns zero if a CCP device is present, -ENODEV otherwise. + * Returns: zero if a CCP device is present, -ENODEV otherwise. */ int ccp_present(void); @@ -38,7 +38,7 @@ int ccp_present(void); /** * ccp_version - get the version of the CCP * - * Returns a positive version number, or zero if no CCP + * Returns: a positive version number, or zero if no CCP */ unsigned int ccp_version(void); @@ -61,9 +61,9 @@ unsigned int ccp_version(void); * will be -EINPROGRESS. Any other "err" value during callback is * the result of the operation. * - * The cmd has been successfully queued if: - * the return code is -EINPROGRESS or - * the return code is -EBUSY and CCP_CMD_MAY_BACKLOG flag is set + * Returns: The cmd has been successfully queued if: + * * the return code is -EINPROGRESS or + * * the return code is -EBUSY and CCP_CMD_MAY_BACKLOG flag is set */ int ccp_enqueue_cmd(struct ccp_cmd *cmd); @@ -89,7 +89,7 @@ static inline int ccp_enqueue_cmd(struct ccp_cmd *cmd) /***** AES engine *****/ /** - * ccp_aes_type - AES key size + * enum ccp_aes_type - AES key size * * @CCP_AES_TYPE_128: 128-bit key * @CCP_AES_TYPE_192: 192-bit key @@ -99,11 +99,12 @@ enum ccp_aes_type { CCP_AES_TYPE_128 = 0, CCP_AES_TYPE_192, CCP_AES_TYPE_256, + /* private: */ CCP_AES_TYPE__LAST, }; /** - * ccp_aes_mode - AES operation mode + * enum ccp_aes_mode - AES operation mode * * @CCP_AES_MODE_ECB: ECB mode * @CCP_AES_MODE_CBC: CBC mode @@ -111,6 +112,10 @@ enum ccp_aes_type { * @CCP_AES_MODE_CFB: CFB mode * @CCP_AES_MODE_CTR: CTR mode * @CCP_AES_MODE_CMAC: CMAC mode + * @CCP_AES_MODE_GHASH: GHASH mode + * @CCP_AES_MODE_GCTR: GCTR mode + * @CCP_AES_MODE_GCM: GCM mode + * @CCP_AES_MODE_GMAC: GMAC mode */ enum ccp_aes_mode { CCP_AES_MODE_ECB = 0, @@ -123,11 +128,12 @@ enum ccp_aes_mode { CCP_AES_MODE_GCTR, CCP_AES_MODE_GCM, CCP_AES_MODE_GMAC, + /* private: */ CCP_AES_MODE__LAST, }; /** - * ccp_aes_mode - AES operation mode + * enum ccp_aes_action - AES operation mode * * @CCP_AES_ACTION_DECRYPT: AES decrypt operation * @CCP_AES_ACTION_ENCRYPT: AES encrypt operation @@ -135,6 +141,7 @@ enum ccp_aes_mode { enum ccp_aes_action { CCP_AES_ACTION_DECRYPT = 0, CCP_AES_ACTION_ENCRYPT, + /* private: */ CCP_AES_ACTION__LAST, }; /* Overloaded field */ @@ -146,6 +153,7 @@ enum ccp_aes_action { * @type: AES operation key size * @mode: AES operation mode * @action: AES operation (decrypt/encrypt) + * @authsize: AES block request size * @key: key to be used for this AES operation * @key_len: length in bytes of key * @iv: IV to be used for this AES operation @@ -156,6 +164,7 @@ enum ccp_aes_action { * @cmac_final: indicates final operation when running in CMAC mode * @cmac_key: K1/K2 key used in final CMAC operation * @cmac_key_len: length in bytes of cmac_key + * @aad_len: length in bytes of Additional Authenticated Data * * Variables required to be set when calling ccp_enqueue_cmd(): * - type, mode, action, key, key_len, src, dst, src_len @@ -192,7 +201,7 @@ struct ccp_aes_engine { /***** XTS-AES engine *****/ /** - * ccp_xts_aes_unit_size - XTS unit size + * enum ccp_xts_aes_unit_size - XTS unit size * * @CCP_XTS_AES_UNIT_SIZE_16: Unit size of 16 bytes * @CCP_XTS_AES_UNIT_SIZE_512: Unit size of 512 bytes @@ -206,11 +215,13 @@ enum ccp_xts_aes_unit_size { CCP_XTS_AES_UNIT_SIZE_1024, CCP_XTS_AES_UNIT_SIZE_2048, CCP_XTS_AES_UNIT_SIZE_4096, + /* private: */ CCP_XTS_AES_UNIT_SIZE__LAST, }; /** * struct ccp_xts_aes_engine - CCP XTS AES operation + * @type: ccp_aes_type - AES key size * @action: AES operation (decrypt/encrypt) * @unit_size: unit size of the XTS operation * @key: key to be used for this XTS AES operation @@ -247,11 +258,13 @@ struct ccp_xts_aes_engine { /***** SHA engine *****/ /** - * ccp_sha_type - type of SHA operation + * enum ccp_sha_type - type of SHA operation * * @CCP_SHA_TYPE_1: SHA-1 operation * @CCP_SHA_TYPE_224: SHA-224 operation * @CCP_SHA_TYPE_256: SHA-256 operation + * @CCP_SHA_TYPE_384: SHA-384 operation + * @CCP_SHA_TYPE_512: SHA-512 operation */ enum ccp_sha_type { CCP_SHA_TYPE_1 = 1, @@ -259,6 +272,7 @@ enum ccp_sha_type { CCP_SHA_TYPE_256, CCP_SHA_TYPE_384, CCP_SHA_TYPE_512, + /* private: */ CCP_SHA_TYPE__LAST, }; @@ -384,7 +398,7 @@ struct ccp_rsa_engine { /***** Passthru engine *****/ /** - * ccp_passthru_bitwise - type of bitwise passthru operation + * enum ccp_passthru_bitwise - type of bitwise passthru operation * * @CCP_PASSTHRU_BITWISE_NOOP: no bitwise operation performed * @CCP_PASSTHRU_BITWISE_AND: perform bitwise AND of src with mask @@ -398,11 +412,12 @@ enum ccp_passthru_bitwise { CCP_PASSTHRU_BITWISE_OR, CCP_PASSTHRU_BITWISE_XOR, CCP_PASSTHRU_BITWISE_MASK, + /* private: */ CCP_PASSTHRU_BITWISE__LAST, }; /** - * ccp_passthru_byteswap - type of byteswap passthru operation + * enum ccp_passthru_byteswap - type of byteswap passthru operation * * @CCP_PASSTHRU_BYTESWAP_NOOP: no byte swapping performed * @CCP_PASSTHRU_BYTESWAP_32BIT: swap bytes within 32-bit words @@ -412,6 +427,7 @@ enum ccp_passthru_byteswap { CCP_PASSTHRU_BYTESWAP_NOOP = 0, CCP_PASSTHRU_BYTESWAP_32BIT, CCP_PASSTHRU_BYTESWAP_256BIT, + /* private: */ CCP_PASSTHRU_BYTESWAP__LAST, }; @@ -450,8 +466,8 @@ struct ccp_passthru_engine { * @byte_swap: byteswap operation to perform * @mask: mask to be applied to data * @mask_len: length in bytes of mask - * @src: data to be used for this operation - * @dst: data produced by this operation + * @src_dma: data to be used for this operation + * @dst_dma: data produced by this operation * @src_len: length in bytes of data used for this operation * @final: indicate final pass-through operation * @@ -478,7 +494,7 @@ struct ccp_passthru_nomap_engine { #define CCP_ECC_MAX_OUTPUTS 3 /** - * ccp_ecc_function - type of ECC function + * enum ccp_ecc_function - type of ECC function * * @CCP_ECC_FUNCTION_MMUL_384BIT: 384-bit modular multiplication * @CCP_ECC_FUNCTION_MADD_384BIT: 384-bit modular addition @@ -564,8 +580,9 @@ struct ccp_ecc_point_math { * @function: ECC function to perform * @mod: ECC modulus * @mod_len: length in bytes of modulus - * @mm: module math parameters - * @pm: point math parameters + * @u: union for math parameters + * @u.mm: module math parameters + * @u.pm: point math parameters * @ecc_result: result of the ECC operation * * Variables required to be set when calling ccp_enqueue_cmd(): @@ -589,11 +606,11 @@ struct ccp_ecc_engine { /** - * ccp_engine - CCP operation identifiers + * enum ccp_engine - CCP operation identifiers * * @CCP_ENGINE_AES: AES operation - * @CCP_ENGINE_XTS_AES: 128-bit XTS AES operation - * @CCP_ENGINE_RSVD1: unused + * @CCP_ENGINE_XTS_AES_128: 128-bit XTS AES operation + * @CCP_ENGINE_DES3: 3DES operation * @CCP_ENGINE_SHA: SHA operation * @CCP_ENGINE_RSA: RSA operation * @CCP_ENGINE_PASSTHRU: pass-through operation @@ -609,6 +626,7 @@ enum ccp_engine { CCP_ENGINE_PASSTHRU, CCP_ENGINE_ZLIB_DECOMPRESS, CCP_ENGINE_ECC, + /* private: */ CCP_ENGINE__LAST, }; diff --git a/include/linux/cgroup-defs.h b/include/linux/cgroup-defs.h index 7a631a257613..3754d697854b 100644 --- a/include/linux/cgroup-defs.h +++ b/include/linux/cgroup-defs.h @@ -527,7 +527,10 @@ struct cgroup { int nr_threaded_children; /* # of live threaded child cgroups */ - /* sequence number for cgroup.kill, serialized by css_set_lock. */ + /* + * Sequence number for cgroup.kill. Incremented with both cgroup_mutex + * and css_set_lock held. Readers hold either one. + */ unsigned int kill_seq; struct kernfs_node *kn; /* cgroup kernfs entry */ diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h index 5dfa915a630e..2afb4cb2bb4f 100644 --- a/include/linux/cgroup.h +++ b/include/linux/cgroup.h @@ -154,6 +154,9 @@ struct cgroup *cgroup_get_from_path(const char *path); struct cgroup *cgroup_get_from_fd(int fd); struct cgroup *cgroup_v1v2_get_from_fd(int fd); +/* Was this controller blocked from v1 hierarchies by cgroup_no_v1= ? */ +bool cgroup1_ssid_disabled(int ssid); + int cgroup_attach_task_all(struct task_struct *from, struct task_struct *); int cgroup_transfer_tasks(struct cgroup *to, struct cgroup *from); diff --git a/include/linux/cleanup.h b/include/linux/cleanup.h index b1b5698cbf1b..1fb8058b897d 100644 --- a/include/linux/cleanup.h +++ b/include/linux/cleanup.h @@ -261,10 +261,6 @@ const volatile void * __must_check_fn(const volatile void *val) * CLASS(name, var)(args...): * declare the variable @var as an instance of the named class * - * CLASS_INIT(name, var, init_expr): - * declare the variable @var as an instance of the named class with - * custom initialization expression. - * * Ex. * * DEFINE_CLASS(fdget, struct fd, fdput(_T), fdget(fd), int fd) @@ -302,9 +298,6 @@ static __always_inline class_##_name##_t class_##_name##ext##_constructor(_init_ class_##_name##_t var __cleanup(class_##_name##_destructor) = \ class_##_name##_constructor -#define CLASS_INIT(_name, _var, _init_expr) \ - class_##_name##_t _var __cleanup(class_##_name##_destructor) = (_init_expr) - #define __scoped_class(_name, var, _label, args...) \ for (CLASS(_name, var)(args); ; ({ goto _label; })) \ if (0) { \ diff --git a/include/linux/clk-provider.h b/include/linux/clk-provider.h index dc8e161f9c2c..a355d1ee5189 100644 --- a/include/linux/clk-provider.h +++ b/include/linux/clk-provider.h @@ -7,6 +7,7 @@ #define __LINUX_CLK_PROVIDER_H #include <dt-bindings/clock/clock.h> +#include <linux/bits.h> #include <linux/of.h> #include <linux/of_clk.h> @@ -50,6 +51,7 @@ struct clk; struct clk_hw; struct clk_core; struct dentry; +struct module; /** * struct clk_rate_request - Structure encoding the clk constraints that @@ -351,8 +353,9 @@ struct clk_init_data { * @core: pointer to the struct clk_core instance that points back to this * struct clk_hw instance * - * @clk: pointer to the per-user struct clk instance that can be used to call - * into the clk API + * @clk: pointer to the per-user struct clk instance. This will be removed at + * some point in the future, Please consider the field obsolete and do not use + * it. Use clk_hw_get_clk() to get a per-user struct clk from a struct clk_hw. * * @init: pointer to struct clk_init_data that contains the init data shared * with the common clock framework. This pointer will be set to NULL once @@ -755,7 +758,7 @@ struct clk_divider { spinlock_t *lock; }; -#define clk_div_mask(width) ((1 << (width)) - 1) +#define clk_div_mask(width) GENMASK((width) - 1, 0) #define to_clk_divider(_hw) container_of(_hw, struct clk_divider, hw) #define CLK_DIVIDER_ONE_BASED BIT(0) @@ -1474,10 +1477,24 @@ static inline struct clk_hw *__clk_get_hw(struct clk *clk) } #endif -struct clk *clk_hw_get_clk(struct clk_hw *hw, const char *con_id); +struct clk *__clk_hw_get_clk(struct clk_hw *hw, const char *con_id, + struct module *owner); struct clk *devm_clk_hw_get_clk(struct device *dev, struct clk_hw *hw, const char *con_id); +/** + * clk_hw_get_clk - get clk consumer given a clk_hw + * @hw: clk_hw associated with the clk being consumed + * @con_id: connection ID string on device + * + * Return: new clk consumer + * This is the function to be used by providers which need + * to get a consumer clk and act on the clock element + * Calls to this function must be balanced with calls to clk_put() + */ +#define clk_hw_get_clk(hw, con_id) \ + __clk_hw_get_clk((hw), (con_id), THIS_MODULE) + unsigned int clk_hw_get_num_parents(const struct clk_hw *hw); struct clk_hw *clk_hw_get_parent(const struct clk_hw *hw); struct clk_hw *clk_hw_get_parent_by_index(const struct clk_hw *hw, diff --git a/include/linux/clk/tegra.h b/include/linux/clk/tegra.h index 3650e926e93f..f1034e14c1a4 100644 --- a/include/linux/clk/tegra.h +++ b/include/linux/clk/tegra.h @@ -170,7 +170,7 @@ struct tegra210_clk_emc_provider { struct module *owner; struct device *dev; - struct tegra210_clk_emc_config *configs; + struct tegra210_clk_emc_config *configs __counted_by_ptr(num_configs); unsigned int num_configs; int (*set_rate)(struct device *dev, diff --git a/include/linux/clk/ti.h b/include/linux/clk/ti.h index 54a3fa370004..61c4abf19995 100644 --- a/include/linux/clk/ti.h +++ b/include/linux/clk/ti.h @@ -51,7 +51,7 @@ struct clk_omap_reg { * @autoidle_mask: mask of the DPLL autoidle mode bitfield in @autoidle_reg * @freqsel_mask: mask of the DPLL jitter correction bitfield in @control_reg * @dcc_mask: mask of the DPLL DCC correction bitfield @mult_div1_reg - * @dcc_rate: rate atleast which DCC @dcc_mask must be set + * @dcc_rate: rate at least which DCC @dcc_mask must be set * @idlest_mask: mask of the DPLL idle status bitfield in @idlest_reg * @lpmode_mask: mask of the DPLL low-power mode bitfield in @control_reg * @m4xen_mask: mask of the DPLL M4X multiplier bitfield in @control_reg diff --git a/include/linux/compaction.h b/include/linux/compaction.h index 66a2f70e9e01..691c09f0a697 100644 --- a/include/linux/compaction.h +++ b/include/linux/compaction.h @@ -81,10 +81,6 @@ static inline unsigned long compact_gap(unsigned int order) return min(2UL << order, COMPACT_CLUSTER_MAX); } -static inline int current_is_kcompactd(void) -{ - return current->flags & PF_KCOMPACTD; -} #ifdef CONFIG_COMPACTION @@ -103,7 +99,7 @@ extern void compaction_defer_reset(struct zone *zone, int order, bool compaction_zonelist_suitable(struct alloc_context *ac, int order, int alloc_flags, gfp_t gfp_mask); - +bool current_is_kcompactd(void); extern void __meminit kcompactd_run(int nid); extern void __meminit kcompactd_stop(int nid); extern void wakeup_kcompactd(pg_data_t *pgdat, int order, int highest_zoneidx); @@ -120,6 +116,11 @@ static inline bool compaction_suitable(struct zone *zone, int order, return false; } +static inline bool current_is_kcompactd(void) +{ + return false; +} + static inline void kcompactd_run(int nid) { } diff --git a/include/linux/compiler.h b/include/linux/compiler.h index cb2f6050bdf7..0ea7150dba7b 100644 --- a/include/linux/compiler.h +++ b/include/linux/compiler.h @@ -239,6 +239,23 @@ void ftrace_likely_update(struct ftrace_likely_data *f, int val, # define TYPEOF_UNQUAL(exp) __typeof__(exp) #endif +/* + * TYPEOF_NO_ADDRESS_SPACE() - typeof() without the address space qualifiers + * + * No operator strips only the address space qualifiers: typeof() keeps every + * qualifier and TYPEOF_UNQUAL() drops every qualifier, const and volatile + * included. + * + * Approximate one by dropping the qualifiers for sparse only, as it is the + * only one that knows about address spaces. The compiler keeps seeing the fully + * qualified type, so a missing const or volatile will still throw a warning. + */ +#ifdef __CHECKER__ +# define TYPEOF_NO_ADDRESS_SPACE(exp) TYPEOF_UNQUAL(exp) +#else +# define TYPEOF_NO_ADDRESS_SPACE(exp) __typeof__(exp) +#endif + #endif /* __KERNEL__ */ #if defined(CONFIG_CFI) && !defined(__DISABLE_EXPORTS) && !defined(BUILD_VDSO) @@ -275,6 +292,11 @@ static inline void *offset_to_ptr(const int *off) #define __ADDRESSABLE(sym) \ ___ADDRESSABLE(sym, __section(".discard.addressable")) +/* Enforce static storage duration. */ +#define ASSERT_STATIC_STORAGE(name) \ + static typeof(name) * const __always_unused \ + name##_storage_check = &(name) + /* * This returns a constant expression while determining if an argument is * a constant expression, most importantly without evaluating the argument. diff --git a/include/linux/configfs.h b/include/linux/configfs.h index ef65c75beeaa..5bead9173ec1 100644 --- a/include/linux/configfs.h +++ b/include/linux/configfs.h @@ -66,8 +66,11 @@ struct config_item_type { struct module *ct_owner; const struct configfs_item_operations *ct_item_ops; const struct configfs_group_operations *ct_group_ops; - struct configfs_attribute **ct_attrs; - struct configfs_bin_attribute **ct_bin_attrs; + union { + struct configfs_attribute **ct_attrs; + const struct configfs_attribute *const *ct_attrs_const; + }; + const struct configfs_bin_attribute *const *ct_bin_attrs; }; /** @@ -160,41 +163,41 @@ struct configfs_bin_attribute { ssize_t (*write)(struct config_item *, const void *, size_t); }; -#define CONFIGFS_BIN_ATTR(_pfx, _name, _priv, _maxsz) \ -static struct configfs_bin_attribute _pfx##attr_##_name = { \ - .cb_attr = { \ - .ca_name = __stringify(_name), \ - .ca_mode = S_IRUGO | S_IWUSR, \ - .ca_owner = THIS_MODULE, \ - }, \ - .cb_private = _priv, \ - .cb_max_size = _maxsz, \ - .read = _pfx##_name##_read, \ - .write = _pfx##_name##_write, \ +#define CONFIGFS_BIN_ATTR(_pfx, _name, _priv, _maxsz) \ +static const struct configfs_bin_attribute _pfx##attr_##_name = { \ + .cb_attr = { \ + .ca_name = __stringify(_name), \ + .ca_mode = S_IRUGO | S_IWUSR, \ + .ca_owner = THIS_MODULE, \ + }, \ + .cb_private = _priv, \ + .cb_max_size = _maxsz, \ + .read = _pfx##_name##_read, \ + .write = _pfx##_name##_write, \ } -#define CONFIGFS_BIN_ATTR_RO(_pfx, _name, _priv, _maxsz) \ -static struct configfs_bin_attribute _pfx##attr_##_name = { \ - .cb_attr = { \ - .ca_name = __stringify(_name), \ - .ca_mode = S_IRUGO, \ - .ca_owner = THIS_MODULE, \ - }, \ - .cb_private = _priv, \ - .cb_max_size = _maxsz, \ - .read = _pfx##_name##_read, \ +#define CONFIGFS_BIN_ATTR_RO(_pfx, _name, _priv, _maxsz) \ +static const struct configfs_bin_attribute _pfx##attr_##_name = { \ + .cb_attr = { \ + .ca_name = __stringify(_name), \ + .ca_mode = S_IRUGO, \ + .ca_owner = THIS_MODULE, \ + }, \ + .cb_private = _priv, \ + .cb_max_size = _maxsz, \ + .read = _pfx##_name##_read, \ } -#define CONFIGFS_BIN_ATTR_WO(_pfx, _name, _priv, _maxsz) \ -static struct configfs_bin_attribute _pfx##attr_##_name = { \ - .cb_attr = { \ - .ca_name = __stringify(_name), \ - .ca_mode = S_IWUSR, \ - .ca_owner = THIS_MODULE, \ - }, \ - .cb_private = _priv, \ - .cb_max_size = _maxsz, \ - .write = _pfx##_name##_write, \ +#define CONFIGFS_BIN_ATTR_WO(_pfx, _name, _priv, _maxsz) \ +static const struct configfs_bin_attribute _pfx##attr_##_name = { \ + .cb_attr = { \ + .ca_name = __stringify(_name), \ + .ca_mode = S_IWUSR, \ + .ca_owner = THIS_MODULE, \ + }, \ + .cb_private = _priv, \ + .cb_max_size = _maxsz, \ + .write = _pfx##_name##_write, \ } /* @@ -220,8 +223,8 @@ struct configfs_group_operations { struct config_group *(*make_group)(struct config_group *group, const char *name); void (*disconnect_notify)(struct config_group *group, struct config_item *item); void (*drop_item)(struct config_group *group, struct config_item *item); - bool (*is_visible)(struct config_item *item, struct configfs_attribute *attr, int n); - bool (*is_bin_visible)(struct config_item *item, struct configfs_bin_attribute *attr, + bool (*is_visible)(struct config_item *item, const struct configfs_attribute *attr, int n); + bool (*is_bin_visible)(struct config_item *item, const struct configfs_bin_attribute *attr, int n); }; diff --git a/include/linux/console.h b/include/linux/console.h index d624200cfc17..502d1abe3f50 100644 --- a/include/linux/console.h +++ b/include/linux/console.h @@ -173,7 +173,7 @@ static inline void con_debug_leave(void) { } * @CON_BRL: Indicates a braille device which is exempt from * receiving the printk spam for obvious reasons. * @CON_EXTENDED: The console supports the extended output format of - * /dev/kmesg which requires a larger output buffer. + * /dev/kmsg which requires a larger output buffer. * @CON_SUSPENDED: Indicates if a console is suspended. If true, the * printing callbacks must not be called. * @CON_NBCON: Console can operate outside of the legacy style console_lock diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 7b38ee2e7913..74af57b9406b 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -6,9 +6,20 @@ #include <linux/mm.h> #include <linux/fs.h> #include <linux/sched/coredump.h> +#include <uapi/linux/coredump.h> #include <asm/siginfo.h> #ifdef CONFIG_COREDUMP +/** + * enum coredump_state - what happened while the coredump was written + * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump + * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it + */ +enum coredump_state { + COREDUMP_STATE_STARTED = (1U << 0), + COREDUMP_STATE_TRUNCATED = (1U << 1), +}; + struct core_vma_metadata { unsigned long start, end; vm_flags_t flags; @@ -21,12 +32,20 @@ struct coredump_params { const kernel_siginfo_t *siginfo; struct file *file; unsigned long limit; - /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */ - unsigned long mm_flags; + /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */ + u64 memory_types; /* Snapshot of dumpable at dump start. */ enum task_dumpable dumpable; int cpu; + /* COREDUMP_* options negotiated with the coredump server. */ + u64 mask; + /* COREDUMP_STATE_* raised while the coredump is written. */ + enum coredump_state state; + /* Record header scratch, NULL unless the coredump is a record stream. */ + struct coredump_record_header *record_hdr; + /* Bytes handed to the file, record headers included. */ loff_t written; + /* Offset in the coredump, record headers excluded. */ loff_t pos; loff_t to_skip; int vma_count; @@ -41,13 +60,13 @@ extern unsigned int core_file_note_size_limit; * These are the only things you should do on a core-file: use only these * functions to write out all the necessary info. */ -extern void dump_skip_to(struct coredump_params *cprm, unsigned long to); -extern void dump_skip(struct coredump_params *cprm, size_t nr); -extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr); -extern int dump_align(struct coredump_params *cprm, int align); -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len); -extern void vfs_coredump(const kernel_siginfo_t *siginfo); +void dump_skip_to(struct coredump_params *cprm, unsigned long to); +void dump_skip(struct coredump_params *cprm, size_t nr); +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr); +bool dump_align(struct coredump_params *cprm, int align); +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len); +void vfs_coredump(const kernel_siginfo_t *siginfo); /* * Logging for the coredump code, ratelimited. diff --git a/include/linux/cpufreq.h b/include/linux/cpufreq.h index 35ce665edfd8..d3d0d9d02aa4 100644 --- a/include/linux/cpufreq.h +++ b/include/linux/cpufreq.h @@ -420,6 +420,9 @@ struct cpufreq_driver { /* Will be called after the driver is fully initialized */ void (*ready)(struct cpufreq_policy *policy); + /* Return the capacity reference frequency for policy. */ + unsigned int (*scale_freq_ref)(struct cpufreq_policy *policy); + struct freq_attr **attr; /* platform specific boost support code */ diff --git a/include/linux/cpumask.h b/include/linux/cpumask.h index 4c8bb6953107..bf89bb3f30f6 100644 --- a/include/linux/cpumask.h +++ b/include/linux/cpumask.h @@ -121,12 +121,20 @@ extern struct cpumask __cpu_enabled_mask; extern struct cpumask __cpu_present_mask; extern struct cpumask __cpu_active_mask; extern struct cpumask __cpu_dying_mask; + +#ifdef CONFIG_PREFERRED_CPU +extern struct cpumask __cpu_preferred_mask; +#else +#define __cpu_preferred_mask __cpu_active_mask +#endif + #define cpu_possible_mask ((const struct cpumask *)&__cpu_possible_mask) #define cpu_online_mask ((const struct cpumask *)&__cpu_online_mask) #define cpu_enabled_mask ((const struct cpumask *)&__cpu_enabled_mask) #define cpu_present_mask ((const struct cpumask *)&__cpu_present_mask) #define cpu_active_mask ((const struct cpumask *)&__cpu_active_mask) #define cpu_dying_mask ((const struct cpumask *)&__cpu_dying_mask) +#define cpu_preferred_mask ((const struct cpumask *)&__cpu_preferred_mask) extern atomic_t __num_online_cpus; extern unsigned int __num_possible_cpus; @@ -825,6 +833,24 @@ bool cpumask_intersects(const struct cpumask *src1p, const struct cpumask *src2p } /** + * cpumask_intersects_and - (*src1p & *src2p & *src3p) != 0 + * @src1p: the first input + * @src2p: the second input + * @src3p: the third input + * + * Return: true if AND of the three cpumasks is non-empty, + * otherwise false + */ +static __always_inline +bool cpumask_intersects_and(const struct cpumask *src1p, + const struct cpumask *src2p, + const struct cpumask *src3p) +{ + return bitmap_intersects_and(cpumask_bits(src1p), cpumask_bits(src2p), + cpumask_bits(src3p), small_cpumask_bits); +} + +/** * cpumask_subset - (*src1p & ~*src2p) == 0 * @src1p: the first input * @src2p: the second input @@ -1163,6 +1189,12 @@ void init_cpu_possible(const struct cpumask *src); #define set_cpu_active(cpu, active) assign_cpu((cpu), &__cpu_active_mask, (active)) #define set_cpu_dying(cpu, dying) assign_cpu((cpu), &__cpu_dying_mask, (dying)) +#ifdef CONFIG_PREFERRED_CPU +#define set_cpu_preferred(cpu, preferred) assign_cpu((cpu), &__cpu_preferred_mask, (preferred)) +#else +#define set_cpu_preferred(cpu, preferred) do { } while (0) +#endif + void set_cpu_online(unsigned int cpu, bool online); void set_cpu_possible(unsigned int cpu, bool possible); @@ -1257,6 +1289,11 @@ static __always_inline bool cpu_dying(unsigned int cpu) return cpumask_test_cpu(cpu, cpu_dying_mask); } +static __always_inline bool cpu_preferred(unsigned int cpu) +{ + return cpumask_test_cpu(cpu, cpu_preferred_mask); +} + #else #define num_online_cpus() 1U @@ -1295,6 +1332,11 @@ static __always_inline bool cpu_dying(unsigned int cpu) return false; } +static __always_inline bool cpu_preferred(unsigned int cpu) +{ + return cpu == 0; +} + #endif /* NR_CPUS > 1 */ #define cpu_is_offline(cpu) unlikely(!cpu_online(cpu)) diff --git a/include/linux/cred.h b/include/linux/cred.h index 6ef1750c93e2..49c26af37349 100644 --- a/include/linux/cred.h +++ b/include/linux/cred.h @@ -180,12 +180,18 @@ static inline bool cap_ambient_invariant_ok(const struct cred *cred) static inline const struct cred *override_creds(const struct cred *override_cred) { - return rcu_replace_pointer(current->cred, override_cred, 1); + const struct cred *old = current->cred; + + current->cred = override_cred; + return old; } static inline const struct cred *revert_creds(const struct cred *revert_cred) { - return rcu_replace_pointer(current->cred, revert_cred, 1); + const struct cred *override_cred = current->cred; + + current->cred = revert_cred; + return override_cred; } DEFINE_CLASS(override_creds, @@ -293,11 +299,10 @@ DEFINE_FREE(put_cred, struct cred *, if (!IS_ERR_OR_NULL(_T)) put_cred(_T)) /** * current_cred - Access the current task's subjective credentials * - * Access the subjective credentials of the current task. RCU-safe, - * since nobody else can modify it. + * Access the subjective credentials of the current task. + * Nobody else can modify it. */ -#define current_cred() \ - rcu_dereference_protected(current->cred, 1) +#define current_cred() (current->cred) /** * current_real_cred - Access the current task's objective credentials diff --git a/include/linux/damon.h b/include/linux/damon.h index 0c8b7ddef9ab..63050eb2206a 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -117,7 +117,6 @@ struct damon_target { * @DAMOS_MIGRATE_HOT: Migrate the regions prioritizing warmer regions. * @DAMOS_MIGRATE_COLD: Migrate the regions prioritizing colder regions. * @DAMOS_STAT: Do nothing but count the stat. - * @NR_DAMOS_ACTIONS: Total number of DAMOS actions * * The support of each action is up to running &struct damon_operations. * Refer to 'Operation Action' section of Documentation/mm/damon/design.rst for @@ -137,7 +136,6 @@ enum damos_action { DAMOS_MIGRATE_HOT, DAMOS_MIGRATE_COLD, DAMOS_STAT, /* Do nothing but only record the stat */ - NR_DAMOS_ACTIONS, }; /** @@ -153,9 +151,7 @@ enum damos_action { * @DAMOS_QUOTA_INACTIVE_MEM_BP: Inactive to total LRU memory ratio. * @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a * node in basis points (0-10000). - * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics. - * - * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported. + * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio. */ enum damos_quota_goal_metric { DAMOS_QUOTA_USER_INPUT, @@ -167,12 +163,13 @@ enum damos_quota_goal_metric { DAMOS_QUOTA_ACTIVE_MEM_BP, DAMOS_QUOTA_INACTIVE_MEM_BP, DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, - NR_DAMOS_QUOTA_GOAL_METRICS, + DAMOS_QUOTA_HUGEPAGE_MEM_BP, }; /** * struct damos_quota_goal - DAMOS scheme quota auto-tuning goal. * @metric: Metric to be used for representing the goal. + * @complement: Use the complement of the metric. * @target_value: Target value of @metric to achieve with the tuning. * @current_value: Current value of @metric. * @nid: Node id. @@ -195,6 +192,7 @@ enum damos_quota_goal_metric { */ struct damos_quota_goal { enum damos_quota_goal_metric metric; + bool complement; unsigned long target_value; unsigned long current_value; /* metric-dependent fields */ @@ -311,12 +309,10 @@ struct damos_quota { * * @DAMOS_WMARK_NONE: Ignore the watermarks of the given scheme. * @DAMOS_WMARK_FREE_MEM_RATE: Free memory rate of the system in [0,1000]. - * @NR_DAMOS_WMARK_METRICS: Total number of DAMOS watermark metrics */ enum damos_wmark_metric { DAMOS_WMARK_NONE, DAMOS_WMARK_FREE_MEM_RATE, - NR_DAMOS_WMARK_METRICS, }; /** @@ -357,7 +353,8 @@ struct damos_watermarks { * Total bytes that passed ops layer-handled DAMOS filters. * @qt_exceeds: Total number of times the quota of the scheme has exceeded. * @nr_snapshots: - * Total number of DAMON snapshots that the scheme has tried. + * Total number of DAMON snapshots that the scheme is completely + * tried to be applied. * * "Tried an action to a region" in this context means the DAMOS core logic * determined the region as eligible to apply the action. The access pattern @@ -396,14 +393,14 @@ struct damos_stat { * @DAMOS_FILTER_TYPE_UNMAPPED: Unmapped pages. * @DAMOS_FILTER_TYPE_ADDR: Address range. * @DAMOS_FILTER_TYPE_TARGET: Data Access Monitoring target. - * @NR_DAMOS_FILTER_TYPES: Number of filter types. + * @DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: probe_hits weighted sum range. * - * All types except &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET - * are handled by the underlying &struct damon_operations as a part of scheme - * action trying, and therefore accounted as 'tried'. In contrast, - * &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET filters are handled - * by the core layer before trying of the action, and therefore not accounted - * as 'tried'. + * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and + * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct + * damon_operations as a part of scheme action trying, and therefore accounted + * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET + * and &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM filters are handled by the core layer + * before trying of the action, and therefore not accounted as 'tried'. * * Support for the operations-handled filters depends on the running * &struct damon_operations. @@ -417,7 +414,7 @@ enum damos_filter_type { DAMOS_FILTER_TYPE_UNMAPPED, DAMOS_FILTER_TYPE_ADDR, DAMOS_FILTER_TYPE_TARGET, - NR_DAMOS_FILTER_TYPES, + DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, }; /** @@ -431,6 +428,8 @@ enum damos_filter_type { * &damon_ctx->adaptive_targets if @type is * DAMOS_FILTER_TYPE_TARGET. * @sz_range: Size range if @type is DAMOS_FILTER_TYPE_HUGEPAGE_SIZE. + * @range_min: Minimum value of range arguments. + * @range_max: Maximum value of range arguments. * * Before applying the &damos->action to a memory region, DAMOS checks if each * byte of the region matches to this given condition and avoid applying the @@ -447,6 +446,10 @@ struct damos_filter { struct damon_addr_range addr_range; int target_idx; struct damon_size_range sz_range; + struct { + unsigned long range_min; + unsigned long range_max; + }; }; /* private: */ /* List head for siblings. */ @@ -548,7 +551,7 @@ struct damos_migrate_dests { * * After applying the &action to each region, &stat is updated. * - * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots be same to or + * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots equals or is * greater than it, the scheme is deactivated. */ struct damos { @@ -627,6 +630,7 @@ enum damon_ops_id { * @update: Update operations-related data structures. * @prepare_access_checks: Prepare next access check of target regions. * @check_accesses: Check the accesses to target regions. + * @prep_probes: Prepare applying probes for each region. * @apply_probes: Apply probes for each region. * @get_scheme_score: Get the score of a region for a scheme. * @apply_scheme: Apply a DAMON-based operation scheme. @@ -654,6 +658,9 @@ enum damon_ops_id { * last preparation and update the number of observed accesses of each region. * It should also return max number of observed accesses that made as a result * of its update. The value will be used for regions adjustment threshold. + * @prep_probes should execute required &struct damon_prep for next &struct + * damon_probe applications to each region. It should also set + * &damon_region->sampling_addr of each region if ``set_samples`` is true. * @apply_probes should apply the data attribute probes to each region and * accordingly update the probe hits counter of the region. It should also * set &damon_region->sampling_addr of each region if ``set_samples`` is true. @@ -676,6 +683,7 @@ struct damon_operations { void (*update)(struct damon_ctx *context); void (*prepare_access_checks)(struct damon_ctx *context); unsigned int (*check_accesses)(struct damon_ctx *context); + void (*prep_probes)(struct damon_ctx *context, bool set_samples); unsigned int (*apply_probes)(struct damon_ctx *context, bool set_samples, bool return_max_wsum); int (*get_scheme_score)(struct damon_ctx *context, @@ -740,14 +748,41 @@ struct damon_intervals_goal { }; /** + * enum damon_prep_action - DAMON probing preparation action. + * + * @DAMON_PREP_SET_PGIDLE: Set the probing memory as idle page. + */ +enum damon_prep_action { + DAMON_PREP_SET_PGIDLE, +}; + +/** + * struct damon_prep - DAMON probing preparation request. + * + * @action: Action to do to the probing memory for the preparation. + */ +struct damon_prep { + enum damon_prep_action action; +/* private: */ + /* siblings list. */ + struct list_head list; +}; + +/** * enum damon_filter_type - Type of &struct damon_filter * - * @DAMON_FILTER_TYPE_ANON: Anonymous pages. - * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_ANON: Anonymous pages. + * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. + * @DAMON_FILTER_TYPE_PGIDLE_SET: Pgidle is set. + * @DAMON_FILTER_TYPE_HUGEPAGE_SIZE: Page is part of a hugepage. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, + DAMON_FILTER_TYPE_PGIDLE_UNSET, + DAMON_FILTER_TYPE_PGIDLE_SET, + DAMON_FILTER_TYPE_HUGEPAGE_SIZE, }; /** @@ -757,6 +792,8 @@ enum damon_filter_type { * @matching: Whether this filter is for the type-matching ones. * @allow: Whether the @type-@matching ones should pass this filter. * @memcg_id: Memcg id of the question if @type is DAMON_FILTER_MEMCG. + * @range_min: Minimum value of range arguments. + * @range_max: Maximum value of range arguments. */ struct damon_filter { enum damon_filter_type type; @@ -764,6 +801,10 @@ struct damon_filter { bool allow; union { u64 memcg_id; + struct { + unsigned long range_min; + unsigned long range_max; + }; }; /* private: */ /* Siblings list. */ @@ -778,6 +819,8 @@ struct damon_filter { struct damon_probe { unsigned int weight; /* private: */ + /* Preparation actions to apply to each probing memory. */ + struct list_head preps; /* Filters for assessing if a given region is for this probe. */ struct list_head filters; /* Siblings list. */ @@ -787,7 +830,8 @@ struct damon_probe { /** * struct damon_attrs - Monitoring attributes for accuracy/overhead control. * - * @sample_interval: The time between access samplings. + * @sample_interval: The time between access samplings. Zero is + * accepted. * @aggr_interval: The time between monitor results aggregations. * @ops_update_interval: The time between monitoring operations updates. * @intervals_goal: Intervals auto-tuning goal. @@ -957,6 +1001,12 @@ static inline unsigned long damon_sz_region(struct damon_region *r) return r->ar.end - r->ar.start; } +#define damon_for_each_prep(p, probe) \ + list_for_each_entry(p, &(probe)->preps, list) + +#define damon_for_each_prep_safe(p, next, probe) \ + list_for_each_entry_safe(p, next, &(probe)->preps, list) + #define damon_for_each_filter(f, p) \ list_for_each_entry(f, &(p)->filters, list) @@ -1010,6 +1060,9 @@ static inline unsigned long damon_sz_region(struct damon_region *r) #ifdef CONFIG_DAMON +struct damon_prep *damon_new_prep(enum damon_prep_action action); +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep); + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow); void damon_add_filter(struct damon_probe *probe, struct damon_filter *f); @@ -1023,7 +1076,7 @@ unsigned int damon_nr_accesses_mvsum(struct damon_region *r, struct damon_ctx *ctx); unsigned char damon_probe_hits_mvsum(int probe_idx, struct damon_region *r, struct damon_ctx *ctx); -unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, +unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv, struct damon_ctx *ctx); int damon_set_regions(struct damon_target *t, struct damon_addr_range *ranges, @@ -1037,7 +1090,7 @@ bool damos_filter_for_ops(enum damos_filter_type type); void damos_destroy_filter(struct damos_filter *f); struct damos_quota_goal *damos_new_quota_goal( - enum damos_quota_goal_metric metric, + enum damos_quota_goal_metric metric, bool complement, unsigned long target_value); void damos_add_quota_goal(struct damos_quota *q, struct damos_quota_goal *g); void damos_destroy_quota_goal(struct damos_quota_goal *goal); @@ -1054,7 +1107,7 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); -bool damon_targets_empty(struct damon_ctx *ctx); +int damon_set_target_pid(struct damon_target *t, int pid); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); @@ -1065,7 +1118,6 @@ int damon_set_attrs(struct damon_ctx *ctx, struct damon_attrs *attrs); void damon_set_schemes(struct damon_ctx *ctx, struct damos **schemes, ssize_t nr_schemes); int damon_commit_ctx(struct damon_ctx *old_ctx, struct damon_ctx *new_ctx); -int damon_nr_running_ctxs(void); bool damon_is_registered_ops(enum damon_ops_id id); int damon_register_ops(struct damon_operations *ops); int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id); diff --git a/include/linux/dax.h b/include/linux/dax.h index fe6c3ded1b50..f2d47975d905 100644 --- a/include/linux/dax.h +++ b/include/linux/dax.h @@ -155,8 +155,6 @@ int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc); int dax_folio_reset_order(struct folio *folio); -struct page *dax_layout_busy_page(struct address_space *mapping); -struct page *dax_layout_busy_page_range(struct address_space *mapping, loff_t start, loff_t end); dax_entry_t dax_lock_folio(struct folio *folio); void dax_unlock_folio(struct folio *folio, dax_entry_t cookie); dax_entry_t dax_lock_mapping_entry(struct address_space *mapping, @@ -173,16 +171,6 @@ static inline int fs_dax_get(struct dax_device *dax_dev, void *holder, { return -EOPNOTSUPP; } -static inline struct page *dax_layout_busy_page(struct address_space *mapping) -{ - return NULL; -} - -static inline struct page *dax_layout_busy_page_range(struct address_space *mapping, pgoff_t start, pgoff_t nr_pages) -{ - return NULL; -} - static inline int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc) { diff --git a/include/linux/dcache.h b/include/linux/dcache.h index 4b1ff99608e0..adf239f8205f 100644 --- a/include/linux/dcache.h +++ b/include/linux/dcache.h @@ -116,6 +116,8 @@ struct dentry { * possible! */ + /* lockdep tracking of DCACHE_PAR_LOOKUP locks */ + struct lockdep_map lookup_map; struct list_head d_lru; /* LRU list */ struct hlist_node d_sib; /* child of parent list */ struct hlist_head d_children; /* our children */ @@ -236,7 +238,9 @@ enum dentry_flags { DCACHE_PAR_LOOKUP = BIT(24), /* being looked up (with parent locked shared) */ DCACHE_DENTRY_CURSOR = BIT(25), DCACHE_NORCU = BIT(26), /* No RCU delay for freeing */ - DCACHE_PERSISTENT = BIT(27) + DCACHE_PERSISTENT = BIT(27), +/* 28, 29, 30 free */ + DCACHE_PRIVATE = BIT(31) /* fs-specific flag */ }; #define DCACHE_MANAGED_DENTRY \ @@ -257,7 +261,9 @@ extern void d_delete(struct dentry *); extern struct dentry * d_alloc(struct dentry *, const struct qstr *); extern struct dentry * d_alloc_anon(struct super_block *); extern struct dentry * d_alloc_parallel(struct dentry *, const struct qstr *); +extern struct dentry * d_alloc_trylock(struct dentry *, struct qstr *); extern struct dentry * d_splice_alias(struct inode *, struct dentry *); +struct dentry *d_duplicate(struct dentry *dentry); /* weird procfs mess; *NOT* exported */ extern struct dentry * d_splice_alias_ops(struct inode *, struct dentry *, const struct dentry_operations *); @@ -553,6 +559,36 @@ static inline int simple_positive(const struct dentry *dentry) unsigned long vfs_pressure_ratio(unsigned long val); /** + * d_lookup_release - release ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry is to be passed to another thread which + * will drop the in-lookup lock, then d_lookup_release() must be called + * to tell lockdep that this thread no lock holds the lock. The + * thread that receives the lock must call d_lookup_acquire() to + * acquire the lock. + */ +static inline void d_lookup_release(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_release(&dentry->lookup_map); +} + +/** + * d_lookup_acquire - acquire ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry was passed to this thread, the + * d_lookup_acquire() must be called to tell lockdep that this + * thread now owns the DCACHE_PAR_LOOKUP lock. + */ +static inline void d_lookup_acquire(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_acquire_try(&dentry->lookup_map); +} + +/** * d_inode - Get the actual inode of this dentry * @dentry: The dentry to query * diff --git a/include/linux/device-id/ap.h b/include/linux/device-id/ap.h index 0992333a34db..e050abebbf3d 100644 --- a/include/linux/device-id/ap.h +++ b/include/linux/device-id/ap.h @@ -4,7 +4,6 @@ #ifdef __KERNEL__ #include <linux/types.h> -typedef unsigned long kernel_ulong_t; #endif #define AP_DEVICE_ID_MATCH_CARD_TYPE 0x01 @@ -14,7 +13,6 @@ typedef unsigned long kernel_ulong_t; struct ap_device_id { __u16 match_flags; /* which fields to match against */ __u8 dev_type; /* device type */ - kernel_ulong_t driver_info; }; #endif /* ifndef LINUX_DEVICE_ID_AP_H */ diff --git a/include/linux/device-id/apr.h b/include/linux/device-id/apr.h deleted file mode 100644 index f282608ea018..000000000000 --- a/include/linux/device-id/apr.h +++ /dev/null @@ -1,21 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#ifndef LINUX_DEVICE_ID_APR_H -#define LINUX_DEVICE_ID_APR_H - -#ifdef __KERNEL__ -#include <linux/types.h> -typedef unsigned long kernel_ulong_t; -#endif - -#define APR_NAME_SIZE 32 -#define APR_MODULE_PREFIX "apr:" - -struct apr_device_id { - char name[APR_NAME_SIZE]; - __u32 domain_id; - __u32 svc_id; - __u32 svc_version; - kernel_ulong_t driver_data; /* Data private to the driver */ -}; - -#endif /* ifndef LINUX_DEVICE_ID_APR_H */ diff --git a/include/linux/device-id/arm_smccc.h b/include/linux/device-id/arm_smccc.h new file mode 100644 index 000000000000..a8579832fc1c --- /dev/null +++ b/include/linux/device-id/arm_smccc.h @@ -0,0 +1,19 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __LINUX_DEVICE_ID_ARM_SMCCC_H +#define __LINUX_DEVICE_ID_ARM_SMCCC_H + +#ifdef __KERNEL__ +#include <linux/types.h> +#endif + +#define ARM_SMCCC_MODULE_PREFIX "arm_smccc:" + +/** + * struct arm_smccc_device_id - Arm SMCCC bus device identifier + * @func_id: SMCCC function identifier + */ +struct arm_smccc_device_id { + __u32 func_id; +}; + +#endif /* __LINUX_DEVICE_ID_ARM_SMCCC_H */ diff --git a/include/linux/device-id/scmi.h b/include/linux/device-id/scmi.h new file mode 100644 index 000000000000..1b4ccfa9dcc5 --- /dev/null +++ b/include/linux/device-id/scmi.h @@ -0,0 +1,17 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +#ifndef LINUX_DEVICE_ID_SCMI_H +#define LINUX_DEVICE_ID_SCMI_H + +#ifdef __KERNEL__ +#include <linux/types.h> +#endif + +#define SCMI_NAME_SIZE 32 +#define SCMI_MODULE_PREFIX "scmi:" + +struct scmi_device_id { + __u8 protocol_id; + char name[SCMI_NAME_SIZE]; +}; + +#endif /* ifndef LINUX_DEVICE_ID_SCMI_H */ diff --git a/include/linux/device/bus.h b/include/linux/device/bus.h index a38f7229b8f4..89e156b9acb5 100644 --- a/include/linux/device/bus.h +++ b/include/linux/device/bus.h @@ -113,6 +113,7 @@ struct bus_type { bool need_parent_lock; }; +int __must_check companion_bus_register(const struct bus_type *bus); int __must_check bus_register(const struct bus_type *bus); void bus_unregister(const struct bus_type *bus); diff --git a/include/linux/device/driver.h b/include/linux/device/driver.h index 768a1334c0a1..29fbc01ef06f 100644 --- a/include/linux/device/driver.h +++ b/include/linux/device/driver.h @@ -297,4 +297,35 @@ static int __init __driver##_init(void) \ } \ device_initcall(__driver##_init); +/** + * subsys_driver() - Helper macro for drivers that don't do anything special + * in init/exit but have to register earlier, at subsys_initcall level, when + * built in. This eliminates a lot of boilerplate. Each driver may only use + * this macro once, and calling it replaces the init/exit boilerplate. + * + * @__driver: driver name + * @__register: register function for this driver type + * @__unregister: unregister function for this driver type + * @...: Additional arguments to be passed to __register and __unregister. + * + * This is meant to be a direct parallel of module_driver() above, but with + * the init call promoted to subsys_initcall() for built-in drivers so they + * are available earlier during boot. When built as a module subsys_initcall() + * collapses to module_init() and the normal module init/exit paths are used. + * + * Use this macro to construct bus specific macros for registering drivers, + * and do not use it on its own. + */ +#define subsys_driver(__driver, __register, __unregister, ...) \ +static int __init __driver##_init(void) \ +{ \ + return __register(&(__driver), ##__VA_ARGS__); \ +} \ +subsys_initcall(__driver##_init) \ +static void __exit __driver##_exit(void) \ +{ \ + __unregister(&(__driver), ##__VA_ARGS__); \ +} \ +module_exit(__driver##_exit) + #endif /* _DEVICE_DRIVER_H_ */ diff --git a/include/linux/dibs.h b/include/linux/dibs.h index d3e0777f25ae..0c10c224bcca 100644 --- a/include/linux/dibs.h +++ b/include/linux/dibs.h @@ -261,12 +261,12 @@ struct dibs_dev_ops { * @vlan_id: deprecated, ignored if device does not support vlan * Upon return in addition the following fields will be valid: * @dmb_tok: for usage by remote and local devices and clients - * @cpu_addr: allocated buffer + * @cpu_addr: allocated, zerorized buffer * @idx: dmb index, unique per dibs device * @dma_addr: to be used by device driver,if applicable * - * Allocate a dmb buffer and register it with this device and for this - * client. + * Allocate and zerorize a dmb buffer and register it with this device + * and for this client. * Return: zero on success */ int (*register_dmb)(struct dibs_dev *dev, struct dibs_dmb *dmb, diff --git a/include/linux/dma-buf.h b/include/linux/dma-buf.h index d1203da56fc5..d15b2b31d3c9 100644 --- a/include/linux/dma-buf.h +++ b/include/linux/dma-buf.h @@ -567,6 +567,7 @@ void dma_buf_unpin(struct dma_buf_attachment *attach); struct dma_buf *dma_buf_export(const struct dma_buf_export_info *exp_info); int dma_buf_fd(struct dma_buf *dmabuf, int flags); +void dma_buf_fd_install(struct dma_buf *dmabuf, int fd); struct dma_buf *dma_buf_get(int fd); void dma_buf_put(struct dma_buf *dmabuf); diff --git a/include/linux/dma-fence-array.h b/include/linux/dma-fence-array.h index 1b1d87579c38..0c49d7ccefb6 100644 --- a/include/linux/dma-fence-array.h +++ b/include/linux/dma-fence-array.h @@ -28,7 +28,6 @@ struct dma_fence_array_cb { /** * struct dma_fence_array - fence to represent an array of fences * @base: fence base class - * @lock: spinlock for fence handling * @num_fences: number of fences in the array * @num_pending: fences in the array still pending * @fences: array of the fences diff --git a/include/linux/dma-fence-chain.h b/include/linux/dma-fence-chain.h index df3beadf1515..705c4394ac0d 100644 --- a/include/linux/dma-fence-chain.h +++ b/include/linux/dma-fence-chain.h @@ -20,7 +20,6 @@ * @prev: previous fence of the chain * @prev_seqno: original previous seqno before garbage collection * @fence: encapsulated fence - * @lock: spinlock for fence handling */ struct dma_fence_chain { struct dma_fence base; @@ -81,9 +80,8 @@ dma_fence_chain_contained(struct dma_fence *fence) } /** - * dma_fence_chain_alloc - * - * Returns a new struct dma_fence_chain object or NULL on failure. + * dma_fence_chain_alloc - Returns a new &struct dma_fence_chain object or + * %NULL on failure. * * This specialized allocator has to be a macro for its allocations to be * accounted separately (to have a separate alloc_tag). The typecast is @@ -93,7 +91,8 @@ dma_fence_chain_contained(struct dma_fence *fence) kmalloc_obj(struct dma_fence_chain) /** - * dma_fence_chain_free + * dma_fence_chain_free - Frees an allocated but not used + * &struct dma_fence_chain object. * @chain: chain node to free * * Frees up an allocated but not used struct dma_fence_chain object. This diff --git a/include/linux/dma-fence.h b/include/linux/dma-fence.h index 158cd609f103..dd07d128adc3 100644 --- a/include/linux/dma-fence.h +++ b/include/linux/dma-fence.h @@ -141,6 +141,9 @@ struct dma_fence_ops { * compute the name at runtime, without having it to store permanently * for each fence, or build a cache of some sort. * + * The returned string is RCU protected and can be freed after the fence + * signaled and a RCU grace period passed. + * * This callback is mandatory. */ const char * (*get_driver_name)(struct dma_fence *fence); @@ -153,6 +156,9 @@ struct dma_fence_ops { * having it to store permanently for each fence, or build a cache of * some sort. * + * The returned string is RCU protected and can be freed after the fence + * signaled and a RCU grace period passed. + * * This callback is mandatory. */ const char * (*get_timeline_name)(struct dma_fence *fence); @@ -501,7 +507,7 @@ dma_fence_test_signaled_flag(struct dma_fence *fence) * Returns true if the fence was already signaled, false if not. Since this * function doesn't enable signaling, it is not guaranteed to ever return * true if dma_fence_add_callback(), dma_fence_wait() or - * dma_fence_enable_sw_signaling() haven't been called before. + * dma_fence_enable_signaling() haven't been called before. * * This function requires &dma_fence.lock to be held. * diff --git a/include/linux/dma-map-ops.h b/include/linux/dma-map-ops.h index 8fae2b7deb20..afaf2c12189b 100644 --- a/include/linux/dma-map-ops.h +++ b/include/linux/dma-map-ops.h @@ -60,7 +60,7 @@ struct dma_map_ops { int (*dma_supported)(struct device *dev, u64 mask); u64 (*get_required_mask)(struct device *dev); size_t (*max_mapping_size)(struct device *dev); - size_t (*opt_mapping_size)(void); + size_t (*max_opt_mapping_size)(void); unsigned long (*get_merge_boundary)(struct device *dev); }; @@ -143,7 +143,7 @@ static inline struct page *dma_alloc_contiguous(struct device *dev, size_t size, static inline void dma_free_contiguous(struct device *dev, struct page *page, size_t size) { - __free_pages(page, get_order(size)); + free_pages_exact(page_address(page), size); } #endif /* CONFIG_DMA_CMA*/ diff --git a/include/linux/dma-mapping.h b/include/linux/dma-mapping.h index a3e880649fa4..efde0e5a58dc 100644 --- a/include/linux/dma-mapping.h +++ b/include/linux/dma-mapping.h @@ -207,7 +207,7 @@ int dma_set_coherent_mask(struct device *dev, u64 mask); u64 dma_get_required_mask(struct device *dev); bool dma_addressing_limited(struct device *dev); size_t dma_max_mapping_size(struct device *dev); -size_t dma_opt_mapping_size(struct device *dev); +size_t dma_max_opt_mapping_size(struct device *dev); unsigned long dma_get_merge_boundary(struct device *dev); struct sg_table *dma_alloc_noncontiguous(struct device *dev, size_t size, enum dma_data_direction dir, gfp_t gfp, unsigned long attrs); @@ -326,7 +326,7 @@ static inline size_t dma_max_mapping_size(struct device *dev) { return 0; } -static inline size_t dma_opt_mapping_size(struct device *dev) +static inline size_t dma_max_opt_mapping_size(struct device *dev) { return 0; } diff --git a/include/linux/dmaengine.h b/include/linux/dmaengine.h index fe33a20abc61..f322e501ff62 100644 --- a/include/linux/dmaengine.h +++ b/include/linux/dmaengine.h @@ -1760,15 +1760,6 @@ void dma_run_dependencies(struct dma_async_tx_descriptor *tx); #define dma_request_channel(mask, x, y) \ __dma_request_channel(&(mask), x, y, NULL) -/* Deprecated, please use dma_request_chan() directly */ -static inline struct dma_chan * __deprecated -dma_request_slave_channel(struct device *dev, const char *name) -{ - struct dma_chan *ch = dma_request_chan(dev, name); - - return IS_ERR(ch) ? NULL : ch; -} - static inline struct dma_chan *dma_request_slave_channel_compat(const dma_cap_mask_t mask, dma_filter_fn fn, void *fn_param, diff --git a/include/linux/edac.h b/include/linux/edac.h index e6b4e51130e5..f7a8218f9cc0 100644 --- a/include/linux/edac.h +++ b/include/linux/edac.h @@ -598,9 +598,6 @@ struct mem_ctl_info { int op_state; struct dentry *debugfs; - u8 fake_inject_layer[EDAC_MAX_LAYERS]; - bool fake_inject_ue; - u16 fake_inject_count; /* * Memory Controller hierarchy diff --git a/include/linux/entry-common.h b/include/linux/entry-common.h index 6574b7183c01..fa2854fed1f2 100644 --- a/include/linux/entry-common.h +++ b/include/linux/entry-common.h @@ -102,7 +102,14 @@ static __always_inline long syscall_trace_enter(struct pt_regs *regs, unsigned l if (unlikely(work & SYSCALL_WORK_SYSCALL_TRACEPOINT)) trace_syscall_enter(regs); - if (unlikely(audit_context())) + /* + * The config check works around broken compilers which fail to + * eliminate the dead code in case of CONFIG_AUDITSYSCALL=n as they + * insist on creating a always false runtime condition based on + * audit_context() which returns NULL in that case. The explicit + * IS_ENABLED() check makes that madness go away. + */ + if (IS_ENABLED(CONFIG_AUDITSYSCALL) && unlikely(audit_context())) syscall_enter_audit(regs); return true; diff --git a/include/linux/ethtool.h b/include/linux/ethtool.h index 12683b5d125e..c3a41fbd5426 100644 --- a/include/linux/ethtool.h +++ b/include/linux/ethtool.h @@ -944,6 +944,7 @@ struct kernel_ethtool_ts_info { #define ETHTOOL_OP_NEEDS_RTNL_SPAUSEPARAM BIT(6) #define ETHTOOL_OP_NEEDS_RTNL_RSS BIT(7) #define ETHTOOL_OP_NEEDS_RTNL_GLINK BIT(8) +#define ETHTOOL_OP_NEEDS_RTNL_TEST BIT(9) /** * struct ethtool_ops - optional netdev operations @@ -981,6 +982,7 @@ struct kernel_ethtool_ts_info { * - netdev_update_features() * - netif_set_real_num_tx_queues() * - ethtool_op_get_link() (syncs link watch under rtnl_lock) + * - netif_open() / netif_close() (used by @self_test) * * @get_drvinfo: Report driver/device information. Modern drivers no * longer have to implement this callback. Most fields are @@ -1023,7 +1025,9 @@ struct kernel_ethtool_ts_info { * types should be set in @supported_coalesce_params. * Returns a negative error code or zero. * @get_ringparam: Report ring sizes - * @set_ringparam: Set ring sizes. Returns a negative error code or zero. + * @set_ringparam: Set ring sizes. The &struct ethtool_ringparam argument is + * also an output; drivers which normalize requested sizes must update it + * with the applied sizes. Returns a negative error code or zero. * @get_pause_stats: Report pause frame statistics. Drivers must not zero * statistics which they don't report. The stats structure is initialized * to ETHTOOL_STAT_NOT_SET indicating driver does not report statistics. @@ -1057,6 +1061,12 @@ struct kernel_ethtool_ts_info { * @get_sset_count: Get number of strings that @get_strings will write. * @get_rxnfc: Get RX flow classification rules. Returns a negative * error code or zero. + * Note that for %ETHTOOL_GRXCLSRLALL rule_cnt and size of the arrays + * is user-provided, and not guaranteed to match what driver would + * have reported via %ETHTOOL_GRXCLSRLCNT. Drivers must return -%EMSGSIZE + * when rule_cnt is too small. rule_locs is %NULL when rule_cnt is zero. + * On success drivers must set rule_cnt to the number of locations they + * filled in, the core copies out exactly that many. * @set_rxnfc: Set RX flow classification rules. Returns a negative * error code or zero. * @flash_device: Write a firmware image to device's flash memory. diff --git a/include/linux/f2fs_fs.h b/include/linux/f2fs_fs.h index bb2b6cd5d507..3081702b1ddb 100644 --- a/include/linux/f2fs_fs.h +++ b/include/linux/f2fs_fs.h @@ -14,9 +14,9 @@ #define F2FS_SUPER_OFFSET 1024 /* byte-size offset */ #define F2FS_MIN_LOG_SECTOR_SIZE 9 /* 9 bits for 512 bytes */ #define F2FS_MAX_LOG_SECTOR_SIZE PAGE_SHIFT /* Max is Block Size */ -#define F2FS_LOG_SECTORS_PER_BLOCK (PAGE_SHIFT - 9) /* log number for sector/blk */ -#define F2FS_BLKSIZE PAGE_SIZE /* support only block == page */ -#define F2FS_BLKSIZE_BITS PAGE_SHIFT /* bits for F2FS_BLKSIZE */ +#define F2FS_MIN_LOG_BLOCKSIZE 12 +#define F2FS_MIN_BLKSIZE 4096UL +#define F2FS_MAX_BLKSIZE PAGE_SIZE #define F2FS_MAX_EXTENSION 64 /* # of extension entries */ #define F2FS_EXTENSION_LEN 8 /* max size of extension */ @@ -24,19 +24,25 @@ #define NEW_ADDR ((block_t)-1) /* used as block_t addresses */ #define COMPRESS_ADDR ((block_t)-2) /* used as compressed data flag */ -#define F2FS_BLKSIZE_MASK (F2FS_BLKSIZE - 1) -#define F2FS_BYTES_TO_BLK(bytes) ((unsigned long long)(bytes) >> F2FS_BLKSIZE_BITS) -#define F2FS_BLK_TO_BYTES(blk) ((unsigned long long)(blk) << F2FS_BLKSIZE_BITS) -#define F2FS_BLK_END_BYTES(blk) (F2FS_BLK_TO_BYTES(blk + 1) - 1) -#define F2FS_BLK_ALIGN(x) (F2FS_BYTES_TO_BLK((x) + F2FS_BLKSIZE - 1)) +#define F2FS_BLKSIZE(sbi) ((sbi)->blocksize) +#define F2FS_BLKSIZE_BITS(sbi) ((sbi)->log_blocksize) +#define F2FS_BLKSIZE_MASK(sbi) (F2FS_BLKSIZE(sbi) - 1) +#define F2FS_LOG_SECTORS_PER_BLOCK(sbi) (F2FS_BLKSIZE_BITS(sbi) - 9) +#define F2FS_BLKS_PER_PAGE(sbi) (PAGE_SIZE / F2FS_BLKSIZE(sbi)) +#define F2FS_BYTES_TO_BLK(sbi, bytes) \ + ((unsigned long long)(bytes) >> F2FS_BLKSIZE_BITS(sbi)) +#define F2FS_BLK_TO_BYTES(sbi, blk) \ + ((unsigned long long)(blk) << F2FS_BLKSIZE_BITS(sbi)) +#define F2FS_BLK_END_BYTES(sbi, blk) \ + (F2FS_BLK_TO_BYTES(sbi, (blk) + 1) - 1) +#define F2FS_BLK_ALIGN(sbi, bytes) \ + F2FS_BYTES_TO_BLK(sbi, (unsigned long long)(bytes) + \ + F2FS_BLKSIZE(sbi) - 1) /* 0, 1(node nid), 2(meta nid) are reserved node id */ #define F2FS_RESERVED_NODE_NUM 3 #define F2FS_ROOT_INO(sbi) ((sbi)->root_ino_num) -#define F2FS_NODE_INO(sbi) ((sbi)->node_ino_num) -#define F2FS_META_INO(sbi) ((sbi)->meta_ino_num) -#define F2FS_COMPRESS_INO(sbi) (NM_I(sbi)->max_nid) #define F2FS_MAX_QUOTAS 3 @@ -214,20 +220,27 @@ struct f2fs_checkpoint { unsigned char sit_nat_version_bitmap[]; } __packed; -#define CP_CHKSUM_OFFSET (F2FS_BLKSIZE - sizeof(__le32)) /* default chksum offset in checkpoint */ #define CP_MIN_CHKSUM_OFFSET \ (offsetof(struct f2fs_checkpoint, sit_nat_version_bitmap)) /* * For orphan inode management + * + * The number of inode entries in an orphan block depends on the filesystem + * block size. Its exact on-disk layout is: + * + * 0 blocksize - 16 blocksize + * +--------------------------+--------------------------+ + * | ino[0] ... ino[n - 1] | struct f2fs_orphan_footer | + * +--------------------------+--------------------------+ + * + * n = (blocksize - sizeof(struct f2fs_orphan_footer)) / sizeof(__le32) */ -#define F2FS_ORPHANS_PER_BLOCK ((F2FS_BLKSIZE - 4 * sizeof(__le32)) / sizeof(__le32)) - -#define GET_ORPHAN_BLOCKS(n) (((n) + F2FS_ORPHANS_PER_BLOCK - 1) / \ - F2FS_ORPHANS_PER_BLOCK) - struct f2fs_orphan_block { - __le32 ino[F2FS_ORPHANS_PER_BLOCK]; /* inode numbers */ + DECLARE_FLEX_ARRAY(__le32, ino); +} __packed; + +struct f2fs_orphan_footer { __le32 reserved; /* reserved */ __le16 blk_addr; /* block index in current CP */ __le16 blk_count; /* Number of orphan inode blocks in CP */ @@ -260,26 +273,14 @@ struct node_footer { } __packed; /* Address Pointers in an Inode */ -#define DEF_ADDRS_PER_INODE ((F2FS_BLKSIZE - OFFSET_OF_END_OF_I_EXT \ - - SIZE_OF_I_NID \ - - sizeof(struct node_footer)) / sizeof(__le32)) -#define CUR_ADDRS_PER_INODE(inode) (DEF_ADDRS_PER_INODE - \ - get_extra_isize(inode)) +#define F2FS_DEF_ADDRS_PER_INODE(blocksize) \ + (((blocksize) - OFFSET_OF_END_OF_I_EXT - SIZE_OF_I_NID - \ + sizeof(struct node_footer)) / sizeof(__le32)) #define DEF_NIDS_PER_INODE 5 /* Node IDs in an Inode */ #define ADDRS_PER_INODE(inode) addrs_per_page(inode, true) /* Address Pointers in a Direct Block */ -#define DEF_ADDRS_PER_BLOCK ((F2FS_BLKSIZE - sizeof(struct node_footer)) / sizeof(__le32)) #define ADDRS_PER_BLOCK(inode) addrs_per_page(inode, false) -/* Node IDs in an Indirect Block */ -#define NIDS_PER_BLOCK ((F2FS_BLKSIZE - sizeof(struct node_footer)) / sizeof(__le32)) - -#define ADDRS_PER_PAGE(folio, inode) (addrs_per_page(inode, IS_INODE(folio))) - -#define NODE_DIR1_BLOCK (DEF_ADDRS_PER_INODE + 1) -#define NODE_DIR2_BLOCK (DEF_ADDRS_PER_INODE + 2) -#define NODE_IND1_BLOCK (DEF_ADDRS_PER_INODE + 3) -#define NODE_IND2_BLOCK (DEF_ADDRS_PER_INODE + 4) -#define NODE_DIND_BLOCK (DEF_ADDRS_PER_INODE + 5) +#define ADDRS_PER_PAGE(folio, inode) (addrs_per_page(inode, IS_INODE(F2FS_I_SB(inode), folio))) #define F2FS_INLINE_XATTR 0x01 /* file inline xattr flag */ #define F2FS_INLINE_DATA 0x02 /* file inline data flag */ @@ -339,18 +340,26 @@ struct f2fs_inode { */ __le32 i_extra_end[0]; /* for attribute size calculation */ } __packed; - __le32 i_addr[DEF_ADDRS_PER_INODE]; /* Pointers to data blocks */ + DECLARE_FLEX_ARRAY(__le32, i_addr); /* data block pointers */ }; - __le32 i_nid[DEF_NIDS_PER_INODE]; /* direct(2), indirect(2), - double_indirect(1) node id */ + /* + * __le32 i_nid[DEF_NIDS_PER_INODE]; + * direct(2), indirect(2), double_indirect(1) node IDs + * + * It is stored immediately before the node footer at the end of the + * filesystem block. Its offset depends on the filesystem block size, so + * locate it dynamically with F2FS_INODE_NIDS(). + */ } __packed; struct direct_node { - __le32 addr[DEF_ADDRS_PER_BLOCK]; /* array of data block address */ + /* The address count depends on the filesystem block size. */ + DECLARE_FLEX_ARRAY(__le32, addr); /* array of data block address */ } __packed; struct indirect_node { - __le32 nid[NIDS_PER_BLOCK]; /* array of data block address */ + /* The node ID count depends on the filesystem block size. */ + DECLARE_FLEX_ARRAY(__le32, nid); /* array of data block address */ } __packed; enum { @@ -369,14 +378,18 @@ struct f2fs_node { struct direct_node dn; struct indirect_node in; }; - struct node_footer footer; + /* + * struct node_footer footer; + * + * It is stored at the end of the filesystem block, after the inode or + * direct/indirect node data. Its offset depends on the filesystem block + * size, so locate it dynamically with F2FS_NODE_FOOTER(). + */ } __packed; /* * For NAT entries */ -#define NAT_ENTRY_PER_BLOCK (F2FS_BLKSIZE / sizeof(struct f2fs_nat_entry)) - struct f2fs_nat_entry { __u8 version; /* latest version of cached nat entry */ __le32 ino; /* inode number */ @@ -384,7 +397,8 @@ struct f2fs_nat_entry { } __packed; struct f2fs_nat_block { - struct f2fs_nat_entry entries[NAT_ENTRY_PER_BLOCK]; + /* The entry count depends on the filesystem block size. */ + DECLARE_FLEX_ARRAY(struct f2fs_nat_entry, entries); } __packed; /* @@ -396,8 +410,6 @@ struct f2fs_nat_block { * Not allow to change this. */ #define SIT_VBLOCK_MAP_SIZE 64 -#define SIT_ENTRY_PER_BLOCK (F2FS_BLKSIZE / sizeof(struct f2fs_sit_entry)) - /* * F2FS uses 4 bytes to represent block address. As a result, supported size of * disk is 16 TB for a 4K page size and 64 TB for a 16K page size and it equals @@ -424,8 +436,13 @@ struct f2fs_sit_entry { __le64 mtime; /* segment age for cleaning */ } __packed; +/* + * The on-disk SIT block is a filesystem-block-sized array of SIT entries. + * Its entry count depends on the filesystem block size, so it must be + * calculated by the caller rather than implied by this C structure. + */ struct f2fs_sit_block { - struct f2fs_sit_entry entries[SIT_ENTRY_PER_BLOCK]; + DECLARE_FLEX_ARRAY(struct f2fs_sit_entry, entries); } __packed; /* @@ -595,15 +612,7 @@ typedef __le32 f2fs_hash_t; * dentry, when converting inline dentry we should handle this carefully. */ -/* the number of dentry in a block */ -#define NR_DENTRY_IN_BLOCK ((BITS_PER_BYTE * F2FS_BLKSIZE) / \ - ((SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * BITS_PER_BYTE + 1)) #define SIZE_OF_DIR_ENTRY 11 /* by byte */ -#define SIZE_OF_DENTRY_BITMAP ((NR_DENTRY_IN_BLOCK + BITS_PER_BYTE - 1) / \ - BITS_PER_BYTE) -#define SIZE_OF_RESERVED (F2FS_BLKSIZE - ((SIZE_OF_DIR_ENTRY + \ - F2FS_SLOT_LEN) * \ - NR_DENTRY_IN_BLOCK + SIZE_OF_DENTRY_BITMAP)) #define MIN_INLINE_DENTRY_SIZE 40 /* just include '.' and '..' entries */ /* One directory entry slot representing F2FS_SLOT_LEN-sized file name */ @@ -614,14 +623,21 @@ struct f2fs_dir_entry { __u8 file_type; /* file type */ } __packed; -/* Block-sized directory entry block */ -struct f2fs_dentry_block { - /* validity bitmap for directory entries in each block */ - __u8 dentry_bitmap[SIZE_OF_DENTRY_BITMAP]; - __u8 reserved[SIZE_OF_RESERVED]; - struct f2fs_dir_entry dentry[NR_DENTRY_IN_BLOCK]; - __u8 filename[NR_DENTRY_IN_BLOCK][F2FS_SLOT_LEN]; -} __packed; +/* + * A dentry block is laid out as follows, where the number of entries and all + * offsets are determined by the filesystem block size at runtime: + * + * 0 blocksize + * +--------+----------+-------------------+-----------------------+ + * | bitmap | reserved | dir_entry[entries]| filename[entries][8] | + * +--------+----------+-------------------+-----------------------+ + * + * entries = (BITS_PER_BYTE * blocksize) / + * ((SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * BITS_PER_BYTE + 1) + * bitmap_size = DIV_ROUND_UP(entries, BITS_PER_BYTE) + * reserved_size = blocksize - bitmap_size - + * (SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * entries + */ #define F2FS_DEF_PROJID 0 /* default project ID */ diff --git a/include/linux/fault-inject.h b/include/linux/fault-inject.h index 58fd14c82270..5c74748a53f3 100644 --- a/include/linux/fault-inject.h +++ b/include/linux/fault-inject.h @@ -19,6 +19,13 @@ enum fault_flags { #include <linux/ratelimit.h> /* + * Length of the debugfs directory name embedded in struct fault_attr. + * Chosen to accommodate every in-tree caller of fault_create_debugfs_attr() + * (the longest is "fail_dma_array_full", 19 chars) with generous headroom. + */ +#define FAULT_ATTR_DNAME_LEN 64 + +/* * For explanation of the elements of this struct, see * Documentation/fault-injection/fault-injection.rst */ @@ -37,7 +44,7 @@ struct fault_attr { unsigned long count; struct ratelimit_state ratelimit_state; - struct dentry *dname; + char dname[FAULT_ATTR_DNAME_LEN]; }; #define FAULT_ATTR_INITIALIZER { \ @@ -47,7 +54,6 @@ struct fault_attr { .stacktrace_depth = 32, \ .ratelimit_state = RATELIMIT_STATE_INIT_DISABLED, \ .verbose = 2, \ - .dname = NULL, \ } #define DECLARE_FAULT_ATTR(name) struct fault_attr name = FAULT_ATTR_INITIALIZER diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index c45306a9f007..a46781058729 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -25,7 +25,7 @@ struct fdtable { unsigned int max_fds; - struct file __rcu **fd; /* current fd array */ + struct file __rcu **fd __counted_by_ptr(max_fds); /* current fd array */ unsigned long *close_on_exec; unsigned long *open_fds; unsigned long *full_fds_bits; @@ -101,11 +101,22 @@ struct task_struct; void put_files_struct(struct files_struct *fs); int unshare_files(void); +void switch_files_struct(struct task_struct *tsk, struct files_struct *files); +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp); +enum fd_range_flags { + /* Leave behind all descriptors outside of the specified range. */ + FD_RANGE_EXCEPT = (1U << 0), + + /* Only select descriptors that have close-on-exec set. */ + FD_RANGE_CLOEXEC_ONLY = (1U << 1), +}; + struct fd_range { unsigned int from, to; + enum fd_range_flags flags; }; struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy; -void do_close_on_exec(struct files_struct *); +void close_cloexec_files(struct files_struct *); int iterate_fd(struct files_struct *, unsigned, int (*)(const void *, struct file *, unsigned), const void *); diff --git a/include/linux/file.h b/include/linux/file.h index 27484b444d31..41c3c0be1064 100644 --- a/include/linux/file.h +++ b/include/linux/file.h @@ -12,6 +12,7 @@ #include <linux/errno.h> #include <linux/cleanup.h> #include <linux/err.h> +#include <linux/vfsdebug.h> struct file; @@ -129,117 +130,84 @@ extern unsigned int sysctl_nr_open_min, sysctl_nr_open_max; /* * fd_prepare: Combined fd + file allocation cleanup class. - * @err: Error code to indicate if allocation succeeded. - * @__fd: Allocated fd (may not be accessed directly) - * @__file: Allocated struct file pointer (may not be accessed directly) + * @fd: Allocated fd + * @file: Allocated struct file pointer * * Allocates an fd and a file together. On error paths, automatically cleans * up whichever resource was successfully allocated. Allows flexible file * allocation with different functions per usage. * - * Do not use directly. + * Do not declare directly, use FD_PREPARE(). */ struct fd_prepare { - s32 err; - s32 __fd; /* do not access directly */ - struct file *__file; /* do not access directly */ + int fd; + struct file *file; }; -/* Typedef for fd_prepare cleanup guards. */ -typedef struct fd_prepare class_fd_prepare_t; - -/* - * Accessors for fd_prepare class members. - * _Generic() is used for zero-cost type safety. - */ -#define fd_prepare_fd(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__fd)) - -#define fd_prepare_file(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__file)) - /* Do not use directly. */ -static inline void class_fd_prepare_destructor(const struct fd_prepare *fdf) +static __always_inline void __fd_prepare_cleanup(const struct fd_prepare *fdf) { - if (unlikely(fdf->__fd >= 0)) - put_unused_fd(fdf->__fd); - if (unlikely(!IS_ERR_OR_NULL(fdf->__file))) - fput(fdf->__file); + if (unlikely(fdf->fd >= 0)) { + put_unused_fd(fdf->fd); + fput(fdf->file); + } } /* Do not use directly. */ -static inline int class_fd_prepare_lock_err(const struct fd_prepare *fdf) +static __always_inline struct fd_prepare __fd_prepare(int fd, struct file *file) { - if (unlikely(fdf->err)) - return fdf->err; - if (unlikely(fdf->__fd < 0)) - return fdf->__fd; - if (unlikely(IS_ERR(fdf->__file))) - return PTR_ERR(fdf->__file); - if (unlikely(!fdf->__file)) - return -ENOMEM; - return 0; + if (fd >= 0 && IS_ERR_OR_NULL(file)) { + int err = file ? PTR_ERR(file) : -ENOMEM; + + put_unused_fd(fd); + fd = err; + file = NULL; + } + return (struct fd_prepare){ .fd = fd, .file = file }; } /* - * __FD_PREPARE_INIT - Helper to initialize fd_prepare class. - * @_fd_flags: flags for get_unused_fd_flags() - * @_file_owned: expression that returns struct file * - * - * Returns a struct fd_prepare with fd, file, and err set. - * If fd allocation fails, fd will be negative and err will be set. If - * fd succeeds but file_init_expr fails, file will be ERR_PTR and err - * will be set. The err field is the single source of truth for error - * checking. - */ -#define __FD_PREPARE_INIT(_fd_flags, _file_owned) \ - ({ \ - struct fd_prepare fdf = { \ - .__fd = get_unused_fd_flags((_fd_flags)), \ - }; \ - if (likely(fdf.__fd >= 0)) \ - fdf.__file = (_file_owned); \ - fdf.err = ACQUIRE_ERR(fd_prepare, &fdf); \ - fdf; \ - }) - -/* - * FD_PREPARE - Macro to declare and initialize an fd_prepare variable. + * FD_PREPARE - Declare and initialize an fd_prepare instance. * - * Declares and initializes an fd_prepare variable with automatic - * cleanup. No separate scope required - cleanup happens when variable - * goes out of scope. + * This allocates a new fd and only evaluates @_file_owned if the + * allocation succeeded. Cleanup happens when the variable goes out of + * scope and the guard releases whichever of the descriptor and the file + * was allocated. If fd_publish() was called the fd and file are + * published and cleanup becomes a nop. * - * @_fdf: name of struct fd_prepare variable to define + * @_fdf: name of the const struct fd_prepare pointer to define * @_fd_flags: flags for get_unused_fd_flags() * @_file_owned: struct file to take ownership of (can be expression) */ +#define __FD_PREPARE(_guard, _fdf, _fd_flags, _file_owned) \ + struct fd_prepare _guard __cleanup(__fd_prepare_cleanup) = ({ \ + int __fd = get_unused_fd_flags(_fd_flags); \ + __fd_prepare(__fd, __fd < 0 ? NULL : (_file_owned)); \ + }); \ + const struct fd_prepare *const _fdf = &_guard + #define FD_PREPARE(_fdf, _fd_flags, _file_owned) \ - CLASS_INIT(fd_prepare, _fdf, __FD_PREPARE_INIT(_fd_flags, _file_owned)) + __FD_PREPARE(__UNIQUE_ID(fd_prepare), _fdf, _fd_flags, _file_owned) /* * fd_publish - Publish prepared fd and file to the fd table. - * @_fdf: struct fd_prepare variable + * @fdf: struct fd_prepare pointer defined by FD_PREPARE() */ -#define fd_publish(_fdf) \ - ({ \ - struct fd_prepare *fdp = &(_fdf); \ - VFS_WARN_ON_ONCE(fdp->err); \ - VFS_WARN_ON_ONCE(fdp->__fd < 0); \ - VFS_WARN_ON_ONCE(IS_ERR_OR_NULL(fdp->__file)); \ - fd_install(fdp->__fd, fdp->__file); \ - retain_and_null_ptr(fdp->__file); \ - take_fd(fdp->__fd); \ - }) +static __always_inline int fd_publish(const struct fd_prepare *fdf) +{ + /* Callers only get a const view, the guard itself is writable. */ + struct fd_prepare *guard = (struct fd_prepare *)fdf; + + VFS_WARN_ON_ONCE(guard->fd < 0); + fd_install(guard->fd, guard->file); + return take_fd(guard->fd); +} /* Do not use directly. */ -#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ - ({ \ - FD_PREPARE(_fdf, _fd_flags, _file_owned); \ - s32 ret = _fdf.err; \ - if (likely(!ret)) \ - ret = fd_publish(_fdf); \ - ret; \ +#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ + ({ \ + FD_PREPARE(_fdf, _fd_flags, _file_owned); \ + _fdf->fd < 0 ? _fdf->fd : fd_publish(_fdf); \ }) /* diff --git a/include/linux/fileattr.h b/include/linux/fileattr.h index 58044b598016..09e32b84e02a 100644 --- a/include/linux/fileattr.h +++ b/include/linux/fileattr.h @@ -74,7 +74,7 @@ static inline bool fileattr_has_fsx(const struct file_kattr *fa) } int vfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int ioctl_getflags(struct file *file, unsigned int __user *argp); int ioctl_setflags(struct file *file, unsigned int __user *argp); diff --git a/include/linux/filter.h b/include/linux/filter.h index 4a9bc6a848f2..e42eccb0990e 100644 --- a/include/linux/filter.h +++ b/include/linux/filter.h @@ -98,6 +98,11 @@ struct ctl_table_header; /* BPF program can access up to 512 bytes of stack space. */ #define MAX_BPF_STACK 512 +/* + * Stack budget of a program on a JIT that lays out frames of that size. + * The interpreter and JITs without such support keep MAX_BPF_STACK. + */ +#define MAX_BPF_STACK_JIT 2048 /* Helper macros for filter block array initializers. */ @@ -1237,6 +1242,8 @@ bool bpf_jit_inlines_helper_call(s32 imm); bool bpf_jit_supports_subprog_tailcalls(void); bool bpf_jit_supports_percpu_insn(void); bool bpf_jit_supports_kfunc_call(void); +bool bpf_jit_supports_callx(void); +bool bpf_jit_supports_kfunc_ret_reg_pair(void); bool bpf_jit_supports_stack_args(void); bool bpf_jit_supports_arena_args(void); bool bpf_jit_supports_far_kfunc_call(void); @@ -1245,8 +1252,41 @@ bool bpf_jit_supports_ptr_xchg(void); bool bpf_jit_supports_arena(void); bool bpf_jit_supports_insn(struct bpf_insn *insn, bool in_arena); bool bpf_jit_supports_private_stack(void); +bool bpf_jit_supports_large_stack(void); bool bpf_jit_supports_timed_may_goto(void); bool bpf_jit_supports_fsession(void); + +struct bpf_jit_arg_abi { + /* Argument registers of the kernel convention. */ + u8 nr_arg_regs; + /* Round the register number up to an even one for 16-byte alignment. */ + bool even_reg_align; + /* Round the stack slot up to an even one for 16-byte alignment. */ + bool even_stack_align; + /* An argument may straddle the last register and the stack. */ + bool split_at_boundary; + /* A later argument may reuse a register a stack-passed one skipped. */ + bool backfill_after_stack; +}; + +const struct bpf_jit_arg_abi *bpf_jit_arg_abi(void); +u32 bpf_jit_place_args(const struct bpf_jit_arg_abi *abi, + const struct btf_func_model *fm, u8 *pos_of_slot); + +/* The JIT's scratch register, in place of an argument slot. */ +#define BPF_JIT_ARG_TMP 0xff + +/* Every argument slot moves at most once, and the scratch goes out and back. */ +#define BPF_JIT_MAX_ARG_MOVES (MAX_BPF_FUNC_ARG_SLOTS + 2) + +struct bpf_jit_arg_move { + u8 dst; + u8 src; +}; + +u32 bpf_jit_plan_arg_moves(const struct bpf_jit_arg_abi *abi, + const struct btf_func_model *fm, + struct bpf_jit_arg_move *moves); u64 bpf_arch_uaddress_limit(void); void arch_bpf_stack_walk(bool (*consume_fn)(void *cookie, u64 ip, u64 sp, u64 bp), void *cookie); u64 arch_bpf_timed_may_goto(void); @@ -1376,7 +1416,6 @@ bpf_jit_binary_alloc(unsigned int proglen, u8 **image_ptr, void bpf_jit_binary_free(struct bpf_binary_header *hdr); u64 bpf_jit_alloc_exec_limit(void); void *bpf_jit_alloc_exec(unsigned long size); -void *bpf_jit_alloc_exec_rw(unsigned long size); void bpf_jit_free_exec(void *addr); void bpf_jit_free(struct bpf_prog *fp); struct bpf_binary_header * @@ -1898,6 +1937,11 @@ static __always_inline long __bpf_xdp_redirect_map(struct bpf_map *map, u64 inde return XDP_REDIRECT; } +int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + void *search_res, u32 len, u64 flags); +int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + const void *from, u32 len, u64 flags); + #ifdef CONFIG_NET int __bpf_skb_load_bytes(const struct sk_buff *skb, u32 offset, void *to, u32 len); int __bpf_skb_store_bytes(struct sk_buff *skb, u32 offset, const void *from, diff --git a/include/linux/firewire.h b/include/linux/firewire.h index fd35a6570cd8..dbb5c825a1cc 100644 --- a/include/linux/firewire.h +++ b/include/linux/firewire.h @@ -172,10 +172,10 @@ struct fw_attribute_group { }; enum fw_device_quirk { - // See afa1282a35d3 ("firewire: core: check for 1394a compliant IRM, fix inaccessibility of Sony camcorder"). + // See 10389536742c ("firewire: core: check for 1394a compliant IRM, fix inaccessibility of Sony camcorder"). FW_DEVICE_QUIRK_IRM_IS_1394_1995_ONLY = BIT(0), - // See a509e43ff338 ("firewire: core: fix unstable I/O with Canon camcorder"). + // See 6044565af458 ("firewire: core: fix unstable I/O with Canon camcorder"). FW_DEVICE_QUIRK_IRM_IGNORES_BUS_MANAGER = BIT(1), // MOTU Audio Express transfers acknowledge packet with 0x10 for pending state. @@ -224,7 +224,7 @@ struct fw_device { struct mutex client_list_mutex; struct list_head client_list; - const u32 *config_rom; + const u32 *config_rom __counted_by_ptr(config_rom_length); size_t config_rom_length; int config_rom_retries; unsigned is_local:1; @@ -298,16 +298,25 @@ union fw_transaction_callback { fw_transaction_callback_with_tstamp_t with_tstamp; }; -/* - * This callback handles an inbound request subaction. It is called in - * RCU read-side context, therefore must not sleep. +/** + * typedef fw_address_callback_t - Function to handle the request of the asynchronous transaction. + * @card: the card instance which receives the request + * @request: the request instance. + * @tcode: the transaction code + * @destination: the destination node ID + * @source: the source node ID + * @generation: the bus generation in which the request was sent + * @offset: the destination offset in source node. + * @data: the request content if available. + * @length: the length of data. + * @callback_data: the data registered with this function. * - * The callback should not initiate outbound request subactions directly. - * Otherwise there is a danger of recursion of inbound and outbound - * transactions from and to the local node. + * This callback handles an inbound request subaction. * * The callback is responsible that fw_send_response() is called on the @request, except for FCP * registers for which the core takes care of that. + * + * Context: Process context. */ typedef void (*fw_address_callback_t)(struct fw_card *card, struct fw_request *request, @@ -322,22 +331,24 @@ struct fw_packet { int generation; u32 header[4]; size_t header_length; - void *payload; + void *payload __counted_by_ptr(payload_length); size_t payload_length; dma_addr_t payload_bus; bool payload_mapped; u32 timestamp; + // Used to handle the local-to-local packets in the AT request/response contexts. + struct list_head link_for_local; + /* * This callback is called when the packet transmission has completed. * For successful transmission, the status code is the ack received * from the destination. Otherwise it is one of the juju-specific * rcodes: RCODE_SEND_ERROR, _CANCELLED, _BUSY, _GENERATION, _NO_ACK. - * The callback can be called from workqueue and thus must never block. + * The callback is called from a workqueue. It is not preferable to block it so long. */ fw_packet_callback_t callback; int ack; - struct list_head link; void *driver_data; }; @@ -359,6 +370,11 @@ struct fw_transaction { union fw_transaction_callback callback; bool with_tstamp; void *callback_data; + + // For some error cases. + struct work_struct error_work; + int rcode; + u32 response_timestamp; }; struct fw_address_handler { @@ -411,9 +427,8 @@ void __fw_send_request(struct fw_card *card, struct fw_transaction *t, int tcode * A variation of __fw_send_request() to generate callback for response subaction without time * stamp. * - * The callback is invoked in the workqueue context in most cases. However, if an error is detected - * before queueing or the destination address refers to the local node, it is invoked in the - * current context instead. + * After the transaction is completed successfully or unsuccessfully, the @callback will be called + * in process context. */ static inline void fw_send_request(struct fw_card *card, struct fw_transaction *t, int tcode, int destination_id, int generation, int speed, @@ -444,9 +459,8 @@ static inline void fw_send_request(struct fw_card *card, struct fw_transaction * * * A variation of __fw_send_request() to generate callback for response subaction with time stamp. * - * The callback is invoked in the workqueue context in most cases. However, if an error is detected - * before queueing or the destination address refers to the local node, it is invoked in the current - * context instead. + * After the transaction is completed successfully or unsuccessfully, the @callback will be called + * in process context. */ static inline void fw_send_request_with_tstamp(struct fw_card *card, struct fw_transaction *t, int tcode, int destination_id, int generation, int speed, unsigned long long offset, diff --git a/include/linux/firmware/cirrus/cs_dsp.h b/include/linux/firmware/cirrus/cs_dsp.h index 4e3baa557068..6aa1e1b2b4a5 100644 --- a/include/linux/firmware/cirrus/cs_dsp.h +++ b/include/linux/firmware/cirrus/cs_dsp.h @@ -96,7 +96,7 @@ struct cs_dsp_alg_region { struct cs_dsp_coeff_ctl { struct list_head list; struct cs_dsp *dsp; - void *cache; + void *cache __counted_by_ptr(len); const char *fw_name; /* Subname is needed to match with firmware */ const char *subname; diff --git a/include/linux/firmware/imx/se_api.h b/include/linux/firmware/imx/se_api.h new file mode 100644 index 000000000000..b1c4c9115d7b --- /dev/null +++ b/include/linux/firmware/imx/se_api.h @@ -0,0 +1,14 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * Copyright 2025 NXP + */ + +#ifndef __SE_API_H__ +#define __SE_API_H__ + +#include <linux/types.h> + +#define SOC_ID_OF_IMX8ULP 0x084d +#define SOC_ID_OF_IMX93 0x9300 + +#endif /* __SE_API_H__ */ diff --git a/include/linux/firmware/qcom/qcom_scm.h b/include/linux/firmware/qcom/qcom_scm.h index 5747bd191bf1..a0a6bc0229c4 100644 --- a/include/linux/firmware/qcom/qcom_scm.h +++ b/include/linux/firmware/qcom/qcom_scm.h @@ -64,35 +64,6 @@ bool qcom_scm_is_available(void); int qcom_scm_set_cold_boot_addr(void *entry); int qcom_scm_set_warm_boot_addr(void *entry); void qcom_scm_cpu_power_down(u32 flags); -int qcom_scm_set_remote_state(u32 state, u32 id); - -struct qcom_scm_pas_context { - struct device *dev; - u32 pas_id; - phys_addr_t mem_phys; - size_t mem_size; - void *ptr; - dma_addr_t phys; - ssize_t size; - bool use_tzmem; -}; - -struct qcom_scm_pas_context *devm_qcom_scm_pas_context_alloc(struct device *dev, - u32 pas_id, - phys_addr_t mem_phys, - size_t mem_size); -int qcom_scm_pas_init_image(u32 pas_id, const void *metadata, size_t size, - struct qcom_scm_pas_context *ctx); -void qcom_scm_pas_metadata_release(struct qcom_scm_pas_context *ctx); -int qcom_scm_pas_mem_setup(u32 pas_id, phys_addr_t addr, phys_addr_t size); -int qcom_scm_pas_auth_and_reset(u32 pas_id); -int qcom_scm_pas_shutdown(u32 pas_id); -bool qcom_scm_pas_supported(u32 pas_id); -struct resource_table *qcom_scm_pas_get_rsc_table(struct qcom_scm_pas_context *ctx, - void *input_rt, size_t input_rt_size, - size_t *output_rt_size); - -int qcom_scm_pas_prepare_and_auth_reset(struct qcom_scm_pas_context *ctx); int qcom_scm_io_readl(phys_addr_t addr, unsigned int *val); int qcom_scm_io_writel(phys_addr_t addr, unsigned int val); diff --git a/include/linux/firmware/xlnx-zynqmp-ufs.h b/include/linux/firmware/xlnx-zynqmp-ufs.h index d3538dd5822a..00383dd835f2 100644 --- a/include/linux/firmware/xlnx-zynqmp-ufs.h +++ b/include/linux/firmware/xlnx-zynqmp-ufs.h @@ -9,17 +9,17 @@ #define __FIRMWARE_XLNX_ZYNQMP_UFS_H__ #if IS_REACHABLE(CONFIG_ZYNQMP_FIRMWARE) -int zynqmp_pm_is_mphy_tx_rx_config_ready(bool *is_ready); -int zynqmp_pm_is_sram_init_done(bool *is_done); +int zynqmp_pm_wait_mphy_tx_rx_config_ready(u32 timeout_us); +int zynqmp_pm_wait_sram_init_done(u32 timeout_us); int zynqmp_pm_set_sram_bypass(void); int zynqmp_pm_get_ufs_calibration_values(u32 *val); #else -static inline int zynqmp_pm_is_mphy_tx_rx_config_ready(bool *is_ready) +static inline int zynqmp_pm_wait_mphy_tx_rx_config_ready(u32 timeout_us) { return -ENODEV; } -static inline int zynqmp_pm_is_sram_init_done(bool *is_done) +static inline int zynqmp_pm_wait_sram_init_done(u32 timeout_us) { return -ENODEV; } diff --git a/include/linux/firmware/xlnx-zynqmp.h b/include/linux/firmware/xlnx-zynqmp.h index f4b3df614c77..41a1cd810bda 100644 --- a/include/linux/firmware/xlnx-zynqmp.h +++ b/include/linux/firmware/xlnx-zynqmp.h @@ -277,11 +277,6 @@ enum rpu_oper_mode { PM_RPU_MODE_SPLIT = 1, }; -enum rpu_boot_mem { - PM_RPU_BOOTMEM_LOVEC = 0, - PM_RPU_BOOTMEM_HIVEC = 1, -}; - enum rpu_tcm_comb { PM_RPU_TCM_SPLIT = 0, PM_RPU_TCM_COMB = 1, diff --git a/include/linux/fs.h b/include/linux/fs.h index f9d1e05e8ae6..3db90996756f 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -1436,10 +1436,10 @@ static inline void i_gid_write(struct inode *inode, gid_t gid) * @idmap: idmap of the mount the inode was found from * @inode: inode to map * - * Return: whe inode's i_uid mapped down according to @idmap. + * Return: the inode's i_uid mapped down according to @idmap. * If the inode's i_uid has no mapping INVALID_VFSUID is returned. */ -static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, +static inline vfsuid_t i_uid_into_vfsuid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsuid(idmap, i_user_ns(inode), inode->i_uid); @@ -1456,7 +1456,7 @@ static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, * * Return: true if @inode's i_uid field needs to be updated, false if not. */ -static inline bool i_uid_needs_update(struct mnt_idmap *idmap, +static inline bool i_uid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1474,7 +1474,7 @@ static inline bool i_uid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_uid field translating the vfsuid of any idmapped * mount into the filesystem kuid. */ -static inline void i_uid_update(struct mnt_idmap *idmap, +static inline void i_uid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1491,7 +1491,7 @@ static inline void i_uid_update(struct mnt_idmap *idmap, * Return: the inode's i_gid mapped down according to @idmap. * If the inode's i_gid has no mapping INVALID_VFSGID is returned. */ -static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, +static inline vfsgid_t i_gid_into_vfsgid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsgid(idmap, i_user_ns(inode), inode->i_gid); @@ -1508,7 +1508,7 @@ static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, * * Return: true if @inode's i_gid field needs to be updated, false if not. */ -static inline bool i_gid_needs_update(struct mnt_idmap *idmap, +static inline bool i_gid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1526,7 +1526,7 @@ static inline bool i_gid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_gid field translating the vfsgid of any idmapped * mount into the filesystem kgid. */ -static inline void i_gid_update(struct mnt_idmap *idmap, +static inline void i_gid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1544,7 +1544,7 @@ static inline void i_gid_update(struct mnt_idmap *idmap, * an idmapped mount map the caller's fsuid according to @idmap. */ static inline void inode_fsuid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_uid = mapped_fsuid(idmap, i_user_ns(inode)); } @@ -1558,7 +1558,7 @@ static inline void inode_fsuid_set(struct inode *inode, * an idmapped mount map the caller's fsgid according to @idmap. */ static inline void inode_fsgid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_gid = mapped_fsgid(idmap, i_user_ns(inode)); } @@ -1575,7 +1575,7 @@ static inline void inode_fsgid_set(struct inode *inode, * Return: true if fsuid and fsgid is mapped, false if not. */ static inline bool fsuidgid_has_mapping(struct super_block *sb, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { struct user_namespace *fs_userns = sb->s_user_ns; kuid_t kuid; @@ -1755,25 +1755,25 @@ static inline bool file_write_not_started(const struct file *file) return sb_write_not_started(file_inode(file)->i_sb); } -bool inode_owner_or_capable(struct mnt_idmap *idmap, +bool inode_owner_or_capable(const struct mnt_idmap *idmap, const struct inode *inode); /* * VFS helper functions.. */ -int vfs_create(struct mnt_idmap *, struct dentry *, umode_t, +int vfs_create(const struct mnt_idmap *, struct dentry *, umode_t, struct delegated_inode *); -struct dentry *vfs_mkdir(struct mnt_idmap *, struct inode *, +struct dentry *vfs_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, struct delegated_inode *); -int vfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t, struct delegated_inode *); -int vfs_symlink(struct mnt_idmap *, struct inode *, +int vfs_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *, struct delegated_inode *); -int vfs_link(struct dentry *, struct mnt_idmap *, struct inode *, +int vfs_link(struct dentry *, const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_rmdir(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_rmdir(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_unlink(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); /** @@ -1787,7 +1787,7 @@ int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, * @flags: rename flags */ struct renamedata { - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; struct dentry *old_parent; struct dentry *old_dentry; struct dentry *new_parent; @@ -1798,14 +1798,14 @@ struct renamedata { int vfs_rename(struct renamedata *); -static inline int vfs_whiteout(struct mnt_idmap *idmap, +static inline int vfs_whiteout(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry) { return vfs_mknod(idmap, dir, dentry, S_IFCHR | WHITEOUT_MODE, WHITEOUT_DEV, NULL); } -struct file *kernel_tmpfile_open(struct mnt_idmap *idmap, +struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap, const struct path *parentpath, umode_t mode, int open_flag, const struct cred *cred); @@ -1830,12 +1830,12 @@ extern long compat_ptr_ioctl(struct file *file, unsigned int cmd, /* * VFS file helper functions. */ -void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode, +void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode, const struct inode *dir, umode_t mode); extern bool may_open_dev(const struct path *path); -umode_t mode_strip_sgid(struct mnt_idmap *idmap, +umode_t mode_strip_sgid(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode); -bool in_group_or_capable(struct mnt_idmap *idmap, +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid); /* @@ -1994,26 +1994,26 @@ enum fs_update_time { struct inode_operations { struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); const char * (*get_link) (struct dentry *, struct inode *, struct delayed_call *); - int (*permission) (struct mnt_idmap *, struct inode *, int); + int (*permission) (const struct mnt_idmap *, struct inode *, int); struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); int (*readlink) (struct dentry *, char __user *,int); - int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); int (*link) (struct dentry *,struct inode *,struct dentry *); int (*unlink) (struct inode *,struct dentry *); - int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *, const char *); - struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *, + struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); int (*rmdir) (struct inode *,struct dentry *); - int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t,dev_t); - int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, + int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); - int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); - int (*getattr) (struct mnt_idmap *, const struct path *, + int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *); + int (*getattr) (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); ssize_t (*listxattr) (struct dentry *, char *, size_t); int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start, @@ -2024,13 +2024,13 @@ struct inode_operations { int (*atomic_open)(struct inode *, struct dentry *, struct file *, unsigned open_flag, umode_t create_mode); - int (*tmpfile) (struct mnt_idmap *, struct inode *, + int (*tmpfile) (const struct mnt_idmap *, struct inode *, struct file *, umode_t); - struct posix_acl *(*get_acl)(struct mnt_idmap *, struct dentry *, + struct posix_acl *(*get_acl)(const struct mnt_idmap *, struct dentry *, int); - int (*set_acl)(struct mnt_idmap *, struct dentry *, + int (*set_acl)(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); - int (*fileattr_set)(struct mnt_idmap *idmap, + int (*fileattr_set)(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); struct offset_ctx *(*get_offset_ctx)(struct inode *inode); @@ -2173,7 +2173,7 @@ extern loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos, (inode)->i_rdev == WHITEOUT_DEV) #define IS_ANON_FILE(inode) ((inode)->i_flags & S_ANON_INODE) -static inline bool HAS_UNMAPPED_ID(struct mnt_idmap *idmap, +static inline bool HAS_UNMAPPED_ID(const struct mnt_idmap *idmap, struct inode *inode) { return !vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) || @@ -2459,7 +2459,7 @@ struct filename { static_assert(offsetof(struct filename, iname) % sizeof(long) == 0); static_assert(sizeof(struct filename) % 64 == 0); -static inline struct mnt_idmap *file_mnt_idmap(const struct file *file) +static inline const struct mnt_idmap *file_mnt_idmap(const struct file *file) { return mnt_idmap(file->f_path.mnt); } @@ -2483,7 +2483,7 @@ static inline bool is_idmapped_mnt(const struct vfsmount *mnt) } int vfs_truncate(const struct path *, loff_t); -int do_truncate(struct mnt_idmap *, struct dentry *, loff_t start, +int do_truncate(const struct mnt_idmap *, struct dentry *, loff_t start, unsigned int time_attrs, struct file *filp); extern int vfs_fallocate(struct file *file, int mode, loff_t offset, loff_t len); @@ -2707,10 +2707,10 @@ static inline int bmap(struct inode *inode, sector_t *block) } #endif -int notify_change(struct mnt_idmap *, struct dentry *, +int notify_change(const struct mnt_idmap *, struct dentry *, struct iattr *, struct delegated_inode *); -int inode_permission(struct mnt_idmap *, struct inode *, int); -int generic_permission(struct mnt_idmap *, struct inode *, int); +int inode_permission(const struct mnt_idmap *, struct inode *, int); +int generic_permission(const struct mnt_idmap *, struct inode *, int); static inline int file_permission(struct file *file, int mask) { return inode_permission(file_mnt_idmap(file), @@ -2721,12 +2721,12 @@ static inline int path_permission(const struct path *path, int mask) return inode_permission(mnt_idmap(path->mnt), d_inode(path->dentry), mask); } -int __check_sticky(struct mnt_idmap *idmap, struct inode *dir, +int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode); -int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir, +int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *victim, bool isdir); -int may_create_dentry(struct mnt_idmap *idmap, +int may_create_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *child); static inline bool execute_ok(struct inode *inode) @@ -3045,9 +3045,9 @@ static inline struct inode *new_inode_pseudo(struct super_block *sb) } extern struct inode *new_inode(struct super_block *sb); extern void free_inode_nonrcu(struct inode *inode); -extern int setattr_should_drop_suidgid(struct mnt_idmap *, struct inode *); +extern int setattr_should_drop_suidgid(const struct mnt_idmap *, struct inode *); extern int file_remove_privs(struct file *); -int setattr_should_drop_sgid(struct mnt_idmap *idmap, +int setattr_should_drop_sgid(const struct mnt_idmap *idmap, const struct inode *inode); /* @@ -3204,7 +3204,7 @@ extern int page_symlink(struct inode *inode, const char *symname, int len); extern const struct inode_operations page_symlink_inode_operations; extern void kfree_link(void *); void fill_mg_cmtime(struct kstat *stat, u32 request_mask, struct inode *inode); -void generic_fillattr(struct mnt_idmap *, u32, struct inode *, struct kstat *); +void generic_fillattr(const struct mnt_idmap *, u32, struct inode *, struct kstat *); void generic_fill_statx_attr(struct inode *inode, struct kstat *stat); void generic_fill_statx_atomic_writes(struct kstat *stat, unsigned int unit_min, @@ -3261,9 +3261,9 @@ extern int dcache_dir_open(struct inode *, struct file *); extern int dcache_dir_close(struct inode *, struct file *); extern loff_t dcache_dir_lseek(struct file *, loff_t, int); extern int dcache_readdir(struct file *, struct dir_context *); -extern int simple_setattr(struct mnt_idmap *, struct dentry *, +extern int simple_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); -extern int simple_getattr(struct mnt_idmap *, const struct path *, +extern int simple_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern int simple_statfs(struct dentry *, struct kstatfs *); extern int simple_open(struct inode *inode, struct file *file); @@ -3276,7 +3276,7 @@ void simple_rename_timestamp(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); extern int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); -extern int simple_rename(struct mnt_idmap *, struct inode *, +extern int simple_rename(const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); extern void simple_recursive_removal(struct dentry *, @@ -3397,11 +3397,11 @@ static inline bool generic_ci_validate_strict_name(struct inode *dir, } #endif -int may_setattr(struct mnt_idmap *idmap, struct inode *inode, +int may_setattr(const struct mnt_idmap *idmap, struct inode *inode, unsigned int ia_valid); -int setattr_prepare(struct mnt_idmap *, struct dentry *, struct iattr *); +int setattr_prepare(const struct mnt_idmap *, struct dentry *, struct iattr *); extern int inode_newsize_ok(const struct inode *, loff_t offset); -void setattr_copy(struct mnt_idmap *, struct inode *inode, +void setattr_copy(const struct mnt_idmap *, struct inode *inode, const struct iattr *attr); extern int file_update_time(struct file *file); @@ -3578,7 +3578,7 @@ static inline bool is_sxid(umode_t mode) return mode & (S_ISUID | S_ISGID); } -static inline int check_sticky(struct mnt_idmap *idmap, +static inline int check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode) { if (!(dir->i_mode & S_ISVTX)) @@ -3653,23 +3653,6 @@ extern int vfs_fadvise(struct file *file, loff_t offset, loff_t len, extern int generic_fadvise(struct file *file, loff_t offset, loff_t len, int advice); -static inline bool vfs_empty_path(int dfd, const char __user *path) -{ - char c; - - if (dfd < 0) - return false; - - /* We now allow NULL to be used for empty path. */ - if (!path) - return true; - - if (unlikely(get_user(c, path))) - return false; - - return !c; -} - int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter); static inline bool extensible_ioctl_valid(unsigned int cmd_a, diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h index 0d6c8a6d7be2..c920aba5177c 100644 --- a/include/linux/fs_context.h +++ b/include/linux/fs_context.h @@ -150,6 +150,10 @@ extern int vfs_parse_fs_param_source(struct fs_context *fc, struct fs_parameter *param); extern void fc_drop_locked(struct fs_context *fc); +extern int get_tree_super(struct fs_context *fc, + int (*test)(struct super_block *, struct fs_context *), + int (*fill_super)(struct super_block *sb, + struct fs_context *fc)); extern int get_tree_nodev(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)); diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h index 4c91a019972b..ee524c863fa9 100644 --- a/include/linux/fscache-cache.h +++ b/include/linux/fscache-cache.h @@ -67,7 +67,7 @@ struct fscache_cache_ops { /* Change the size of a data object */ void (*resize_cookie)(struct netfs_cache_resources *cres, - loff_t new_size); + uoff_t new_size); /* Invalidate an object */ bool (*invalidate_cookie)(struct fscache_cookie *cookie); diff --git a/include/linux/fscache.h b/include/linux/fscache.h index 58fdb9605425..f2d958bd1f48 100644 --- a/include/linux/fscache.h +++ b/include/linux/fscache.h @@ -112,7 +112,7 @@ struct fscache_cookie { struct list_head proc_link; /* Link in proc list */ struct list_head commit_link; /* Link in commit queue */ struct work_struct work; /* Commit/relinq/withdraw work */ - loff_t object_size; /* Size of the netfs object */ + uoff_t object_size; /* Size of the netfs object */ unsigned long unused_at; /* Time at which unused (jiffies) */ unsigned long flags; #define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */ @@ -147,6 +147,23 @@ struct fscache_cookie { }; }; +enum fscache_extent_type { + FSCACHE_EXTENT_DATA, + FSCACHE_EXTENT_ZERO, +} __mode(byte); + +/* + * Cache occupancy information. + */ +struct fscache_occupancy { + unsigned long long query_from; /* Point to query from */ + unsigned long long query_to; /* Point to query to */ + unsigned long long cached_from[2]; /* Point at which cache extents start */ + unsigned long long cached_to[2]; /* Point at which cache extents end */ + unsigned int granularity; /* Granularity desired */ + enum fscache_extent_type cached_type[2]; /* Type of cache extent */ +}; + /* * slow-path functions for when there is actually caching available, and the * netfs does actually have a valid token @@ -163,22 +180,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie( u8, const void *, size_t, const void *, size_t, - loff_t); + uoff_t); extern void __fscache_use_cookie(struct fscache_cookie *, bool); -extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *); +extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *); extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool); -extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t); -extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int); +extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t); +extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int); extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *); extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *); void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond); -extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t); +extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t); /** * fscache_acquire_volume - Register a volume as desiring caching services @@ -249,7 +266,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { if (!fscache_volume_valid(volume)) return NULL; @@ -286,7 +303,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie, */ static inline void fscache_unuse_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_valid(cookie)) __fscache_unuse_cookie(cookie, aux_data, object_size); @@ -327,7 +344,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie) */ static inline void fscache_update_aux(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { void *p = fscache_get_aux(cookie); @@ -343,7 +360,7 @@ extern atomic_t fscache_n_updates; static inline void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { #ifdef CONFIG_FSCACHE_STATS atomic_inc(&fscache_n_updates); @@ -369,7 +386,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data */ static inline void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_enabled(cookie)) __fscache_update_cookie(cookie, aux_data, object_size); @@ -386,7 +403,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, * description. */ static inline -void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { if (fscache_cookie_enabled(cookie)) __fscache_resize_cookie(cookie, new_size); @@ -413,7 +430,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) */ static inline void fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t size, unsigned int flags) + const void *aux_data, uoff_t size, unsigned int flags) { if (fscache_cookie_enabled(cookie)) __fscache_invalidate(cookie, aux_data, size, flags); @@ -502,7 +519,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres) */ static inline int fscache_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -561,7 +578,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres, */ static inline int fscache_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -581,7 +598,7 @@ int fscache_write(struct netfs_cache_resources *cres, * waiting. */ static inline void fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len, + uoff_t start, size_t len, bool caching) { if (caching) @@ -615,7 +632,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping, */ static inline void fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool caching) diff --git a/include/linux/ftrace.h b/include/linux/ftrace.h index 02bc5027523a..bd76a16a63af 100644 --- a/include/linux/ftrace.h +++ b/include/linux/ftrace.h @@ -866,8 +866,9 @@ unsigned long ftrace_get_addr_new(struct dyn_ftrace *rec); unsigned long ftrace_get_addr_curr(struct dyn_ftrace *rec); extern ftrace_func_t ftrace_trace_function; +struct trace_array; -int ftrace_regex_open(struct ftrace_ops *ops, int flag, +int ftrace_regex_open(struct trace_array *tr, struct ftrace_ops *ops, int flag, struct inode *inode, struct file *file); ssize_t ftrace_filter_write(struct file *file, const char __user *ubuf, size_t cnt, loff_t *ppos); @@ -1077,7 +1078,7 @@ static inline unsigned long ftrace_location(unsigned long ip) * have them defined when ftrace is not enabled, but these * functions may still be called. Use a macro instead of inline. */ -#define ftrace_regex_open(ops, flag, inod, file) ({ -ENODEV; }) +#define ftrace_regex_open(tr, ops, flag, inode, file) ({ -ENODEV; }) #define ftrace_set_early_filter(ops, buf, enable) do { } while (0) #define ftrace_set_filter_ip(ops, ip, remove, reset) ({ -ENODEV; }) #define ftrace_set_filter_ips(ops, ips, cnt, remove, reset) ({ -ENODEV; }) diff --git a/include/linux/generic_pt/common.h b/include/linux/generic_pt/common.h index 07ef1c8341a4..1b39b27f0cfa 100644 --- a/include/linux/generic_pt/common.h +++ b/include/linux/generic_pt/common.h @@ -213,4 +213,10 @@ enum { PT_FEAT_X86_64_AMD_ENCRYPT_TABLES = PT_FEAT_FMT_START, }; +struct pt_bcm2712 { + struct pt_common common; + u8 bigpage_lg2; + u8 superpage_lg2; +}; + #endif diff --git a/include/linux/generic_pt/iommu.h b/include/linux/generic_pt/iommu.h index dd0edd02a48a..13ebb72a67de 100644 --- a/include/linux/generic_pt/iommu.h +++ b/include/linux/generic_pt/iommu.h @@ -346,6 +346,18 @@ struct pt_iommu_x86_64_hw_info { IOMMU_FORMAT(x86_64, x86_64_pt); +struct pt_iommu_bcm2712_cfg { + struct pt_iommu_cfg common; + u8 bigpage_lg2; + u8 superpage_lg2; +}; + +struct pt_iommu_bcm2712_hw_info { + phys_addr_t pt_base; +}; + +IOMMU_FORMAT(bcm2712, bcm2712pt); + #undef IOMMU_PROTOTYPES #undef IOMMU_FORMAT #endif diff --git a/include/linux/gfp.h b/include/linux/gfp.h index 872bc53f32ec..aa4601805f21 100644 --- a/include/linux/gfp.h +++ b/include/linux/gfp.h @@ -329,7 +329,7 @@ void *alloc_pages_exact_noprof(size_t size, gfp_t gfp_mask) __alloc_size(1); void free_pages_exact(void *virt, size_t size); -__meminit void *alloc_pages_exact_nid_noprof(int nid, size_t size, gfp_t gfp_mask) __alloc_size(2); +void *alloc_pages_exact_nid_noprof(int nid, size_t size, gfp_t gfp_mask) __alloc_size(2); #define alloc_pages_exact_nid(...) \ alloc_hooks(alloc_pages_exact_nid_noprof(__VA_ARGS__)) diff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h index 190191411009..bfd4c43ed777 100644 --- a/include/linux/gfp_types.h +++ b/include/linux/gfp_types.h @@ -244,7 +244,7 @@ enum { * definitely preferable to use the flag rather than opencode endless * loop around allocator. * Allocating pages from the buddy with __GFP_NOFAIL and order > 1 is - * not supported. Please consider using kvmalloc() instead. + * discouraged. Please consider using kvmalloc() instead if possible. */ #define __GFP_IO ((__force gfp_t)___GFP_IO) #define __GFP_FS ((__force gfp_t)___GFP_FS) diff --git a/include/linux/gpu_buddy.h b/include/linux/gpu_buddy.h index 2c36124bb696..ddc4eca84175 100644 --- a/include/linux/gpu_buddy.h +++ b/include/linux/gpu_buddy.h @@ -43,8 +43,8 @@ /** * GPU_BUDDY_CLEAR_ALLOCATION - Prefer pre-cleared (zeroed) memory * - * Attempt to allocate from the clear tree first. If insufficient clear - * memory is available, falls back to dirty memory. Useful when the + * Attempt to allocate outside dirty-tracked ranges first. If insufficient + * clear memory is available, falls back to dirty memory. Useful when the * caller needs zeroed memory and wants to avoid GPU clear operations. */ #define GPU_BUDDY_CLEAR_ALLOCATION BIT(3) @@ -53,8 +53,8 @@ * GPU_BUDDY_CLEARED - Mark returned blocks as cleared * * Used with gpu_buddy_free_list() to indicate that the memory being - * freed has been cleared (zeroed). The blocks will be placed in the - * clear tree for future GPU_BUDDY_CLEAR_ALLOCATION requests. + * freed has been cleared (zeroed). The blocks will be removed from the + * dirty tracker for future GPU_BUDDY_CLEAR_ALLOCATION requests. */ #define GPU_BUDDY_CLEARED BIT(4) @@ -67,15 +67,17 @@ */ #define GPU_BUDDY_TRIM_DISABLE BIT(5) -enum gpu_buddy_free_tree { - GPU_BUDDY_CLEAR_TREE = 0, - GPU_BUDDY_DIRTY_TREE, - GPU_BUDDY_MAX_FREE_TREES, +/* + * Clear/dirty state of a free block. Ordered so a numerically larger value + * is "more clear" (DIRTY < MIXED < CLEAR) which lets subtree_block_state be + * maintained as a simple max-augment over the per-order free tree. + */ +enum gpu_block_state { + GPU_BLOCK_DIRTY = 0, + GPU_BLOCK_MIXED = 1, + GPU_BLOCK_CLEAR = 2, }; -#define for_each_free_tree(tree) \ - for ((tree) = 0; (tree) < GPU_BUDDY_MAX_FREE_TREES; (tree)++) - /** * struct gpu_buddy_block - Block within a buddy allocator * @@ -103,6 +105,13 @@ struct gpu_buddy_block { #define GPU_BUDDY_ALLOCATED (1 << 10) #define GPU_BUDDY_FREE (2 << 10) #define GPU_BUDDY_SPLIT (3 << 10) +/* + * GPU_BUDDY_HEADER_CLEAR has two roles: + * - FREE state: set when the block's full range is cleared (dirty + * tracker confirmed no overlap). + * - ALLOCATED state: set when the block was served from cleared memory, + * informing the caller that no GPU clear pass is needed. + */ #define GPU_BUDDY_HEADER_CLEAR GENMASK_ULL(9, 9) /* Free to be used, if needed in the future */ #define GPU_BUDDY_HEADER_UNUSED GENMASK_ULL(8, 6) @@ -128,14 +137,45 @@ struct gpu_buddy_block { struct list_head link; }; /* private: */ - struct list_head tmp_link; + enum gpu_block_state subtree_block_state; unsigned int subtree_max_alignment; + struct list_head tmp_link; + bool has_clear; }; /* Order-zero must be at least SZ_4K */ #define GPU_BUDDY_MAX_ORDER (63 - 12) /** + * struct gpu_dirty_extent - a contiguous dirty address range + * + * Tracks a single contiguous address range whose memory content is known + * to be dirty. Extents are non-overlapping and stored in an augmented + * red-black tree sorted by @start. The augmented value @subtree_max_size + * allows O(log N) search for an extent of at least a given size. + */ +struct gpu_dirty_extent { +/* private: */ + struct rb_node rb; + u64 start; + u64 end; + u64 subtree_max_size; +}; + +/** + * struct gpu_dirty_tracker - tracks dirty address intervals + * + * Maintains a set of non-overlapping dirty extents as an augmented + * red-black tree. + */ +struct gpu_dirty_tracker { +/* private: */ + struct rb_root root; + /* Total bytes of dirty memory currently tracked. */ + u64 total_dirty; +}; + +/** * struct gpu_buddy - GPU binary buddy allocator * * The buddy allocator provides efficient power-of-two memory allocation @@ -152,20 +192,21 @@ struct gpu_buddy_block { * @chunk_size: Minimum allocation granularity in bytes. Must be at least SZ_4K. * @size: Total size of the address space managed by this allocator in bytes. * @avail: Total free space currently available for allocation in bytes. - * @clear_avail: Free space available in the clear tree (zeroed memory) in bytes. - * This is a subset of @avail. * @lock_dep_map: Annotates gpu_buddy API with a driver provided lock. */ struct gpu_buddy { /* private: */ + /* Tracker of dirty address ranges (decoupled from free_tree). */ + struct gpu_dirty_tracker dirty; /* - * Array of red-black trees for free block management. - * Indexed as free_trees[clear/dirty][order] where: - * - Index 0 (GPU_BUDDY_CLEAR_TREE): blocks with zeroed content - * - Index 1 (GPU_BUDDY_DIRTY_TREE): blocks with unknown content - * Each tree holds free blocks of the corresponding order. + * One RB-tree per order containing all free blocks (clear and + * dirty alike). The augment field subtree_block_state (a max over + * the subtree of each block's state) lets clear allocations + * find the right-most fully-clear or mixed block in O(log N). + * Dirty free blocks coexist here but are also indexed by the + * @dirty tracker for fast dirty allocation lookups. */ - struct rb_root **free_trees; + struct rb_root *free_tree; /* * Array of root blocks representing the top-level blocks of the * binary tree(s). Multiple roots exist when the total size is not @@ -194,7 +235,6 @@ struct gpu_buddy { u64 chunk_size; u64 size; u64 avail; - u64 clear_avail; #ifdef CONFIG_LOCKDEP struct lockdep_map *lock_dep_map; #endif @@ -226,17 +266,32 @@ struct gpu_buddy { * * Ensure driver lock is held. */ -static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm) +static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm) { if (mm->lock_dep_map) lockdep_assert(lock_is_held_type(mm->lock_dep_map, 0)); } #else -static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm) +static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm) { } #endif +/** + * gpu_buddy_clear_avail - free space that is clear (zeroed), in bytes + * @mm: gpu buddy allocator + * + * A subset of @mm->avail. Derived on demand as @mm->avail minus the bytes + * the dirty tracker records as dirty, so it is always consistent with the + * tracker without a cached field to keep in sync. Zero for a fresh pool, + * which is fully dirty. + */ +static inline u64 gpu_buddy_clear_avail(const struct gpu_buddy *mm) +{ + gpu_buddy_driver_lock_held(mm); + return mm->avail - mm->dirty.total_dirty; +} + static inline u64 gpu_buddy_block_offset(const struct gpu_buddy_block *block) { diff --git a/include/linux/hazptr.h b/include/linux/hazptr.h new file mode 100644 index 000000000000..d1670121947a --- /dev/null +++ b/include/linux/hazptr.h @@ -0,0 +1,325 @@ +// SPDX-License-Identifier: LGPL-2.1-or-later +// +// SPDX-FileCopyrightText: 2024 Mathieu Desnoyers <mathieu.desnoyers@efficios.com> + +#ifndef _LINUX_HAZPTR_H +#define _LINUX_HAZPTR_H + +/* + * hazptr: Hazard Pointers + * + * This API provides existence guarantees of objects through hazard + * pointers. + * + * Its main benefit over RCU is that it allows fast reclaim of + * HP-protected pointers without needing to wait for a grace period. + * + * References: + * + * [1]: M. M. Michael, "Hazard pointers: safe memory reclamation for + * lock-free objects," in IEEE Transactions on Parallel and + * Distributed Systems, vol. 15, no. 6, pp. 491-504, June 2004 + */ + +#include <linux/percpu.h> +#include <linux/types.h> +#include <linux/cleanup.h> +#include <linux/sched.h> + +/* 4 slots (each sizeof(hazptr_slot_item)) fit in a single 64-byte cache line. */ +#define NR_HAZPTR_PERCPU_SLOTS 4 + +/* The current hazard pointer wildcard. */ +extern void *hazptr_wildcard; + +/* + * Hazard pointer slot. + */ +struct hazptr_slot { + void *addr; +}; + +struct hazptr_overflow_list; + +struct hazptr_backup_slot { + struct hlist_node overflow_node; + struct hazptr_slot slot; + /* Overflow list where the backup slot is added. */ + struct hazptr_overflow_list *overflow_list; +}; + +struct hazptr_ctx { + struct hazptr_slot *slot; + /* Backup slot in case all per-CPU slots are used. */ + struct hazptr_backup_slot backup_slot; + struct hlist_node preempt_node; +#ifdef CONFIG_HAZPTR_DEBUG + bool detach_task, detach_cpu; /* Whether the ctx has been detached from task/cpu. */ + int acquire_pid, acquire_cpu; /* Note the task and cpu number at acquire. */ + unsigned long acquire_caller; /* Acquire instruction pointer. */ +#endif +}; + +struct hazptr_slot_ctx { + struct hazptr_ctx *ctx; +}; + +struct hazptr_slot_item { + struct hazptr_slot slot; + struct hazptr_slot_ctx ctx; +}; + +struct hazptr_percpu_slots { + struct hazptr_slot_item items[NR_HAZPTR_PERCPU_SLOTS]; +} ____cacheline_aligned; + +DECLARE_PER_CPU(struct hazptr_percpu_slots, hazptr_percpu_slots); + +void *__hazptr_acquire(struct hazptr_ctx *ctx, void * const *addr_p); + +/** + * hazptr_synchronize: Wait for release from hazard-pointer protection + * + * @addr: The address to be released from hazard-pointer protection + * + * Wait for the specified @addr to be released from protection from all + * hazard pointers. The caller should make @addr inaccessible to all + * hazard-pointer readers before invoking this function. + * + * Must be called from preemptible context. + */ +void hazptr_synchronize(void *addr); + +/* + * hazptr_chain_backup_slot: Chain backup slot into overflow list. + * + * Set backup slot address to @addr, and chain it into the overflow + * list. + */ +struct hazptr_slot *hazptr_chain_backup_slot(struct hazptr_ctx *ctx); + +/* + * hazptr_unchain_backup_slot: Unchain backup slot from overflow list. + */ +void hazptr_unchain_backup_slot(struct hazptr_ctx *ctx); + +static inline +bool hazptr_slot_is_backup(struct hazptr_ctx *ctx, struct hazptr_slot *slot) +{ + return slot == &ctx->backup_slot.slot; +} + +/* Internal helper. */ +static inline +void hazptr_promote_to_backup_slot(struct hazptr_ctx *ctx, struct hazptr_slot *slot) +{ + struct hazptr_slot *backup_slot; + + backup_slot = hazptr_chain_backup_slot(ctx); + /* + * Move hazard pointer from the per-CPU slot to the + * backup slot. This requires hazard pointer + * synchronize to iterate on per-CPU slots with + * load-acquire before iterating on the overflow list. + */ + WRITE_ONCE(backup_slot->addr, slot->addr); + /* + * store-release orders store to backup slot addr before + * store to per-CPU slot addr. + */ + smp_store_release(&slot->addr, NULL); + /* Use the backup slot for context. */ + ctx->slot = backup_slot; +} + +/** + * hazptr_detach - Allow a hazard pointer to be released in some other context + * + * @ctx: The hazard-pointer context to be detached. + * + * By default, a given hazptr_acquire() and the corresponding + * hazptr_release() must run in a single execution context, for example, + * the context of a single task or a single interrupt handler. When you + * have acquired a hazard pointer in one context and need to release it + * in another, you must invoke hazptr_detach() on that hazard pointer's + * context. It is permissible to invoke hazptr_detach() multiple times + * on the same @ctx while it is protecting the same pointer, however, + * the first invocation absolutely must be in the same context that did + * the hazptr_acquire(), and must take place after the return from that + * hazptr_acquire(). + * + * For example, if a hazard pointer is acquired by a task and released + * by a timer handler, that task would need to pass the hazard pointer's + * context to hazptr_detach() after return from the hazptr_acquire() and + * before arming the timer (or at least before the handler had a chance + * to access that hazard-pointer context). + */ +static inline +void hazptr_detach(struct hazptr_ctx *ctx) +{ + struct hazptr_slot *slot; + + guard(preempt)(); + slot = ctx->slot; + if (!slot->addr) + return; +#ifdef CONFIG_HAZPTR_DEBUG + ctx->detach_task = ctx->detach_cpu = true; +#endif + if (unlikely(hazptr_slot_is_backup(ctx, slot))) + return; + hazptr_promote_to_backup_slot(ctx, slot); +} + +static inline +void hazptr_note_context_switch(void) +{ + struct hazptr_percpu_slots *percpu_slots = this_cpu_ptr(&hazptr_percpu_slots); + unsigned int idx; + + for (idx = 0; idx < NR_HAZPTR_PERCPU_SLOTS; idx++) { + struct hazptr_slot_item *item = &percpu_slots->items[idx]; + struct hazptr_slot *slot = &item->slot; + + if (!slot->addr) + continue; +#ifdef CONFIG_HAZPTR_DEBUG + item->ctx.ctx->detach_cpu = true; +#endif + hazptr_promote_to_backup_slot(item->ctx.ctx, slot); + } +} + +/** + * hazptr_acquire - Load pointer at address and protect with hazard pointer. + * + * @ctx: The hazard-pointer context to be passed to hazptr_release(). + * @addr_p: Pointer to the pointer that is to be hazard-pointer protected. + * + * Load @addr_p, and protect the loaded pointer with hazard pointer. + * This protection is roughly similar to (but way faster than) that of a + * reference counter, and ends with a later call to hazptr_release(). + * + * This protection is unconditional, and has limitations similar to + * that of unconditional reference-counter acquisition. In particular, + * although holding a hazard pointer prevents a hazard-pointer-protected + * object from being freed, it does not prevent that object from being + * removed from a linked data structure, and does not prevent other + * hazard-pointer-protected objects referenced by this object from being + * both removed and freed. At which point, invoking hazptr_acquire() + * on these dangling pointers would be a bug. On the other hand, use of + * hazptr_acquire() is safe for immortal pointers to objects that do not + * themselves contain pointers to hazard-pointer-protected objects. + * Other (more complex) use cases are also possible. + * + * By default, the call to hazptr_release() must be running in the same + * execution context as the corresponding hazptr_acquire(), for example, + * within the same task or interrupt handler. When it is necessary to + * instead call hazptr_release() from some other context, pass @ctx to + * hazptr_detach() in the original context after invoking hazptr_acquire() + * but before making the hazard pointer available to that other context. + * + * It is not permissible to invoke hazptr_acquire() twice on the same @ctx + * without an intervening hazptr_release(). + * + * Returns a non-NULL protected address if the loaded pointer is non-NULL. + * Returns NULL if the loaded pointer is NULL. + * + * On success the protected hazptr slot is stored in @ctx->slot. + */ +static inline +void *hazptr_acquire(struct hazptr_ctx *ctx, void * const *addr_p) +{ + struct hazptr_percpu_slots *percpu_slots; + struct hazptr_slot_item *slot_item; + struct hazptr_slot *slot; + void *addr; + + guard(preempt)(); + percpu_slots = this_cpu_ptr(&hazptr_percpu_slots); + slot_item = &percpu_slots->items[0]; + slot = &slot_item->slot; +#ifdef CONFIG_HAZPTR_DEBUG + ctx->detach_cpu = ctx->detach_task = false; + ctx->acquire_pid = current->pid; + ctx->acquire_cpu = smp_processor_id(); + ctx->acquire_caller = _THIS_IP_; +#endif + if (unlikely(slot->addr)) + return __hazptr_acquire(ctx, addr_p); + WRITE_ONCE(slot->addr, READ_ONCE(hazptr_wildcard)); /* Store B */ + + /* Memory ordering: Store B before Load A. */ + smp_mb(); + + /* + * Load @addr_p after storing wildcard to the hazard pointer slot. + */ + addr = READ_ONCE(*addr_p); /* Load A */ + + /* + * We don't care about ordering of Store C. It will simply + * replace the wildcard by a more specific address. If addr is + * NULL, we simply store NULL into the slot. + */ + WRITE_ONCE(slot->addr, addr); /* Store C */ + slot_item->ctx.ctx = ctx; + ctx->slot = slot; + return addr; +} + +#ifdef CONFIG_HAZPTR_DEBUG +/* Called with preemption disabled. */ +static inline +void hazptr_release_debug(struct hazptr_ctx *ctx, void *addr) +{ + int pid = current->pid, cpu = smp_processor_id(); + bool warn_remote_cpu = !ctx->detach_cpu && ctx->acquire_cpu != cpu, + warn_remote_task = !ctx->detach_task && ctx->acquire_pid != pid; + + WARN_ONCE(warn_remote_cpu || warn_remote_task, + "Hazard Pointer (addr=%p) released on remote %s without %s. Acquire: caller=%pS, pid=%d, cpu=%d. Release: pid=%d, cpu=%d.", + addr, + warn_remote_task ? "task" : "cpu", + warn_remote_task ? "being detached from task" : "context switch", + (void *) ctx->acquire_caller, ctx->acquire_pid, ctx->acquire_cpu, pid, cpu); +} +#else +static inline void hazptr_release_debug(struct hazptr_ctx *ctx, void *addr) { } +#endif + +/** + * hazptr_release - Release the specified hazard pointer + * + * @ctx: The hazard-pointer context that was passed to hazptr_acquire(). + * @addr_p: The pointer that is to be hazard-pointer unprotected. + * + * Release the protected hazard pointer recorded in @ctx. + * + * By default, hazptr_release() must execute in the same execution context + * that invoked the corresponding hazptr_acquire(), for example, within the + * same task or the same interrupt handler. However, if this restriction + * is problematic for your use case, please see hazptr_detach(). + * + * It is permissible (though unwise from a maintainability viewpoint) + * to invoke hazptr_release() twice on the same @ctx without an intervening + * hazptr_acquire(). + */ +static inline +void hazptr_release(struct hazptr_ctx *ctx, void *addr) +{ + struct hazptr_slot *slot; + + if (!addr) + return; + guard(preempt)(); + hazptr_release_debug(ctx, addr); + slot = ctx->slot; + smp_store_release(&slot->addr, NULL); + if (unlikely(hazptr_slot_is_backup(ctx, slot))) + hazptr_unchain_backup_slot(ctx); +} + +void hazptr_init(void); + +#endif /* _LINUX_HAZPTR_H */ diff --git a/include/linux/hdmi.h b/include/linux/hdmi.h index 8dab78e1f61b..b80a5ee63bb2 100644 --- a/include/linux/hdmi.h +++ b/include/linux/hdmi.h @@ -27,6 +27,18 @@ #include <linux/types.h> #include <linux/device.h> +enum hdmi_version { + HDMI_VERSION_UNKNOWN, + HDMI_VERSION_1_0, + HDMI_VERSION_1_1, + HDMI_VERSION_1_2, + HDMI_VERSION_1_3, + HDMI_VERSION_1_4, + HDMI_VERSION_2_0, + HDMI_VERSION_2_1, + HDMI_VERSION_2_2, +}; + enum hdmi_packet_type { HDMI_PACKET_TYPE_NULL = 0x00, HDMI_PACKET_TYPE_AUDIO_CLOCK_REGEN = 0x01, diff --git a/include/linux/hiddev.h b/include/linux/hiddev.h index 2164c03d2c72..8e9f8a33e359 100644 --- a/include/linux/hiddev.h +++ b/include/linux/hiddev.h @@ -13,6 +13,7 @@ #ifndef _HIDDEV_H #define _HIDDEV_H +#include <linux/refcount.h> #include <uapi/linux/hiddev.h> @@ -24,6 +25,7 @@ struct hiddev { int minor; int exist; int open; + refcount_t refcount; struct mutex existancelock; wait_queue_head_t wait; struct hid_device *hid; diff --git a/include/linux/hrtimer.h b/include/linux/hrtimer.h index 29072d89e5cb..cad8482337cb 100644 --- a/include/linux/hrtimer.h +++ b/include/linux/hrtimer.h @@ -72,7 +72,7 @@ enum hrtimer_mode { */ struct hrtimer_sleeper { struct hrtimer timer; - struct task_struct *task; + struct task_struct *__private task; }; static inline void hrtimer_set_expires(struct hrtimer *timer, ktime_t time) @@ -320,6 +320,14 @@ extern int schedule_hrtimeout_range_clock(ktime_t *expires, const enum hrtimer_mode mode, clockid_t clock_id); extern int schedule_hrtimeout(ktime_t *expires, const enum hrtimer_mode mode); +static inline struct task_struct *hrtimer_sleeper_task_get(struct hrtimer_sleeper *sl) +{ + return READ_ONCE(ACCESS_PRIVATE(sl, task)); +} +static inline void hrtimer_sleeper_task_set(struct hrtimer_sleeper *sl, struct task_struct *t) +{ + WRITE_ONCE(ACCESS_PRIVATE(sl, task), t); +} /* Soft interrupt function to run the hrtimer queues: */ extern void hrtimer_run_queues(void); diff --git a/include/linux/hrtimer_rearm.h b/include/linux/hrtimer_rearm.h index 17a81826bd9a..12e42721c8e0 100644 --- a/include/linux/hrtimer_rearm.h +++ b/include/linux/hrtimer_rearm.h @@ -45,7 +45,7 @@ hrtimer_rearm_deferred_user_irq(unsigned long *tif_work, const unsigned long tif clear_thread_flag(TIF_HRTIMER_REARM); __hrtimer_rearm_deferred(); /* Don't go into the loop if HRTIMER_REARM was the only flag */ - *tif_work &= ~TIF_HRTIMER_REARM; + *tif_work &= ~_TIF_HRTIMER_REARM; return !*tif_work; } return false; diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h index c745f7ad2298..8ca0fa3be2ac 100644 --- a/include/linux/huge_mm.h +++ b/include/linux/huge_mm.h @@ -510,8 +510,6 @@ change_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma, int hugepage_madvise(struct vm_area_struct *vma, vm_flags_t *vm_flags, int advice); -int madvise_collapse(struct vm_area_struct *vma, unsigned long start, - unsigned long end, bool *lock_dropped); void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start, unsigned long end, struct vm_area_struct *next); spinlock_t *__pmd_trans_huge_lock(pmd_t *pmd, struct vm_area_struct *vma); @@ -715,13 +713,6 @@ static inline int hugepage_madvise(struct vm_area_struct *vma, return -EINVAL; } -static inline int madvise_collapse(struct vm_area_struct *vma, - unsigned long start, - unsigned long end, bool *lock_dropped) -{ - return -EINVAL; -} - static inline void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start, unsigned long end, diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 16c4c4caa126..5029c7241863 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -7,11 +7,11 @@ #include <linux/mm_types.h> #include <linux/mmdebug.h> #include <linux/fs.h> -#include <linux/hugetlb_inline.h> #include <linux/cgroup.h> #include <linux/page_ref.h> #include <linux/list.h> #include <linux/kref.h> +#include <linux/atomic.h> #include <linux/pgtable.h> #include <linux/gfp.h> #include <linux/userfaultfd_k.h> @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio); extern int movable_gigantic_pages __read_mostly; extern int sysctl_hugetlb_shm_group __read_mostly; -extern struct list_head huge_boot_pages[MAX_NUMNODES]; void hugetlb_bootmem_struct_page_init(void); void hugetlb_bootmem_alloc(void); @@ -253,14 +252,14 @@ extern void __hugetlb_zap_end(struct vm_area_struct *vma, static inline void hugetlb_zap_begin(struct vm_area_struct *vma, unsigned long *start, unsigned long *end) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) __hugetlb_zap_begin(vma, start, end); } static inline void hugetlb_zap_end(struct vm_area_struct *vma, struct zap_details *details) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) __hugetlb_zap_end(vma, details); } @@ -509,6 +508,9 @@ struct hugetlbfs_inode_info { struct inode vfs_inode; struct resv_map *resv_map; unsigned int seals; +#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING + atomic64_t pmd_sharing_count; +#endif }; static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode) @@ -516,6 +518,39 @@ static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode) return container_of(inode, struct hugetlbfs_inode_info, vfs_inode); } +#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING +static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) +{ + atomic64_set(&HUGETLBFS_I(inode)->pmd_sharing_count, 0); +} + +static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) +{ + atomic64_inc(&HUGETLBFS_I(inode)->pmd_sharing_count); +} + +static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) +{ + atomic64_dec(&HUGETLBFS_I(inode)->pmd_sharing_count); +} + +static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode) +{ + return atomic64_read(&HUGETLBFS_I(inode)->pmd_sharing_count) != 0; +} +#else +static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) {} + +static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) {} + +static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) {} + +static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode) +{ + return false; +} +#endif + extern const struct vm_operations_struct hugetlb_vm_ops; struct file *hugetlb_file_setup(const char *name, size_t size, vma_flags_t acct, int creat_flags, int page_size_log); @@ -676,10 +711,6 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_HVO 0x0001 -#define HUGE_BOOTMEM_ZONES_VALID 0x0002 -#define HUGE_BOOTMEM_CMA 0x0004 - int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); void wait_for_freed_hugetlb_folios(void); @@ -699,7 +730,8 @@ enum hugetlb_alloc_flag { #define HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS BIT(HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS_BIT) struct folio *hugetlb_alloc_folio(struct hstate *h, - struct mempolicy_interpreted *mpoli, u8 alloc_flags); + struct mempolicy_interpreted *mpoli, struct mm_struct *mm, + u8 alloc_flags); struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, unsigned long addr, bool cow_from_owner); struct folio *alloc_hugetlb_folio_nodemask(struct hstate *h, int preferred_nid, diff --git a/include/linux/hugetlb_inline.h b/include/linux/hugetlb_inline.h deleted file mode 100644 index 5c29cd3223a1..000000000000 --- a/include/linux/hugetlb_inline.h +++ /dev/null @@ -1,28 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#ifndef _LINUX_HUGETLB_INLINE_H -#define _LINUX_HUGETLB_INLINE_H - -#include <linux/mm.h> - -#ifdef CONFIG_HUGETLB_PAGE - -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) -{ - return vma_flags_test(flags, VMA_HUGETLB_BIT); -} - -#else - -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) -{ - return false; -} - -#endif - -static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) -{ - return is_vma_hugetlb_flags(&vma->flags); -} - -#endif diff --git a/include/linux/hwmon.h b/include/linux/hwmon.h index dd713e193d0c..a3a7d27f3b5f 100644 --- a/include/linux/hwmon.h +++ b/include/linux/hwmon.h @@ -134,6 +134,8 @@ enum hwmon_in_attributes { hwmon_in_max, hwmon_in_lcrit, hwmon_in_crit, + hwmon_in_lemergency, + hwmon_in_emergency, hwmon_in_average, hwmon_in_lowest, hwmon_in_highest, @@ -144,6 +146,8 @@ enum hwmon_in_attributes { hwmon_in_max_alarm, hwmon_in_lcrit_alarm, hwmon_in_crit_alarm, + hwmon_in_lemergency_alarm, + hwmon_in_emergency_alarm, hwmon_in_rated_min, hwmon_in_rated_max, hwmon_in_beep, @@ -156,6 +160,8 @@ enum hwmon_in_attributes { #define HWMON_I_MAX BIT(hwmon_in_max) #define HWMON_I_LCRIT BIT(hwmon_in_lcrit) #define HWMON_I_CRIT BIT(hwmon_in_crit) +#define HWMON_I_LEMERGENCY BIT(hwmon_in_lemergency) +#define HWMON_I_EMERGENCY BIT(hwmon_in_emergency) #define HWMON_I_AVERAGE BIT(hwmon_in_average) #define HWMON_I_LOWEST BIT(hwmon_in_lowest) #define HWMON_I_HIGHEST BIT(hwmon_in_highest) @@ -166,6 +172,8 @@ enum hwmon_in_attributes { #define HWMON_I_MAX_ALARM BIT(hwmon_in_max_alarm) #define HWMON_I_LCRIT_ALARM BIT(hwmon_in_lcrit_alarm) #define HWMON_I_CRIT_ALARM BIT(hwmon_in_crit_alarm) +#define HWMON_I_LEMERGENCY_ALARM BIT(hwmon_in_lemergency_alarm) +#define HWMON_I_EMERGENCY_ALARM BIT(hwmon_in_emergency_alarm) #define HWMON_I_RATED_MIN BIT(hwmon_in_rated_min) #define HWMON_I_RATED_MAX BIT(hwmon_in_rated_max) #define HWMON_I_BEEP BIT(hwmon_in_beep) @@ -178,6 +186,7 @@ enum hwmon_curr_attributes { hwmon_curr_max, hwmon_curr_lcrit, hwmon_curr_crit, + hwmon_curr_emergency, hwmon_curr_average, hwmon_curr_lowest, hwmon_curr_highest, @@ -188,6 +197,7 @@ enum hwmon_curr_attributes { hwmon_curr_max_alarm, hwmon_curr_lcrit_alarm, hwmon_curr_crit_alarm, + hwmon_curr_emergency_alarm, hwmon_curr_rated_min, hwmon_curr_rated_max, hwmon_curr_beep, @@ -199,6 +209,7 @@ enum hwmon_curr_attributes { #define HWMON_C_MAX BIT(hwmon_curr_max) #define HWMON_C_LCRIT BIT(hwmon_curr_lcrit) #define HWMON_C_CRIT BIT(hwmon_curr_crit) +#define HWMON_C_EMERGENCY BIT(hwmon_curr_emergency) #define HWMON_C_AVERAGE BIT(hwmon_curr_average) #define HWMON_C_LOWEST BIT(hwmon_curr_lowest) #define HWMON_C_HIGHEST BIT(hwmon_curr_highest) @@ -209,6 +220,7 @@ enum hwmon_curr_attributes { #define HWMON_C_MAX_ALARM BIT(hwmon_curr_max_alarm) #define HWMON_C_LCRIT_ALARM BIT(hwmon_curr_lcrit_alarm) #define HWMON_C_CRIT_ALARM BIT(hwmon_curr_crit_alarm) +#define HWMON_C_EMERGENCY_ALARM BIT(hwmon_curr_emergency_alarm) #define HWMON_C_RATED_MIN BIT(hwmon_curr_rated_min) #define HWMON_C_RATED_MAX BIT(hwmon_curr_rated_max) #define HWMON_C_BEEP BIT(hwmon_curr_beep) diff --git a/include/linux/i2c.h b/include/linux/i2c.h index 14ab4d3055af..6a847fc88d25 100644 --- a/include/linux/i2c.h +++ b/include/linux/i2c.h @@ -174,8 +174,23 @@ i2c_smbus_write_word_swapped(const struct i2c_client *client, } /* Returns the number of read bytes */ -s32 i2c_smbus_read_block_data(const struct i2c_client *client, - u8 command, u8 *values); +s32 __i2c_smbus_read_block_data(const struct i2c_client *client, + u8 command, u8 length, u8 *values); +/* + * This monstrosity allows to call i2c_smbus_read_block_data() with either + * 3 or 4 arguments and will be removed once all users have been switched + * to the 4 argument version. + */ +#define __i2c_smbus_read_block_data_3arg(client, cmd, values) \ + __i2c_smbus_read_block_data(client, cmd, I2C_SMBUS_BLOCK_MAX, values) +#define __i2c_smbus_read_block_data_4arg(client, cmd, length, values) \ + __i2c_smbus_read_block_data(client, cmd, length, values) +#define __i2c_smbus_read_block_data_impl(_1, _2, _3, _4, impl, ...) impl +#define i2c_smbus_read_block_data(client, cmd, varargs...) \ + __i2c_smbus_read_block_data_impl(client, cmd, varargs, \ + __i2c_smbus_read_block_data_4arg, \ + __i2c_smbus_read_block_data_3arg) \ + (client, cmd, varargs) s32 i2c_smbus_write_block_data(const struct i2c_client *client, u8 command, u8 length, const u8 *values); /* Returns the number of read bytes */ @@ -742,6 +757,9 @@ struct i2c_adapter { struct rt_mutex mux_lock; int timeout; /* in jiffies */ +#ifdef CONFIG_I2C_DYNAMIC_TIMEOUT + int user_timeout; /* I2C_TIMEOUT ioctl value in jiffies */ +#endif int retries; struct device dev; /* the adapter device */ unsigned long locked_flags; /* owned by the I2C core */ @@ -913,14 +931,24 @@ unsigned int i2c_adapter_depth(struct i2c_adapter *adapter); void i2c_parse_fw_timings(struct device *dev, struct i2c_timings *t, bool use_defaults); +#ifdef CONFIG_I2C_DYNAMIC_TIMEOUT +void i2c_update_timeout(struct i2c_adapter *adap, u32 bus_freq_hz, + size_t len, unsigned int safety_coeff, + unsigned int min_usec); +#else +static inline void i2c_update_timeout(struct i2c_adapter *adap, u32 bus_freq_hz, + size_t len, unsigned int safety_coeff, + unsigned int min_usec) {} +#endif + /* Return the functionality mask */ static inline u32 i2c_get_functionality(struct i2c_adapter *adap) { return adap->algo->functionality(adap); } -/* Return 1 if adapter supports everything we need, 0 if not. */ -static inline int i2c_check_functionality(struct i2c_adapter *adap, u32 func) +/* Return true if adapter supports everything we need, false if not. */ +static inline bool i2c_check_functionality(struct i2c_adapter *adap, u32 func) { return (func & i2c_get_functionality(adap)) == func; } diff --git a/include/linux/i3c/device.h b/include/linux/i3c/device.h index 0f065b883ee0..e30ea2cb13cb 100644 --- a/include/linux/i3c/device.h +++ b/include/linux/i3c/device.h @@ -79,6 +79,10 @@ struct i3c_xfer { enum i3c_error_code err; }; +/* For splitting 'cmd' from struct i3c_xfer */ +#define I3C_HDR_CMD_RNW BIT(7) +#define I3C_HDR_CMD_CODE GENMASK(6, 0) + /** * enum i3c_dcr - I3C DCR values * @I3C_DCR_GENERIC_DEVICE: generic I3C device diff --git a/include/linux/i3c/hub.h b/include/linux/i3c/hub.h new file mode 100644 index 000000000000..a368ea9e5ef7 --- /dev/null +++ b/include/linux/i3c/hub.h @@ -0,0 +1,92 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Copyright 2026 NXP + * Generic hub definitions and helper interfaces. + */ +#ifndef _LINUX_I3C_HUB_H +#define _LINUX_I3C_HUB_H + +#include <linux/i3c/master.h> +#include <linux/mutex.h> + +/** + * struct i3c_hub - Generic I3C hub context + * @ops: Vendor callbacks for port connection control + * @hub_dev: I3C device representing the hub on the parent bus + * @lock: Serializes hub port routing/forwarding; its lockdep class is keyed + * per hub nesting depth in i3c_hub_init(). + */ +struct i3c_hub { + const struct i3c_hub_ops *ops; + struct i3c_device *hub_dev; + struct mutex lock; /* Serializes hub port routing. */ +}; + +struct i3c_hub_controller { + struct i3c_master_controller *parent; + struct i3c_master_controller controller; + struct i3c_hub *hub; +}; + +struct i3c_hub_ops { + void (*enable_port)(struct i3c_master_controller *controller); + void (*disable_port)(struct i3c_master_controller *controller); +}; + +/** + * i3c_hub_enable_port() - Enable hub connection for a controller + * @controller: Virtual controller representing a hub port + * + * Retrieves hub context from controller drvdata and invokes the vendor + * callback to enable the associated port connection. + */ +static inline void i3c_hub_enable_port(struct i3c_master_controller *controller) +{ + struct i3c_hub_controller *hub_controller; + struct i3c_hub *hub; + + hub_controller = dev_get_drvdata(&controller->dev); + if (!hub_controller || !hub_controller->hub) + return; + + hub = hub_controller->hub; + + if (hub && hub->ops && hub->ops->enable_port) + hub->ops->enable_port(controller); +} + +/** + * i3c_hub_disable_port() - Disable hub connection for a controller + * @controller: Virtual controller representing a hub port + * + * Retrieves hub context from controller drvdata and invokes the vendor + * callback to disable the associated port connection. + */ +static inline void i3c_hub_disable_port(struct i3c_master_controller *controller) +{ + struct i3c_hub_controller *hub_controller; + struct i3c_hub *hub; + + hub_controller = dev_get_drvdata(&controller->dev); + if (!hub_controller || !hub_controller->hub) + return; + + hub = hub_controller->hub; + + if (hub && hub->ops && hub->ops->disable_port) + hub->ops->disable_port(controller); +} + +/* + * Controller operations used by the virtual controllers created for hub + * target ports. Hub drivers pass this to i3c_master_register_fwnode(). + */ +extern const struct i3c_master_controller_ops i3c_hub_master_ops; + +void i3c_hub_init(struct i3c_hub *hub, + const struct i3c_hub_ops *ops, + struct i3c_device *hub_dev); + +int i3c_hub_reserve_parent_addrslots_from_dt(struct i3c_hub_controller *hubc, + struct device_node *node); +#endif diff --git a/include/linux/i3c/master.h b/include/linux/i3c/master.h index f7ceec2b4477..d58070a72a49 100644 --- a/include/linux/i3c/master.h +++ b/include/linux/i3c/master.h @@ -536,6 +536,8 @@ struct i3c_master_controller_ops { * Bit 0: SETDASA * Bit 1: SETAASA * All other bits are reserved. + * @instance: Zero-based instance number of the Bus Controller as defined by the + * DisCo specification I3C Target Address (_ADR) Encoding * @wq: freezable workqueue which can be used by master * drivers if they need to postpone operations that need to take place * in a thread context. Typical examples are Hot Join processing which @@ -573,6 +575,7 @@ struct i3c_master_controller { } boardinfo; struct i3c_bus bus; u8 addr_method; + u8 instance; struct workqueue_struct *wq; struct work_struct hj_work; struct work_struct reg_work; @@ -649,9 +652,18 @@ DEFINE_FREE(i3c_master_dma_unmap_single, void *, int i3c_master_reattach_i3c_dev_locked(struct i3c_dev_desc *dev, u8 old_dyn_addr); +int i3c_master_send_ccc_cmd(struct i3c_master_controller *master, + struct i3c_ccc_cmd *cmd); +bool i3c_master_supports_ccc_cmd(struct i3c_master_controller *master, + const struct i3c_ccc_cmd *cmd); int i3c_master_set_info(struct i3c_master_controller *master, const struct i3c_device_info *info); +int i3c_master_register_fwnode(struct i3c_master_controller *master, + struct device *parent, + struct fwnode_handle *fwnode, + const struct i3c_master_controller_ops *ops, + bool secondary); int i3c_master_register(struct i3c_master_controller *master, struct device *parent, const struct i3c_master_controller_ops *ops, @@ -775,4 +787,12 @@ void i3c_for_each_bus_locked(int (*fn)(struct i3c_bus *bus, void *data), int i3c_register_notifier(struct notifier_block *nb); int i3c_unregister_notifier(struct notifier_block *nb); +enum i3c_addr_slot_status +i3c_bus_get_addr_slot_status(struct i3c_bus *bus, u16 addr); + +void i3c_bus_set_addr_slot_status(struct i3c_bus *bus, u16 addr, + enum i3c_addr_slot_status status); + +void i3c_bus_maintenance_lock(struct i3c_bus *bus); +void i3c_bus_maintenance_unlock(struct i3c_bus *bus); #endif /* I3C_MASTER_H */ diff --git a/include/linux/idr.h b/include/linux/idr.h index 789e23e67444..e2a4b6298511 100644 --- a/include/linux/idr.h +++ b/include/linux/idr.h @@ -16,6 +16,7 @@ #include <linux/gfp.h> #include <linux/percpu.h> #include <linux/cleanup.h> +#include <linux/compiler.h> struct idr { struct radix_tree_root idr_rt; @@ -269,7 +270,9 @@ struct ida { #define IDA_INIT(name) { \ .xa = XARRAY_INIT(name, IDA_INIT_FLAGS) \ } -#define DEFINE_IDA(name) struct ida name = IDA_INIT(name) +#define DEFINE_IDA(name) \ + struct ida name = IDA_INIT(name); \ + ASSERT_STATIC_STORAGE(name) int ida_alloc_range(struct ida *, unsigned int min, unsigned int max, gfp_t); void ida_free(struct ida *, unsigned int id); diff --git a/include/linux/ieee80211-eht.h b/include/linux/ieee80211-eht.h index b62297a978e7..a699d108eee6 100644 --- a/include/linux/ieee80211-eht.h +++ b/include/linux/ieee80211-eht.h @@ -482,6 +482,7 @@ struct ieee80211_multi_link_elem { #define IEEE80211_MLC_BASIC_PRES_MLD_ID 0x0200 #define IEEE80211_MLC_BASIC_PRES_EXT_MLD_CAPA_OP 0x0400 #define IEEE80211_MLC_BASIC_PRES_ENH_CRIT_UPD 0x0800 +#define IEEE80211_MLC_BASIC_PRES_AGE_OF_BSS_LOAD 0x1000 #define IEEE80211_MED_SYNC_DELAY_DURATION 0x00ff #define IEEE80211_MED_SYNC_DELAY_SYNC_OFDM_ED_THRESH 0x0f00 @@ -550,6 +551,8 @@ struct ieee80211_mle_basic_common_info { } __packed; #define IEEE80211_MLC_PREQ_PRES_MLD_ID 0x0010 +#define IEEE80211_MLC_PREQ_PRES_MLD_MAC_ADDR 0x0020 +#define IEEE80211_MLC_PREQ_PRES_NEIGH_AP_MLD_MAC_ADDR 0x0040 struct ieee80211_mle_preq_common_info { u8 len; @@ -560,6 +563,7 @@ struct ieee80211_mle_preq_common_info { #define IEEE80211_MLC_RECONF_PRES_EML_CAPA 0x0020 #define IEEE80211_MLC_RECONF_PRES_MLD_CAPA_OP 0x0040 #define IEEE80211_MLC_RECONF_PRES_EXT_MLD_CAPA_OP 0x0080 +#define IEEE80211_MLC_RECONF_PRES_TARGET_AP_MLD_ADDR 0x0100 /* no fixed fields in RECONF */ @@ -766,7 +770,7 @@ static inline u16 ieee80211_mle_get_mld_capa_op(const u8 *data) #define IEEE80211_EHT_ML_EXT_MLD_CAPA_NSTR_UPDATE 0x0020 #define IEEE80211_EHT_ML_EXT_MLD_CAPA_EMLSR_ENA_ON_ONE_LINK 0x0040 #define IEEE80211_EHT_ML_EXT_MLD_CAPA_BTM_MLD_RECO_MULTI_AP 0x0080 -/* defined by UHR Draft P802.11bn_D1.3 Figure 9-1147 */ +/* defined by UHR Draft P802.11bn_D1.5 Figure 9-1168 */ #define IEEE80211_UHR_ML_EXT_MLD_CAPA_ML_PM 0x0100 /** @@ -928,11 +932,17 @@ static inline bool ieee80211_mle_size_ok(const u8 *data, size_t len) common += 2; if (control & IEEE80211_MLC_BASIC_PRES_ENH_CRIT_UPD) common += 1; + if (control & IEEE80211_MLC_BASIC_PRES_AGE_OF_BSS_LOAD) + common += 1; break; case IEEE80211_ML_CONTROL_TYPE_PREQ: common += sizeof(struct ieee80211_mle_preq_common_info); if (control & IEEE80211_MLC_PREQ_PRES_MLD_ID) common += 1; + if (control & IEEE80211_MLC_PREQ_PRES_MLD_MAC_ADDR) + common += ETH_ALEN; + if (control & IEEE80211_MLC_PREQ_PRES_NEIGH_AP_MLD_MAC_ADDR) + common += ETH_ALEN; break; case IEEE80211_ML_CONTROL_TYPE_RECONF: common += 1; @@ -944,6 +954,8 @@ static inline bool ieee80211_mle_size_ok(const u8 *data, size_t len) common += 2; if (control & IEEE80211_MLC_RECONF_PRES_EXT_MLD_CAPA_OP) common += 2; + if (control & IEEE80211_MLC_RECONF_PRES_TARGET_AP_MLD_ADDR) + common += ETH_ALEN; break; case IEEE80211_ML_CONTROL_TYPE_TDLS: common += sizeof(struct ieee80211_mle_tdls_common_info); @@ -1003,6 +1015,7 @@ enum ieee80211_mle_subelems { #define IEEE80211_MLE_STA_CONTROL_BSS_PARAM_CHANGE_CNT_PRESENT 0x0800 #define IEEE80211_MLE_STA_CONTROL_ENH_CRIT_UPD_PRESENT 0x1000 #define IEEE80211_MLE_STA_CONTROL_AP_CONDUCTED_TX_PWR_PRESENT 0x2000 +#define IEEE80211_MLE_STA_CONTROL_AGE_OF_BSS_LOAD_PRESENT 0x4000 struct ieee80211_mle_per_sta_profile { __le16 control; @@ -1054,6 +1067,9 @@ static inline bool ieee80211_mle_basic_sta_prof_size_ok(const u8 *data, if (control & IEEE80211_MLE_STA_CONTROL_AP_CONDUCTED_TX_PWR_PRESENT) info_len += 1; + if (control & IEEE80211_MLE_STA_CONTROL_AGE_OF_BSS_LOAD_PRESENT) + info_len += 1; + return prof->sta_info_len >= info_len && fixed + prof->sta_info_len - 1 <= len; } diff --git a/include/linux/ieee80211-mesh.h b/include/linux/ieee80211-mesh.h index 7eb15834531c..9e548b9173df 100644 --- a/include/linux/ieee80211-mesh.h +++ b/include/linux/ieee80211-mesh.h @@ -361,8 +361,7 @@ ieee80211_mesh_hwmp_perr_get_rcode(const u8 *ie, u8 dst_idx) /* IEEE Std 802.11-2016 9.4.2.113 PREQ element */ static inline bool ieee80211_mesh_preq_size_ok(const u8 *pos, u8 elen) { - struct ieee80211_mesh_hwmp_preq_bottom *preq_elem_bottom = - ieee80211_mesh_hwmp_preq_get_bottom(pos); + struct ieee80211_mesh_hwmp_preq_bottom *preq_elem_bottom; u8 target_count; int needed; @@ -378,6 +377,7 @@ static inline bool ieee80211_mesh_preq_size_ok(const u8 *pos, u8 elen) if (elen < needed) return false; + preq_elem_bottom = ieee80211_mesh_hwmp_preq_get_bottom(pos); target_count = preq_elem_bottom->target_count; /* IEEE Std 802.11-2016 Table 14-10 to 14-16 */ if (target_count < 1) diff --git a/include/linux/ieee80211-uhr.h b/include/linux/ieee80211-uhr.h index 665d4b3a5b41..19e9260d0db8 100644 --- a/include/linux/ieee80211-uhr.h +++ b/include/linux/ieee80211-uhr.h @@ -17,32 +17,32 @@ #define IEEE80211_UHR_OPER_PARAMS_PEDCA_ENA 0x0004 #define IEEE80211_UHR_OPER_PARAMS_DBE_ENA 0x0008 #define IEEE80211_UHR_OPER_PARAMS_DBE_BW 0x0070 -#define IEEE80211_UHR_OPER_PARAMS_DUO_PRES 0x0080 -#define IEEE80211_UHR_OPER_PARAMS_DPS_PRES 0x0100 -#define IEEE80211_UHR_OPER_PARAMS_NPCA_PRES 0x0200 -#define IEEE80211_UHR_OPER_PARAMS_PEDCA_PRES 0x0400 -#define IEEE80211_UHR_OPER_PARAMS_DBE_PRES 0x0800 +#define IEEE80211_UHR_OPER_PARAMS_ELR_RX 0x0080 +#define IEEE80211_UHR_OPER_PARAMS_DUO_PRES 0x0100 +#define IEEE80211_UHR_OPER_PARAMS_DPS_PRES 0x0200 +#define IEEE80211_UHR_OPER_PARAMS_NPCA_PRES 0x0400 +#define IEEE80211_UHR_OPER_PARAMS_PEDCA_PRES 0x0800 +#define IEEE80211_UHR_OPER_PARAMS_DBE_PRES 0x1000 struct ieee80211_uhr_operation { __le16 params; - u8 basic_mcs_nss_set[4]; u8 variable[]; } __packed; -#define IEEE80211_UHR_NPCA_PARAMS_PRIMARY_CHAN_OFFS 0x0000000F -#define IEEE80211_UHR_NPCA_PARAMS_MIN_DUR_THRESH 0x000000F0 -#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_DELAY 0x00003F00 -#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_BACK_DELAY 0x000FC000 -#define IEEE80211_UHR_NPCA_PARAMS_INIT_QSRC 0x00300000 -#define IEEE80211_UHR_NPCA_PARAMS_MOPLEN 0x00400000 -#define IEEE80211_UHR_NPCA_PARAMS_DIS_SUBCH_BMAP_PRES 0x00800000 +#define IEEE80211_UHR_NPCA_PARAMS_PRIMARY_CHAN 0x000000FF +#define IEEE80211_UHR_NPCA_PARAMS_MIN_DUR_THRESH 0x00000F00 +#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_DELAY 0x0003F000 +#define IEEE80211_UHR_NPCA_PARAMS_SWITCH_BACK_DELAY 0x00FC0000 +#define IEEE80211_UHR_NPCA_PARAMS_INIT_QSRC 0x03000000 +#define IEEE80211_UHR_NPCA_PARAMS_MOPLEN 0x04000000 +#define IEEE80211_UHR_NPCA_PARAMS_DIS_SUBCH_BMAP_PRES 0x08000000 /** * struct ieee80211_uhr_npca_info - npca operation information * * This structure is the "NPCA Operation Parameters field format" of "UHR - * Operation Element" fields as described in P802.11bn_D1.3 - * subclause 9.4.2.353. See Figure 9-aa4. + * Operation Element" fields as described in P802.11bn_D1.5 + * subclause 9.4.2.355.2, see Figure 9-aa6. * * Refer to IEEE80211_UHR_NPCA* * @params: @@ -97,7 +97,7 @@ struct ieee80211_uhr_npca_info { * struct ieee80211_uhr_dps_info - DPS operation information * * This structure is the "DPS Operation Parameter field" of "UHR - * Operation Element" fields as described in P802.11bn_D1.3 + * Operation Element" fields as described in P802.11bn_D1.5 * subclause 9.4.1.87. See Figure 9-207u. * * Refer to IEEE80211_UHR_DPS* @@ -211,8 +211,8 @@ static inline int ieee80211_uhr_dbe_bw_mhz(enum ieee80211_uhr_dbe_oper_bw bw) * struct ieee80211_uhr_dbe_info - DBE operation information * * This structure is the "DBE Operation Parameters field" of - * "UHR Operation Element" fields as described in P802.11bn_D1.3 - * subclause 9.4.2.353. See Figure 9-aa6. + * "UHR Operation Element" fields as described in P802.11bn_D1.5 + * subclause 9.4.2.355.4, see Figure 9-aa9. * * Refer to IEEE80211_UHR_DBE_OPER* * @params: @@ -234,25 +234,23 @@ struct ieee80211_uhr_dbe_info { __le16 dis_subch_bmap[]; } __packed; -#define IEEE80211_UHR_P_EDCA_ECWMIN 0x0F -#define IEEE80211_UHR_P_EDCA_ECWMAX 0xF0 -#define IEEE80211_UHR_P_EDCA_AIFSN 0x000F -#define IEEE80211_UHR_P_EDCA_CW_DS 0x0030 -#define IEEE80211_UHR_P_EDCA_PSRC_THRESHOLD 0x01C0 -#define IEEE80211_UHR_P_EDCA_QSRC_THRESHOLD 0x0600 +#define IEEE80211_UHR_P_EDCA_ECWMIN 0x0007 +#define IEEE80211_UHR_P_EDCA_ECWMAX 0x0038 +#define IEEE80211_UHR_P_EDCA_AIFSN 0x01C0 +#define IEEE80211_UHR_P_EDCA_CW_DS 0x0600 +#define IEEE80211_UHR_P_EDCA_PSRC_THRESHOLD 0x3800 +#define IEEE80211_UHR_P_EDCA_QSRC_THRESHOLD 0xC000 /** * struct ieee80211_uhr_p_edca_info - P-EDCA operation information * * This structure is the "P-EDCA Operation Parameters field" of - * "UHR Operation Element" fields as described in P802.11bn_D1.3 - * subclause 9.4.2.353. See Figure 9-aa5. + * "UHR Operation Element" fields as described in P802.11bn_D1.5 + * subclause 9.4.2.355.3, see Figure 9-aa8. * * Refer to IEEE80211_UHR_P_EDCA* - * @p_edca_ec: P-EDCA ECWmin and ECWmax. - * These fields indicate the CWmin and CWmax values used by a - * P-EDCA STA during P-EDCA contention. - * @params: AIFSN, CW DS, PSRC threshold, and QSRC threshold. + * @params: ECWmin, ECWmax, AIFSN, CW DS, PSRC threshold, and QSRC threshold. + * - CWmin/CWmax values used by a P-EDCA STA during P-EDCA contention. * - The AIFSN field indicates the AIFSN value used by a P-EDCA STA * during P-EDCA contention. * - The CW DS field indicates the value used for randomization of the @@ -266,7 +264,6 @@ struct ieee80211_uhr_dbe_info { * value 0 is reserved. */ struct ieee80211_uhr_p_edca_info { - u8 p_edca_ec; __le16 params; } __packed; @@ -302,7 +299,7 @@ static inline bool ieee80211_uhr_oper_size_ok(const u8 *data, u8 len) } } - /* P-EDCA Operation Parameters (fixed 3 bytes) */ + /* P-EDCA Operation Parameters */ if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_PEDCA_PRES)) { needed += sizeof(struct ieee80211_uhr_p_edca_info); if (len < needed) @@ -341,6 +338,9 @@ ieee80211_uhr_npca_info(const struct ieee80211_uhr_operation *oper) if (!(oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_NPCA_ENA))) return NULL; + if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DUO_PRES)) + pos += 1; + if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DPS_PRES)) pos += sizeof(struct ieee80211_uhr_dps_info); @@ -372,6 +372,9 @@ ieee80211_uhr_oper_dbe_info(const struct ieee80211_uhr_operation *oper) if (!(oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DBE_ENA))) return NULL; + if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DUO_PRES)) + pos += 1; + if (oper->params & cpu_to_le16(IEEE80211_UHR_OPER_PARAMS_DPS_PRES)) pos += sizeof(struct ieee80211_uhr_dps_info); @@ -415,28 +418,42 @@ ieee80211_uhr_oper_dbe_info(const struct ieee80211_uhr_operation *oper) #define IEEE80211_UHR_MAC_CAP2_UHR_OM_PU_TO_LOW 0xC0 #define IEEE80211_UHR_MAC_CAP3_UHR_OM_PU_TO_HIGH 0x03 -#define IEEE80211_UHR_MAC_CAP3_PARAM_UPD_ADV_NOTIF_INTV 0x1C -#define IEEE80211_UHR_MAC_CAP3_UPD_IND_TIM_INTV_LOW 0xE0 +#define IEEE80211_UHR_MAC_CAP3_PARAM_UPD_ADV_NOTIF_INTV 0x7C +#define IEEE80211_UHR_MAC_CAP3_UPD_IND_TIM_INTV_LOW 0x80 -#define IEEE80211_UHR_MAC_CAP4_UPD_IND_TIM_INTV_HIGH 0x03 -#define IEEE80211_UHR_MAC_CAP4_BOUNDED_ESS 0x04 -#define IEEE80211_UHR_MAC_CAP4_BTM_ASSURANCE 0x08 -#define IEEE80211_UHR_MAC_CAP4_CO_BF_SUPP 0x10 +#define IEEE80211_UHR_MAC_CAP4_UPD_IND_TIM_INTV_HIGH 0x0F +#define IEEE80211_UHR_MAC_CAP4_BOUNDED_ESS 0x10 +#define IEEE80211_UHR_MAC_CAP4_BTM_ASSURANCE 0x20 +#define IEEE80211_UHR_MAC_CAP4_CO_BF_SUPP 0x40 +#define IEEE80211_UHR_MAC_CAP4_CO_SR_SUPP 0x80 + +#define IEEE80211_UHR_MAC_CAP5_MAPC_ENH_MSRMT_SUPP 0x01 #define IEEE80211_UHR_MAC_CAP_DBE_MAX_BW 0x07 #define IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_160_PRES 0x08 #define IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_320_PRES 0x10 +/* struct ieee80211_uhr_cap_dbe::bwcap[].cap */ +#define IEEE80211_UHR_MAC_CAP_DBE_CAP_NUM_SND_DIMS 0x07 +#define IEEE80211_UHR_MAC_CAP_DBE_CAP_NON_OFDMA_UL_MUMIMO 0x08 +#define IEEE80211_UHR_MAC_CAP_DBE_CAP_MU_BEAMFORMER 0x10 +#define IEEE80211_UHR_MAC_CAP_DBE_CAP_BEAMFORMEE_SS 0xe0 + struct ieee80211_uhr_cap_dbe { u8 cap; + u8 max_switch_time_period; + u8 mode_change_intvl; /* present 0, 1 or 2 times depending on _PRES bits */ - struct ieee80211_eht_mcs_nss_supp_bw eht_mcs_map[]; + struct { + struct ieee80211_eht_mcs_nss_supp_bw eht_mcs_map; + u8 cap; + } __packed bwcap[]; } __packed; /** * enum ieee80211_uhr_dbe_max_supported_bw - DBE Maximum Supported Bandwidth * - * As per spec P802.11bn_D1.3 "Table 9-bb5—Encoding of the DBE Maximum + * As per spec P802.11bn_D1.5 Table 9-bb8 "Encoding of the DBE Maximum * Supported Bandwidth field". * * @IEEE80211_UHR_DBE_MAX_BW_40: Indicates 40 MHz DBE max supported bw @@ -520,10 +537,10 @@ static inline bool ieee80211_uhr_capa_size_ok(const u8 *data, u8 len, dbe = (const void *)cap->variable; if (dbe->cap & IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_160_PRES) - needed += sizeof(dbe->eht_mcs_map[0]); + needed += sizeof(dbe->bwcap[0]); if (dbe->cap & IEEE80211_UHR_MAC_CAP_DBE_EHT_MCS_MAP_320_PRES) - needed += sizeof(dbe->eht_mcs_map[0]); + needed += sizeof(dbe->bwcap[0]); } return len >= needed; diff --git a/include/linux/ieee80211.h b/include/linux/ieee80211.h index 26e674038865..73ac5336a17a 100644 --- a/include/linux/ieee80211.h +++ b/include/linux/ieee80211.h @@ -1828,6 +1828,7 @@ enum ieee80211_eid_ext { WLAN_EID_EXT_BANDWIDTH_INDICATION = 135, WLAN_EID_EXT_KNOWN_STA_IDENTIFCATION = 136, WLAN_EID_EXT_NON_AP_STA_REG_CON = 137, + WLAN_EID_EXT_CIP_CAPA = 150, WLAN_EID_EXT_UHR_OPER = 151, WLAN_EID_EXT_UHR_CAPA = 152, WLAN_EID_EXT_MACP = 153, @@ -2421,8 +2422,7 @@ static inline bool ieee80211_is_bufferable_mmpdu(struct sk_buff *skb) /* action frame - additionally check for non-bufferable FTM */ - if (mgmt->u.action.category != WLAN_CATEGORY_PUBLIC && - mgmt->u.action.category != WLAN_CATEGORY_PROTECTED_DUAL_OF_ACTION) + if (mgmt->u.action.category != WLAN_CATEGORY_PUBLIC) return true; if (mgmt->u.action.action_code == WLAN_PUB_ACTION_FTM_REQUEST || @@ -2900,4 +2900,25 @@ static inline bool ieee80211_check_tim(const struct ieee80211_tim_ie *tim, __ieee80211_check_tim(tim, tim_len, aid); } +/** + * enum ieee80211_cip_cap_fields - CIP Capabilities element fields + * @IEEE80211_CIP_CAP_MIC_PADDING: The MIC padding subfield in the CIP + * capabilities + * @IEEE80211_CIP_CAP_PROTECTED_CTRL_FRAME_ONLY: The MIC Padding For Protected + * Control Frames Only bit in CIP capabilities + */ +enum ieee80211_cip_cap_fields { + IEEE80211_CIP_CAP_MIC_PADDING = 0x0F, + IEEE80211_CIP_CAP_PROTECTED_CTRL_FRAME_ONLY = 0x10, +}; + +/** + * struct ieee80211_cip_cap - CIP Capabilities element + * + * @v: The value of the Control Integrity Protocol Capability element + */ +struct ieee80211_cip_cap { + u8 v; +} __packed; + #endif /* LINUX_IEEE80211_H */ diff --git a/include/linux/if_vlan.h b/include/linux/if_vlan.h index 20cc16ea4e5a..4846032bf4ff 100644 --- a/include/linux/if_vlan.h +++ b/include/linux/if_vlan.h @@ -365,6 +365,9 @@ static inline int __vlan_insert_inner_tag(struct sk_buff *skb, const u8 meta_len = mac_len > ETH_TLEN ? skb_metadata_len(skb) : 0; struct vlan_ethhdr *veth; + if (unlikely(!pskb_may_pull(skb, mac_len))) + return -EINVAL; + if (skb_cow_head(skb, meta_len + VLAN_HLEN) < 0) return -ENOMEM; diff --git a/include/linux/igmp.h b/include/linux/igmp.h index 3a2d35a9f307..e075611344ef 100644 --- a/include/linux/igmp.h +++ b/include/linux/igmp.h @@ -14,6 +14,7 @@ #include <linux/timer.h> #include <linux/in.h> #include <linux/ip.h> +#include <linux/net.h> #include <linux/refcount.h> #include <linux/sockptr.h> #include <uapi/linux/igmp.h> @@ -57,20 +58,21 @@ struct ip_mc_socklist { }; struct ip_sf_list { - struct ip_sf_list *sf_next; + struct ip_sf_list __rcu *sf_next; unsigned long sf_count[2]; /* include/exclude counts */ __be32 sf_inaddr; unsigned char sf_gsresp; /* include in g & s response? */ unsigned char sf_oldin; /* change state */ unsigned char sf_crcount; /* retrans. left to send */ + struct rcu_head rcu; }; struct ip_mc_list { struct in_device *interface; __be32 multiaddr; unsigned int sfmode; - struct ip_sf_list *sources; - struct ip_sf_list *tomb; + struct ip_sf_list __rcu *sources; + struct ip_sf_list __rcu *tomb; unsigned long sfcount[2]; union { struct ip_mc_list *next; @@ -272,7 +274,7 @@ extern int ip_mc_source(int add, int omode, struct sock *sk, struct ip_mreq_source *mreqs, int ifindex); extern int ip_mc_msfilter(struct sock *sk, struct ip_msfilter *msf,int ifindex); extern int ip_mc_msfget(struct sock *sk, struct ip_msfilter *msf, - sockptr_t optval, sockptr_t optlen); + sockopt_t *opt); extern int ip_mc_gsfget(struct sock *sk, struct group_filter *gsf, sockptr_t optval, size_t offset); extern int ip_mc_sf_allow(const struct sock *sk, __be32 local, __be32 rmt, diff --git a/include/linux/input/samsung-keypad.h b/include/linux/input/samsung-keypad.h deleted file mode 100644 index ab6b97114c08..000000000000 --- a/include/linux/input/samsung-keypad.h +++ /dev/null @@ -1,39 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-or-later */ -/* - * Samsung Keypad platform data definitions - * - * Copyright (C) 2010 Samsung Electronics Co.Ltd - * Author: Joonyoung Shim <jy0922.shim@samsung.com> - */ - -#ifndef __SAMSUNG_KEYPAD_H -#define __SAMSUNG_KEYPAD_H - -#include <linux/input/matrix_keypad.h> - -#define SAMSUNG_MAX_ROWS 8 -#define SAMSUNG_MAX_COLS 8 - -/** - * struct samsung_keypad_platdata - Platform device data for Samsung Keypad. - * @keymap_data: pointer to &matrix_keymap_data. - * @rows: number of keypad row supported. - * @cols: number of keypad col supported. - * @no_autorepeat: disable key autorepeat. - * @wakeup: controls whether the device should be set up as wakeup source. - * @cfg_gpio: configure the GPIO. - * - * Initialisation data specific to either the machine or the platform - * for the device driver to use or call-back when configuring gpio. - */ -struct samsung_keypad_platdata { - const struct matrix_keymap_data *keymap_data; - unsigned int rows; - unsigned int cols; - bool no_autorepeat; - bool wakeup; - - void (*cfg_gpio)(unsigned int rows, unsigned int cols); -}; - -#endif /* __SAMSUNG_KEYPAD_H */ diff --git a/include/linux/intel_vsec.h b/include/linux/intel_vsec.h index 843cda8f8644..917d9397a993 100644 --- a/include/linux/intel_vsec.h +++ b/include/linux/intel_vsec.h @@ -90,13 +90,25 @@ enum intel_vsec_quirks { * @read_telem: when specified, called by client driver to access PMT * data (instead of direct copy). * * dev: device reference for the callback's use - * * guid: ID of data to acccss + * * guid: ID of data to access * * data: buffer for the data to be copied * * off: offset into the requested buffer * * count: size of buffer + * @read_reg: when specified called by client driver to read PMT state + * * dev: device reference for the callback's use + * * guid: ID of data to access + * * reg_data: register data + * * offset: offset of register to read + * @write_reg: when specified called by client driver to write PMT state + * * dev: device reference for the callback's use + * * guid: ID of data to access + * * reg_data: register data + * * offset: offset of register to write */ struct pmt_callbacks { int (*read_telem)(struct device *dev, u32 guid, u64 *data, loff_t off, u32 count); + int (*read_reg)(struct device *dev, u32 guid, u32 *reg_data, u32 offset); + int (*write_reg)(struct device *dev, u32 guid, u32 reg_data, u32 offset); }; struct vsec_feature_dependency { diff --git a/include/linux/interrupt.h b/include/linux/interrupt.h index 3bf969ad8fe0..52bb684090c0 100644 --- a/include/linux/interrupt.h +++ b/include/linux/interrupt.h @@ -573,9 +573,11 @@ enum * _ IRQ_POLL: irq_poll_cpu_dead() migrates the queue * * _ (HR)TIMER_SOFTIRQ: (hr)timers_dead_cpu() migrates the queue + * + * _ BLOCK_SOFTIRQ: blk_softirq_cpu_dead() completes the remaining requests */ -#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(IRQ_POLL_SOFTIRQ) |\ - BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ)) +#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(BLOCK_SOFTIRQ) |\ + BIT(IRQ_POLL_SOFTIRQ) | BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ)) /* map softirq index to softirq name. update 'softirq_to_name' in diff --git a/include/linux/interrupt_rc.h b/include/linux/interrupt_rc.h index b9a7f05ecf42..a9ed937a80e7 100644 --- a/include/linux/interrupt_rc.h +++ b/include/linux/interrupt_rc.h @@ -20,11 +20,8 @@ /* Per-CPU interrupt disabling state for local_interrupt_{disable,enable}(). */ DECLARE_PER_CPU(unsigned long, local_interrupt_disable_state); -static __always_inline void __local_interrupt_disable(void) +static __always_inline void __local_interrupt_save_state(unsigned long flags) { - unsigned long flags; - - local_irq_save(flags); raw_cpu_write(local_interrupt_disable_state, flags); } @@ -36,9 +33,9 @@ static __always_inline void __local_interrupt_enable(void) } #ifndef INSTANTIATE_EXPORTED_INTERRUPT_DISABLE -static __always_inline void _local_interrupt_disable(void) +static __always_inline void _local_interrupt_save_state(unsigned long flags) { - __local_interrupt_disable(); + __local_interrupt_save_state(flags); } static __always_inline void _local_interrupt_enable(void) @@ -46,27 +43,30 @@ static __always_inline void _local_interrupt_enable(void) __local_interrupt_enable(); } #else -extern void _local_interrupt_disable(void); +extern void _local_interrupt_save_state(unsigned long flags); extern void _local_interrupt_enable(void); #endif #else /* !MODULE */ -extern void _local_interrupt_disable(void); +extern void _local_interrupt_save_state(unsigned long flags); extern void _local_interrupt_enable(void); #endif /* !MODULE */ +#define hardirq_disable_enter() __preempt_count_add_return(HARDIRQ_DISABLE_OFFSET) +#define hardirq_disable_exit() __preempt_count_sub_return(HARDIRQ_DISABLE_OFFSET) + static inline void local_interrupt_disable(void) { int new_count; + unsigned long flags; WARN_ON_ONCE(in_nmi()); + local_irq_save(flags); new_count = hardirq_disable_enter(); - /* Interrupts can happen here, but it's OK, see __irq_exit_rcu(). */ - if ((new_count & HARDIRQ_DISABLE_MASK) == HARDIRQ_DISABLE_OFFSET) - _local_interrupt_disable(); + _local_interrupt_save_state(flags); } static inline void local_interrupt_enable(void) diff --git a/include/linux/io_uring.h b/include/linux/io_uring.h index d1aa4edfc2a5..505caa99c621 100644 --- a/include/linux/io_uring.h +++ b/include/linux/io_uring.h @@ -2,10 +2,42 @@ #ifndef _LINUX_IO_URING_H #define _LINUX_IO_URING_H +#include <linux/io_uring_types.h> #include <linux/sched.h> #include <linux/xarray.h> #include <uapi/linux/io_uring.h> +static inline void req_set_fail(struct io_kiocb *req) +{ + req->flags |= REQ_F_FAIL; + if (req->flags & REQ_F_CQE_SKIP) { + req->flags &= ~REQ_F_CQE_SKIP; + req->flags |= REQ_F_SKIP_LINK_CQES; + } +} + +static inline void io_req_set_res(struct io_kiocb *req, s32 res, u32 cflags) +{ + req->cqe.res = res; + req->cqe.flags = cflags; +} + +static inline u32 ctx_cqe32_flags(struct io_ring_ctx *ctx) +{ + if (ctx->flags & IORING_SETUP_CQE_MIXED) + return IORING_CQE_F_32; + return 0; +} + +static inline void io_req_set_res32(struct io_kiocb *req, s32 res, u32 cflags, + __u64 extra1, __u64 extra2) +{ + req->cqe.res = res; + req->cqe.flags = cflags | ctx_cqe32_flags(req->ctx); + req->big_cqe.extra1 = extra1; + req->big_cqe.extra2 = extra2; +} + #if defined(CONFIG_IO_URING) void __io_uring_cancel(bool cancel_all); void __io_uring_free(struct task_struct *tsk); diff --git a/include/linux/io_uring/cmd.h b/include/linux/io_uring/cmd.h index 42801f0b6456..e18d4af62e1a 100644 --- a/include/linux/io_uring/cmd.h +++ b/include/linux/io_uring/cmd.h @@ -3,6 +3,7 @@ #define _LINUX_IO_URING_CMD_H #include <uapi/linux/io_uring.h> +#include <linux/io_uring.h> #include <linux/io_uring_types.h> #include <linux/blk-mq.h> @@ -41,6 +42,24 @@ static inline void io_uring_cmd_private_sz_check(size_t cmd_sz) ((pdu_type *)&(cmd)->pdu) \ ) +static inline void io_uring_cmd_set_res(struct io_uring_cmd *cmd, s32 ret) +{ + struct io_kiocb *req = cmd_to_io_kiocb(cmd); + + if (ret < 0) + req_set_fail(req); + io_req_set_res(req, ret, 0); +} + +static inline void io_uring_cmd_set_res32(struct io_uring_cmd *cmd, s32 ret, u64 res2) +{ + struct io_kiocb *req = cmd_to_io_kiocb(cmd); + + if (ret < 0) + req_set_fail(req); + io_req_set_res32(req, ret, 0, res2, 0); +} + #if defined(CONFIG_IO_URING) int io_uring_cmd_import_fixed(u64 ubuf, unsigned long len, int rw, struct iov_iter *iter, @@ -56,11 +75,12 @@ int io_uring_cmd_import_fixed_vec(struct io_uring_cmd *ioucmd, * Completes the request, i.e. posts an io_uring CQE and deallocates @ioucmd * and the corresponding io_uring request. * + * io_uring_cmd_set_res()/io_uring_cmd_set_res32() must be called first. + * * Note: the caller should never hard code @issue_flags and is only allowed * to pass the mask provided by the core io_uring code. */ -void __io_uring_cmd_done(struct io_uring_cmd *cmd, s32 ret, u64 res2, - unsigned issue_flags, bool is_cqe32); +void __io_uring_cmd_done(struct io_uring_cmd *, unsigned issue_flags); void __io_uring_cmd_do_in_task(struct io_uring_cmd *ioucmd, io_req_tw_func_t task_work_cb, @@ -116,8 +136,8 @@ static inline int io_uring_cmd_import_fixed_vec(struct io_uring_cmd *ioucmd, { return -EOPNOTSUPP; } -static inline void __io_uring_cmd_done(struct io_uring_cmd *cmd, s32 ret, - u64 ret2, unsigned issue_flags, bool is_cqe32) +static inline void __io_uring_cmd_done(struct io_uring_cmd *cmd, + unsigned issue_flags) { } static inline void __io_uring_cmd_do_in_task(struct io_uring_cmd *ioucmd, @@ -205,13 +225,15 @@ static inline void *io_uring_cmd_ctx_handle(struct io_uring_cmd *cmd) static inline void io_uring_cmd_done(struct io_uring_cmd *ioucmd, s32 ret, unsigned issue_flags) { - return __io_uring_cmd_done(ioucmd, ret, 0, issue_flags, false); + io_uring_cmd_set_res(ioucmd, ret); + __io_uring_cmd_done(ioucmd, issue_flags); } static inline void io_uring_cmd_done32(struct io_uring_cmd *ioucmd, s32 ret, u64 res2, unsigned issue_flags) { - return __io_uring_cmd_done(ioucmd, ret, res2, issue_flags, true); + io_uring_cmd_set_res32(ioucmd, ret, res2); + __io_uring_cmd_done(ioucmd, issue_flags); } #endif /* _LINUX_IO_URING_CMD_H */ diff --git a/include/linux/io_uring_types.h b/include/linux/io_uring_types.h index 39629ee77b91..90fea94ad202 100644 --- a/include/linux/io_uring_types.h +++ b/include/linux/io_uring_types.h @@ -522,6 +522,8 @@ struct io_ring_ctx { /* protected by ->completion_lock */ unsigned nr_req_allocated; + /* pending SEND_ZC notifications, protected by ->uring_lock */ + unsigned nr_notifs; #ifdef CONFIG_NET_RX_BUSY_POLL struct list_head napi_list; /* track busy poll napi_id */ diff --git a/include/linux/iomap.h b/include/linux/iomap.h index bc7ae6327dbf..59718f73c15a 100644 --- a/include/linux/iomap.h +++ b/include/linux/iomap.h @@ -483,13 +483,35 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, #define IOMAP_IOEND_BOUNDARY (1U << 2) /* is direct I/O */ #define IOMAP_IOEND_DIRECT (1U << 3) +/* generate integrity (PI) information */ +#ifdef CONFIG_BLK_DEV_INTEGRITY +#define IOMAP_IOEND_INTEGRITY (1U << 4) +#else +#define IOMAP_IOEND_INTEGRITY 0 +#endif /* CONFIG_BLK_DEV_INTEGRITY */ /* * Flags that if set on either ioend prevent the merge of two ioends. * (IOMAP_IOEND_BOUNDARY also prevents merges, but only one-way) */ #define IOMAP_IOEND_NOMERGE_FLAGS \ - (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT) + (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT | \ + IOMAP_IOEND_INTEGRITY) + +/* ioend flags directly implied by iomap flags */ +static inline u16 iomap_ioend_flags(const struct iomap *iomap) +{ + unsigned int flags = 0; + + if (iomap->type == IOMAP_UNWRITTEN) + flags |= IOMAP_IOEND_UNWRITTEN; + if (iomap->flags & IOMAP_F_SHARED) + flags |= IOMAP_IOEND_SHARED; + if (iomap->flags & IOMAP_F_INTEGRITY) + flags |= IOMAP_IOEND_INTEGRITY; + + return flags; +} /* * Structure for writeback I/O completions. @@ -500,6 +522,7 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, struct iomap_ioend { struct list_head io_list; /* next ioend in chain */ u16 io_flags; /* IOMAP_IOEND_* */ + u32 io_bvec_offset; /* offset into first bvec */ struct inode *io_inode; /* file being written to */ size_t io_size; /* size of the extent */ atomic_t io_remaining; /* completetion defer count */ @@ -517,6 +540,13 @@ static inline struct iomap_ioend *iomap_ioend_from_bio(struct bio *bio) return container_of(bio, struct iomap_ioend, io_bio); } +#define BVEC_ITER_IOEND(_ioend) \ +{ \ + .bi_sector = (_ioend)->io_sector, \ + .bi_size = (_ioend)->io_size, \ + .bi_offset = (_ioend)->io_bvec_offset, \ +} + struct iomap_writeback_ops { /* * Performs writeback on the passed in range @@ -565,6 +595,7 @@ void iomap_finish_ioends(struct iomap_ioend *ioend, int error); void iomap_ioend_try_merge(struct iomap_ioend *ioend, struct list_head *more_ioends); void iomap_sort_ioends(struct list_head *ioend_list); +int iomap_ioend_integrity_verify(struct iomap_ioend *ioend); ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, loff_t pos, loff_t end_pos, unsigned int dirty_len); int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error); @@ -577,6 +608,11 @@ void iomap_finish_folio_write(struct inode *inode, struct folio *folio, int iomap_writeback_folio(struct iomap_writepage_ctx *wpc, struct folio *folio); int iomap_writepages(struct iomap_writepage_ctx *wpc); +void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize, + void (*submit_ioend)(struct iomap_ioend *ioend)); +void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio, + int error); + struct iomap_read_folio_ctx { const struct iomap_read_ops *ops; struct folio *cur_folio; diff --git a/include/linux/iommu-dma.h b/include/linux/iommu-dma.h index 060f6e23ab3c..fae5d50e4f27 100644 --- a/include/linux/iommu-dma.h +++ b/include/linux/iommu-dma.h @@ -39,7 +39,7 @@ int iommu_dma_get_sgtable(struct device *dev, struct sg_table *sgt, void *cpu_addr, dma_addr_t dma_addr, size_t size, unsigned long attrs); unsigned long iommu_dma_get_merge_boundary(struct device *dev); -size_t iommu_dma_opt_mapping_size(void); +size_t iommu_dma_max_opt_mapping_size(void); size_t iommu_dma_max_mapping_size(struct device *dev); void iommu_dma_free(struct device *dev, size_t size, void *cpu_addr, dma_addr_t handle, unsigned long attrs); diff --git a/include/linux/irq-entry-common.h b/include/linux/irq-entry-common.h index 0bb6c03481fa..2be273bb68f0 100644 --- a/include/linux/irq-entry-common.h +++ b/include/linux/irq-entry-common.h @@ -346,22 +346,7 @@ typedef struct irqentry_state { * * Conditional reschedule with additional sanity checks. */ -void raw_irqentry_exit_cond_resched(void); - -#ifdef CONFIG_PREEMPT_DYNAMIC -#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched -#define irqentry_exit_cond_resched_dynamic_disabled NULL -DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched); -#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)() -#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched); -void dynamic_irqentry_exit_cond_resched(void); -#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched() -#endif -#else /* CONFIG_PREEMPT_DYNAMIC */ -#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched() -#endif /* CONFIG_PREEMPT_DYNAMIC */ +void irqentry_exit_cond_resched(void); /** * irqentry_enter_from_kernel_mode - Establish state before invoking the irq handler diff --git a/include/linux/irqchip/arm-gic-v5.h b/include/linux/irqchip/arm-gic-v5.h index 2c2fb39f049c..299e75bc9525 100644 --- a/include/linux/irqchip/arm-gic-v5.h +++ b/include/linux/irqchip/arm-gic-v5.h @@ -63,20 +63,34 @@ #define GICV5_OUTER_SHARE 0b10 #define GICV5_INNER_SHARE 0b11 +#define GICV5_AIDR_COMPONENT_IRS 0b00 +#define GICV5_AIDR_COMPONENT_ITS 0b01 +#define GICV5_AIDR_COMPONENT_IWB 0b10 + +#define GICV5_AIDR_ARCH_MAJ_REV_V5 0 +#define GICV5_AIDR_ARCH_MIN_REV_V0 0 + /* * IRS registers and tables structures */ #define GICV5_IRS_IDR0 0x0000 #define GICV5_IRS_IDR1 0x0004 #define GICV5_IRS_IDR2 0x0008 +#define GICV5_IRS_IDR3 0x000c +#define GICV5_IRS_IDR4 0x0010 #define GICV5_IRS_IDR5 0x0014 #define GICV5_IRS_IDR6 0x0018 #define GICV5_IRS_IDR7 0x001c +#define GICV5_IRS_IIDR 0x0040 +#define GICV5_IRS_AIDR 0x0044 #define GICV5_IRS_CR0 0x0080 #define GICV5_IRS_CR1 0x0084 #define GICV5_IRS_SYNCR 0x00c0 #define GICV5_IRS_SYNC_STATUSR 0x00c4 +#define GICV5_IRS_SPI_VMR 0x0100 #define GICV5_IRS_SPI_SELR 0x0108 +#define GICV5_IRS_SPI_DOMAINR 0x010c +#define GICV5_IRS_SPI_RESAMPLER 0x0110 #define GICV5_IRS_SPI_CFGR 0x0114 #define GICV5_IRS_SPI_STATUSR 0x0118 #define GICV5_IRS_PE_SELR 0x0140 @@ -86,11 +100,51 @@ #define GICV5_IRS_IST_CFGR 0x0190 #define GICV5_IRS_IST_STATUSR 0x0194 #define GICV5_IRS_MAP_L2_ISTR 0x01c0 - +#define GICV5_IRS_VMT_BASER 0x0200 +#define GICV5_IRS_VMT_CFGR 0x0210 +#define GICV5_IRS_VMT_STATUSR 0x0214 +#define GICV5_IRS_VPE_SELR 0x0240 +#define GICV5_IRS_VPE_DBR 0x0248 +#define GICV5_IRS_VPE_HPPIR 0x0250 +#define GICV5_IRS_VPE_CR0 0x0258 +#define GICV5_IRS_VPE_STATUSR 0x025c +#define GICV5_IRS_VM_DBR 0x0280 +#define GICV5_IRS_VM_SELR 0x0288 +#define GICV5_IRS_VM_STATUSR 0x028c +#define GICV5_IRS_VMAP_L2_VMTR 0x02c0 +#define GICV5_IRS_VMAP_VMR 0x02c8 +#define GICV5_IRS_VMAP_VISTR 0x02d0 +#define GICV5_IRS_VMAP_L2_VISTR 0x02d8 +#define GICV5_IRS_VMAP_VPER 0x02e0 +#define GICV5_IRS_SAVE_VMR 0x0300 +#define GICV5_IRS_SAVE_VM_STATUSR 0x0308 +#define GICV5_IRS_MEC_IDR 0x0340 +#define GICV5_IRS_MEC_MECID_R 0x0344 +#define GICV5_IRS_MPAM_IDR 0x0380 +#define GICV5_IRS_MPAM_PARTID_R 0x0384 +#define GICV5_IRS_SWERR_STATUSR 0x03c0 +#define GICV5_IRS_SWERR_SYNDROMER0 0x03c8 +#define GICV5_IRS_SWERR_SYNDROMER1 0x03d0 + +#define GICV5_IRS_IDR0_IRSID GENMASK(31, 16) +#define GICV5_IRS_IDR0_SWE BIT(12) +#define GICV5_IRS_IDR0_MPAM BIT(11) +#define GICV5_IRS_IDR0_MEC BIT(10) +#define GICV5_IRS_IDR0_SETLPI BIT(9) +#define GICV5_IRS_IDR0_VIRT_ONE_N BIT(8) +#define GICV5_IRS_IDR0_ONE_N BIT(7) #define GICV5_IRS_IDR0_VIRT BIT(6) +#define GICV5_IRS_IDR0_PA_RANGE GENMASK(5, 2) +#define GICV5_IRS_IDR0_INT_DOM GENMASK(1, 0) + +#define GICV5_IRS_IDR0_INT_DOM_SECURE 0b00 +#define GICV5_IRS_IDR0_INT_DOM_NON_SECURE 0b01 +#define GICV5_IRS_IDR0_INT_DOM_EL3 0b10 +#define GICV5_IRS_IDR0_INT_DOM_REALM 0b11 #define GICV5_IRS_IDR1_PRIORITY_BITS GENMASK(22, 20) #define GICV5_IRS_IDR1_IAFFID_BITS GENMASK(19, 16) +#define GICV5_IRS_IDR1_PE_CNT GENMASK(15, 0) #define GICV5_IRS_IDR1_PRIORITY_BITS_1BITS 0b000 #define GICV5_IRS_IDR1_PRIORITY_BITS_2BITS 0b001 @@ -106,13 +160,30 @@ #define GICV5_IRS_IDR2_LPI BIT(5) #define GICV5_IRS_IDR2_ID_BITS GENMASK(4, 0) +#define GICV5_IRS_IST_L2SZ_SUPPORT_4KB(r) FIELD_GET(BIT(0), (r)) +#define GICV5_IRS_IST_L2SZ_SUPPORT_16KB(r) FIELD_GET(BIT(1), (r)) +#define GICV5_IRS_IST_L2SZ_SUPPORT_64KB(r) FIELD_GET(BIT(2), (r)) + +#define GICV5_IRS_IDR3_VMT_LEVELS BIT(10) +#define GICV5_IRS_IDR3_VM_ID_BITS GENMASK(9, 5) +#define GICV5_IRS_IDR3_VMD_SZ GENMASK(4, 1) +#define GICV5_IRS_IDR3_VMD BIT(0) + +#define GICV5_IRS_IDR4_VPE_ID_BITS GENMASK(9, 6) +#define GICV5_IRS_IDR4_VPED_SZ GENMASK(5, 0) + #define GICV5_IRS_IDR5_SPI_RANGE GENMASK(24, 0) #define GICV5_IRS_IDR6_SPI_IRS_RANGE GENMASK(24, 0) #define GICV5_IRS_IDR7_SPI_BASE GENMASK(23, 0) -#define GICV5_IRS_IST_L2SZ_SUPPORT_4KB(r) FIELD_GET(BIT(11), (r)) -#define GICV5_IRS_IST_L2SZ_SUPPORT_16KB(r) FIELD_GET(BIT(12), (r)) -#define GICV5_IRS_IST_L2SZ_SUPPORT_64KB(r) FIELD_GET(BIT(13), (r)) +#define GICV5_IRS_IIDR_PRODUCT_ID GENMASK(31, 20) +#define GICV5_IRS_IIDR_VARIANT GENMASK(19, 16) +#define GICV5_IRS_IIDR_REVISION GENMASK(15, 12) +#define GICV5_IRS_IIDR_IMPLEMENTER GENMASK(11, 0) + +#define GICV5_IRS_AIDR_COMPONENT GENMASK(11, 8) +#define GICV5_IRS_AIDR_ARCHMAJORREV GENMASK(7, 4) +#define GICV5_IRS_AIDR_ARCHMINORREV GENMASK(3, 0) #define GICV5_IRS_CR0_IDLE BIT(1) #define GICV5_IRS_CR0_IRSEN BIT(0) @@ -135,21 +206,39 @@ #define GICV5_IRS_SYNC_STATUSR_IDLE BIT(0) -#define GICV5_IRS_SPI_STATUSR_V BIT(1) -#define GICV5_IRS_SPI_STATUSR_IDLE BIT(0) +#define GICV5_IRS_SPI_VMR_VIRT BIT_ULL(63) +#define GICV5_IRS_SPI_VMR_VM_ID GENMASK_ULL(15, 0) #define GICV5_IRS_SPI_SELR_ID GENMASK(23, 0) +#define GICV5_IRS_SPI_DOMAINR_DOMAIN GENMASK(1, 0) + +#define GICV5_IRS_SPI_DOMAINR_DOMAIN_SECURE 0b00 +#define GICV5_IRS_SPI_DOMAINR_DOMAIN_NON_SECURE 0b01 +#define GICV5_IRS_SPI_DOMAINR_DOMAIN_EL3 0b10 +#define GICV5_IRS_SPI_DOMAINR_DOMAIN_REALM 0b11 + +#define GICV5_IRS_SPI_RESAMPLER_ID GENMASK(23, 0) + #define GICV5_IRS_SPI_CFGR_TM BIT(0) +#define GICV5_IRS_SPI_CFGR_TM_EDGE 0b0 +#define GICV5_IRS_SPI_CFGR_TM_LEVEL 0b1 + +#define GICV5_IRS_SPI_STATUSR_V BIT(1) +#define GICV5_IRS_SPI_STATUSR_IDLE BIT(0) + #define GICV5_IRS_PE_SELR_IAFFID GENMASK(15, 0) +#define GICV5_IRS_PE_STATUSR_ONLINE BIT(2) #define GICV5_IRS_PE_STATUSR_V BIT(1) #define GICV5_IRS_PE_STATUSR_IDLE BIT(0) #define GICV5_IRS_PE_CR0_DPS BIT(0) -#define GICV5_IRS_IST_STATUSR_IDLE BIT(0) +#define GICV5_IRS_IST_BASER_ADDR_MASK GENMASK_ULL(55, 6) +#define GICV5_IRS_IST_BASER_VALID BIT_ULL(0) +#define GICV5_IRS_IST_BASER_ADDR_SHIFT 6ULL #define GICV5_IRS_IST_CFGR_STRUCTURE BIT(16) #define GICV5_IRS_IST_CFGR_ISTSZ GENMASK(8, 7) @@ -167,15 +256,111 @@ #define GICV5_IRS_IST_CFGR_L2SZ_16K 0b01 #define GICV5_IRS_IST_CFGR_L2SZ_64K 0b10 -#define GICV5_IRS_IST_BASER_ADDR_MASK GENMASK_ULL(55, 6) -#define GICV5_IRS_IST_BASER_VALID BIT_ULL(0) +#define GICV5_IRS_IST_STATUSR_IDLE BIT(0) #define GICV5_IRS_MAP_L2_ISTR_ID GENMASK(23, 0) +#define GICV5_IRS_VMT_BASER_ADDR GENMASK_ULL(55, 3) +#define GICV5_IRS_VMT_BASER_ADDR_SHIFT 3ULL +#define GICV5_IRS_VMT_BASER_VALID BIT_ULL(0) + +#define GICV5_IRS_VMT_CFGR_STRUCTURE_TWO_LEVEL 0b1 +#define GICV5_IRS_VMT_CFGR_STRUCTURE_LINEAR 0b0 + +#define GICV5_IRS_VMT_CFGR_STRUCTURE BIT(16) +#define GICV5_IRS_VMT_CFGR_VM_ID_BITS GENMASK(4, 0) + +#define GICV5_IRS_VMT_STATUSR_IDLE BIT(0) + +#define GICV5_IRS_VPE_SELR_S BIT_ULL(63) +#define GICV5_IRS_VPE_SELR_VPE_ID GENMASK_ULL(47, 32) +#define GICV5_IRS_VPE_SELR_VM_ID GENMASK_ULL(15, 0) + +#define GICV5_IRS_VPE_DBR_DBV BIT_ULL(63) +#define GICV5_IRS_VPE_DBR_REQ_DB BIT_ULL(62) +#define GICV5_IRS_VPE_DBR_DBPM GENMASK_ULL(36, 32) +#define GICV5_IRS_VPE_DBR_INTID GENMASK_ULL(23, 0) + +#define GICV5_IRS_VPE_HPPIR_HPPIV BIT_ULL(32) +#define GICV5_IRS_VPE_HPPIR_TYPE GENMASK_ULL(31, 29) +#define GICV5_IRS_VPE_HPPIR_ID GENMASK_ULL(23, 0) + +#define GICV5_IRS_VPE_CR0_DPS BIT(0) + +#define GICV5_IRS_VPE_STATUSR_V BIT(1) +#define GICV5_IRS_VPE_STATUSR_IDLE BIT(0) + +#define GICV5_IRS_VM_DBR_EN BIT_ULL(63) +#define GICV5_IRS_VM_DBR_VPE_ID GENMASK_ULL(15, 0) + +#define GICV5_IRS_VM_SELR_VM_ID GENMASK(15, 0) + +#define GICV5_IRS_VM_STATUSR_V BIT(1) +#define GICV5_IRS_VM_STATUSR_IDLE BIT(0) + +#define GICV5_IRS_VMAP_L2_VMTR_M BIT_ULL(63) +#define GICV5_IRS_VMAP_L2_VMTR_VM_ID GENMASK_ULL(15, 0) + +#define GICV5_IRS_VMAP_VMR_M BIT_ULL(63) +#define GICV5_IRS_VMAP_VMR_U BIT_ULL(62) +#define GICV5_IRS_VMAP_VMR_VM_ID GENMASK_ULL(15, 0) + +#define GICV5_IRS_VMAP_VISTR_M BIT_ULL(63) +#define GICV5_IRS_VMAP_VISTR_U BIT_ULL(62) +#define GICV5_IRS_VMAP_VISTR_VM_ID GENMASK_ULL(47, 32) +#define GICV5_IRS_VMAP_VISTR_TYPE GENMASK_ULL(31, 29) + +#define GICV5_IRS_VMAP_L2_VISTR_M BIT_ULL(63) +#define GICV5_IRS_VMAP_L2_VISTR_VM_ID GENMASK_ULL(47, 32) +#define GICV5_IRS_VMAP_L2_VISTR_TYPE GENMASK_ULL(31, 29) +#define GICV5_IRS_VMAP_L2_VISTR_ID GENMASK_ULL(23, 0) + +#define GICV5_IRS_VMAP_VPER_M BIT_ULL(63) +#define GICV5_IRS_VMAP_VPER_VM_ID GENMASK_ULL(47, 32) +#define GICV5_IRS_VMAP_VPER_VPE_ID GENMASK_ULL(15, 0) + +#define GICV5_IRS_SAVE_VMR_VM_ID GENMASK_ULL(15, 0) +#define GICV5_IRS_SAVE_VMR_Q BIT_ULL(62) +#define GICV5_IRS_SAVE_VMR_S BIT_ULL(63) + +#define GICV5_IRS_SAVE_VM_STATUSR_IDLE BIT(0) +#define GICV5_IRS_SAVE_VM_STATUSR_Q BIT(1) + +#define GICV5_IRS_MEC_IDR_MECIDSIZE GENMASK(3, 0) + +#define GICV5_IRS_MEC_MECID_R_MECID GENMASK(15, 0) + +#define GICV5_IRS_MPAM_IDR_HAS_MPAM_SP BIT(24) +#define GICV5_IRS_MPAM_IDR_PMG_MAX GENMASK(23, 16) +#define GICV5_IRS_MPAM_IDR_PARTID_MAX GENMASK(15, 0) + +#define GICV5_IRS_MPAM_PARTID_R_IDLE BIT(31) +#define GICV5_IRS_MPAM_PARTID_R_MPAM_SP GENMASK(25, 24) +#define GICV5_IRS_MPAM_PARTID_R_PMG GENMASK(23, 16) +#define GICV5_IRS_MPAM_PARTID_R_PARTID GENMASK(15, 0) + +#define GICV5_IRS_SWERR_STATUSR_IMP_EC GENMASK_ULL(31, 24) +#define GICV5_IRS_SWERR_STATUSR_EC GENMASK_ULL(23, 16) +#define GICV5_IRS_SWERR_STATUSR_OF BIT_ULL(3) +#define GICV5_IRS_SWERR_STATUSR_S1V BIT_ULL(2) +#define GICV5_IRS_SWERR_STATUSR_S0V BIT_ULL(1) +#define GICV5_IRS_SWERR_STATUSR_V BIT_ULL(0) + +#define GICV5_IRS_SWERR_SYNDROMER0_VIRTUAL BIT_ULL(63) +#define GICV5_IRS_SWERR_SYNDROMER0_TYPE GENMASK_ULL(62, 60) +#define GICV5_IRS_SWERR_SYNDROMER0_ID GENMASK_ULL(55, 32) +#define GICV5_IRS_SWERR_SYNDROMER0_VM_ID GENMASK_ULL(15, 0) + +#define GICV5_IRS_SWERR_SYNDROMER1_ADDR GENMASK_ULL(55, 3) + #define GICV5_ISTL1E_VALID BIT_ULL(0) +#define GICV5_IRS_ISTL1E_SIZE 8UL #define GICV5_ISTL1E_L2_ADDR_MASK GENMASK_ULL(55, 12) +#define GICV5_IRS_SETLPIR 0x0000 +#define GICV5_IRS_SETLPIR_ID GENMASK(23, 0) + /* * ITS registers and tables structures */ @@ -298,6 +483,42 @@ #define GICV5_GSI_IWB_WIRE GENMASK(15, 0) /* + * CoreSight identification registers - ordered by increasing offset. + */ +#define GICV5_CORESIGHT_DEVARCH 0xffbc +#define GICV5_CORESIGHT_PIDR4 0xffd0 +#define GICV5_CORESIGHT_PIDR5 0xffd4 +#define GICV5_CORESIGHT_PIDR6 0xffd8 +#define GICV5_CORESIGHT_PIDR7 0xffdc +#define GICV5_CORESIGHT_PIDR0 0xffe0 +#define GICV5_CORESIGHT_PIDR1 0xffe4 +#define GICV5_CORESIGHT_PIDR2 0xffe8 +#define GICV5_CORESIGHT_PIDR3 0xffec +#define GICV5_CORESIGHT_CIDR0 0xfff0 +#define GICV5_CORESIGHT_CIDR1 0xfff4 +#define GICV5_CORESIGHT_CIDR2 0xfff8 +#define GICV5_CORESIGHT_CIDR3 0xfffc + +#define GICV5_CORESIGHT_DEVARCH_VAL \ + (FIELD_PREP(GENMASK(31, 21), 0x23b) | \ + BIT(20) | \ + 0x5a19) + +#define GICV5_CORESIGHT_PIDR4_JEP106_CONT 0x04 +#define GICV5_CORESIGHT_PIDR5_RES0 0x00 +#define GICV5_CORESIGHT_PIDR6_RES0 0x00 +#define GICV5_CORESIGHT_PIDR7_RES0 0x00 +#define GICV5_CORESIGHT_PIDR0_PART_0 0x4b +#define GICV5_CORESIGHT_PIDR1_DES_0_PART_1 0xb0 +#define GICV5_CORESIGHT_PIDR2_DES_1 0x0b +#define GICV5_CORESIGHT_PIDR3_REVAND_CMOD 0x00 + +#define GICV5_CORESIGHT_CIDR0_VAL 0x0d +#define GICV5_CORESIGHT_CIDR1_VAL 0xf0 +#define GICV5_CORESIGHT_CIDR2_VAL 0x05 +#define GICV5_CORESIGHT_CIDR3_VAL 0xb1 + +/* * Global Data structures and functions */ struct gicv5_chip_data { @@ -332,6 +553,8 @@ struct gicv5_irs_chip_data { raw_spinlock_t spi_config_lock; }; +#define IRS_FLAGS_NON_COHERENT BIT(0) + static inline int gicv5_wait_for_op_s_atomic(void __iomem *addr, u32 offset, const char *reg_s, u32 mask, u32 *val) @@ -379,6 +602,7 @@ void __init gicv5_free_lpi_domain(void); int gicv5_irs_of_probe(struct device_node *parent); int gicv5_irs_acpi_probe(void); +struct gicv5_irs_chip_data *gicv5_irs_get_chip_data(void); void gicv5_irs_remove(void); int gicv5_irs_enable(void); void gicv5_irs_its_probe(void); @@ -387,10 +611,13 @@ int gicv5_irs_cpu_to_iaffid(int cpu_id, u16 *iaffid); struct gicv5_irs_chip_data *gicv5_irs_lookup_by_spi_id(u32 spi_id); int gicv5_spi_irq_set_type(struct irq_data *d, unsigned int type); int gicv5_irs_iste_alloc(u32 lpi); +unsigned int gicv5_irs_l2_sz(u32 l2sz); void gicv5_irs_syncr(void); /* Embedded in kvm.arch */ struct gicv5_vpe { + int db; + bool db_fired; bool resident; }; @@ -429,4 +656,15 @@ void gicv5_deinit_lpis(void); void __init gicv5_its_of_probe(struct device_node *parent); void __init gicv5_its_acpi_probe(void); + +enum gicv5_vcpu_cmd { + VMT_L2_MAP, /* Map in a L2 VMT - *may* happen on VM init */ + VMTE_MAKE_VALID, /* Make the VMTE valid */ + VMTE_MAKE_INVALID, /* Make the VMTE (et al.) invalid */ + VPE_MAKE_VALID, /* No corresponding invalid */ + SPI_VIST_MAKE_VALID, /* No corresponding invalid */ + LPI_VIST_MAKE_VALID, /* Triggered by a guest */ + LPI_VIST_MAKE_INVALID, /* Triggered by a guest */ +}; + #endif diff --git a/include/linux/irqchip/arm-vgic-info.h b/include/linux/irqchip/arm-vgic-info.h index 67d9d960273b..f05370e2debf 100644 --- a/include/linux/irqchip/arm-vgic-info.h +++ b/include/linux/irqchip/arm-vgic-info.h @@ -38,6 +38,11 @@ struct gic_kvm_info { bool has_v4_1; /* Deactivation impared, subpar stuff */ bool no_hw_deactivation; + /* GICv5 IRS base */ + struct { + void __iomem *base; + bool non_coherent; + } gicv5_irs; }; #ifdef CONFIG_KVM diff --git a/include/linux/irqdomain.h b/include/linux/irqdomain.h index 73c25d40846c..3ba75a4ed3da 100644 --- a/include/linux/irqdomain.h +++ b/include/linux/irqdomain.h @@ -752,24 +752,6 @@ static inline void msi_device_domain_free_wired(struct irq_domain *domain, unsig } #endif -static inline struct irq_domain *irq_domain_add_linear(struct device_node *of_node, - unsigned int size, - const struct irq_domain_ops *ops, - void *host_data) -{ - struct irq_domain_info info = { - .fwnode = of_fwnode_handle(of_node), - .size = size, - .hwirq_max = size, - .ops = ops, - .host_data = host_data, - }; - struct irq_domain *d; - - d = irq_domain_instantiate(&info); - return IS_ERR(d) ? NULL : d; -} - #else /* CONFIG_IRQ_DOMAIN */ static inline void irq_dispose_mapping(unsigned int virq) { } static inline struct irq_domain *irq_find_matching_fwnode(struct fwnode_handle *fwnode, diff --git a/include/linux/kdev_t.h b/include/linux/kdev_t.h index 4856706fbfeb..2dbbd47f1e68 100644 --- a/include/linux/kdev_t.h +++ b/include/linux/kdev_t.h @@ -9,7 +9,7 @@ #define MAJOR(dev) ((unsigned int) ((dev) >> MINORBITS)) #define MINOR(dev) ((unsigned int) ((dev) & MINORMASK)) -#define MKDEV(ma,mi) (((ma) << MINORBITS) | (mi)) +#define MKDEV(ma, mi) (((dev_t)(ma) << MINORBITS) | (mi)) #define print_dev_t(buffer, dev) \ sprintf((buffer), "%u:%u\n", MAJOR(dev), MINOR(dev)) diff --git a/include/linux/kernel-page-flags.h b/include/linux/kernel-page-flags.h index 196778a087c4..fe5ab6e50bd7 100644 --- a/include/linux/kernel-page-flags.h +++ b/include/linux/kernel-page-flags.h @@ -11,7 +11,6 @@ #define KPF_RESERVED 32 #define KPF_MLOCKED 33 #define KPF_OWNER_2 34 -#define KPF_PRIVATE 35 #define KPF_PRIVATE_2 36 #define KPF_OWNER_PRIVATE 37 #define KPF_ARCH 38 diff --git a/include/linux/kernel.h b/include/linux/kernel.h index 24414c79e59a..4b11d1dc0a67 100644 --- a/include/linux/kernel.h +++ b/include/linux/kernel.h @@ -43,30 +43,10 @@ struct completion; struct user; #ifdef CONFIG_PREEMPT_VOLUNTARY_BUILD - extern int __cond_resched(void); # define might_resched() __cond_resched() - -#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) - -extern int __cond_resched(void); - -DECLARE_STATIC_CALL(might_resched, __cond_resched); - -static __always_inline void might_resched(void) -{ - static_call_mod(might_resched)(); -} - -#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) - -extern int dynamic_might_resched(void); -# define might_resched() dynamic_might_resched() - #else - # define might_resched() do { } while (0) - #endif /* CONFIG_PREEMPT_* */ #ifdef CONFIG_DEBUG_ATOMIC_SLEEP diff --git a/include/linux/kernel_stat.h b/include/linux/kernel_stat.h index 9ca6c2259dfe..c1e85550bf12 100644 --- a/include/linux/kernel_stat.h +++ b/include/linux/kernel_stat.h @@ -196,6 +196,17 @@ static inline void kcpustat_cpu_fetch(struct kernel_cpustat *dst, int cpu) } #endif /* !CONFIG_VIRT_CPU_ACCOUNTING_GEN */ +static inline u64 kcpustat_field_total(enum cpu_usage_stat usage, const struct cpumask *cpus) +{ + u64 total = 0; + int cpu; + + for_each_cpu(cpu, cpus) + total += kcpustat_field(usage, cpu); + + return total; +} + extern void account_user_time(struct task_struct *, u64); extern void account_guest_time(struct task_struct *, u64); extern void account_system_time(struct task_struct *, int, u64); diff --git a/include/linux/kexec.h b/include/linux/kexec.h index 0af8ae4fdd08..f970ca2c8ce9 100644 --- a/include/linux/kexec.h +++ b/include/linux/kexec.h @@ -259,21 +259,6 @@ int kexec_kernel_verify_pe_sig(const char *kernel, unsigned long kernel_len); extern int kexec_add_buffer(struct kexec_buf *kbuf); int kexec_locate_mem_hole(struct kexec_buf *kbuf); -#ifndef arch_kexec_locate_mem_hole -/** - * arch_kexec_locate_mem_hole - Find free memory to place the segments. - * @kbuf: Parameters for the memory search. - * - * On success, kbuf->mem will have the start address of the memory region found. - * - * Return: 0 on success, negative errno on error. - */ -static inline int arch_kexec_locate_mem_hole(struct kexec_buf *kbuf) -{ - return kexec_locate_mem_hole(kbuf); -} -#endif - #ifndef arch_kexec_apply_relocations_add /* * arch_kexec_apply_relocations_add - apply relocations of type RELA @@ -418,7 +403,7 @@ struct kimage { #endif struct { - struct kexec_segment *scratch; + struct kexec_segment *bootmem; phys_addr_t fdt; } kho; diff --git a/include/linux/kexec_handover.h b/include/linux/kexec_handover.h index 46de86dc343e..ef2b188c8ed0 100644 --- a/include/linux/kexec_handover.h +++ b/include/linux/kexec_handover.h @@ -36,15 +36,15 @@ int kho_retrieve_subtree(const char *name, phys_addr_t *phys, size_t *size); void kho_memory_init(void); void kho_memory_init_early(void); -void kho_populate(phys_addr_t fdt_phys, u64 fdt_len, phys_addr_t scratch_phys, - u64 scratch_len); +void kho_populate(phys_addr_t fdt_phys, u64 fdt_len, phys_addr_t bootmem_phys, + u64 bootmem_len); -bool kho_scratch_overlap(phys_addr_t phys, size_t size); +bool kho_bootmem_overlap(phys_addr_t phys, size_t size); -static inline enum migratetype kho_scratch_migratetype(unsigned long pfn, +static inline enum migratetype kho_bootmem_migratetype(unsigned long pfn, enum migratetype mt) { - if (kho_scratch_overlap(PFN_PHYS(pfn), pageblock_nr_pages << PAGE_SHIFT)) + if (kho_bootmem_overlap(PFN_PHYS(pfn), pageblock_nr_pages << PAGE_SHIFT)) return MIGRATE_CMA; return mt; } @@ -123,16 +123,16 @@ static inline void kho_memory_init(void) { } static inline void kho_memory_init_early(void) { } static inline void kho_populate(phys_addr_t fdt_phys, u64 fdt_len, - phys_addr_t scratch_phys, u64 scratch_len) + phys_addr_t bootmem_phys, u64 bootmem_len) { } -static inline bool kho_scratch_overlap(phys_addr_t phys, size_t size) +static inline bool kho_bootmem_overlap(phys_addr_t phys, size_t size) { return false; } -static inline enum migratetype kho_scratch_migratetype(unsigned long pfn, +static inline enum migratetype kho_bootmem_migratetype(unsigned long pfn, enum migratetype mt) { return mt; diff --git a/include/linux/key.h b/include/linux/key.h index 81b8f05c6898..bd10fe45819d 100644 --- a/include/linux/key.h +++ b/include/linux/key.h @@ -440,6 +440,8 @@ extern key_ref_t keyring_search(key_ref_t keyring, extern int keyring_restrict(key_ref_t keyring, const char *type, const char *restriction); +extern void key_register_bpf_keyring(struct key *keyring); + extern struct key *key_lookup(key_serial_t id); static inline key_serial_t key_serial(const struct key *key) diff --git a/include/linux/kho/abi/pci.h b/include/linux/kho/abi/pci.h new file mode 100644 index 000000000000..9485ed73c351 --- /dev/null +++ b/include/linux/kho/abi/pci.h @@ -0,0 +1,66 @@ +/* SPDX-License-Identifier: GPL-2.0 */ + +/* + * Copyright (c) 2026, Google LLC. + * David Matlack <dmatlack@google.com> + */ + +#ifndef _LINUX_KHO_ABI_PCI_H +#define _LINUX_KHO_ABI_PCI_H + +#include <linux/bug.h> +#include <linux/compiler.h> +#include <linux/types.h> + +/** + * DOC: PCI File-Lifecycle Bound (FLB) Live Update ABI + * + * This header defines the ABI for preserving core PCI state across kexec using + * Live Update File-Lifecycle Bound (FLB) data. + * + * This interface is a contract. Any modification to any of the serialization + * structs defined here constitutes a breaking change. Such changes require + * incrementing the version number in the PCI_LUO_FLB_VERSION number. + */ + +#define PCI_LUO_FLB_COMPATIBLE "pci" +#define PCI_LUO_FLB_VERSION 2 + +/** + * struct pci_dev_ser - Serialized state about a single PCI device. + * + * @domain: The device's PCI domain number (segment). + * @bdf: The device's PCI bus, device, and function number. + * @refcount: Reference count used by the PCI core to keep track of whether it + * is done using a device's struct pci_dev_ser. The value of the + * refcount is equal to the number of preserved devices at or below + * it in the PCI hierarchy when the struct pci_dev_ser is in use, and + * 0 otherwise. + */ +struct pci_dev_ser { + u32 domain; + u16 bdf; + u16 refcount; +} __packed; + +/** + * struct pci_ser - PCI Subsystem Live Update State + * + * This struct tracks state about all devices that are being preserved across + * a Live Update for the next kernel. It contains only state owned by the PCI + * core; the state a driver needs to resume its device is preserved separately + * by that driver. + * + * @version: The version of the "pci" FLB struct. This field must never be + * deleted, moved, or resized, as the kernel depends on always being + * able to check the struct pci_ser version number. + * @nr_devices: The number of devices that were preserved. + * @devices: Physical address of the first KHO block containing pci_dev_ser. + */ +struct pci_ser { + u32 version; + u32 nr_devices; + u64 devices; +} __packed; + +#endif /* _LINUX_KHO_ABI_PCI_H */ diff --git a/include/linux/kmod.h b/include/linux/kmod.h index 9a07c3215389..b9474a62a568 100644 --- a/include/linux/kmod.h +++ b/include/linux/kmod.h @@ -2,17 +2,9 @@ #ifndef __LINUX_KMOD_H__ #define __LINUX_KMOD_H__ -/* - * include/linux/kmod.h - */ - -#include <linux/umh.h> -#include <linux/gfp.h> -#include <linux/stddef.h> +#include <linux/compiler_attributes.h> #include <linux/errno.h> -#include <linux/compiler.h> -#include <linux/workqueue.h> -#include <linux/sysctl.h> +#include <linux/types.h> #ifdef CONFIG_MODULES /* modprobe exit status on success, -ve on error. Return value diff --git a/include/linux/kprobes.h b/include/linux/kprobes.h index 8c4f3bb24429..e6de7ae55bda 100644 --- a/include/linux/kprobes.h +++ b/include/linux/kprobes.h @@ -181,6 +181,7 @@ struct kprobe_blacklist_entry { struct list_head list; unsigned long start_addr; unsigned long end_addr; + struct rcu_head rcu; }; #ifdef CONFIG_KPROBES diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index 03bfc92864b6..c0748f7e40ea 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -722,27 +722,6 @@ static inline int kvm_arch_vcpu_memslots_id(struct kvm_vcpu *vcpu) } #endif -#ifndef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES -static inline bool kvm_arch_has_private_mem(struct kvm *kvm) -{ - return false; -} -#endif - -#ifdef CONFIG_KVM_GUEST_MEMFD -bool kvm_arch_supports_gmem_init_shared(struct kvm *kvm); - -static inline u64 kvm_gmem_get_supported_flags(struct kvm *kvm) -{ - u64 flags = GUEST_MEMFD_FLAG_MMAP; - - if (!kvm || kvm_arch_supports_gmem_init_shared(kvm)) - flags |= GUEST_MEMFD_FLAG_INIT_SHARED; - - return flags; -} -#endif - #ifndef kvm_arch_has_readonly_mem static inline bool kvm_arch_has_readonly_mem(struct kvm *kvm) { @@ -791,7 +770,6 @@ struct kvm { /* The current active memslot set for each address space */ struct kvm_memslots __rcu *memslots[KVM_MAX_NR_ADDRESS_SPACES]; struct xarray vcpu_array; - DECLARE_BITMAP(vcpu_ids, KVM_MAX_VCPU_IDS); /* * Protected by slots_lock, but can be read outside if an * incorrect answer is acceptable. @@ -855,6 +833,8 @@ struct kvm { gfn_t mmu_invalidate_range_start; gfn_t mmu_invalidate_range_end; + unsigned long gpc_invalidate_seq; + struct list_head devices; u64 manual_dirty_log_protect; struct dentry *debugfs_dentry; @@ -872,7 +852,7 @@ struct kvm { #ifdef CONFIG_HAVE_KVM_PM_NOTIFIER struct notifier_block pm_notifier; #endif -#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES +#ifdef CONFIG_KVM_VM_MEMORY_ATTRIBUTES /* Protected by slots_lock (for writes) and RCU (for reads) */ struct xarray mem_attr_array; #endif @@ -2324,13 +2304,18 @@ static inline bool kvm_test_request(int req, struct kvm_vcpu *vcpu) return test_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests); } -static inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu) +static __always_inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu) { + BUILD_BUG_ON(req == KVM_REQ_VM_DEAD); + clear_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests); } -static inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu) +static __always_inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu) { + /* Once a VM is dead, it needs to stay dead. */ + BUILD_BUG_ON(req == KVM_REQ_VM_DEAD); + if (kvm_test_request(req, vcpu)) { kvm_clear_request(req, vcpu); @@ -2560,48 +2545,80 @@ static inline bool kvm_memslot_is_gmem_only(const struct kvm_memory_slot *slot) return slot->flags & KVM_MEMSLOT_GMEM_ONLY; } -#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES -static inline unsigned long kvm_get_memory_attributes(struct kvm *kvm, gfn_t gfn) +#ifdef CONFIG_KVM_VM_MEMORY_ATTRIBUTES +static inline unsigned long kvm_get_vm_memory_attributes(struct kvm *kvm, gfn_t gfn) { return xa_to_value(xa_load(&kvm->mem_attr_array, gfn)); } -bool kvm_range_has_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end, - unsigned long mask, unsigned long attrs); -bool kvm_arch_pre_set_memory_attributes(struct kvm *kvm, - struct kvm_gfn_range *range); -bool kvm_arch_post_set_memory_attributes(struct kvm *kvm, - struct kvm_gfn_range *range); +bool kvm_range_has_vm_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end, + unsigned long mask, unsigned long attrs); +bool kvm_arch_pre_set_vm_memory_attributes(struct kvm *kvm, + struct kvm_gfn_range *range); +bool kvm_arch_post_set_vm_memory_attributes(struct kvm *kvm, + struct kvm_gfn_range *range); -static inline bool kvm_mem_is_private(struct kvm *kvm, gfn_t gfn) +static inline bool kvm_vm_is_private_gfn(struct kvm *kvm, gfn_t gfn) { - return kvm_get_memory_attributes(kvm, gfn) & KVM_MEMORY_ATTRIBUTE_PRIVATE; + return kvm_get_vm_memory_attributes(kvm, gfn) & KVM_MEMORY_ATTRIBUTE_PRIVATE; +} +#endif /* CONFIG_KVM_VM_MEMORY_ATTRIBUTES */ + +#ifdef kvm_arch_has_private_mem +extern bool gmem_in_place_conversion; + +typedef bool (kvm_is_private_gfn_t)(struct kvm *kvm, gfn_t gfn); +DECLARE_STATIC_CALL(__kvm_is_private_gfn, kvm_is_private_gfn_t); + +static inline bool kvm_is_private_gfn(struct kvm *kvm, gfn_t gfn) +{ + return static_call(__kvm_is_private_gfn)(kvm, gfn); } #else -static inline bool kvm_mem_is_private(struct kvm *kvm, gfn_t gfn) +#define gmem_in_place_conversion false + +static inline bool kvm_arch_has_private_mem(struct kvm *kvm) +{ + return false; +} + +static inline bool kvm_is_private_gfn(struct kvm *kvm, gfn_t gfn) { return false; } -#endif /* CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES */ +#endif /* kvm_arch_has_private_mem */ #ifdef CONFIG_KVM_GUEST_MEMFD +bool kvm_gmem_is_private_gfn(struct kvm *kvm, gfn_t gfn); +bool kvm_arch_supports_gmem_init_shared(struct kvm *kvm); + +static inline u64 kvm_gmem_get_supported_flags(struct kvm *kvm) +{ + u64 flags = GUEST_MEMFD_FLAG_MMAP; + + if (!kvm || kvm_arch_supports_gmem_init_shared(kvm)) + flags |= GUEST_MEMFD_FLAG_INIT_SHARED; + + return flags; +} + int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot, - gfn_t gfn, kvm_pfn_t *pfn, struct page **page, - int *max_order); + gfn_t gfn, kvm_pfn_t *pfn, int *max_order); #else static inline int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot, gfn_t gfn, - kvm_pfn_t *pfn, struct page **page, - int *max_order) + kvm_pfn_t *pfn, int *max_order) { KVM_BUG_ON(1, kvm); return -EIO; } #endif /* CONFIG_KVM_GUEST_MEMFD */ -#ifdef CONFIG_HAVE_KVM_ARCH_GMEM_CONVERT int kvm_arch_gmem_make_private(struct kvm *kvm, gfn_t gfn, kvm_pfn_t pfn, kvm_pfn_t nr_pages); +void kvm_arch_gmem_make_shared(kvm_pfn_t pfn, kvm_pfn_t nr_pages); +#ifndef CONFIG_HAVE_KVM_ARCH_GMEM_CONVERT +#define kvm_arch_has_gmem_convert() false #endif #ifdef CONFIG_HAVE_KVM_ARCH_GMEM_POPULATE @@ -2643,6 +2660,7 @@ void kvm_arch_gmem_invalidate_range(struct kvm *kvm, struct kvm_gfn_range *range #endif #ifdef CONFIG_KVM_GENERIC_PRE_FAULT_MEMORY +int kvm_arch_pre_fault_allowed(struct kvm_vcpu *vcpu); long kvm_arch_vcpu_pre_fault_memory(struct kvm_vcpu *vcpu, struct kvm_pre_fault_memory *range); #endif diff --git a/include/linux/libata.h b/include/linux/libata.h index 313e96173b19..3bd74d79f812 100644 --- a/include/linux/libata.h +++ b/include/linux/libata.h @@ -76,7 +76,7 @@ enum ata_quirks { __ATA_QUIRK_MAX_SEC, /* Limit max sectors */ __ATA_QUIRK_MAX_TRIM_128M, /* Limit max trim size to 128M */ __ATA_QUIRK_NO_NCQ_ON_ATI, /* Disable NCQ on ATI chipset */ - __ATA_QUIRK_NO_LPM_ON_ATI, /* Disable LPM on ATI chipset */ + __ATA_QUIRK_NO_LPM_ON_ATI_AND_AMD, /* Disable LPM on ATI and AMD chipsets */ __ATA_QUIRK_NO_ID_DEV_LOG, /* Identify device log missing */ __ATA_QUIRK_NO_LOG_DIR, /* Do not read log directory */ __ATA_QUIRK_NO_FUA, /* Do not use FUA */ @@ -115,7 +115,7 @@ enum { ATA_QUIRK_MAX_SEC = BIT_ULL(__ATA_QUIRK_MAX_SEC), ATA_QUIRK_MAX_TRIM_128M = BIT_ULL(__ATA_QUIRK_MAX_TRIM_128M), ATA_QUIRK_NO_NCQ_ON_ATI = BIT_ULL(__ATA_QUIRK_NO_NCQ_ON_ATI), - ATA_QUIRK_NO_LPM_ON_ATI = BIT_ULL(__ATA_QUIRK_NO_LPM_ON_ATI), + ATA_QUIRK_NO_LPM_ON_ATI_AND_AMD = BIT_ULL(__ATA_QUIRK_NO_LPM_ON_ATI_AND_AMD), ATA_QUIRK_NO_ID_DEV_LOG = BIT_ULL(__ATA_QUIRK_NO_ID_DEV_LOG), ATA_QUIRK_NO_LOG_DIR = BIT_ULL(__ATA_QUIRK_NO_LOG_DIR), ATA_QUIRK_NO_FUA = BIT_ULL(__ATA_QUIRK_NO_FUA), @@ -1147,6 +1147,7 @@ extern struct ata_host *ata_host_alloc_pinfo(struct device *dev, extern void ata_host_get(struct ata_host *host); extern void ata_host_put(struct ata_host *host); extern int ata_host_start(struct ata_host *host); +void ata_host_undo_start(struct ata_host *host); extern int ata_host_register(struct ata_host *host, const struct scsi_host_template *sht); extern int ata_host_activate(struct ata_host *host, int irq, diff --git a/include/linux/list.h b/include/linux/list.h index 77fb62f79928..e3753695e76c 100644 --- a/include/linux/list.h +++ b/include/linux/list.h @@ -1172,7 +1172,7 @@ static inline void hlist_move_list(struct hlist_head *old, { new->first = old->first; if (new->first) - new->first->pprev = &new->first; + WRITE_ONCE(new->first->pprev, &new->first); old->first = NULL; } @@ -1189,10 +1189,10 @@ static inline void hlist_splice_init(struct hlist_head *from, struct hlist_head *to) { if (to->first) - to->first->pprev = &last->next; + WRITE_ONCE(to->first->pprev, &last->next); last->next = to->first; to->first = from->first; - from->first->pprev = &to->first; + WRITE_ONCE(from->first->pprev, &to->first); from->first = NULL; } diff --git a/include/linux/liveupdate.h b/include/linux/liveupdate.h index 63ea5417de84..6051abc0612c 100644 --- a/include/linux/liveupdate.h +++ b/include/linux/liveupdate.h @@ -25,6 +25,7 @@ struct file; /** * struct liveupdate_file_op_args - Arguments for file operation callbacks. * @handler: The file handler being called. + * @session: The session this file belongs to. * @retrieve_status: The retrieve status for the 'can_finish / finish' * operation. A value of 0 means the retrieve has not been * attempted, a positive value means the retrieve was @@ -45,6 +46,7 @@ struct file; */ struct liveupdate_file_op_args { struct liveupdate_file_handler *handler; + struct liveupdate_session *session; int retrieve_status; struct file *file; u64 serialized_data; @@ -247,6 +249,14 @@ void liveupdate_flb_put_incoming(struct liveupdate_flb *flb); int liveupdate_flb_get_outgoing(struct liveupdate_flb *flb, void **objp); void liveupdate_flb_put_outgoing(struct liveupdate_flb *flb); +/* kernel can internally retrieve files */ +int liveupdate_get_file_incoming(struct liveupdate_session *s, u64 token, + struct file **filep); + +/* Get a token for an outgoing file, or -ENOENT if file is not preserved */ +int liveupdate_get_token_outgoing(struct liveupdate_session *s, + struct file *file, u64 *tokenp); + #else /* CONFIG_LIVEUPDATE */ static inline bool liveupdate_enabled(void) @@ -299,5 +309,17 @@ static inline void liveupdate_flb_put_outgoing(struct liveupdate_flb *flb) { } +static inline int liveupdate_get_file_incoming(struct liveupdate_session *s, + u64 token, struct file **filep) +{ + return -EOPNOTSUPP; +} + +static inline int liveupdate_get_token_outgoing(struct liveupdate_session *s, + struct file *file, u64 *tokenp) +{ + return -EOPNOTSUPP; +} + #endif /* CONFIG_LIVEUPDATE */ #endif /* _LINUX_LIVEUPDATE_H */ diff --git a/include/linux/llc.h b/include/linux/llc.h index 944e9e213112..c03bb12bcc9c 100644 --- a/include/linux/llc.h +++ b/include/linux/llc.h @@ -9,9 +9,4 @@ #include <uapi/linux/llc.h> -#define LLC_SAP_DYN_START 0xC0 -#define LLC_SAP_DYN_STOP 0xDE -#define LLC_SAP_DYN_TRIES 4 - -#define llc_ui_skb_cb(__skb) ((struct sockaddr_llc *)&((__skb)->cb[0])) #endif /* __LINUX_LLC_H */ diff --git a/include/linux/lsm_audit.h b/include/linux/lsm_audit.h index 584db296e43b..5cf0b4795065 100644 --- a/include/linux/lsm_audit.h +++ b/include/linux/lsm_audit.h @@ -44,7 +44,7 @@ struct lsm_network_audit { struct lsm_ioctlop_audit { struct path path; - u16 cmd; + unsigned int cmd; }; struct lsm_ibpkey_audit { @@ -78,6 +78,7 @@ struct common_audit_data { #define LSM_AUDIT_DATA_NOTIFICATION 16 #define LSM_AUDIT_DATA_ANONINODE 17 #define LSM_AUDIT_DATA_NLMSGTYPE 18 +#define LSM_AUDIT_DATA_NS 19 union { struct path path; struct dentry *dentry; @@ -100,6 +101,10 @@ struct common_audit_data { int reason; const char *anonclass; u16 nlmsg_type; + struct { + u32 ns_type; + u64 ns_id; + } ns; } u; /* this union contains LSM specific data */ union { diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h index 65c9609ec207..ef16e5343e09 100644 --- a/include/linux/lsm_hook_defs.h +++ b/include/linux/lsm_hook_defs.h @@ -36,6 +36,7 @@ LSM_HOOK(int, 0, binder_transfer_file, const struct cred *from, LSM_HOOK(int, 0, ptrace_access_check, struct task_struct *child, unsigned int mode) LSM_HOOK(int, 0, ptrace_traceme, struct task_struct *parent) +LSM_HOOK(int, 0, mem_foll_force, const struct cred *subject, bool opened_by_owner) LSM_HOOK(int, 0, capget, const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, kernel_cap_t *permitted) LSM_HOOK(int, 0, capset, struct cred *new, const struct cred *old, @@ -94,7 +95,7 @@ LSM_HOOK(int, 0, path_mkdir, const struct path *dir, struct dentry *dentry, LSM_HOOK(int, 0, path_rmdir, const struct path *dir, struct dentry *dentry) LSM_HOOK(int, 0, path_mknod, const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev) -LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, const struct mnt_idmap *idmap, struct dentry *dentry) LSM_HOOK(int, 0, path_truncate, const struct path *path) LSM_HOOK(int, 0, path_symlink, const struct path *dir, struct dentry *dentry, @@ -120,59 +121,60 @@ LSM_HOOK(int, -EOPNOTSUPP, inode_init_security, struct inode *inode, int *xattr_count) LSM_HOOK(int, 0, inode_init_security_anon, struct inode *inode, const struct qstr *name, const struct inode *context_inode) -LSM_HOOK(int, 0, inode_create, struct inode *dir, struct dentry *dentry, - umode_t mode) -LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_create, const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode) +LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, const struct mnt_idmap *idmap, struct inode *inode) -LSM_HOOK(int, 0, inode_link, struct dentry *old_dentry, struct inode *dir, - struct dentry *new_dentry) +LSM_HOOK(int, 0, inode_link, const struct mnt_idmap *idmap, + struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry) LSM_HOOK(int, 0, inode_unlink, struct inode *dir, struct dentry *dentry) -LSM_HOOK(int, 0, inode_symlink, struct inode *dir, struct dentry *dentry, - const char *old_name) -LSM_HOOK(int, 0, inode_mkdir, struct inode *dir, struct dentry *dentry, - umode_t mode) +LSM_HOOK(int, 0, inode_symlink, const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, const char *old_name) +LSM_HOOK(int, 0, inode_mkdir, const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode) LSM_HOOK(int, 0, inode_rmdir, struct inode *dir, struct dentry *dentry) -LSM_HOOK(int, 0, inode_mknod, struct inode *dir, struct dentry *dentry, - umode_t mode, dev_t dev) +LSM_HOOK(int, 0, inode_mknod, const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode, dev_t dev) LSM_HOOK(int, 0, inode_rename, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry) LSM_HOOK(int, 0, inode_readlink, struct dentry *dentry) LSM_HOOK(int, 0, inode_follow_link, struct dentry *dentry, struct inode *inode, bool rcu) -LSM_HOOK(int, 0, inode_permission, struct inode *inode, int mask) -LSM_HOOK(int, 0, inode_setattr, struct mnt_idmap *idmap, struct dentry *dentry, +LSM_HOOK(int, 0, inode_permission, const struct mnt_idmap *idmap, + struct inode *inode, int mask) +LSM_HOOK(int, 0, inode_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) -LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) LSM_HOOK(int, 0, inode_getattr, const struct path *path) LSM_HOOK(int, 0, inode_xattr_skipcap, const char *name) -LSM_HOOK(int, 0, inode_setxattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_setxattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(void, LSM_RET_VOID, inode_post_setxattr, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(int, 0, inode_getxattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_listxattr, struct dentry *dentry) -LSM_HOOK(int, 0, inode_removexattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_removexattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) LSM_HOOK(void, LSM_RET_VOID, inode_post_removexattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_file_setattr, struct dentry *dentry, struct file_kattr *fa) LSM_HOOK(int, 0, inode_file_getattr, struct dentry *dentry, struct file_kattr *fa) -LSM_HOOK(int, 0, inode_set_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_set_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) LSM_HOOK(void, LSM_RET_VOID, inode_post_set_acl, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) -LSM_HOOK(int, 0, inode_get_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_get_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(int, 0, inode_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) LSM_HOOK(int, 0, inode_need_killpriv, struct dentry *dentry) -LSM_HOOK(int, 0, inode_killpriv, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_killpriv, const struct mnt_idmap *idmap, struct dentry *dentry) -LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, struct mnt_idmap *idmap, +LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) LSM_HOOK(int, -EOPNOTSUPP, inode_setsecurity, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -187,7 +189,7 @@ LSM_HOOK(int, 0, inode_setintegrity, const struct inode *inode, enum lsm_integrity_type type, const void *value, size_t size) LSM_HOOK(int, 0, kernfs_init_security, struct kernfs_node *kn_dir, struct kernfs_node *kn) -LSM_HOOK(int, 0, file_permission, struct file *file, int mask) +LSM_HOOK(int, 0, file_permission, const struct file *file, int mask) LSM_HOOK(int, 0, file_alloc_security, struct file *file) LSM_HOOK(void, LSM_RET_VOID, file_release, struct file *file) LSM_HOOK(void, LSM_RET_VOID, file_free_security, struct file *file) @@ -265,6 +267,9 @@ LSM_HOOK(int, -ENOSYS, task_prctl, int option, unsigned long arg2, LSM_HOOK(void, LSM_RET_VOID, task_to_inode, struct task_struct *p, struct inode *inode) LSM_HOOK(int, 0, userns_create, const struct cred *cred) +LSM_HOOK(int, 0, namespace_init, struct ns_common *ns) +LSM_HOOK(void, LSM_RET_VOID, namespace_free, struct ns_common *ns) +LSM_HOOK(int, 0, namespace_install, const struct nsset *nsset, struct ns_common *ns) LSM_HOOK(int, 0, ipc_permission, struct kern_ipc_perm *ipcp, short flag) LSM_HOOK(void, LSM_RET_VOID, ipc_getlsmprop, struct kern_ipc_perm *ipcp, struct lsm_prop *prop) diff --git a/include/linux/lsm_hooks.h b/include/linux/lsm_hooks.h index c4488c4a6d8a..13621e9e233e 100644 --- a/include/linux/lsm_hooks.h +++ b/include/linux/lsm_hooks.h @@ -112,6 +112,7 @@ struct lsm_blob_sizes { unsigned int lbs_ipc; unsigned int lbs_key; unsigned int lbs_msg_msg; + unsigned int lbs_ns; unsigned int lbs_perf_event; unsigned int lbs_task; unsigned int lbs_xattr_count; /* num xattr slots in new_xattrs array */ diff --git a/include/linux/mailbox/riscv-rpmi-message.h b/include/linux/mailbox/riscv-rpmi-message.h index e135c6564d0c..d5362b5821f9 100644 --- a/include/linux/mailbox/riscv-rpmi-message.h +++ b/include/linux/mailbox/riscv-rpmi-message.h @@ -93,6 +93,7 @@ static inline int rpmi_to_linux_error(int rpmi_error) /* RPMI service group IDs */ #define RPMI_SRVGRP_SYSTEM_MSI 0x00002 #define RPMI_SRVGRP_CLOCK 0x00008 +#define RPMI_SRVGRP_DEVICE_POWER 0x00009 /* RPMI clock service IDs */ enum rpmi_clock_service_id { @@ -119,6 +120,16 @@ enum rpmi_sysmsi_service_id { RPMI_SYSMSI_SRV_ID_MAX_COUNT }; +/* RPMI device power service IDs */ +enum rpmi_device_power_service_id { + RPMI_DP_SRV_ENABLE_NOTIFICATION = 0x01, + RPMI_DP_SRV_GET_NUM_DOMAINS = 0x02, + RPMI_DP_SRV_GET_ATTRS = 0x03, + RPMI_DP_SRV_SET_STATE = 0x04, + RPMI_DP_SRV_GET_STATE = 0x05, + RPMI_DP_SRV_ID_MAX_COUNT, +}; + /* RPMI Linux mailbox attribute IDs */ enum rpmi_mbox_attribute_id { RPMI_MBOX_ATTR_SPEC_VERSION, diff --git a/include/linux/maple_tree.h b/include/linux/maple_tree.h index e595ae5cd0ee..e30e57f5f3df 100644 --- a/include/linux/maple_tree.h +++ b/include/linux/maple_tree.h @@ -9,6 +9,7 @@ */ #include <linux/kernel.h> +#include <linux/compiler.h> #include <linux/rcupdate.h> #include <linux/spinlock.h> @@ -297,7 +298,8 @@ struct maple_tree { #endif #define DEFINE_MTREE(name) \ - struct maple_tree name = MTREE_INIT(name, 0) + struct maple_tree name = MTREE_INIT(name, 0); \ + ASSERT_STATIC_STORAGE(name) #define mtree_lock(mt) spin_lock((&(mt)->ma_lock)) #define mtree_lock_nested(mas, subclass) \ diff --git a/include/linux/mdio-mux.h b/include/linux/mdio-mux.h index a5d58f221939..e2af9cb8e017 100644 --- a/include/linux/mdio-mux.h +++ b/include/linux/mdio-mux.h @@ -1,5 +1,5 @@ /* - * MDIO bus multiplexer framwork. + * MDIO bus multiplexer framework. * * This file is subject to the terms and conditions of the GNU General Public * License. See the file "COPYING" in the main directory of this archive diff --git a/include/linux/memblock.h b/include/linux/memblock.h index d62db9e776cf..ea288cf21400 100644 --- a/include/linux/memblock.h +++ b/include/linux/memblock.h @@ -46,11 +46,11 @@ extern unsigned long long max_possible_pfn; * @MEMBLOCK_RSRV_KERN: memory region that is reserved for kernel use, * either explictitly with memblock_reserve_kern() or via memblock * allocation APIs. All memblock allocations set this flag. - * @MEMBLOCK_KHO_SCRATCH: memory region that kexec can pass to the next - * kernel in handover mode. During early boot, we do not know about all - * memory reservations yet, so we get scratch memory from the previous - * kernel that we know is good to use. It is the only memory that - * allocations may happen from in this phase. + * @MEMBLOCK_KHO_NOPRSRV: memory region with no preservation from kexec + * handover. During early boot, we do not know about all memory preservations + * yet, so we get memory with no preservations from the previous kernel that we + * know is good to use. It is the only memory that allocations may happen from + * in this phase. * @MEMBLOCK_RSRV_HUGETLB: memory is reserved for hugetlb pages */ enum memblock_flags { @@ -61,7 +61,7 @@ enum memblock_flags { MEMBLOCK_DRIVER_MANAGED = 0x8, /* always detected via a driver */ MEMBLOCK_RSRV_NOINIT = 0x10, /* don't initialize struct pages */ MEMBLOCK_RSRV_KERN = 0x20, /* memory reserved for kernel use */ - MEMBLOCK_KHO_SCRATCH = 0x40, /* scratch memory for kexec handover */ + MEMBLOCK_KHO_NOPRSRV = 0x40, /* memory with no KHO preservations */ MEMBLOCK_RSRV_HUGETLB = 0x80, /* memory reserved for hugetlb pages */ }; @@ -158,11 +158,10 @@ int memblock_mark_nomap(phys_addr_t base, phys_addr_t size); int memblock_clear_nomap(phys_addr_t base, phys_addr_t size); int memblock_reserved_mark_noinit(phys_addr_t base, phys_addr_t size); int memblock_reserved_mark_kern(phys_addr_t base, phys_addr_t size); -int memblock_mark_kho_scratch(phys_addr_t base, phys_addr_t size); -int memblock_clear_kho_scratch(phys_addr_t base, phys_addr_t size); +int memblock_mark_kho_noprsrv(phys_addr_t base, phys_addr_t size); +int memblock_clear_kho_noprsrv(phys_addr_t base, phys_addr_t size); void memblock_free(void *ptr, size_t size); -void reset_all_zones_managed_pages(void); /* Low level functions */ void __next_mem_range(u64 *idx, int nid, enum memblock_flags flags, @@ -301,9 +300,9 @@ static inline bool memblock_is_driver_managed(struct memblock_region *m) return m->flags & MEMBLOCK_DRIVER_MANAGED; } -static inline bool memblock_is_kho_scratch(struct memblock_region *m) +static inline bool memblock_is_kho_noprsrv(struct memblock_region *m) { - return m->flags & MEMBLOCK_KHO_SCRATCH; + return m->flags & MEMBLOCK_KHO_NOPRSRV; } int memblock_search_pfn_nid(unsigned long pfn, unsigned long *start_pfn, @@ -614,12 +613,12 @@ static inline void early_memtest(phys_addr_t start, phys_addr_t end) { } static inline void memtest_report_meminfo(struct seq_file *m) { } #endif -#ifdef CONFIG_MEMBLOCK_KHO_SCRATCH -void memblock_set_kho_scratch_only(void); -void memblock_clear_kho_scratch_only(void); +#ifdef CONFIG_KEXEC_HANDOVER +void memblock_set_kho_noprsrv_only(void); +void memblock_clear_kho_noprsrv_only(void); #else -static inline void memblock_set_kho_scratch_only(void) { } -static inline void memblock_clear_kho_scratch_only(void) { } +static inline void memblock_set_kho_noprsrv_only(void) { } +static inline void memblock_clear_kho_noprsrv_only(void) { } #endif #endif /* _LINUX_MEMBLOCK_H */ diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 7d1c0ce189a8..74110a324f9e 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -23,6 +23,7 @@ #include <linux/writeback.h> #include <linux/page-flags.h> #include <linux/shrinker.h> +#include <linux/irq_work_types.h> struct mem_cgroup; struct obj_cgroup; @@ -66,11 +67,6 @@ struct mem_cgroup_reclaim_cookie { #define MEM_CGROUP_ID_SHIFT 16 -struct mem_cgroup_private_id { - int id; - refcount_t ref; -}; - struct memcg_vmstats_percpu; struct memcg1_events_percpu; struct memcg_vmstats; @@ -87,50 +83,48 @@ struct mem_cgroup_reclaim_iter { * per-node information in memory controller. */ struct mem_cgroup_per_node { - /* Keep the read-only fields at the start */ + /* Set when the memcg is created, then only read. */ + __cacheline_group_begin_aligned(memcg_pn_read_mostly); struct mem_cgroup *memcg; /* Back pointer, we cannot */ /* use container_of */ struct lruvec_stats_percpu __percpu *lruvec_stats_percpu; struct lruvec_stats *lruvec_stats; struct shrinker_info __rcu *shrinker_info; + struct obj_cgroup __rcu *objcg; + + __cacheline_group_end_aligned(memcg_pn_read_mostly); -#ifdef CONFIG_MEMCG_V1 /* - * Memcg-v1 only stuff in middle as buffer between read mostly fields - * and update often fields to avoid false sharing. If v1 stuff is - * not present, an explicit padding is needed. + * Keep lruvec on its own lines. Sharing them with lru_zone_size[] + * regressed, see commit f59adcf59332 ("mm: memcg: add cacheline + * padding after lruvec in mem_cgroup_per_node"). */ - - struct rb_node tree_node; /* RB tree node */ - unsigned long usage_in_excess;/* Set to the value by which */ - /* the soft limit is exceeded*/ - bool on_tree; -#else - CACHELINE_PADDING(_pad1_); -#endif - - /* Fields which get updated often at the end. */ + __cacheline_group_begin_aligned(memcg_pn_lruvec); struct lruvec lruvec; - CACHELINE_PADDING(_pad2_); - unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; + __cacheline_group_end_aligned(memcg_pn_lruvec); + + /* Written on every LRU update and on every reclaim iteration. */ + __cacheline_group_begin_aligned(memcg_pn_write_hot); + long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; struct mem_cgroup_reclaim_iter iter; +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + /* slab stats for nmi context */ + atomic_t slab_reclaimable; + atomic_t slab_unreclaimable; +#endif + __cacheline_group_end_aligned(memcg_pn_write_hot); + /* Touched only when the memcg is reparented or freed. */ + __cacheline_group_begin_aligned(memcg_pn_cold); /* - * objcg is wiped out as a part of the objcg repaprenting process. * orig_objcg preserves a pointer (and a reference) to the original - * objcg until the end of live of memcg. + * objcg until the end of life of memcg. */ - struct obj_cgroup __rcu *objcg; struct obj_cgroup *orig_objcg; /* list of inherited objcgs, protected by objcg_lock */ struct list_head objcg_list; - -#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC - /* slab stats for nmi context */ - atomic_t slab_reclaimable; - atomic_t slab_unreclaimable; -#endif + __cacheline_group_end_aligned(memcg_pn_cold); }; struct mem_cgroup_threshold { @@ -186,6 +180,7 @@ struct obj_cgroup { struct percpu_ref refcnt; struct mem_cgroup *memcg; atomic_t nr_charged_bytes; + refcount_t private_id_ref; union { struct list_head list; /* protected by objcg_lock */ struct rcu_head rcu; @@ -202,9 +197,6 @@ struct obj_cgroup { struct mem_cgroup { struct cgroup_subsys_state css; - /* Private memcg ID. Used to ID objects that outlive the cgroup */ - struct mem_cgroup_private_id id; - /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -213,70 +205,59 @@ struct mem_cgroup { struct page_counter memsw; /* v1 only */ }; - /* registered local peak watchers */ - struct list_head memory_peaks; - struct list_head swap_peaks; - spinlock_t peaks_lock; - - /* Range enforcement for interrupt charges */ - struct work_struct high_work; - -#ifdef CONFIG_ZSWAP - unsigned long zswap_max; - + /* Written on the charge, reclaim and socket paths. */ + __cacheline_group_begin_aligned(memcg_write_hot); /* - * Prevent pages from this memcg from being written back from zswap to - * swap, and from being swapped out on zswap store failures. + * Hint of reclaim pressure for socket memory management. Note + * that this indicator should NOT be used in legacy cgroup mode + * where socket memory is accounted/charged separately. */ - bool zswap_writeback; + u64 socket_pressure; +#if BITS_PER_LONG < 64 + seqlock_t socket_pressure_seqlock; #endif - - /* vmpressure notifications */ - struct vmpressure vmpressure; - /* - * Should the OOM killer kill all belonging tasks, had it kill one? + * memory.events is bumped for this memcg and all its ancestors, so a + * busy child dirties every ancestor. */ - bool oom_group; - - /* memory.events and memory.events.local */ - struct cgroup_file events_file; - struct cgroup_file events_local_file; - - /* handle for "memory.swap.events" */ - struct cgroup_file swap_events_file; - - /* memory.stat */ - struct memcg_vmstats *vmstats; - - /* memory.events */ atomic_long_t memory_events[MEMCG_NR_MEMORY_EVENTS]; atomic_long_t memory_events_local[MEMCG_NR_MEMORY_EVENTS]; + /* vmpressure notifications. Written on every reclaim iteration. */ + struct vmpressure vmpressure; + #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC /* MEMCG_KMEM for nmi context */ atomic_t kmem_stat; #endif + + /* Range enforcement for interrupt charges */ + struct irq_work high_irq_work; + struct work_struct high_work; + + __cacheline_group_end_aligned(memcg_write_hot); + /* - * Hint of reclaim pressure for socket memroy management. Note - * that this indicator should NOT be used in legacy cgroup mode - * where socket memory is accounted/charged separately. + * Off the charge and fault paths. Not write free: cgwb_domain is + * written on every writeout completion and mm_list on fork, exit and + * MGLRU aging. They are grouped here so those writes cannot land on + * a line that the fast paths read. */ - u64 socket_pressure; -#if BITS_PER_LONG < 64 - seqlock_t socket_pressure_seqlock; -#endif - int kmemcg_id; + __cacheline_group_begin_aligned(memcg_cold); + /* registered local peak watchers */ + struct list_head memory_peaks; + struct list_head swap_peaks; + spinlock_t peaks_lock; -#ifdef CONFIG_CGROUP_WRITEBACK - struct list_head cgwb_list; -#endif + /* memory.events and memory.events.local */ + struct cgroup_file events_file; + struct cgroup_file events_local_file; - /* Keep the hot per-CPU stats pointer away from memory event counters. */ - struct memcg_vmstats_percpu __percpu *vmstats_percpu - ____cacheline_aligned_in_smp; + /* handle for "memory.swap.events" */ + struct cgroup_file swap_events_file; #ifdef CONFIG_CGROUP_WRITEBACK + struct list_head cgwb_list; struct wb_domain cgwb_domain; struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT]; #endif @@ -285,16 +266,17 @@ struct mem_cgroup { /* per-memcg mm_struct list */ struct lru_gen_mm_list mm_list; #endif + __cacheline_group_end_aligned(memcg_cold); #ifdef CONFIG_MEMCG_V1 + /* v1 only. Not grouped: v1 is legacy, sorting it is not worth it. */ + /* Legacy consumer-oriented counters */ struct page_counter kmem; /* v1 only */ struct page_counter tcpmem; /* v1 only */ struct memcg1_events_percpu __percpu *events_percpu; - unsigned long soft_limit; - /* protected by memcg_oom_lock */ bool oom_lock; int under_oom; @@ -325,6 +307,44 @@ struct mem_cgroup { int swappiness; #endif /* CONFIG_MEMCG_V1 */ + /* + * Set when the memcg is created and cleared when it is offlined. + * Never written on a hot path. + */ + __cacheline_group_begin_aligned(memcg_read_mostly); + /* Read on every stat update */ + struct memcg_vmstats_percpu __percpu *vmstats_percpu; + + /* memory.stat */ + struct memcg_vmstats *vmstats; + +#ifdef CONFIG_ZSWAP + unsigned long zswap_max; +#endif + + /* The objcg holding private memcg ID. */ + struct obj_cgroup *private_id_objcg; + + /* Private memcg ID. Used to ID objects that outlive the cgroup */ + int private_id; + + int kmemcg_id; + + /* + * Should the OOM killer kill all belonging tasks, had it kill one? + */ + bool oom_group; + +#ifdef CONFIG_ZSWAP + /* + * Prevent pages from this memcg from being written back from zswap to + * swap, and from being swapped out on zswap store failures. + */ + bool zswap_writeback; +#endif + /* Not padded: nodeinfo[] is read-mostly too, let it share the line. */ + __cacheline_group_end(memcg_read_mostly); + struct mem_cgroup_per_node *nodeinfo[]; }; @@ -377,10 +397,10 @@ enum objext_flags { * * The caller must ensure that the returned memcg won't be released. */ -static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg) { lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex)); - return READ_ONCE(objcg->memcg); + return objcg ? READ_ONCE(objcg->memcg) : NULL; } /* @@ -391,7 +411,7 @@ static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) * or NULL. This function assumes that the folio is known to have a * proper object cgroup pointer. */ -static inline struct obj_cgroup *folio_objcg(struct folio *folio) +static inline struct obj_cgroup *folio_objcg(const struct folio *folio) { unsigned long memcg_data = folio->memcg_data; @@ -429,11 +449,11 @@ static inline struct obj_cgroup *folio_objcg(struct folio *folio) * Note: The caller should hold an rcu read lock or cgroup_mutex to protect * memcg associated with a folio from being released. */ -static inline struct mem_cgroup *folio_memcg(struct folio *folio) +static inline struct mem_cgroup *folio_memcg(const struct folio *folio) { struct obj_cgroup *objcg = folio_objcg(folio); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } /* @@ -442,7 +462,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio) * * Returns true if folio is charged to a memory cgroup, otherwise returns false. */ -static inline bool folio_memcg_charged(struct folio *folio) +static inline bool folio_memcg_charged(const struct folio *folio) { return folio->memcg_data != 0; } @@ -462,7 +482,7 @@ static inline bool folio_memcg_charged(struct folio *folio) * A caller should hold an rcu read lock to protect memcg associated with a * page from being released. */ -static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) +static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio) { /* * Because folio->memcg_data might be changed asynchronously @@ -476,17 +496,17 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } -static inline struct mem_cgroup *page_memcg_check(struct page *page) +static inline struct mem_cgroup *page_memcg_check(const struct page *page) { if (PageTail(page)) return NULL; - return folio_memcg_check((struct folio *)page); + return folio_memcg_check((const struct folio *)page); } -static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg) { struct mem_cgroup *memcg; @@ -508,19 +528,19 @@ retry: * that the folio has an associated memory cgroup. It's not safe to call * this function against some types of folios, e.g. slab folios. */ -static inline bool folio_memcg_kmem(struct folio *folio) +static inline bool folio_memcg_kmem(const struct folio *folio) { VM_BUG_ON_PGFLAGS(PageTail(&folio->page), &folio->page); VM_BUG_ON_FOLIO(folio->memcg_data & MEMCG_DATA_OBJEXTS, folio); return folio->memcg_data & MEMCG_DATA_KMEM; } -static inline bool PageMemcgKmem(struct page *page) +static inline bool PageMemcgKmem(const struct page *page) { return folio_memcg_kmem(page_folio(page)); } -static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) +static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg) { return (memcg == root_mem_cgroup); } @@ -536,7 +556,7 @@ static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) * and do not honour sc->memcg can use this to early-return 0 in per-memcg * contexts. */ -static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc) +static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc) { return !sc->memcg || mem_cgroup_is_root(sc->memcg); } @@ -551,8 +571,8 @@ static inline bool mem_cgroup_disabled(void) return !cgroup_subsys_enabled(memory_cgrp_subsys); } -static inline void mem_cgroup_protection(struct mem_cgroup *root, - struct mem_cgroup *memcg, +static inline void mem_cgroup_protection(const struct mem_cgroup *root, + const struct mem_cgroup *memcg, unsigned long *min, unsigned long *low, unsigned long *usage) @@ -606,8 +626,8 @@ static inline void mem_cgroup_protection(struct mem_cgroup *root, void mem_cgroup_calculate_protection(struct mem_cgroup *root, struct mem_cgroup *memcg); -static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { /* * The root memcg doesn't account charges, and doesn't support @@ -618,8 +638,8 @@ static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, memcg == target; } -static inline bool mem_cgroup_below_low(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_low(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { if (mem_cgroup_unprotected(target, memcg)) return false; @@ -628,8 +648,8 @@ static inline bool mem_cgroup_below_low(struct mem_cgroup *target, page_counter_read(&memcg->memory); } -static inline bool mem_cgroup_below_min(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_min(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { if (mem_cgroup_unprotected(target, memcg)) return false; @@ -662,7 +682,8 @@ static inline int mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, return __mem_cgroup_charge(folio, mm, gfp); } -int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp); +int mem_cgroup_charge_hugetlb(struct folio *folio, struct mm_struct *mm, + gfp_t gfp); int mem_cgroup_swapin_charge_folio(struct folio *folio, unsigned short id, struct mm_struct *mm, gfp_t gfp); @@ -702,7 +723,7 @@ void mem_cgroup_migrate(struct folio *old, struct folio *new); * @pgdat combination. This can be the node lruvec, if the memory * controller is disabled. */ -static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, +static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg, struct pglist_data *pgdat) { struct mem_cgroup_per_node *mz; @@ -743,7 +764,7 @@ out: * their binding is stable if the returned lruvec matches the one the caller has * locked. Useful for lock batching. */ -static inline struct lruvec *folio_lruvec(struct folio *folio) +static inline struct lruvec *folio_lruvec(const struct folio *folio) { struct mem_cgroup *memcg = folio_memcg(folio); @@ -757,11 +778,11 @@ struct mem_cgroup *get_mem_cgroup_from_mm(struct mm_struct *mm); struct mem_cgroup *get_mem_cgroup_from_current(void); -struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio); +struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio); -struct lruvec *folio_lruvec_lock(struct folio *folio); -struct lruvec *folio_lruvec_lock_irq(struct folio *folio); -struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +struct lruvec *folio_lruvec_lock(const struct folio *folio); +struct lruvec *folio_lruvec_lock_irq(const struct folio *folio); +struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flags); static inline @@ -820,16 +841,16 @@ void mem_cgroup_iter_break(struct mem_cgroup *, struct mem_cgroup *); void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, int (*)(struct task_struct *, void *), void *arg); -static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) +static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg) { if (mem_cgroup_disabled()) return 0; - return memcg->id.id; + return memcg->private_id; } struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id); -static inline u64 mem_cgroup_id(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg) { return memcg ? cgroup_id(memcg->css.cgroup) : 0; } @@ -841,14 +862,14 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m) return mem_cgroup_from_css(seq_css(m)); } -static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) +static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) { - struct mem_cgroup_per_node *mz; + const struct mem_cgroup_per_node *mz; if (mem_cgroup_disabled()) return NULL; - mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); return mz->memcg; } @@ -858,13 +879,13 @@ static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) * * Returns the parent memcg, or NULL if this is the root. */ -static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg) +static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg) { return mem_cgroup_from_css(memcg->css.parent); } -static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg, - struct mem_cgroup *root) +static inline bool mem_cgroup_is_descendant(const struct mem_cgroup *memcg, + const struct mem_cgroup *root) { if (root == memcg) return true; @@ -872,7 +893,7 @@ static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg, } static inline bool mm_match_cgroup(struct mm_struct *mm, - struct mem_cgroup *memcg) + const struct mem_cgroup *memcg) { struct mem_cgroup *task_memcg; bool match = false; @@ -885,8 +906,8 @@ static inline bool mm_match_cgroup(struct mm_struct *mm, return match; } -struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio); -ino_t page_cgroup_ino(struct page *page); +struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(const struct folio *folio); +ino_t page_cgroup_ino(const struct page *page); static inline bool mem_cgroup_online(struct mem_cgroup *memcg) { @@ -899,13 +920,18 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru, int zid, long nr_pages); static inline -unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, +unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, enum lru_list lru, int zone_idx) { - struct mem_cgroup_per_node *mz; + long val; + const struct mem_cgroup_per_node *mz; + + mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); + val = READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + if (WARN_ON_ONCE(val < 0)) + return 0; - mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - return READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + return val; } void __mem_cgroup_handle_over_high(gfp_t gfp_mask); @@ -916,9 +942,9 @@ static inline void mem_cgroup_handle_over_high(gfp_t gfp_mask) __mem_cgroup_handle_over_high(gfp_mask); } -unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg); +unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg); -void mem_cgroup_print_oom_context(struct mem_cgroup *memcg, +void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, struct task_struct *p); void mem_cgroup_print_oom_meminfo(struct mem_cgroup *memcg); @@ -931,7 +957,7 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg); void mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, int val); -static inline void mod_memcg_page_state(struct page *page, +static inline void mod_memcg_page_state(const struct page *page, enum memcg_stat_item idx, int val) { struct mem_cgroup *memcg; @@ -946,17 +972,21 @@ static inline void mod_memcg_page_state(struct page *page, rcu_read_unlock(); } -unsigned long memcg_events(struct mem_cgroup *memcg, int event); -unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx); -unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item); +unsigned long memcg_events(const struct mem_cgroup *memcg, int event); +unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx); +unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item); bool memcg_stat_item_valid(int idx); bool memcg_vm_event_item_valid(enum vm_event_item idx); -unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx); -unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +unsigned long lruvec_page_state(const struct lruvec *lruvec, + enum node_stat_item idx); +unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx); -unsigned long lruvec_page_state_local(struct lruvec *lruvec, +unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx); +void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val); + void mem_cgroup_flush_stats(struct mem_cgroup *memcg); void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg); @@ -965,7 +995,7 @@ void mod_lruvec_kmem_state(void *p, enum node_stat_item idx, int val); void count_memcg_events(struct mem_cgroup *memcg, enum vm_event_item idx, unsigned long count); -static inline void count_memcg_folio_events(struct folio *folio, +static inline void count_memcg_folio_events(const struct folio *folio, enum vm_event_item idx, unsigned long nr) { struct mem_cgroup *memcg; @@ -1050,51 +1080,56 @@ void mem_cgroup_flush_workqueue(void); extern int mem_cgroup_init(void); #else /* CONFIG_MEMCG */ +static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg) +{ + return NULL; +} + #define MEM_CGROUP_ID_SHIFT 0 #define root_mem_cgroup (NULL) -static inline struct mem_cgroup *folio_memcg(struct folio *folio) +static inline struct mem_cgroup *folio_memcg(const struct folio *folio) { return NULL; } -static inline bool folio_memcg_charged(struct folio *folio) +static inline bool folio_memcg_charged(const struct folio *folio) { return false; } -static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) +static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio) { return NULL; } -static inline struct mem_cgroup *page_memcg_check(struct page *page) +static inline struct mem_cgroup *page_memcg_check(const struct page *page) { return NULL; } -static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg) { return NULL; } -static inline bool folio_memcg_kmem(struct folio *folio) +static inline bool folio_memcg_kmem(const struct folio *folio) { return false; } -static inline bool PageMemcgKmem(struct page *page) +static inline bool PageMemcgKmem(const struct page *page) { return false; } -static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) +static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg) { return true; } -static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc) +static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc) { return true; } @@ -1119,8 +1154,8 @@ static inline void memcg_memory_event_mm(struct mm_struct *mm, { } -static inline void mem_cgroup_protection(struct mem_cgroup *root, - struct mem_cgroup *memcg, +static inline void mem_cgroup_protection(const struct mem_cgroup *root, + const struct mem_cgroup *memcg, unsigned long *min, unsigned long *low, unsigned long *usage) @@ -1133,19 +1168,19 @@ static inline void mem_cgroup_calculate_protection(struct mem_cgroup *root, { } -static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return true; } -static inline bool mem_cgroup_below_low(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_low(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return false; } -static inline bool mem_cgroup_below_min(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_min(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return false; } @@ -1156,9 +1191,10 @@ static inline int mem_cgroup_charge(struct folio *folio, return 0; } -static inline int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp) +static inline int mem_cgroup_charge_hugetlb(struct folio *folio, + struct mm_struct *mm, gfp_t gfp) { - return 0; + return 0; } static inline int mem_cgroup_swapin_charge_folio(struct folio *folio, @@ -1184,25 +1220,25 @@ static inline void mem_cgroup_migrate(struct folio *old, struct folio *new) { } -static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, +static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg, struct pglist_data *pgdat) { return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec(struct folio *folio) +static inline struct lruvec *folio_lruvec(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); return &pgdat->__lruvec; } -static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg) +static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg) { return NULL; } static inline bool mm_match_cgroup(struct mm_struct *mm, - struct mem_cgroup *memcg) + const struct mem_cgroup *memcg) { return true; } @@ -1217,7 +1253,7 @@ static inline struct mem_cgroup *get_mem_cgroup_from_current(void) return NULL; } -static inline struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio) +static inline struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio) { return NULL; } @@ -1250,7 +1286,7 @@ static inline void mem_cgroup_put(struct mem_cgroup *memcg) { } -static inline struct lruvec *folio_lruvec_lock(struct folio *folio) +static inline struct lruvec *folio_lruvec_lock(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1259,7 +1295,7 @@ static inline struct lruvec *folio_lruvec_lock(struct folio *folio) return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio) +static inline struct lruvec *folio_lruvec_lock_irq(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1268,7 +1304,7 @@ static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio) return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +static inline struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flagsp) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1296,7 +1332,7 @@ static inline void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, { } -static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) +static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg) { return 0; } @@ -1308,7 +1344,7 @@ static inline struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id) return NULL; } -static inline u64 mem_cgroup_id(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg) { return 0; } @@ -1323,7 +1359,7 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m) return NULL; } -static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) +static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) { return NULL; } @@ -1334,19 +1370,20 @@ static inline bool mem_cgroup_online(struct mem_cgroup *memcg) } static inline -unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, +unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, enum lru_list lru, int zone_idx) { return 0; } -static inline unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) +static inline unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg) { return 0; } static inline void -mem_cgroup_print_oom_context(struct mem_cgroup *memcg, struct task_struct *p) +mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, + struct task_struct *p) { } @@ -1375,17 +1412,17 @@ static inline void mod_memcg_state(struct mem_cgroup *memcg, { } -static inline void mod_memcg_page_state(struct page *page, +static inline void mod_memcg_page_state(const struct page *page, enum memcg_stat_item idx, int val) { } -static inline unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx) +static inline unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx) { return 0; } -static inline unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item) +static inline unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item) { return 0; } @@ -1400,24 +1437,29 @@ static inline bool memcg_vm_event_item_valid(enum vm_event_item idx) return false; } -static inline unsigned long lruvec_page_state(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state(lruvec_pgdat(lruvec), idx); } -static inline unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state_monotonic(lruvec_pgdat(lruvec), idx); } -static inline unsigned long lruvec_page_state_local(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state(lruvec_pgdat(lruvec), idx); } +static inline void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val) +{ +} + static inline void mem_cgroup_flush_stats(struct mem_cgroup *memcg) { } @@ -1440,7 +1482,7 @@ static inline void count_memcg_events(struct mem_cgroup *memcg, { } -static inline void count_memcg_folio_events(struct folio *folio, +static inline void count_memcg_folio_events(const struct folio *folio, enum vm_event_item idx, unsigned long nr) { } @@ -1474,7 +1516,7 @@ static inline void mem_cgroup_flush_workqueue(void) { } static inline int mem_cgroup_init(void) { return 0; } #endif /* CONFIG_MEMCG */ -static inline struct lruvec *parent_lruvec(struct lruvec *lruvec) +static inline struct lruvec *parent_lruvec(const struct lruvec *lruvec) { struct mem_cgroup *memcg; @@ -1537,15 +1579,15 @@ static inline void lruvec_unlock_irqrestore(struct lruvec *lruvec, unsigned long } /* Test requires a stable folio->memcg binding, see folio_memcg() */ -static inline bool folio_matches_lruvec(struct folio *folio, - struct lruvec *lruvec) +static inline bool folio_matches_lruvec(const struct folio *folio, + const struct lruvec *lruvec) { return lruvec_pgdat(lruvec) == folio_pgdat(folio) && lruvec_memcg(lruvec) == folio_memcg(folio); } /* Don't lock again iff page's lruvec locked */ -static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio, +static inline struct lruvec *folio_lruvec_relock_irq(const struct folio *folio, struct lruvec *locked_lruvec) { if (locked_lruvec) { @@ -1559,7 +1601,7 @@ static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio, } /* Don't lock again iff folio's lruvec locked */ -static inline void folio_lruvec_relock_irqsave(struct folio *folio, +static inline void folio_lruvec_relock_irqsave(const struct folio *folio, struct lruvec **lruvecp, unsigned long *flags) { if (*lruvecp) { @@ -1651,7 +1693,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg) write_sequnlock_irqrestore(&memcg->socket_pressure_seqlock, flags); } -static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg) { unsigned int seq; u64 val; @@ -1669,7 +1711,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg) WRITE_ONCE(memcg->socket_pressure, jiffies + HZ); } -static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg) { return READ_ONCE(memcg->socket_pressure); } @@ -1736,7 +1778,7 @@ void __memcg_kmem_uncharge_page(struct page *page, int order); * needs to be used outside of the local scope. */ struct obj_cgroup *current_obj_cgroup(void); -struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio); +struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio); static inline struct obj_cgroup *get_obj_cgroup_from_current(void) { @@ -1782,7 +1824,7 @@ static inline void memcg_kmem_uncharge_page(struct page *page, int order) * A helper for accessing memcg's kmem_id, used for getting * corresponding LRU lists. */ -static inline int memcg_kmem_id(struct mem_cgroup *memcg) +static inline int memcg_kmem_id(const struct mem_cgroup *memcg) { return memcg ? memcg->kmemcg_id : -1; } @@ -1839,7 +1881,7 @@ static inline void __memcg_kmem_uncharge_page(struct page *page, int order) { } -static inline struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio) +static inline struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio) { return NULL; } @@ -1854,7 +1896,7 @@ static inline bool memcg_kmem_online(void) return false; } -static inline int memcg_kmem_id(struct mem_cgroup *memcg) +static inline int memcg_kmem_id(const struct mem_cgroup *memcg) { return -1; } @@ -1870,7 +1912,7 @@ static inline void count_objcg_events(struct obj_cgroup *objcg, { } -static inline ino_t page_cgroup_ino(struct page *page) +static inline ino_t page_cgroup_ino(const struct page *page) { return 0; } @@ -1890,11 +1932,21 @@ static inline bool memcg_is_dying(struct mem_cgroup *memcg) } #endif /* CONFIG_MEMCG */ +#if defined(CONFIG_MEMCG) && defined(CONFIG_LRU_GEN) +void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg); +#else +static inline void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg) +{ +} +#endif + #if defined(CONFIG_MEMCG) && defined(CONFIG_ZSWAP) bool obj_cgroup_may_zswap(struct obj_cgroup *objcg); void obj_cgroup_charge_zswap(struct obj_cgroup *objcg, size_t size); void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size); -bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg); +bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg); #else static inline bool obj_cgroup_may_zswap(struct obj_cgroup *objcg) { @@ -1908,7 +1960,7 @@ static inline void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size) { } -static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) +static inline bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg) { /* if zswap is disabled, do not block pages going to the swapping device */ return true; @@ -1919,13 +1971,9 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) /* Cgroup v1-related declarations */ #ifdef CONFIG_MEMCG_V1 -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned); - bool mem_cgroup_oom_synchronize(bool wait); -static inline bool task_in_memcg_oom(struct task_struct *p) +static inline bool task_in_memcg_oom(const struct task_struct *p) { return p->memcg_in_oom; } @@ -1943,15 +1991,7 @@ static inline void mem_cgroup_exit_user_fault(void) } #else /* CONFIG_MEMCG_V1 */ -static inline -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - return 0; -} - -static inline bool task_in_memcg_oom(struct task_struct *p) +static inline bool task_in_memcg_oom(const struct task_struct *p) { return false; } diff --git a/include/linux/mempool.h b/include/linux/mempool.h index a0fa6d43e0dc..6da502aef2f7 100644 --- a/include/linux/mempool.h +++ b/include/linux/mempool.h @@ -70,6 +70,13 @@ int mempool_alloc_bulk_noprof(struct mempool *pool, void **elem, #define mempool_alloc_bulk(...) \ alloc_hooks(mempool_alloc_bulk_noprof(__VA_ARGS__)) +/* + * Allocate a new element without dipping into the pool's reserves or + * waiting. Returns NULL on failure. + */ +#define mempool_alloc_noreserve(_pool, _gfp) \ + alloc_hooks((_pool)->alloc(_gfp, (_pool)->pool_data)) + void *mempool_alloc_preallocated(struct mempool *pool) __malloc; void mempool_free(void *element, struct mempool *pool); unsigned int mempool_free_bulk(struct mempool *pool, void **elem, diff --git a/include/linux/mfd/88pm886.h b/include/linux/mfd/88pm886.h index 2c24dd3032ab..9e96d2cb92f5 100644 --- a/include/linux/mfd/88pm886.h +++ b/include/linux/mfd/88pm886.h @@ -2,6 +2,7 @@ #ifndef __MFD_88PM886_H #define __MFD_88PM886_H +#include <linux/bits.h> #include <linux/i2c.h> #include <linux/regmap.h> @@ -130,6 +131,12 @@ #define PM886_GPADC_INDEX_TO_BIAS_uA(i) (1 + (i) * 5) /* Battery block register definitions */ +#define PM886_REG_BATTERY_CONFIG1 0x28 +#define PM886_REG_VBUS_EN BIT(7) + +#define PM886_REG_BOOST_CONFIG1 0x6b +#define PM886_REG_BOOST_MASK GENMASK(2, 0) + #define PM886_REG_CLS_CONFIG1 0x71 struct pm886_chip { diff --git a/include/linux/mfd/cs42l43-regs.h b/include/linux/mfd/cs42l43-regs.h index 68831f113589..4c00ceae8b46 100644 --- a/include/linux/mfd/cs42l43-regs.h +++ b/include/linux/mfd/cs42l43-regs.h @@ -1183,6 +1183,7 @@ /* CS42L43B VARIANT REGISTERS */ #define CS42L43B_DEVID_VAL 0x0042A43B +#define CS42L44_DEVID_VAL 0x00042A44 #define CS42L43B_DECIM_VOL_CTRL_CH1_CH2 0x00008280 #define CS42L43B_DECIM_VOL_CTRL_CH3_CH4 0x00008284 diff --git a/include/linux/mfd/da9150/core.h b/include/linux/mfd/da9150/core.h index d116d5f3ef56..369698036d1a 100644 --- a/include/linux/mfd/da9150/core.h +++ b/include/linux/mfd/da9150/core.h @@ -65,6 +65,7 @@ struct da9150 { struct regmap_irq_chip_data *regmap_irq_data; int irq; int irq_base; + bool irq_wake_enabled; }; /* Device I/O - Query Interface for FG and standard register access */ diff --git a/include/linux/mfd/khadas-mcu.h b/include/linux/mfd/khadas-mcu.h index a99ba2ed0e4e..acd3291061b4 100644 --- a/include/linux/mfd/khadas-mcu.h +++ b/include/linux/mfd/khadas-mcu.h @@ -70,6 +70,13 @@ #define KHADAS_MCU_WOL_INIT_START_REG 0x87 /* WO */ #define KHADAS_MCU_CMD_FAN_STATUS_CTRL_REG 0x88 /* WO */ +/* VIM4 specific registers */ +#define KHADAS_MCU_VIM4_REST_CONF_REG 0x2c /* WO - reset EEPROM */ +#define KHADAS_MCU_VIM4_LED_ON_RAM_REG 0x89 /* WO - LED volatile */ +#define KHADAS_MCU_VIM4_FAN_CTRL_REG 0x8a /* WO */ +#define KHADAS_MCU_VIM4_WDT_EN_REG 0x8b /* WO */ +#define KHADAS_MCU_VIM4_SYS_RST_REG 0x91 /* WO */ + enum { KHADAS_BOARD_VIM1 = 0x1, KHADAS_BOARD_VIM2, @@ -80,7 +87,7 @@ enum { /** * struct khadas_mcu - Khadas MCU structure - * @device: device reference used for logs + * @dev: device reference used for logs * @regmap: register map */ struct khadas_mcu { @@ -88,4 +95,14 @@ struct khadas_mcu { struct regmap *regmap; }; +/** + * enum khadas_mcu_type - Khadas MCU hardware variant + * @KHADAS_MCU_GENERIC: VIM1, VIM2, VIM3, Edge, Edge-V (shared register map) + * @KHADAS_MCU_VIM4: VIM4 (extended register map, distinct fan/LED/WDT regs) + */ +enum khadas_mcu_type { + KHADAS_MCU_GENERIC = 1, + KHADAS_MCU_VIM4, +}; + #endif /* MFD_KHADAS_MCU_H */ diff --git a/include/linux/mfd/lm3533.h b/include/linux/mfd/lm3533.h index 69059a7a2ce5..8f72dd41e8f0 100644 --- a/include/linux/mfd/lm3533.h +++ b/include/linux/mfd/lm3533.h @@ -15,9 +15,14 @@ #define LM3533_ATTR_RW(_name) \ DEVICE_ATTR(_name, S_IRUGO | S_IWUSR , show_##_name, store_##_name) +#define LM3533_MAX_CURRENT_MIN 5000 +#define LM3533_MAX_CURRENT_MAX 29800 +#define LM3533_MAX_CURRENT_STEP 800 + struct device; struct gpio_desc; struct regmap; +struct regulator; struct lm3533 { struct device *dev; @@ -25,7 +30,10 @@ struct lm3533 { struct regmap *regmap; struct gpio_desc *hwen; - int irq; + struct regulator *vin_supply; + + u32 boost_ovp; + u32 boost_freq; unsigned have_als:1; unsigned have_backlights:1; @@ -33,67 +41,18 @@ struct lm3533 { }; struct lm3533_ctrlbank { - struct lm3533 *lm3533; + struct regmap *regmap; struct device *dev; int id; }; -struct lm3533_als_platform_data { - unsigned pwm_mode:1; /* PWM input mode (default analog) */ - u8 r_select; /* 1 - 127 (ignored in PWM-mode) */ -}; - -struct lm3533_bl_platform_data { - char *name; - u16 max_current; /* 5000 - 29800 uA (800 uA step) */ - u8 default_brightness; /* 0 - 255 */ - u8 pwm; /* 0 - 0x3f */ -}; - -struct lm3533_led_platform_data { - char *name; - const char *default_trigger; - u16 max_current; /* 5000 - 29800 uA (800 uA step) */ - u8 pwm; /* 0 - 0x3f */ -}; - -enum lm3533_boost_freq { - LM3533_BOOST_FREQ_500KHZ, - LM3533_BOOST_FREQ_1000KHZ, -}; - -enum lm3533_boost_ovp { - LM3533_BOOST_OVP_16V, - LM3533_BOOST_OVP_24V, - LM3533_BOOST_OVP_32V, - LM3533_BOOST_OVP_40V, -}; - -struct lm3533_platform_data { - enum lm3533_boost_ovp boost_ovp; - enum lm3533_boost_freq boost_freq; - - struct lm3533_als_platform_data *als; - - struct lm3533_bl_platform_data *backlights; - int num_backlights; - - struct lm3533_led_platform_data *leds; - int num_leds; -}; - -extern int lm3533_ctrlbank_enable(struct lm3533_ctrlbank *cb); -extern int lm3533_ctrlbank_disable(struct lm3533_ctrlbank *cb); - -extern int lm3533_ctrlbank_set_brightness(struct lm3533_ctrlbank *cb, u8 val); -extern int lm3533_ctrlbank_get_brightness(struct lm3533_ctrlbank *cb, u8 *val); -extern int lm3533_ctrlbank_set_max_current(struct lm3533_ctrlbank *cb, - u16 imax); -extern int lm3533_ctrlbank_set_pwm(struct lm3533_ctrlbank *cb, u8 val); -extern int lm3533_ctrlbank_get_pwm(struct lm3533_ctrlbank *cb, u8 *val); +int lm3533_ctrlbank_enable(struct lm3533_ctrlbank *cb); +int lm3533_ctrlbank_disable(struct lm3533_ctrlbank *cb); -extern int lm3533_read(struct lm3533 *lm3533, u8 reg, u8 *val); -extern int lm3533_write(struct lm3533 *lm3533, u8 reg, u8 val); -extern int lm3533_update(struct lm3533 *lm3533, u8 reg, u8 val, u8 mask); +int lm3533_ctrlbank_set_brightness(struct lm3533_ctrlbank *cb, u32 val); +int lm3533_ctrlbank_get_brightness(struct lm3533_ctrlbank *cb, u32 *val); +int lm3533_ctrlbank_set_max_current(struct lm3533_ctrlbank *cb, u32 imax); +int lm3533_ctrlbank_set_pwm(struct lm3533_ctrlbank *cb, u32 val); +int lm3533_ctrlbank_get_pwm(struct lm3533_ctrlbank *cb, u32 *val); #endif /* __LINUX_MFD_LM3533_H */ diff --git a/include/linux/mfd/motorola-cpcap.h b/include/linux/mfd/motorola-cpcap.h index 981e5777deb7..bb23363eeccd 100644 --- a/include/linux/mfd/motorola-cpcap.h +++ b/include/linux/mfd/motorola-cpcap.h @@ -25,6 +25,13 @@ #define CPCAP_REVISION_2_0 0x10 #define CPCAP_REVISION_2_1 0x11 +enum cpcap_variant { + CPCAP_DEFAULT = 1, + CPCAP_MAPPHONE, + CPCAP_MOT, + CPCAP_MAX +}; + /* CPCAP registers */ #define CPCAP_REG_INT1 0x0000 /* Interrupt 1 */ #define CPCAP_REG_INT2 0x0004 /* Interrupt 2 */ diff --git a/include/linux/mfd/rohm-bd73800.h b/include/linux/mfd/rohm-bd73800.h new file mode 100644 index 000000000000..c6c0d453a40d --- /dev/null +++ b/include/linux/mfd/rohm-bd73800.h @@ -0,0 +1,306 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * Copyright 2026 ROHM Semiconductors. + * + * Author: Matti Vaittinen <matti.vaittinen@fi.rohmeurope.com> + */ + +#ifndef _MFD_BD73800_H +#define _MFD_BD73800_H + +#include <linux/regmap.h> + +enum { + BD73800_BUCK1 = 0, + BD73800_BUCK2, + BD73800_BUCK3, + BD73800_BUCK4, + BD73800_BUCK5, + BD73800_BUCK6, + BD73800_BUCK7, + BD73800_BUCK8, + BD73800_LDO1, + BD73800_LDO2, + BD73800_LDO3, + BD73800_LDO4, +}; + +/* + * All regulators except BUCK 5 have full 8-bit register of valid voltage + * values, including 0. + */ +#define BD73800_NUM_VOLTS (0xff + 1) +/* + * BUCK 5 has two sets of voltage ranges, both having valid voltage selectors + * from 0x0 to 0x7f + */ +#define BD73800_BUCK5_VOLTS (0x80 + 0x80) + +/* BD73800 interrupts */ +enum { + /* INT_STAT_1 register IRQs, ADC and RTC */ + BD73800_INT_ADC_ACCUM_DONE, + BD73800_INT_ADC_ACCUM_OVF, + BD73800_INT_ADC_ACCUM_VAL, + BD73800_INT_ADC_ACCUM_TW, + BD73800_INT_ADC_POW_VAL, + BD73800_INT_RTC0, + BD73800_INT_RTC1, + BD73800_INT_RTC2, + + /* BUCK reg interrupts */ + /* INT_STAT_2 IRQs */ + BD73800_INT_BUCK1_DVS_DONE, + BD73800_INT_BUCK2_DVS_DONE, + BD73800_INT_BUCK3_DVS_DONE, + BD73800_INT_BUCK4_DVS_DONE, + BD73800_INT_BUCK5_DVS_DONE, + BD73800_INT_BUCK6_DVS_DONE, + BD73800_INT_BUCK7_DVS_DONE, + BD73800_INT_BUCK8_DVS_DONE, + /* INT_STAT_3 IRQs */ + BD73800_INT_BUCK1_OCP, + BD73800_INT_BUCK2_OCP, + BD73800_INT_BUCK3_OCP, + BD73800_INT_BUCK4_OCP, + BD73800_INT_BUCK5_OCP, + BD73800_INT_BUCK6_OCP, + BD73800_INT_BUCK7_OCP, + BD73800_INT_BUCK8_OCP, + + /* INT_STAT_4 IRQs, power-button, WDG and reset */ + BD73800_INT_PBTN_LONG_PRESS, + BD73800_INT_PBTN_MID_PRESS, + /* + * The SHORT_PUSH is generated when button is first pressed (longer + * than configured time limit), and then released before the MID_PRESS + * time limit. The SHORT_PRESS is generated immediately when button is + * pressed for longer than configured limit, whether it is released or + * not. + */ + BD73800_INT_PBTN_SHORT_PUSH, + BD73800_INT_PBTN_SHORT_PRESS, + BD73800_INT_WDG, + BD73800_INT_SWRESET, + BD73800_INT_SEQ_DONE, + + /* INT_STAT_5 IRQs, GPIO */ + BD73800_INT_GPIO1, + BD73800_INT_GPIO2, + BD73800_INT_GPIO3, + BD73800_INT_GPIO4, +}; + +#define BD73800_MASK_RUN_EN BIT(2) +#define BD73800_MASK_SUSP_EN BIT(1) +#define BD73800_MASK_IDLE_EN BIT(0) +#define BD73800_MASK_VOLT GENMASK(7, 0) +#define BD73800_MASK_BUCK5_VOLT GENMASK(6, 0) +#define BD73800_MASK_RAMP_DELAY GENMASK(2, 1) +#define BD73800_BUCK5_RANGE_MASK BIT(7) + +/* BD73800 registers */ +enum { + BD73800_REG_PRODUCT_ID = 0x0, + BD73800_REG_MANUFACTURER_ID, + BD73800_REG_REVISION, + BD73800_REG_NVMVERSION, + BD73800_REG_POR_REASON, + BD73800_REG_RESET_REASON1, + BD73800_REG_RESET_REASON2, + BD73800_REG_RESET_REASON3, + BD73800_REG_POW_STATE, + BD73800_REG_WRST_SEL, + BD73800_REG_PS_CTRL_1, + BD73800_REG_PS_CTRL_2, + BD73800_REG_RCVCFG, + BD73800_REG_RCVNUM, + BD73800_REG_CRDCFG, /* 0x0f, followed by undocumented reg */ + + BD73800_REG_BUCK1_ON = 0x11, + BD73800_REG_BUCK1_MODE, + BD73800_REG_BUCK1_VOLT_RUN, + BD73800_REG_BUCK1_VOLT_IDLE, + BD73800_REG_BUCK1_VOLT_SUSP, /* 0x15, followed by undocumented reg */ + + BD73800_REG_BUCK2_ON = 0x17, + BD73800_REG_BUCK2_MODE, + BD73800_REG_BUCK2_VOLT_RUN, + BD73800_REG_BUCK2_VOLT_IDLE, + BD73800_REG_BUCK2_VOLT_SUSP, /* 0x1b */ + + BD73800_REG_BUCK3_ON = 0x1d, + BD73800_REG_BUCK3_MODE, + BD73800_REG_BUCK3_VOLT_RUN, + BD73800_REG_BUCK3_VOLT_IDLE, + BD73800_REG_BUCK3_VOLT_SUSP, /* 0x21 */ + + BD73800_REG_BUCK4_ON = 0x23, + BD73800_REG_BUCK4_MODE, + BD73800_REG_BUCK4_VOLT_RUN, + BD73800_REG_BUCK4_VOLT_IDLE, + BD73800_REG_BUCK4_VOLT_SUSP, /* 0x27 */ + + BD73800_REG_BUCK5_ON = 0x29, + BD73800_REG_BUCK5_MODE, + BD73800_REG_BUCK5_VOLT_RUN, + BD73800_REG_BUCK5_VOLT_IDLE, + BD73800_REG_BUCK5_VOLT_SUSP, /* 0x2d */ + + BD73800_REG_BUCK6_ON = 0x2f, + BD73800_REG_BUCK6_MODE, + BD73800_REG_BUCK6_VOLT_RUN, + BD73800_REG_BUCK6_VOLT_IDLE, + BD73800_REG_BUCK6_VOLT_SUSP, /* 0x33 */ + + BD73800_REG_BUCK7_ON = 0x35, + BD73800_REG_BUCK7_MODE, + BD73800_REG_BUCK7_VOLT_RUN, + BD73800_REG_BUCK7_VOLT_IDLE, + BD73800_REG_BUCK7_VOLT_SUSP, /* 0x39 */ + + BD73800_REG_BUCK8_ON = 0x3b, + BD73800_REG_BUCK8_MODE, + BD73800_REG_BUCK8_VOLT_RUN, + BD73800_REG_BUCK8_VOLT_IDLE, + BD73800_REG_BUCK8_VOLT_SUSP, /* 0x3f */ + + BD73800_REG_LDO1_ON = 0x41, + BD73800_REG_LDO1_VOLT, + BD73800_REG_LDO1_MODE, + BD73800_REG_LDO2_ON, + BD73800_REG_LDO2_VOLT, + BD73800_REG_LDO2_MODE, + BD73800_REG_LDO3_ON, + BD73800_REG_LDO3_VOLT, + BD73800_REG_LDO3_MODE, + BD73800_REG_LDO4_ON, + BD73800_REG_LDO4_VOLT, + BD73800_REG_LDO4_MODE, /* 0x4c */ + + BD73800_REG_GPO_OUT = 0x4e, + BD73800_REG_OUT32K = 0x50, + BD73800_REG_RTC_SEC, + BD73800_REG_RTC_MIN, + BD73800_REG_RTC_HOUR, + BD73800_REG_RTC_WEEK, + BD73800_REG_RTC_DAY, + BD73800_REG_RTC_MONTH, + BD73800_REG_RTC_YEAR, + BD73800_REG_RTC_ALM0_SEC, + BD73800_REG_RTC_ALM0_MIN, + BD73800_REG_RTC_ALM0_HOUR, + BD73800_REG_RTC_ALM0_WEEK, + BD73800_REG_RTC_ALM0_DAY, + BD73800_REG_RTC_ALM0_MONTH, + BD73800_REG_RTC_ALM0_YEAR, + BD73800_REG_RTC_ALM1_SEC, + BD73800_REG_RTC_ALM1_MIN, + BD73800_REG_RTC_ALM1_HOUR, + BD73800_REG_RTC_ALM1_WEEK, + BD73800_REG_RTC_ALM1_DAY, + BD73800_REG_RTC_ALM1_MONTH, + BD73800_REG_RTC_ALM1_YEAR, + BD73800_REG_RTC_ALM2, + BD73800_REG_RTC_CONF, /* 0x69 */ + + BD73800_REG_ADC_CTRL_1 = 0x6b, + BD73800_REG_ADC_CTRL_2, + BD73800_REG_ADC_ACCUM_NUM2, + BD73800_REG_ADC_ACCUM_NUM1, + BD73800_REG_ADC_ACCUM_NUM0, + BD73800_REG_ADC_ACCUM_KICK, + BD73800_REG_ADC_ACCUM_CNT2, + BD73800_REG_ADC_ACCUM_CNT1, + BD73800_REG_ADC_ACCUM_CNT0, + BD73800_REG_ADC_ACCUM_VAL2, + BD73800_REG_ADC_ACCUM_VAL1, + BD73800_REG_ADC_ACCUM_VAL0, + BD73800_REG_ADC_VOL_VAL1, + BD73800_REG_ADC_VOL_VAL0, + BD73800_REG_ADC_CUR_VAL1, + BD73800_REG_ADC_CUR_VAL0, + BD73800_REG_ADC_POW_VAL1, + BD73800_REG_ADC_POW_VAL0, + BD73800_REG_ADC_TEMP_VAL1, + BD73800_REG_ADC_TEMP_VAL0, + BD73800_REG_ADC_ACCUM_VAL_INT_TH4, + BD73800_REG_ADC_ACCUM_VAL_INT_TH3, + BD73800_REG_ADC_ACCUM_VAL_INT_TH2, + BD73800_REG_ADC_ACCUM_VAL_INT_TH1, + BD73800_REG_ADC_ACCUM_VAL_INT_TH0, + BD73800_REG_ADC_WARN_TEMP_INT_TH1, + BD73800_REG_ADC_WARN_TEMP_INT_TH0, + BD73800_REG_ADC_POW_VAL_INT_TH1, + BD73800_REG_ADC_POW_VAL_INT_TH0, /* 0x89 */ + + BD73800_REG_PBTN_CONF = 0x8b, + + BD73800_REG_INT_MAIN_EN = 0x8f, + BD73800_REG_INT_1_EN, + BD73800_REG_INT_2_EN, + BD73800_REG_INT_3_EN, + BD73800_REG_INT_4_EN, + BD73800_REG_INT_5_EN, /* 0x94 */ + + BD73800_REG_INT_MAIN_STAT = 0x96, + BD73800_REG_INT_1_STAT, + BD73800_REG_INT_2_STAT, + BD73800_REG_INT_3_STAT, + BD73800_REG_INT_4_STAT, + BD73800_REG_INT_5_STAT, /* 0x9b */ + + BD73800_REG_INT_MAIN_SRC = 0x9d, + BD73800_REG_INT_1_SRC, + BD73800_REG_INT_2_SRC, + BD73800_REG_INT_3_SRC, + BD73800_REG_INT_4_SRC, + BD73800_REG_INT_5_SRC, /* 0xa2 */ + + BD73800_REG_RST_MASK = 0xaf, + BD73800_MAX_REGISTER, +}; + +#define BD73800_REG_RTC_START BD73800_REG_RTC_SEC +#define BD73800_REG_RTC_ALM_START BD73800_REG_RTC_ALM0_SEC + +/* BD73800 IRQ register masks */ + +#define BD73800_INT_MAIN_EN_ALL GENMASK(4, 0) +#define BD73800_INT_ADC_ACCUM_DONE_MASK BIT(0) +#define BD73800_INT_ADC_ACCUM_OVF_MASK BIT(1) +#define BD73800_INT_ADC_ACCUM_VAL_MASK BIT(2) +#define BD73800_INT_ADC_ACCUM_TW_MASK BIT(3) +#define BD73800_INT_ADC_POW_VAL_MASK BIT(4) +#define BD73800_INT_RTC0_MASK BIT(5) +#define BD73800_INT_RTC1_MASK BIT(6) +#define BD73800_INT_RTC2_MASK BIT(7) +#define BD73800_INT_BUCK1_DVS_DONE_MASK BIT(0) +#define BD73800_INT_BUCK2_DVS_DONE_MASK BIT(1) +#define BD73800_INT_BUCK3_DVS_DONE_MASK BIT(2) +#define BD73800_INT_BUCK4_DVS_DONE_MASK BIT(3) +#define BD73800_INT_BUCK5_DVS_DONE_MASK BIT(4) +#define BD73800_INT_BUCK6_DVS_DONE_MASK BIT(5) +#define BD73800_INT_BUCK7_DVS_DONE_MASK BIT(6) +#define BD73800_INT_BUCK8_DVS_DONE_MASK BIT(7) +#define BD73800_INT_BUCK1_OCP_MASK BIT(0) +#define BD73800_INT_BUCK2_OCP_MASK BIT(1) +#define BD73800_INT_BUCK3_OCP_MASK BIT(2) +#define BD73800_INT_BUCK4_OCP_MASK BIT(3) +#define BD73800_INT_BUCK5_OCP_MASK BIT(4) +#define BD73800_INT_BUCK6_OCP_MASK BIT(5) +#define BD73800_INT_BUCK7_OCP_MASK BIT(6) +#define BD73800_INT_BUCK8_OCP_MASK BIT(7) +#define BD73800_INT_PBTN_LONG_PRESS_MASK BIT(0) +#define BD73800_INT_PBTN_MID_PRESS_MASK BIT(1) +#define BD73800_INT_PBTN_SHORT_PUSH_MASK BIT(2) +#define BD73800_INT_PBTN_SHORT_PRESS_MASK BIT(3) +#define BD73800_INT_WDG_MASK BIT(4) +#define BD73800_INT_SWRESET_MASK BIT(5) +#define BD73800_INT_SEQ_DONE_MASK BIT(6) +#define BD73800_INT_GPIO1_MASK BIT(0) +#define BD73800_INT_GPIO2_MASK BIT(1) +#define BD73800_INT_GPIO3_MASK BIT(2) +#define BD73800_INT_GPIO4_MASK BIT(3) + +#endif /* _MFD_BD73800_H */ diff --git a/include/linux/mfd/rohm-generic.h b/include/linux/mfd/rohm-generic.h index 0a284919a6c3..3ec87428ee97 100644 --- a/include/linux/mfd/rohm-generic.h +++ b/include/linux/mfd/rohm-generic.h @@ -17,6 +17,7 @@ enum rohm_chip_type { ROHM_CHIP_TYPE_BD71837, ROHM_CHIP_TYPE_BD71847, ROHM_CHIP_TYPE_BD72720, + ROHM_CHIP_TYPE_BD73800, ROHM_CHIP_TYPE_BD96801, ROHM_CHIP_TYPE_BD96802, ROHM_CHIP_TYPE_BD96805, diff --git a/include/linux/micrel_phy.h b/include/linux/micrel_phy.h index 9c6f9817383f..bb71b2510c2c 100644 --- a/include/linux/micrel_phy.h +++ b/include/linux/micrel_phy.h @@ -47,11 +47,6 @@ #define MICREL_PHY_FXEN BIT(1) #define MICREL_KSZ8_P1_ERRATA BIT(2) -#define MICREL_KSZ9021_EXTREG_CTRL 0xB -#define MICREL_KSZ9021_EXTREG_DATA_WRITE 0xC -#define MICREL_KSZ9021_RGMII_CLK_CTRL_PAD_SCEW 0x104 -#define MICREL_KSZ9021_RGMII_RX_DATA_PAD_SCEW 0x105 - /* Device specific MII_BMCR (Reg 0) bits */ /* 1 = HP Auto MDI/MDI-X mode, 0 = Microchip Auto MDI/MDI-X mode */ #define KSZ886X_BMCR_HP_MDIX BIT(5) diff --git a/include/linux/minmax.h b/include/linux/minmax.h index a0158db54a04..5ef4d58c0c42 100644 --- a/include/linux/minmax.h +++ b/include/linux/minmax.h @@ -38,9 +38,9 @@ * Note that 'x' is the original expression, and 'ux' is the unique variable * that contains the value. * - * We use 'ux' for pure type checking, and 'x' for when we need to look at the - * value (but without evaluating it for side effects! - * Careful to only ever evaluate it with sizeof() or __builtin_constant_p() etc). + * We use 'ux' for both the type and the value checks, so 'x' itself is only + * expanded twice: once to initialise 'ux', and once quoted in the error + * message. * * Pointers end up being checked by the normal C type rules at the actual * comparison, and these expressions only need to be careful to not cause diff --git a/include/linux/mlx5/eswitch.h b/include/linux/mlx5/eswitch.h index a0dd162baa78..03d3620141c8 100644 --- a/include/linux/mlx5/eswitch.h +++ b/include/linux/mlx5/eswitch.h @@ -222,6 +222,9 @@ static inline bool is_mdev_switchdev_mode(struct mlx5_core_dev *dev) /* The returned number is valid only when the dev is eswitch manager. */ static inline u16 mlx5_eswitch_manager_vport(struct mlx5_core_dev *dev) { + if (MLX5_CAP_ESW(dev, esw_manager_vport_number_valid)) + return MLX5_CAP_ESW(dev, esw_manager_vport_number); + return mlx5_core_is_ecpf_esw_manager(dev) ? MLX5_VPORT_ECPF : MLX5_VPORT_HOST_PF; } diff --git a/include/linux/mm.h b/include/linux/mm.h index dd09c438fa23..3d135b09feeb 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -38,6 +38,7 @@ #include <linux/bitops.h> #include <linux/iommu-debug-pagealloc.h> #include <linux/kcsan-checks.h> +#include <linux/vmemmap-optimization.h> struct mempolicy; struct anon_vma; @@ -57,16 +58,6 @@ static inline unsigned long totalram_pages(void) return (unsigned long)atomic_long_read(&_totalram_pages); } -static inline void totalram_pages_inc(void) -{ - atomic_long_inc(&_totalram_pages); -} - -static inline void totalram_pages_dec(void) -{ - atomic_long_dec(&_totalram_pages); -} - static inline void totalram_pages_add(long count) { atomic_long_add(count, &_totalram_pages); @@ -587,14 +578,6 @@ enum { #define VMA_ACCESS_FLAGS mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, VMA_EXEC_BIT) /* - * Special vmas that are non-mergable, non-mlock()able. - */ - -#define VMA_SPECIAL_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_DONTEXPAND_BIT, \ - VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT) -#define VM_SPECIAL vma_flags_to_legacy(VMA_SPECIAL_FLAGS) - -/* * Physically remapped pages are special. Tell the * rest of the world about it: * IO tells people not to look at these pages @@ -610,9 +593,6 @@ enum { #define VMA_REMAP_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_PFNMAP_BIT, \ VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT) -/* This mask prevents VMA from being scanned with khugepaged */ -#define VM_NO_KHUGEPAGED (VM_SPECIAL | VM_HUGETLB) - /* This mask defines which mm->def_flags a process can inherit its parent */ #define VM_INIT_DEF_MASK VM_NOHUGEPAGE @@ -928,7 +908,6 @@ static inline void vma_numab_state_free(struct vm_area_struct *vma) {} * These must be here rather than mmap_lock.h as dependent on vm_fault type, * declared in this header. */ -#ifdef CONFIG_PER_VMA_LOCK static inline void release_fault_lock(struct vm_fault *vmf) { if (vmf->flags & FAULT_FLAG_VMA_LOCK) @@ -944,17 +923,6 @@ static inline void assert_fault_locked(const struct vm_fault *vmf) else mmap_assert_locked(vmf->vma->vm_mm); } -#else -static inline void release_fault_lock(struct vm_fault *vmf) -{ - mmap_read_unlock(vmf->vma->vm_mm); -} - -static inline void assert_fault_locked(const struct vm_fault *vmf) -{ - mmap_assert_locked(vmf->vma->vm_mm); -} -#endif /* CONFIG_PER_VMA_LOCK */ static inline bool mm_flags_test(int flag, const struct mm_struct *mm) { @@ -1551,11 +1519,6 @@ static inline void vma_set_anonymous(struct vm_area_struct *vma) vma->vm_ops = NULL; } -static inline void vma_desc_set_anonymous(struct vm_area_desc *desc) -{ - desc->vm_ops = NULL; -} - static inline bool vma_is_anonymous(const struct vm_area_struct *vma) { return !vma->vm_ops; @@ -1640,6 +1603,211 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) } /** + * vma_flags_is_hugetlb() - Do the specified VMA flags indicate that the + * VMA is a hugetlb mapping? + * @flags: The VMA flags to test. + * + * Returns: true if the flags indicate a hugetlb mapping, false otherwise. + */ +static inline bool vma_flags_is_hugetlb(const vma_flags_t *flags) +{ + return IS_ENABLED(CONFIG_HUGETLB_PAGE) && + vma_flags_test(flags, VMA_HUGETLB_BIT); +} + +/** + * vma_is_hugetlb() - Is @vma a hugetlb mapping? + * @vma: The VMA to test. + * + * Returns: true if @vma is a hugetlb mapping, false otherwise. + */ +static inline bool vma_is_hugetlb(const struct vm_area_struct *vma) +{ + return vma_flags_is_hugetlb(&vma->flags); +} + +/** + * vma_flags_is_kernel_owned() - Do the specified VMA flags indicate that the + * contents of the VMA are owned by the kernel rather than the core mm? + * @flags: The VMA flags to test. + * + * A kernel-owned mapping is one whose contents are established and controlled + * by the kernel, typically a driver, rather than by the core mm's fault and + * rmap machinery. + * + * The mapping may be memory-mapped I/O, kernel-allocated pages or ordinary + * pages the owner has chosen to map itself (shmem via a PFN map, for instance). + * + * In all cases the core mm must not populate, reclaim, migrate, copy-on-write + * or merge it of its own accord. + * + * Pages mapped this way are not necessarily reference counted or map counted. + * + * Returns: true if the flags indicate a kernel-owned mapping. + */ +static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) +{ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); +} + +/** + * vma_is_kernel_owned() - Are the contents of @vma owned by the kernel? + * @vma: The VMA to test. + * + * See vma_flags_is_kernel_owned() for a description of this property. + * + * Returns: true if the VMA is kernel-owned. + */ +static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) +{ + return vma_flags_is_kernel_owned(&vma->flags); +} + +/** + * vma_flags_is_fixed_mapping() - Do the specified VMA flags indicate that this + * is a fixed mapping that cannot be expanded or merged? + * @flags: The VMA flags to test. + * + * Fixed mappings are those whose size is set at the point of mmap (for + * instance, a kernel-owned mapping of a fixed range of memory), and thus + * cannot be expanded or merged. + * + * Returns: true if the flags indicate a fixed mapping. + */ +static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags) +{ + /* + * VMA_PFNMAP_BIT should imply VMA_DONTEXPAND_BIT, but some callers set + * only the former. + */ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_DONTEXPAND_BIT); +} + +/** + * vma_is_fixed_mapping() - Is this VMA a fixed mapping that cannot be + * expanded or merged? + * @vma: The VMA to test. + * + * See vma_flags_is_fixed_mapping() for a description of this property. + * + * Returns: true if the VMA maps a fixed mapping. + */ +static inline bool vma_is_fixed_mapping(const struct vm_area_struct *vma) +{ + return vma_flags_is_fixed_mapping(&vma->flags); +} + +/** + * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be + * merged with another? + * @flags: The VMA flags to test. + * Returns: true if the flags permit merging, false otherwise. + */ +static inline bool vma_flags_can_merge(const vma_flags_t *flags) +{ + /* + * VMA merging assumes that a VMA's flags and fields completely describe + * its state. + * + * However, kernel-owned mappings may have established state upon mapping + * not embodied in any attribute of the VMA. + * + * Additionally, private (CoW) PFN maps encode the source PFN of the + * range in vma->vm_pgoff, which may otherwise cause spurious merges. + */ + if (vma_flags_is_kernel_owned(flags)) + return false; + /* VMA explicitly marked as being unmergeable. */ + if (vma_flags_is_fixed_mapping(flags)) + return false; + + return true; +} + +/** + * vma_can_merge() - Do @vma's flags permit it to be merged with another VMA? + * @vma: The VMA to test. + * Returns: true if the flags permit merging, otherwise false. + */ +static inline bool vma_can_merge(const struct vm_area_struct *vma) +{ + return vma_flags_can_merge(&vma->flags); +} + +/** + * vma_flags_is_persistent() - Do the specified VMA flags imply that the VMA + * contains persistent data? + * @flags: The VMA flags to test. + * + * Persistent in the sense that - if you write bytes to the mapping - do they + * stay written? + * + * If the kernel or a device could write to the memory independently of + * userland, or the kernel could arbitrarily discard it, then it is not + * persistent. + * + * Returns: true if the flags imply this VMA is persistent, otherwise false. + */ +static inline bool vma_flags_is_persistent(const vma_flags_t *flags) +{ + /* hugetlb is a fixed mapping, but its contents are the user's own. */ + if (vma_flags_is_hugetlb(flags)) + return true; + /* + * MMIO mappings may not store what is written and may be changed by the + * device. Kernel-owned and fixed mappings may be changed by their owner + * without the user having initiated it. + */ + if (vma_flags_is_kernel_owned(flags) || + vma_flags_is_fixed_mapping(flags)) + return false; + /* Droppable memory is discardable by definition. */ + return !vma_flags_test_single_mask(flags, VMA_DROPPABLE); +} + +/** + * vma_is_persistent() - Does the VMA contain persistent data? + * @vma: The VMA to test. + * + * See vma_flags_is_persistent() for details. + * + * Returns: true if the VMA is persistent, otherwise false. + */ +static inline bool vma_is_persistent(const struct vm_area_struct *vma) +{ + return vma_flags_is_persistent(&vma->flags); +} + +/** + * vma_flags_can_gup() - Do the specified VMA flags permit GUP to access the + * mapping's pages? + * @flags: The VMA flags to test. + * + * GUP cannot obtain pages from a PFN map (VMA_PFNMAP_BIT), which may have no + * struct pages behind it, and must not provide access to memory-mapped I/O + * (VMA_IO_BIT). + * + * Returns: true if GUP may access pages from the mapping, otherwise false. + */ +static inline bool vma_flags_can_gup(const vma_flags_t *flags) +{ + return !vma_flags_test_any(flags, VMA_IO_BIT, VMA_PFNMAP_BIT); +} + +/** + * vma_can_gup() - May GUP obtain pages from @vma? + * @vma: The VMA to test. + * + * See vma_flags_can_gup() for details. + * + * Returns: true if GUP may access pages from the mapping, otherwise false. + */ +static inline bool vma_can_gup(const struct vm_area_struct *vma) +{ + return vma_flags_can_gup(&vma->flags); +} + +/** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. * @@ -2075,20 +2243,21 @@ vm_fault_t finish_fault(struct vm_fault *vmf); * * A pagecache page contains an opaque `private' member, which belongs to the * page's address_space. Usually, this is the address of a circular list of - * the page's disk buffers. PG_private must be set to tell the VM to call - * into the filesystem to release these pages. + * the page's disk buffers. It tells the VM to call into the filesystem to + * release these pages. * * A folio may belong to an inode's memory mapping. In this case, * folio->mapping points to the inode, and folio->index is the file * offset of the folio, in units of PAGE_SIZE. * - * If pagecache pages are not associated with an inode, they are said to be - * anonymous pages. These may become associated with the swapcache, and in that - * case PG_swapcache is set, and page->private is an offset into the swapcache. + * If pagecache folios are not associated with an inode, they are said to be + * anonymous folios. These may become associated with the swapcache, and in that + * case PG_swapcache is set, and folio->private is an offset into the swapcache. * * In either case (swapcache or inode backed), the pagecache itself holds one - * reference to the page. Setting PG_private should also increment the - * refcount. The each user mapping also has a reference to the page. + * reference to the folio. Attaching filesystem private data via + * folio_attach_private() also increments the refcount. Each user mapping also + * has a reference to the folio. * * The pagecache pages are stored in a per-mapping radix tree, which is * rooted at mapping->i_pages, and indexed by offset. @@ -2644,12 +2813,23 @@ static inline void set_page_section(struct page *page, unsigned long section) page->flags.f |= (section & SECTIONS_MASK) << SECTIONS_PGSHIFT; } +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ + set_page_section(page, pfn_to_section_nr(pfn)); +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { ASSERT_EXCLUSIVE_BITS(mdf->f, SECTIONS_MASK << SECTIONS_PGSHIFT); return (mdf->f >> SECTIONS_PGSHIFT) & SECTIONS_MASK; } #else /* !SECTION_IN_PAGE_FLAGS */ +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { return 0; @@ -2872,9 +3052,7 @@ static inline void set_page_links(struct page *page, enum zone_type zone, { set_page_zone(page, zone); set_page_node(page, node); -#ifdef SECTION_IN_PAGE_FLAGS - set_page_section(page, pfn_to_section_nr(pfn)); -#endif + set_page_section_from_pfn(page, pfn); } /** @@ -3022,9 +3200,9 @@ static inline bool folio_maybe_mapped_shared(struct folio *folio) * @folio: the folio * * Calculate the expected folio refcount, taking references from the pagecache, - * swapcache, PG_private and page table mappings into account. Useful in - * combination with folio_ref_count() to detect unexpected references (e.g., - * GUP or other temporary references). + * swapcache, private data (folio->private != NULL) and page table mappings into + * account. Useful in combination with folio_ref_count() to detect unexpected + * references (e.g., GUP or other temporary references). * * Does currently not consider references from the LRU cache. If the folio * was isolated from the LRU (which is the case during migration or split), @@ -3062,10 +3240,15 @@ static inline int folio_expected_ref_count(const struct folio *folio) ref_count += folio_test_swapcache(folio) << order; if (!folio_test_anon(folio)) { - /* One reference per page from the pagecache. */ - ref_count += !!folio->mapping << order; - /* One reference from PG_private. */ - ref_count += folio_test_private(folio); + /* + * One reference per page from the pagecache. + * Use data_race() since folio might not be locked. + */ + ref_count += !!data_race(folio->mapping) << order; + /* + * One reference from filesystem private data. + */ + ref_count += folio_has_attached_private(folio); } /* One reference per page table mapping. */ @@ -3325,10 +3508,10 @@ extern int access_process_vm(struct task_struct *tsk, unsigned long addr, extern int access_remote_vm(struct mm_struct *mm, unsigned long addr, void *buf, int len, unsigned int gup_flags); -#ifdef CONFIG_BPF_SYSCALL -extern int copy_remote_vm_str(struct task_struct *tsk, unsigned long addr, - void *buf, int len, unsigned int gup_flags); -#endif +int copy_remote_mm_str(struct mm_struct *mm, unsigned long addr, + void *buf, int len, unsigned int gup_flags); +int copy_remote_vm_str(struct task_struct *tsk, unsigned long addr, + void *buf, int len, unsigned int gup_flags); long get_user_pages_remote(struct mm_struct *mm, unsigned long start, unsigned long nr_pages, @@ -4083,12 +4266,6 @@ static inline void free_reserved_page(struct page *page) free_reserved_pages(page, 0); } -static inline void mark_page_reserved(struct page *page) -{ - SetPageReserved(page); - adjust_managed_page_count(page, -1); -} - static inline void free_reserved_ptdesc(struct ptdesc *pt) { free_reserved_page(ptdesc_page(pt)); @@ -4181,6 +4358,12 @@ void mapping_rmap_tree_insert_after(struct vm_area_struct *vma, struct address_space *mapping); void mapping_rmap_tree_remove(struct vm_area_struct *vma, struct address_space *mapping); +void mapping_rmap_tree_pre_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged); +void mapping_rmap_tree_post_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged); struct vm_area_struct * mapping_rmap_tree_iter_first(struct address_space *mapping, pgoff_t pgoff_start, pgoff_t pgoff_last); @@ -4198,6 +4381,10 @@ void anon_rmap_tree_insert(struct anon_vma_chain *avc, struct anon_vma *anon_vma); void anon_rmap_tree_remove(struct anon_vma_chain *avc, struct anon_vma *anon_vma); +void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged); +void anon_rmap_tree_post_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged); struct anon_vma_chain * anon_rmap_tree_iter_first(struct anon_vma *anon_vma, pgoff_t pgoff_start, pgoff_t pgoff_last); @@ -4408,9 +4595,8 @@ static inline unsigned long vma_pages(const struct vm_area_struct *vma) * If @vma is a MAP_PRIVATE file-backed mapping, then this returns the * page offset within the file. * - * Edge cases: nommu does not abide by these, MAP_PRIVATE-/dev/zero satisfies - * vma_is_anonymous() but has file-backed page offset, and MAP_PRIVATE-pfnmap - * regions have their page offset set to the first PFN in the range. + * Edge cases: nommu does not abide by these and CoW MAP_PRIVATE-pfnmap regions + * have their page offset set to the first PFN in the range. * * Returns: The page offset of the start of @vma. */ @@ -4619,7 +4805,7 @@ static inline void mmap_action_simple_ioremap(struct vm_area_desc *desc, * @desc: The VMA descriptor for the VMA requiring kernel pags to be mapped. * @start: The virtual address from which to map them. * @pages: An array of struct page pointers describing the memory to map. - * @nr_pages: The number of entries in the @pages aray. + * @nr_pages: The number of entries in the @pages array. */ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc, unsigned long start, struct page **pages, @@ -4627,7 +4813,7 @@ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc, { struct mmap_action *action = &desc->action; - action->type = MMAP_MAP_KERNEL_PAGES; + action->type = MMAP_KERNEL_PAGES; action->map_kernel.start = start; action->map_kernel.pages = pages; action->map_kernel.nr_pages = nr_pages; @@ -4651,10 +4837,55 @@ static inline void mmap_action_map_kernel_pages_full(struct vm_area_desc *desc, vma_desc_pages(desc)); } +static inline +void mmap_action_map_discontig_kernel_pages(struct vm_area_desc *desc, + void *init_private, const struct discontig_kernel_page_ops *ops) +{ + struct mmap_action *action = &desc->action; + + action->type = MMAP_DISCONTIG_KERNEL_PAGES; + action->map_kernel_discontig.init_private = init_private; + action->map_kernel_discontig.ops = ops; +} + int mmap_action_prepare(struct vm_area_desc *desc); int mmap_action_complete(struct vm_area_struct *vma, struct mmap_action *action, bool is_compat); +static inline void +discontig_kernel_map_abort(struct discontig_kernel_page_state *state) +{ + state->action = DISCONTIG_KERNEL_PAGE_ABORT; +} + +static inline void +discontig_kernel_map_page(struct discontig_kernel_page_state *state, + struct page *page) +{ + struct folio *folio = page_folio(page); + + if (folio_test_large(folio)) { + VM_WARN_ON_ONCE(page != folio_page(folio, 0)); + state->action = DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE; + state->__folio = folio; + state->__nr_pages = min(state->nr_pages_remain, + folio_nr_pages(folio)); + } else { + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE; + state->__page = page; + state->__nr_pages = 1; + } +} + +static inline void +discontig_kernel_map_page_range(struct discontig_kernel_page_state *state, + struct page **page_arr, unsigned long nr_pages) +{ + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE; + state->__page_arr = page_arr; + state->__nr_pages = nr_pages; +} + /* Look up the first VMA which exactly match the interval vm_start ... vm_end */ static inline struct vm_area_struct *find_exact_vma(struct mm_struct *mm, unsigned long vm_start, unsigned long vm_end) @@ -4772,9 +5003,6 @@ int remap_pfn_range(struct vm_area_struct *vma, unsigned long addr, int vm_insert_page(struct vm_area_struct *, unsigned long addr, struct page *); int vm_insert_pages(struct vm_area_struct *vma, unsigned long addr, struct page **pages, unsigned long *num); -int map_kernel_pages_prepare(struct vm_area_desc *desc); -int map_kernel_pages_complete(struct vm_area_struct *vma, - struct mmap_action *action); int vm_map_pages(struct vm_area_struct *vma, struct page **pages, unsigned long num); int vm_map_pages_zero(struct vm_area_struct *vma, struct page **pages, @@ -4787,8 +5015,6 @@ vm_fault_t vmf_insert_pfn_prot(struct vm_area_struct *vma, unsigned long addr, unsigned long pfn, pgprot_t pgprot); vm_fault_t vmf_insert_mixed(struct vm_area_struct *vma, unsigned long addr, unsigned long pfn); -vm_fault_t vmf_insert_mixed_mkwrite(struct vm_area_struct *vma, - unsigned long addr, unsigned long pfn); int vm_iomap_memory(struct vm_area_struct *vma, phys_addr_t start, unsigned long len); static inline vm_fault_t vmf_insert_page(struct vm_area_struct *vma, @@ -5140,7 +5366,6 @@ static inline void print_vma_addr(char *prefix, unsigned long rip) } #endif -unsigned long section_map_size(void); struct page * __populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); @@ -5159,9 +5384,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); int vmemmap_populate(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); -int vmemmap_populate_hvo(unsigned long start, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize); void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node, unsigned long headsize); void vmemmap_populate_print_last(void); @@ -5196,7 +5418,6 @@ static inline void vmem_altmap_free(struct vmem_altmap *altmap, } #endif -#define VMEMMAP_RESERVE_NR 2 #ifdef CONFIG_ARCH_WANT_OPTIMIZE_DAX_VMEMMAP static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, struct dev_pagemap *pgmap) @@ -5204,6 +5425,9 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, unsigned long nr_pages; unsigned long nr_vmemmap_pages; + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) + return false; + if (!pgmap || !is_power_of_2(sizeof(struct page))) return false; @@ -5213,7 +5437,7 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, * For vmemmap optimization with DAX we need minimum 2 vmemmap * pages. See layout diagram in Documentation/mm/vmemmap_dedup.rst */ - return !altmap && (nr_vmemmap_pages > VMEMMAP_RESERVE_NR); + return !altmap && (nr_vmemmap_pages > VMEMMAP_OPTIMIZATION_PAGES); } /* * If we don't have an architecture override, use the generic rule @@ -5254,6 +5478,8 @@ extern const struct attribute_group memory_failure_attr_group; extern void memory_failure_queue(unsigned long pfn, int flags); void num_poisoned_pages_inc(unsigned long pfn); void num_poisoned_pages_sub(unsigned long pfn, long i); +phys_addr_t range_first_hwpoison(phys_addr_t start, unsigned long size); +phys_addr_t range_last_hwpoison(phys_addr_t start, unsigned long size); #else static inline void memory_failure_queue(unsigned long pfn, int flags) { @@ -5266,6 +5492,18 @@ static inline void num_poisoned_pages_inc(unsigned long pfn) static inline void num_poisoned_pages_sub(unsigned long pfn, long i) { } + +static inline phys_addr_t range_first_hwpoison(phys_addr_t start, + unsigned long size) +{ + return PHYS_ADDR_MAX; +} + +static inline phys_addr_t range_last_hwpoison(phys_addr_t start, + unsigned long size) +{ + return PHYS_ADDR_MAX; +} #endif #if defined(CONFIG_MEMORY_FAILURE) && defined(CONFIG_MEMORY_HOTPLUG) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 621c8653d8f7..ab69b9930893 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -30,6 +30,17 @@ static inline int folio_is_file_lru(const struct folio *folio) return !folio_test_swapbacked(folio); } +/** + * folio_flags_is_file_lru - Should the folio be on a file LRU or anon LRU? + * @flags: The folio's flags. + * + * Just like folio_is_file_lru but take the folio flags directly instead. + */ +static inline int folio_flags_is_file_lru(const unsigned long *flags) +{ + return !test_bit(PG_swapbacked, flags); +} + static __always_inline void __update_lru_size(struct lruvec *lruvec, enum lru_list lru, enum zone_type zid, long nr_pages) @@ -142,10 +153,66 @@ static inline int lru_tier_from_refs(int refs, bool workingset) return workingset ? MAX_NR_TIERS - 1 : order_base_2(refs); } -static inline int folio_lru_refs(const struct folio *folio) +/** + * lru_set_gen_flags - Set the LRU generation number to specified folio flags. + * @flags: pointer to the folio flags + * @gen: generation number, between 0 and (MAX_NR_GENS - 1), inclusive. + */ +static inline void lru_set_gen_flags(unsigned long *flags, int gen) +{ + BUILD_BUG_ON(LRU_GEN_MASK & LRU_REFS_MASK); + VM_WARN_ON_ONCE(gen >= MAX_NR_GENS || gen < 0); + /* Store gen offset by 1, zero means the folio is off-list. */ + *flags &= ~LRU_GEN_MASK; + *flags |= (gen + 1UL) << LRU_GEN_PGOFF; +} + +/** + * lru_get_gen_flags - Return the LRU generation number from folio flags. + * @flags: folio flags + * + * Returns: A number between 0 and (MAX_NR_GENS - 1), inclusive. Returns + * -1 if the flags indicate the folio is off the list (e.g., isolated). + */ +static inline int lru_get_gen_flags(unsigned long flags) +{ + int gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + + /* Exclude the legal -1 from the unsigned MAX_NR_GENS comparison */ + VM_WARN_ON_ONCE(gen != -1 && gen >= MAX_NR_GENS); + return gen; +} + +/** + * lru_set_refs_flags - Set the LRU referenced count to folio flags. + * @flags: pointer to the folio flags + * @refs: referenced / access count number, between 0 and LRU_REFS_MAX, inclusive. + * + * For MGLRU, PG_referenced holds the first ref, and the extra bits hold the + * remaining refs. For classical LRU the extra bits are not used, so it can + * also be seen as the refs count never exceeds 1. In both cases, refs == 1 + * means PG_referenced is set and the extra bits are zero, and refs == 0 means + * PG_referenced and the extra bits are all unset. + */ +static inline void lru_set_refs_flags(unsigned long *flags, unsigned int refs) { - unsigned long flags = READ_ONCE(folio->flags.f); + VM_WARN_ON_ONCE(refs > LRU_REFS_MAX); + BUILD_BUG_ON(LRU_REFS_MAX != (LRU_REFS_MASK >> LRU_REFS_PGOFF) + 1); + *flags &= ~LRU_REFS_FLAGS; + if (!refs) + return; + *flags |= (BIT(PG_referenced) | ((refs - 1UL) << LRU_REFS_PGOFF)); +} + +/** + * lru_get_refs_flags - Return LRU referenced / access count from folio flags. + * @flags: folio flags + * + * Reads the LRU referenced count set by lru_set_refs_flags(). + */ +static inline int lru_get_refs_flags(unsigned long flags) +{ if (!(flags & BIT(PG_referenced))) return 0; /* @@ -155,11 +222,24 @@ static inline int folio_lru_refs(const struct folio *folio) return ((flags & LRU_REFS_MASK) >> LRU_REFS_PGOFF) + 1; } -static inline int folio_lru_gen(const struct folio *folio) +static inline int folio_lru_refs(const struct folio *folio) { - unsigned long flags = READ_ONCE(folio->flags.f); + return lru_get_refs_flags(READ_ONCE(*const_folio_flags(folio, 0))); +} - return ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; +static inline void folio_set_lru_refs(struct folio *folio, unsigned int refs) +{ + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + + do { + new_flags = old_flags; + lru_set_refs_flags(&new_flags, refs); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); +} + +static inline int folio_lru_gen(const struct folio *folio) +{ + return lru_get_gen_flags(READ_ONCE(*const_folio_flags(folio, 0))); } static inline bool lru_gen_is_active(const struct lruvec *lruvec, int gen) @@ -270,7 +350,7 @@ static inline bool lru_gen_add_folio(struct lruvec *lruvec, struct folio *folio, gen = lru_gen_from_seq(seq); flags = (gen + 1UL) << LRU_GEN_PGOFF; /* see the comment on MIN_NR_GENS about PG_active */ - set_mask_bits(&folio->flags.f, LRU_GEN_MASK | BIT(PG_active), flags); + set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK | BIT(PG_active), flags); lru_gen_update_size(lruvec, folio, -1, gen); /* for folio_rotate_reclaimable() */ @@ -295,7 +375,7 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, /* for folio_migrate_flags() */ flags = !reclaiming && lru_gen_is_active(lruvec, gen) ? BIT(PG_active) : 0; - flags = set_mask_bits(&folio->flags.f, LRU_GEN_MASK, flags); + flags = set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK, flags); gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; lru_gen_update_size(lruvec, folio, gen, -1); @@ -304,11 +384,19 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return true; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +/** + * folio_migrate_lru_refs - copy the reference state to a new folio + * @new: the destination folio + * @old: the source folio + * + * Transfer the reference state to @new during migration: the MGLRU + * refs count, including PG_referenced, or just PG_referenced for the + * active/inactive LRU. + */ +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { - unsigned long refs = READ_ONCE(old->flags.f) & LRU_REFS_MASK; - - set_mask_bits(&new->flags.f, LRU_REFS_MASK, refs); + BUILD_BUG_ON(LRU_REFS_MASK & BIT(PG_referenced)); + folio_set_lru_refs(new, folio_lru_refs(old)); } #else /* !CONFIG_LRU_GEN */ @@ -337,9 +425,10 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return false; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { - + if (folio_test_referenced(old)) + folio_set_referenced(new); } #endif /* CONFIG_LRU_GEN */ diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 6d815f6440c9..6141160ec652 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -108,7 +108,7 @@ struct page { }; /** * @private: Mapping-private opaque data. - * Usually used for buffer_heads if PagePrivate. + * Usually used for buffer_heads. * Used for swp_entry_t if swapcache flag set. * Indicates order in the buddy system if PageBuddy * or on pcp_llist. @@ -675,7 +675,7 @@ static inline void ptdesc_pmd_pts_init(struct ptdesc *ptdesc) #define STRUCT_PAGE_MAX_SHIFT (order_base_2(sizeof(struct page))) /* - * page_private can be used on tail pages. However, PagePrivate is only + * page_private can be used on tail pages. However, it is only * checked by the VM on the head page. So page_private on the tail pages * should be used for data that's ancillary to the head page (eg attaching * buffer heads to tail pages after attaching buffer heads to the head page) @@ -759,7 +759,7 @@ static inline struct anon_vma_name *anon_vma_name_alloc(const char *name) /* * While __vma_enter_locked() is working to ensure are no read-locks held on a * VMA (either while acquiring a VMA write lock or marking a VMA detached) we - * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indiciate to + * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indicate to * vma_start_read() that the reference count should be left alone. * * See the comment describing vm_refcnt in vm_area_struct for details as to @@ -815,11 +815,47 @@ struct pfnmap_track_ctx { /* What action should be taken after an .mmap_prepare call is complete? */ enum mmap_action_type { - MMAP_NOTHING, /* Mapping is complete, no further action. */ - MMAP_REMAP_PFN, /* Remap PFN range. */ - MMAP_IO_REMAP_PFN, /* I/O remap PFN range. */ - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_MAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_NOTHING, + MMAP_REMAP_PFN, + MMAP_IO_REMAP_PFN, + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */ +}; + +enum discontig_kernel_page_action { + DISCONTIG_KERNEL_PAGE_ABORT, + DISCONTIG_KERNEL_PAGE_MAP_PAGE, + DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE, + DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE, +}; + +struct discontig_kernel_page_state { + /* Map state. */ + const unsigned long start; /* Start address of VMA. */ + const unsigned long end; /* End address of VMA. */ + unsigned long addr; /* The current address to be mapped. */ + pgoff_t pgoff; /* The current pgoff to be mapped. */ + unsigned long nr_pages_mapped; /* The number of pages mapped. */ + unsigned long nr_pages_remain; /* The number of pages remaining. */ + + /* User-defined state. */ + void *vm_private_data; /* VMA private data. */ + void *private; /* Mapping private data. */ + + /* Users should not touch these, use discontig_kernel_map_*() helpers. */ + enum discontig_kernel_page_action action; + union { + struct page *__page; + struct folio *__folio; + struct page **__page_arr; + }; + unsigned long __nr_pages; +}; + +struct discontig_kernel_page_ops { + int (*init)(void *vm_private_data, void **private); + int (*get)(struct discontig_kernel_page_state *state); }; /* @@ -844,6 +880,10 @@ struct mmap_action { unsigned long nr_pages; pgoff_t pgoff; } map_kernel; + struct { + void *init_private; + const struct discontig_kernel_page_ops *ops; + } map_kernel_discontig; }; enum mmap_action_type type; @@ -950,7 +990,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -966,7 +1005,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + /* * Low 32-bits of anonymous page offset. * See vma_start_anon_pgoff() comment for details. @@ -1003,7 +1042,6 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* * Used to keep track of firstly, whether the VMA is attached, secondly, * if attached, how many read locks are taken, and thirdly, if the @@ -1046,7 +1084,6 @@ struct vm_area_struct { #ifdef CONFIG_DEBUG_LOCK_ALLOC struct lockdep_map vmlock_dep_map; #endif -#endif #ifdef CONFIG_64BIT /* * High 32-bits of anonymous page offset. @@ -1226,7 +1263,7 @@ struct mm_struct { struct mm_mm_cid mm_cid; /* sched_cache related statistics */ - struct sched_cache_stat sc_stat; + struct sched_cache_group *sched_cache_grp; #ifdef CONFIG_MMU atomic_long_t pgtables_bytes; /* size of all page tables */ #endif @@ -1254,7 +1291,6 @@ struct mm_struct { * init_mm.mmlist, and are protected * by mmlist_lock */ -#ifdef CONFIG_PER_VMA_LOCK struct rcuwait vma_writer_wait; /* * This field has lock-like semantics, meaning it is sometimes @@ -1274,7 +1310,7 @@ struct mm_struct { * mmap_lock. */ seqcount_t mm_lock_seq; -#endif + struct futex_mm_data futex; unsigned long hiwater_rss; /* High-watermark of RSS usage */ @@ -1624,8 +1660,9 @@ static inline unsigned int mm_cid_size(void) #endif /* CONFIG_SCHED_MM_CID */ #ifdef CONFIG_SCHED_CACHE -void mm_init_sched(struct mm_struct *mm, - struct sched_cache_time __percpu *pcpu_sched); +int mm_init_sched(struct mm_struct *mm, + struct sched_cache_time __percpu *pcpu_sched); +void mm_destroy_sched(struct mm_struct *mm); static inline int mm_alloc_sched_noprof(struct mm_struct *mm) { @@ -1635,17 +1672,11 @@ static inline int mm_alloc_sched_noprof(struct mm_struct *mm) if (!pcpu_sched) return -ENOMEM; - mm_init_sched(mm, pcpu_sched); - return 0; + return mm_init_sched(mm, pcpu_sched); } #define mm_alloc_sched(...) alloc_hooks(mm_alloc_sched_noprof(__VA_ARGS__)) -static inline void mm_destroy_sched(struct mm_struct *mm) -{ - free_percpu(mm->sc_stat.pcpu_sched); - mm->sc_stat.pcpu_sched = NULL; -} #else /* !CONFIG_SCHED_CACHE */ static inline int mm_alloc_sched(struct mm_struct *mm) { return 0; } @@ -1980,7 +2011,7 @@ enum { /* * MMF_HAS_PINNED: Whether this mm has pinned any pages. This can be either * replaced in the future by mm.pinned_vm when it becomes stable, or grow into - * a counter on its own. We're aggresive on this bit for now: even if the + * a counter on its own. We're aggressive on this bit for now: even if the * pinned pages were unpinned later on, we'll still keep this bit set for the * lifecycle of this mm, just for simplicity. */ diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index bec0eab6ef03..e5553f4a414c 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -76,8 +76,6 @@ static inline void mmap_assert_write_locked(const struct mm_struct *mm) rwsem_assert_held_write(&mm->mmap_lock); } -#ifdef CONFIG_PER_VMA_LOCK - #ifdef CONFIG_LOCKDEP #define __vma_lockdep_map(vma) (&vma->vmlock_dep_map) #else @@ -230,10 +228,14 @@ static inline void vma_refcount_put(struct vm_area_struct *vma) } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int subclass) { @@ -249,22 +251,29 @@ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked(struct vm_area_struct *vma) { return vma_start_read_locked_nested(vma, 0); } +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address); + static inline void vma_end_read(struct vm_area_struct *vma) { vma_refcount_put(vma); } -static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma) +static inline unsigned int __vma_raw_mm_seqnum(const struct vm_area_struct *vma) { const struct mm_struct *mm = vma->vm_mm; @@ -279,7 +288,7 @@ static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma) * * Returns true if write-locked, otherwise false. */ -static inline bool __is_vma_write_locked(struct vm_area_struct *vma) +static inline bool __is_vma_write_locked(const struct vm_area_struct *vma) { /* * current task is holding mmap_write_lock, both vma->vm_lock_seq and @@ -297,6 +306,9 @@ int __vma_start_write(struct vm_area_struct *vma, int state); */ static inline void vma_start_write(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + if (__is_vma_write_locked(vma)) return; @@ -319,6 +331,9 @@ static inline void vma_start_write(struct vm_area_struct *vma) static inline __must_check int vma_start_write_killable(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return 0; + if (__is_vma_write_locked(vma)) return 0; @@ -329,8 +344,13 @@ int vma_start_write_killable(struct vm_area_struct *vma) * vma_assert_write_locked() - assert that @vma holds a VMA write lock. * @vma: The VMA to assert. */ -static inline void vma_assert_write_locked(struct vm_area_struct *vma) +static inline void vma_assert_write_locked(const struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_write_locked(vma->vm_mm); + return; + } + VM_WARN_ON_ONCE_VMA(!__is_vma_write_locked(vma), vma); } @@ -339,10 +359,15 @@ static inline void vma_assert_write_locked(struct vm_area_struct *vma) * lock and is not detached. * @vma: The VMA to assert. */ -static inline void vma_assert_locked(struct vm_area_struct *vma) +static inline void vma_assert_locked(const struct vm_area_struct *vma) { unsigned int refcnt; + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_locked(vma->vm_mm); + return; + } + if (IS_ENABLED(CONFIG_LOCKDEP)) { if (!lock_is_held(__vma_lockdep_map(vma))) vma_assert_write_locked(vma); @@ -385,7 +410,7 @@ static inline void vma_assert_locked(struct vm_area_struct *vma) * With lockdep disabled we may sometimes race with other threads acquiring the * mmap read lock simultaneous with our VMA read lock. */ -static inline void vma_assert_stabilised(struct vm_area_struct *vma) +static inline void vma_assert_stabilised(const struct vm_area_struct *vma) { /* * If another thread owns an mmap lock, it may go away at any time, and @@ -420,7 +445,7 @@ static inline void vma_assert_stabilised(struct vm_area_struct *vma) vma_assert_locked(vma); } -static inline bool vma_is_attached(struct vm_area_struct *vma) +static inline bool vma_is_attached(const struct vm_area_struct *vma) { return refcount_read(&vma->vm_refcnt); } @@ -432,6 +457,9 @@ static inline bool vma_is_attached(struct vm_area_struct *vma) */ static inline void vma_assert_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + WARN_ON_ONCE(!vma_is_attached(vma)); } @@ -442,6 +470,9 @@ static inline void vma_assert_detached(struct vm_area_struct *vma) static inline void vma_mark_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_detached(vma); refcount_set_release(&vma->vm_refcnt, 1); @@ -451,6 +482,9 @@ void __vma_exclude_readers_for_detach(struct vm_area_struct *vma); static inline void vma_mark_detached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_attached(vma); @@ -484,54 +518,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, struct vma_iterator *iter, unsigned long address); -#else /* CONFIG_PER_VMA_LOCK */ - -static inline void mm_lock_seqcount_init(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_begin(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_end(struct mm_struct *mm) {} - -static inline bool mmap_lock_speculate_try_begin(struct mm_struct *mm, unsigned int *seq) -{ - return false; -} - -static inline bool mmap_lock_speculate_retry(struct mm_struct *mm, unsigned int seq) -{ - return true; -} -static inline void vma_lock_init(struct vm_area_struct *vma, bool reset_refcnt) {} -static inline void vma_end_read(struct vm_area_struct *vma) {} -static inline void vma_start_write(struct vm_area_struct *vma) {} -static inline __must_check -int vma_start_write_killable(struct vm_area_struct *vma) { return 0; } -static inline void vma_assert_write_locked(struct vm_area_struct *vma) - { mmap_assert_write_locked(vma->vm_mm); } -static inline bool vma_is_attached(struct vm_area_struct *vma) - { return true; } -static inline void vma_assert_attached(struct vm_area_struct *vma) {} -static inline void vma_assert_detached(struct vm_area_struct *vma) {} -static inline void vma_mark_attached(struct vm_area_struct *vma) {} -static inline void vma_mark_detached(struct vm_area_struct *vma) {} - -static inline struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, - unsigned long address) -{ - return NULL; -} - -static inline void vma_assert_locked(struct vm_area_struct *vma) -{ - mmap_assert_locked(vma->vm_mm); -} - -static inline void vma_assert_stabilised(struct vm_area_struct *vma) -{ - /* If no VMA locks, then either mmap lock suffices to stabilise. */ - mmap_assert_locked(vma->vm_mm); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static inline void vma_assert_can_modify(struct vm_area_struct *vma) { if (vma_is_attached(vma)) @@ -630,6 +616,8 @@ static inline void mmap_read_unlock(struct mm_struct *mm) DEFINE_GUARD(mmap_read_lock, struct mm_struct *, mmap_read_lock(_T), mmap_read_unlock(_T)) DEFINE_GUARD_COND(mmap_read_lock, _try, mmap_read_trylock(_T)) +DEFINE_GUARD(mmap_write_lock, struct mm_struct *, + mmap_write_lock(_T), mmap_write_unlock(_T)) static inline void mmap_read_unlock_non_owner(struct mm_struct *mm) { diff --git a/include/linux/mmc/host.h b/include/linux/mmc/host.h index ba84f02c2a10..032c018650e0 100644 --- a/include/linux/mmc/host.h +++ b/include/linux/mmc/host.h @@ -23,6 +23,7 @@ struct mmc_ios { unsigned int clock; /* clock rate */ unsigned short vdd; unsigned int power_delay_ms; /* waiting for stable power */ + unsigned int power_off_delay_us; /* waiting for power discharge */ /* vdd stores the bit number of the selected voltage range from below. */ @@ -463,6 +464,7 @@ struct mmc_host { #define MMC_CAP2_CRYPTO 0 #endif #define MMC_CAP2_ALT_GPT_TEGRA (1 << 28) /* Host with eMMC that has GPT entry at a non-standard location */ +#define MMC_CAP2_CRYPTO_NO_REPROG (1 << 29) /* Host handles inline crypto key reprogramming */ bool uhs2_sd_tran; /* UHS-II flag for SD_TRAN state */ bool uhs2_app_cmd; /* UHS-II flag for APP command */ @@ -583,11 +585,9 @@ struct mmc_host { struct device_node; -struct mmc_host *mmc_alloc_host(int extra, struct device *); struct mmc_host *devm_mmc_alloc_host(struct device *dev, int extra); int mmc_add_host(struct mmc_host *); void mmc_remove_host(struct mmc_host *); -void mmc_free_host(struct mmc_host *); void mmc_of_parse_clk_phase(struct device *dev, struct mmc_clk_phase_map *map); int mmc_of_parse(struct mmc_host *host); @@ -753,6 +753,8 @@ static inline int mmc_card_uhs2_hd_mode(struct mmc_host *host) int mmc_sd_switch(struct mmc_card *card, bool mode, int group, u8 value, u8 *resp); int mmc_send_status(struct mmc_card *card, u32 *status); +int mmc_send_tuning_timeout(struct mmc_host *host, u32 opcode, int *cmd_error, + u32 timeout_ms); int mmc_send_tuning(struct mmc_host *host, u32 opcode, int *cmd_error); int mmc_send_abort_tuning(struct mmc_host *host, u32 opcode); int mmc_get_ext_csd(struct mmc_card *card, u8 **new_ext_csd); diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 94f9c3ff5416..ecd1ebd49756 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -96,25 +96,6 @@ #define MAX_FOLIO_NR_PAGES (1UL << MAX_FOLIO_ORDER) -/* - * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to - * be naturally aligned with regard to the folio size. - * - * HVO which is only active if the size of struct page is a power of 2. - */ -#define MAX_FOLIO_VMEMMAP_ALIGN \ - (IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP) && \ - is_power_of_2(sizeof(struct page)) ? \ - MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) - -/* - * vmemmap optimization (like HVO) is only possible for page orders that fill - * two or more pages with struct pages. - */ -#define VMEMMAP_TAIL_MIN_ORDER (ilog2(2 * PAGE_SIZE / sizeof(struct page))) -#define __NR_VMEMMAP_TAILS (MAX_FOLIO_ORDER - VMEMMAP_TAIL_MIN_ORDER + 1) -#define NR_VMEMMAP_TAILS (__NR_VMEMMAP_TAILS > 0 ? __NR_VMEMMAP_TAILS : 0) - enum migratetype { MIGRATE_UNMOVABLE, MIGRATE_MOVABLE, @@ -492,11 +473,14 @@ enum lruvec_flags { * folio->flags, masked by LRU_REFS_MASK. */ #define MAX_NR_TIERS 4U +#define LRU_TIER_MIN 0U +#define LRU_TIER_MAX (MAX_NR_TIERS - 1) #ifndef __GENERATING_BOUNDS_H #define LRU_GEN_MASK ((BIT(LRU_GEN_WIDTH) - 1) << LRU_GEN_PGOFF) #define LRU_REFS_MASK ((BIT(LRU_REFS_WIDTH) - 1) << LRU_REFS_PGOFF) +#define LRU_REFS_MAX BIT(LRU_REFS_WIDTH) /* * For folios accessed multiple times through file descriptors, @@ -635,35 +619,32 @@ struct lru_gen_mm_walk { * For each node, memcgs are divided into two generations: the old and the * young. For each generation, memcgs are randomly sharded into multiple bins * to improve scalability. For each bin, the hlist_nulls is virtually divided - * into three segments: the head, the tail and the default. + * into two segments: the tail and the default. * * An onlining memcg is added to the tail of a random bin in the old generation. * The eviction starts at the head of a random bin in the old generation. The * per-node memcg generation counter, whose reminder (mod MEMCG_NR_GENS) indexes * the old generation, is incremented when all its bins become empty. * - * There are four operations: - * 1. MEMCG_LRU_HEAD, which moves a memcg to the head of a random bin in its - * current generation (old or young) and updates its "seg" to "head"; - * 2. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its + * There are three operations: + * 1. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its * current generation (old or young) and updates its "seg" to "tail"; - * 3. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old + * 2. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old * generation, updates its "gen" to "old" and resets its "seg" to "default"; - * 4. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the + * 3. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the * young generation, updates its "gen" to "young" and resets its "seg" to * "default". * * The events that trigger the above operations are: - * 1. Exceeding the soft limit, which triggers MEMCG_LRU_HEAD; - * 2. The first attempt to reclaim a memcg below low, which triggers + * 1. The first attempt to reclaim a memcg below low, which triggers * MEMCG_LRU_TAIL; - * 3. The first attempt to reclaim a memcg offlined or below reclaimable size + * 2. The first attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_TAIL; - * 4. The second attempt to reclaim a memcg offlined or below reclaimable size + * 3. The second attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_YOUNG; - * 5. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; - * 6. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; - * 7. Offlining a memcg, which triggers MEMCG_LRU_OLD. + * 4. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; + * 5. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; + * 6. Offlining a memcg, which triggers MEMCG_LRU_OLD. * * Notes: * 1. Memcg LRU only applies to global reclaim, and the round-robin incrementing @@ -696,7 +677,6 @@ void lru_gen_exit_memcg(struct mem_cgroup *memcg); void lru_gen_online_memcg(struct mem_cgroup *memcg); void lru_gen_offline_memcg(struct mem_cgroup *memcg); void lru_gen_release_memcg(struct mem_cgroup *memcg); -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid); void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid); bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid); void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid); @@ -737,10 +717,6 @@ static inline void lru_gen_release_memcg(struct mem_cgroup *memcg) { } -static inline void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ -} - static inline void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid) { } @@ -1043,6 +1019,42 @@ struct zone { * cma pages is present pages that are assigned for CMA use * (MIGRATE_CMA). * + * pages_with_online_memmap tracks pages within the zone that have + * an online memory map: present pages and memory holes whose + * memory map has been initialized and pfn_to_online_page() + * succeeds. When spanned_pages == pages_with_online_memmap, + * pfn_to_page() can be performed without further checks on any + * PFN within the zone span. + * + * Note: this counter may temporarily undercount when pages with an + * online memory map exist outside the current zone span. Such pages + * are only created during boot, when initializing the memory map of + * pages that do not fall into any zone span. The undercount itself + * can only happen after boot, during memory hotplug, when growing + * the zone to cover such pages and later shrinking it back, which + * may result in a "too small" value. This is safe: it merely + * prevents detecting a contiguous zone. + * + * Here is an example (page numbers are just for illustration + * purposes): + * after boot: + * [ zone span ] + * [ zone pages ] + * spanned=10, initialized=15, online=10 + * online == spanned -> contiguous + * + * growing after hotplug (hotplug 5): + * [ zone span ] + * [ zone pages ] [ zone pages ] + * spanned=30, initialized=20, online=15 + * online != spanned -> not contiguous + * + * shrinking after hotunplug (hotunplug 5 again): + * [ zone span ] + * [ zone pages ] + * spanned=15, initialized=15, online=10 + * online != spanned -> not contiguous although contiguous + * * So present_pages may be used by memory hotplug or memory power * management logic to figure out unmanaged pages by checking * (present_pages - managed_pages). And managed_pages should be used @@ -1067,6 +1079,7 @@ struct zone { atomic_long_t managed_pages; unsigned long spanned_pages; unsigned long present_pages; + unsigned long pages_with_online_memmap; #if defined(CONFIG_MEMORY_HOTPLUG) unsigned long present_early_pages; #endif @@ -1101,9 +1114,6 @@ struct zone { #ifdef CONFIG_UNACCEPTED_MEMORY /* Pages to be accepted. All pages on the list are MAX_PAGE_ORDER */ struct list_head unaccepted_pages; - - /* To be called once the last page in the zone is accepted */ - struct work_struct unaccepted_cleanup; #endif /* zone flags, see below */ @@ -1157,8 +1167,8 @@ struct zone { /* Zone statistics */ atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct page *vmemmap_tails[NR_VMEMMAP_TAILS]; +#ifdef CONFIG_VMEMMAP_OPTIMIZATION + struct page **vmemmap_tails; #endif } ____cacheline_internodealigned_in_smp; @@ -1662,7 +1672,7 @@ extern void init_currently_empty_zone(struct zone *zone, unsigned long start_pfn extern void lruvec_init(struct lruvec *lruvec); -static inline struct pglist_data *lruvec_pgdat(struct lruvec *lruvec) +static inline struct pglist_data *lruvec_pgdat(const struct lruvec *lruvec) { #ifdef CONFIG_MEMCG return lruvec->pgdat; @@ -1694,6 +1704,38 @@ static inline bool zone_is_zone_device(const struct zone *zone) } #endif +/** + * zone_is_contiguous - test whether a zone is contiguous + * @zone: the zone to test. + * + * In a contiguous zone, it is valid to call pfn_to_page() on any PFN in the + * spanned zone without requiring pfn_valid() or pfn_to_online_page() checks. + * + * Note that missing synchronization with memory offlining makes any PFN + * traversal prone to races. + * + * ZONE_DEVICE zones are always marked non-contiguous. + * + * Return: true if contiguous, otherwise false. + */ +static inline bool zone_is_contiguous(const struct zone *zone) +{ + return READ_ONCE(zone->contiguous); +} + +static inline void set_zone_contiguous(struct zone *zone) +{ + if (zone_is_zone_device(zone)) + return; + if (zone->spanned_pages == zone->pages_with_online_memmap) + WRITE_ONCE(zone->contiguous, true); +} + +static inline void clear_zone_contiguous(struct zone *zone) +{ + WRITE_ONCE(zone->contiguous, false); +} + /* * Returns true if a zone has pages managed by the buddy allocator. * All the reclaim decisions have to use this function rather than @@ -2021,19 +2063,23 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; +#ifdef CONFIG_VMEMMAP_OPTIMIZATION + /* + * Normally, sections hold regular (order-0) pages. However, for + * sections with HVO enabled, this tracks the compound page order + * to enable deduplication of redundant vmemmap pages. + */ + unsigned int compound_page_order; +#endif #ifdef CONFIG_PAGE_EXTENSION /* * If SPARSEMEM, pgdat doesn't have page_ext pointer. We use * section. (see page_ext.h about this.) */ struct page_ext *page_ext; - unsigned long pad; #endif - /* - * WARNING: mem_section must be a power-of-2 in size for the - * calculation and use of SECTION_ROOT_MASK to make sense. - */ -}; +/* Sacrifice minor padding space for efficient lookup. */ +} __aligned(2 * sizeof(unsigned long)); #ifdef CONFIG_SPARSEMEM_EXTREME #define SECTIONS_PER_ROOT (PAGE_SIZE / sizeof (struct mem_section)) @@ -2043,7 +2089,6 @@ struct mem_section { #define SECTION_NR_TO_ROOT(sec) ((sec) / SECTIONS_PER_ROOT) #define NR_SECTION_ROOTS DIV_ROUND_UP(NR_MEM_SECTIONS, SECTIONS_PER_ROOT) -#define SECTION_ROOT_MASK (SECTIONS_PER_ROOT - 1) #ifdef CONFIG_SPARSEMEM_EXTREME extern struct mem_section **mem_section; @@ -2067,7 +2112,7 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) if (!mem_section || !mem_section[root]) return NULL; #endif - return &mem_section[root][nr & SECTION_ROOT_MASK]; + return &mem_section[root][nr % SECTIONS_PER_ROOT]; } /* @@ -2083,29 +2128,21 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) * accommodate SECTION_MAP_LAST_BIT. We use BUILD_BUG_ON() to ensure this. */ enum { - SECTION_MARKED_PRESENT_BIT, SECTION_HAS_MEM_MAP_BIT, SECTION_IS_ONLINE_BIT, SECTION_IS_EARLY_BIT, #ifdef CONFIG_ZONE_DEVICE SECTION_TAINT_ZONE_DEVICE_BIT, #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - SECTION_IS_VMEMMAP_PREINIT_BIT, -#endif SECTION_MAP_LAST_BIT, }; -#define SECTION_MARKED_PRESENT BIT(SECTION_MARKED_PRESENT_BIT) #define SECTION_HAS_MEM_MAP BIT(SECTION_HAS_MEM_MAP_BIT) #define SECTION_IS_ONLINE BIT(SECTION_IS_ONLINE_BIT) #define SECTION_IS_EARLY BIT(SECTION_IS_EARLY_BIT) #ifdef CONFIG_ZONE_DEVICE #define SECTION_TAINT_ZONE_DEVICE BIT(SECTION_TAINT_ZONE_DEVICE_BIT) #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -#define SECTION_IS_VMEMMAP_PREINIT BIT(SECTION_IS_VMEMMAP_PREINIT_BIT) -#endif #define SECTION_MAP_MASK (~(BIT(SECTION_MAP_LAST_BIT) - 1)) #define SECTION_NID_SHIFT SECTION_MAP_LAST_BIT @@ -2116,16 +2153,6 @@ static inline struct page *__section_mem_map_addr(struct mem_section *section) return (struct page *)map; } -static inline int present_section(const struct mem_section *section) -{ - return (section && (section->section_mem_map & SECTION_MARKED_PRESENT)); -} - -static inline int present_section_nr(unsigned long nr) -{ - return present_section(__nr_to_section(nr)); -} - static inline int valid_section(const struct mem_section *section) { return (section && (section->section_mem_map & SECTION_HAS_MEM_MAP)); @@ -2141,6 +2168,11 @@ static inline int valid_section_nr(unsigned long nr) return valid_section(__nr_to_section(nr)); } +static inline int early_section_nr(unsigned long nr) +{ + return early_section(__nr_to_section(nr)); +} + static inline int online_section(const struct mem_section *section) { return (section && (section->section_mem_map & SECTION_IS_ONLINE)); @@ -2153,28 +2185,20 @@ static inline int online_device_section(const struct mem_section *section) return section && ((section->section_mem_map & flags) == flags); } -#else -static inline int online_device_section(const struct mem_section *section) -{ - return 0; -} -#endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -static inline int preinited_vmemmap_section(const struct mem_section *section) +static inline struct zone *device_zone(int nid) { - return (section && - (section->section_mem_map & SECTION_IS_VMEMMAP_PREINIT)); + return &NODE_DATA(nid)->node_zones[ZONE_DEVICE]; } - -void sparse_vmemmap_init_nid_early(int nid); #else -static inline int preinited_vmemmap_section(const struct mem_section *section) +static inline int online_device_section(const struct mem_section *section) { return 0; } -static inline void sparse_vmemmap_init_nid_early(int nid) + +static inline struct zone *device_zone(int nid) { + return NULL; } #endif @@ -2193,7 +2217,7 @@ static inline struct mem_section *__pfn_to_section(unsigned long pfn) return __nr_to_section(pfn_to_section_nr(pfn)); } -extern unsigned long __highest_present_section_nr; +extern unsigned long __highest_used_section_nr; static inline int subsection_map_index(unsigned long pfn) { @@ -2241,9 +2265,6 @@ static inline bool pfn_section_first_valid(struct mem_section *ms, unsigned long } #endif -void sparse_init_early_section(int nid, struct page *map, unsigned long pnum, - unsigned long flags); - #ifndef CONFIG_HAVE_ARCH_PFN_VALID /** * pfn_valid - check if there is a valid memory map entry for a PFN @@ -2295,7 +2316,7 @@ static inline unsigned long first_valid_pfn(unsigned long pfn, unsigned long end rcu_read_lock_sched(); - while (nr <= __highest_present_section_nr && pfn < end_pfn) { + while (nr <= __highest_used_section_nr && pfn < end_pfn) { struct mem_section *ms = __pfn_to_section(pfn); if (valid_section(ms) && @@ -2341,27 +2362,20 @@ static inline unsigned long next_valid_pfn(unsigned long pfn, unsigned long end_ #endif -static inline int pfn_in_present_section(unsigned long pfn) -{ - if (pfn_to_section_nr(pfn) >= NR_MEM_SECTIONS) - return 0; - return present_section(__pfn_to_section(pfn)); -} - -static inline unsigned long next_present_section_nr(unsigned long section_nr) +static inline unsigned long next_early_section_nr(unsigned long section_nr) { - while (++section_nr <= __highest_present_section_nr) { - if (present_section_nr(section_nr)) + while (++section_nr <= __highest_used_section_nr) { + if (early_section_nr(section_nr)) return section_nr; } return -1; } -#define for_each_present_section_nr(start, section_nr) \ - for (section_nr = next_present_section_nr(start - 1); \ +#define for_each_early_section_nr(start, section_nr) \ + for (section_nr = next_early_section_nr(start - 1); \ section_nr != -1; \ - section_nr = next_present_section_nr(section_nr)) + section_nr = next_early_section_nr(section_nr)) /* * These are _only_ used during initialisation, therefore they @@ -2377,10 +2391,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr) #else #define pfn_to_nid(pfn) (0) #endif - -#else -#define sparse_vmemmap_init_nid_early(_nid) do {} while (0) -#define pfn_in_present_section pfn_valid #endif /* CONFIG_SPARSEMEM */ /* diff --git a/include/linux/mnt_idmapping.h b/include/linux/mnt_idmapping.h index e71a6070a8f8..78eeef4c2996 100644 --- a/include/linux/mnt_idmapping.h +++ b/include/linux/mnt_idmapping.h @@ -8,8 +8,8 @@ struct mnt_idmap; struct user_namespace; -extern struct mnt_idmap nop_mnt_idmap; -extern struct mnt_idmap invalid_mnt_idmap; +extern const struct mnt_idmap nop_mnt_idmap; +extern const struct mnt_idmap invalid_mnt_idmap; extern struct user_namespace init_user_ns; typedef struct { @@ -121,19 +121,19 @@ static inline bool vfsgid_eq_kgid(vfsgid_t vfsgid, kgid_t kgid) int vfsgid_in_group_p(vfsgid_t vfsgid); -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap); -void mnt_idmap_put(struct mnt_idmap *idmap); +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap); +void mnt_idmap_put(const struct mnt_idmap *idmap); -vfsuid_t make_vfsuid(struct mnt_idmap *idmap, +vfsuid_t make_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kuid_t kuid); -vfsgid_t make_vfsgid(struct mnt_idmap *idmap, +vfsgid_t make_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kgid_t kgid); -kuid_t from_vfsuid(struct mnt_idmap *idmap, +kuid_t from_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid); -kgid_t from_vfsgid(struct mnt_idmap *idmap, +kgid_t from_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid); /** @@ -148,7 +148,7 @@ kgid_t from_vfsgid(struct mnt_idmap *idmap, * * Return: true if @vfsuid has a mapping in the filesystem, false if not. */ -static inline bool vfsuid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsuid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid) { @@ -186,7 +186,7 @@ static inline kuid_t vfsuid_into_kuid(vfsuid_t vfsuid) * * Return: true if @vfsgid has a mapping in the filesystem, false if not. */ -static inline bool vfsgid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsgid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid) { @@ -225,7 +225,7 @@ static inline kgid_t vfsgid_into_kgid(vfsgid_t vfsgid) * * Return: the caller's current fsuid mapped up according to @idmap. */ -static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, +static inline kuid_t mapped_fsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsuid(idmap, fs_userns, VFSUIDT_INIT(current_fsuid())); @@ -244,7 +244,7 @@ static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, * * Return: the caller's current fsgid mapped up according to @idmap. */ -static inline kgid_t mapped_fsgid(struct mnt_idmap *idmap, +static inline kgid_t mapped_fsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsgid(idmap, fs_userns, VFSGIDT_INIT(current_fsgid())); diff --git a/include/linux/mod_devicetable.h b/include/linux/mod_devicetable.h index a397213bedac..d1581bb64729 100644 --- a/include/linux/mod_devicetable.h +++ b/include/linux/mod_devicetable.h @@ -15,7 +15,7 @@ #include "device-id/acpi.h" #include "device-id/amba.h" #include "device-id/ap.h" -#include "device-id/apr.h" +#include "device-id/arm_smccc.h" #include "device-id/auxiliary.h" #include "device-id/bcma.h" #include "device-id/ccw.h" diff --git a/include/linux/module_symbol.h b/include/linux/module_symbol.h index 574609aced99..698d3db2b37e 100644 --- a/include/linux/module_symbol.h +++ b/include/linux/module_symbol.h @@ -7,8 +7,11 @@ enum ksym_flags { KSYM_FLAG_GPL_ONLY = 1 << 0, }; -/* This ignores the intensely annoying "mapping symbols" found in ELF files. */ -static inline bool is_mapping_symbol(const char *str) +/* + * Ignore local labels (.L*, L0*) and mapping symbols ($*). These symbols are + * not useful for the kernel, for example, they should not appear in kallsyms. + */ +static inline bool is_ignored_kernel_symbol(const char *str) { if (str[0] == '.' && str[1] == 'L') return true; diff --git a/include/linux/mount.h b/include/linux/mount.h index acfe7ef86a1b..e90ccafef281 100644 --- a/include/linux/mount.h +++ b/include/linux/mount.h @@ -59,10 +59,10 @@ struct vfsmount { struct dentry *mnt_root; /* root of the mounted tree */ struct super_block *mnt_sb; /* pointer to superblock */ int mnt_flags; - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; } __randomize_layout; -static inline struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) +static inline const struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) { /* Pairs with smp_store_release() in do_idmap_mount(). */ return READ_ONCE(mnt->mnt_idmap); diff --git a/include/linux/mtd/bbm.h b/include/linux/mtd/bbm.h index d890805f5494..67929fdf2dff 100644 --- a/include/linux/mtd/bbm.h +++ b/include/linux/mtd/bbm.h @@ -75,7 +75,7 @@ struct nand_bbt_descr { * with NAND_BBT_CREATE. */ #define NAND_BBT_CREATE_EMPTY 0x00000400 -/* Write bbt if neccecary */ +/* Write bbt if necessary */ #define NAND_BBT_WRITE 0x00002000 /* Read and write back block contents when writing bbt */ #define NAND_BBT_SAVECONTENT 0x00004000 diff --git a/include/linux/mtd/lpc32xx_mlc.h b/include/linux/mtd/lpc32xx_mlc.h deleted file mode 100644 index 35e971be0950..000000000000 --- a/include/linux/mtd/lpc32xx_mlc.h +++ /dev/null @@ -1,17 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-only */ -/* - * Platform data for LPC32xx SoC MLC NAND controller - * - * Copyright © 2012 Roland Stigge - */ - -#ifndef __LINUX_MTD_LPC32XX_MLC_H -#define __LINUX_MTD_LPC32XX_MLC_H - -#include <linux/dmaengine.h> - -struct lpc32xx_mlc_platform_data { - dma_filter_fn dma_filter; -}; - -#endif /* __LINUX_MTD_LPC32XX_MLC_H */ diff --git a/include/linux/mtd/lpc32xx_slc.h b/include/linux/mtd/lpc32xx_slc.h deleted file mode 100644 index a044b806566b..000000000000 --- a/include/linux/mtd/lpc32xx_slc.h +++ /dev/null @@ -1,17 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-only */ -/* - * Platform data for LPC32xx SoC SLC NAND controller - * - * Copyright © 2012 Roland Stigge - */ - -#ifndef __LINUX_MTD_LPC32XX_SLC_H -#define __LINUX_MTD_LPC32XX_SLC_H - -#include <linux/dmaengine.h> - -struct lpc32xx_slc_platform_data { - dma_filter_fn dma_filter; -}; - -#endif /* __LINUX_MTD_LPC32XX_SLC_H */ diff --git a/include/linux/mtd/nand-qpic-common.h b/include/linux/mtd/nand-qpic-common.h index 437448995187..0f40964eefe7 100644 --- a/include/linux/mtd/nand-qpic-common.h +++ b/include/linux/mtd/nand-qpic-common.h @@ -262,7 +262,6 @@ struct bam_transaction { u32 tx_sgl_start; u32 rx_sgl_pos; u32 rx_sgl_start; - ); }; diff --git a/include/linux/mtd/nand.h b/include/linux/mtd/nand.h index 09c8c93e4dba..6936180b6ea5 100644 --- a/include/linux/mtd/nand.h +++ b/include/linux/mtd/nand.h @@ -305,6 +305,8 @@ int nand_ecc_prepare_io_req(struct nand_device *nand, struct nand_page_io_req *req); int nand_ecc_finish_io_req(struct nand_device *nand, struct nand_page_io_req *req); +bool nand_ecc_is_pipelined(const struct nand_device *nand); + bool nand_ecc_is_strong_enough(struct nand_device *nand); #if IS_REACHABLE(CONFIG_MTD_NAND_CORE) diff --git a/include/linux/mtd/rawnand.h b/include/linux/mtd/rawnand.h index 5c70e7bd3ed5..43fefff02025 100644 --- a/include/linux/mtd/rawnand.h +++ b/include/linux/mtd/rawnand.h @@ -1592,7 +1592,7 @@ void nand_wait_ready(struct nand_chip *chip); /* * Free resources held by the NAND device, must be called on error after a - * sucessful nand_scan(). + * successful nand_scan(). */ void nand_cleanup(struct nand_chip *chip); diff --git a/include/linux/mtd/sh_flctl.h b/include/linux/mtd/sh_flctl.h index 78fc2d4218c8..15adf2474fa0 100644 --- a/include/linux/mtd/sh_flctl.h +++ b/include/linux/mtd/sh_flctl.h @@ -60,8 +60,8 @@ * Some hardware uses bits called PULSEx instead of FCKSEL_E and QTSEL_E * to control the clock divider used between the High-Speed Peripheral Clock * and the FLCTL internal clock. If so, use CLK_8_BIT_xxx for connecting 8 bit - * and CLK_16_BIT_xxx for connecting 16 bit bus bandwith NAND chips. For the 16 - * bit version the divider is seperate for the pulse width of high and low + * and CLK_16_BIT_xxx for connecting 16 bit bus bandwidth NAND chips. For the 16 + * bit version the divider is separate for the pulse width of high and low * signals. */ #define PULSE3 (0x1 << 27) diff --git a/include/linux/mtd/spi-nor.h b/include/linux/mtd/spi-nor.h index 4b92494827b1..b3e3c6b10186 100644 --- a/include/linux/mtd/spi-nor.h +++ b/include/linux/mtd/spi-nor.h @@ -25,6 +25,7 @@ #define SPINOR_OP_WRSR 0x01 /* Write status register 1 */ #define SPINOR_OP_RDSR2 0x3f /* Read status register 2 */ #define SPINOR_OP_WRSR2 0x3e /* Write status register 2 */ +#define SPINOR_OP_WRSR2_ALT 0x31 /* Write status register 2 (alternative) */ #define SPINOR_OP_READ 0x03 /* Read data bytes (low frequency) */ #define SPINOR_OP_READ_FAST 0x0b /* Read data bytes (high frequency) */ #define SPINOR_OP_READ_1_1_2 0x3b /* Read data bytes (Dual Output SPI) */ @@ -113,20 +114,16 @@ #define SR_E_ERR BIT(5) #define SR_P_ERR BIT(6) -#define SR1_QUAD_EN_BIT6 BIT(6) - #define SR_BP_SHIFT 2 /* Enhanced Volatile Configuration Register bits */ #define EVCR_QUAD_EN_MICRON BIT(7) /* Micron Quad I/O */ /* Status Register 2 bits. */ -#define SR2_QUAD_EN_BIT1 BIT(1) #define SR2_LB1 BIT(3) /* Security Register Lock Bit 1 */ #define SR2_LB2 BIT(4) /* Security Register Lock Bit 2 */ #define SR2_LB3 BIT(5) /* Security Register Lock Bit 3 */ #define SR2_CMP_BIT6 BIT(6) -#define SR2_QUAD_EN_BIT7 BIT(7) /* Supported SPI protocols */ #define SNOR_PROTO_INST_MASK GENMASK(23, 16) @@ -365,8 +362,6 @@ struct spi_nor_flash_parameter; * @read_dummy: the dummy needed by the read operation * @program_opcode: the program opcode * @sst_write_second: used by the SST write operation - * @flags: flag options for the current SPI NOR (SNOR_F_*) - * @cmd_ext_type: the command opcode extension type for DTR mode. * @read_proto: the SPI protocol for read operations * @write_proto: the SPI protocol for write operations * @reg_proto: the SPI protocol for read_reg/write_reg/erase operations @@ -398,6 +393,7 @@ struct spi_nor { u8 *id; const struct flash_info *info; const struct spi_nor_manufacturer *manufacturer; + const char *partname; u8 addr_nbytes; u8 erase_opcode; u8 read_opcode; @@ -407,8 +403,6 @@ struct spi_nor { enum spi_nor_protocol write_proto; enum spi_nor_protocol reg_proto; bool sst_write_second; - u32 flags; - enum spi_nor_cmd_ext cmd_ext_type; struct sfdp *sfdp; struct dentry *debugfs_root; u8 dfs_sr_cache[2]; diff --git a/include/linux/mtd/spinand.h b/include/linux/mtd/spinand.h index 5f4c00ae72a7..d7c99fcfb85e 100644 --- a/include/linux/mtd/spinand.h +++ b/include/linux/mtd/spinand.h @@ -399,7 +399,7 @@ struct spinand_devid { }; /** - * struct manufacurer_ops - SPI NAND manufacturer specific operations + * struct spinand_manufacturer_ops - SPI NAND manufacturer specific operations * @init: initialize a SPI NAND device * @cleanup: cleanup a SPI NAND device * @@ -438,6 +438,7 @@ extern const struct spinand_manufacturer fmsh_spinand_manufacturer; extern const struct spinand_manufacturer foresee_spinand_manufacturer; extern const struct spinand_manufacturer gigadevice_spinand_manufacturer; extern const struct spinand_manufacturer heyangtek_spinand_manufacturer; +extern const struct spinand_manufacturer issi_spinand_manufacturer; extern const struct spinand_manufacturer macronix_spinand_manufacturer; extern const struct spinand_manufacturer micron_spinand_manufacturer; extern const struct spinand_manufacturer paragon_spinand_manufacturer; diff --git a/include/linux/namei.h b/include/linux/namei.h index 86d657b24fc6..c4436e5c2ba6 100644 --- a/include/linux/namei.h +++ b/include/linux/namei.h @@ -32,8 +32,9 @@ enum { MAX_NESTED_LINKS = 8 }; #define LOOKUP_CREATE BIT(17) /* ... in object creation */ #define LOOKUP_EXCL BIT(18) /* ... in target must not exist */ #define LOOKUP_RENAME_TARGET BIT(19) /* ... in destination of rename() */ +#define LOOKUP_SHARED BIT(20) /* Parent lock is held shared */ -/* 4 spare bits for intent */ +/* 3 spare bits for intent */ /* Scoping flags for lookup. */ #define LOOKUP_NO_SYMLINKS BIT(24) /* No symlink crossing. */ @@ -70,24 +71,24 @@ extern struct dentry *try_lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_unlocked(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_positive_unlocked(struct qstr *, struct dentry *); -struct dentry *lookup_one(struct mnt_idmap *, struct qstr *, struct dentry *); -struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one(const struct mnt_idmap *, struct qstr *, struct dentry *); +struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_creating_killable(struct mnt_idmap *idmap, +struct dentry *start_creating_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing_killable(struct mnt_idmap *idmap, +struct dentry *start_removing_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); struct dentry *start_creating_noperm(struct dentry *parent, struct qstr *name); diff --git a/include/linux/net.h b/include/linux/net.h index 3d82966e2243..e2a866fbcfa5 100644 --- a/include/linux/net.h +++ b/include/linux/net.h @@ -47,6 +47,9 @@ typedef struct sockopt { int optlen; } sockopt_t; +int sockptr_to_sockopt(sockopt_t *opt, sockptr_t optval, sockptr_t optlen, + struct kvec *kvec); + /* * Initialize a user-backed sockopt_t from the (optval, optlen) __user pair of * a getsockopt() callback. Used by transitional __user getsockopt wrappers @@ -70,6 +73,35 @@ static inline int sockopt_init_user(sockopt_t *opt, char __user *optval, return 0; } +/* + * Grow optval to @size, for the options whose reply is sized by a count the + * caller left in optval rather than by optlen. Those write past optlen today + * and userspace relies on it. + * + * Call it before writing through opt->iter_out: it re-anchors the iterator at + * the head of optval. Only a user buffer can be longer than the optlen the + * caller declared, so a kernel-backed optval is refused with -EINVAL. + */ +static inline int sockopt_expand_out(sockopt_t *opt, size_t size) +{ + if (size <= (size_t)opt->optlen) + return 0; + + if (size > INT_MAX) + return -EINVAL; + + /* Re-anchoring reads iter_out.ubuf, so the iterator has to be a user + * buffer that nothing has written through yet. + */ + if (WARN_ON_ONCE(!iter_is_ubuf(&opt->iter_out) || + iov_iter_count(&opt->iter_out) != (size_t)opt->optlen)) + return -EINVAL; + + iov_iter_ubuf(&opt->iter_out, ITER_DEST, opt->iter_out.ubuf, size); + + return 0; +} + struct poll_table_struct; struct pipe_inode_info; struct inode; @@ -166,7 +198,7 @@ struct socket { struct file *file; struct sock *sk; - const struct proto_ops *ops; /* Might change with IPV6_ADDRFORM or MPTCP. */ + const struct proto_ops *ops; /* Might change with MPTCP. */ struct socket_wq wq; }; diff --git a/include/linux/netdevice.h b/include/linux/netdevice.h index 87cafc932e9e..f61c97231309 100644 --- a/include/linux/netdevice.h +++ b/include/linux/netdevice.h @@ -887,6 +887,7 @@ enum net_device_path_type { DEV_PATH_DSA, DEV_PATH_MTK_WDMA, DEV_PATH_TUN, + DEV_PATH_IEEE80211, }; struct net_device_path { @@ -953,6 +954,8 @@ struct net_device_path_ctx { u16 id; __be16 proto; } vlan[NET_DEVICE_PATH_VLAN_MAX]; + + bool ieee80211; }; enum tc_setup_type { @@ -1135,13 +1138,16 @@ struct netdev_net_notifier { * struct netdev_hw_addr_list *uc, * struct netdev_hw_addr_list *mc); * Async version of ndo_set_rx_mode which runs in process context - * with rtnl_lock and netdev_lock_ops(dev) held. The uc/mc parameters + * under the netdev instance lock for "ops locked" drivers, or + * rtnl_lock for all other drivers. The uc/mc parameters * are snapshots of the address lists - iterate with * netdev_hw_addr_list_for_each(ha, uc). Return 0 on success or a * negative errno to request a retry via the core backoff. * * void (*ndo_work)(struct net_device *dev, unsigned long events); * Run deferred work scheduled with netdev_work_sched(@events). + * Runs in process context under the netdev instance lock for "ops + * locked" drivers, or rtnl_lock for all other drivers. * * int (*ndo_set_mac_address)(struct net_device *dev, void *addr); * This function is called when the Media Access Control address @@ -1151,11 +1157,6 @@ struct netdev_net_notifier { * int (*ndo_validate_addr)(struct net_device *dev); * Test if Media Access Control address is valid for the device. * - * int (*ndo_do_ioctl)(struct net_device *dev, struct ifreq *ifr, int cmd); - * Old-style ioctl entry point. This is used internally by the - * ieee802154 subsystem but is no longer called by the device - * ioctl handler. - * * int (*ndo_siocbond)(struct net_device *dev, struct ifreq *ifr, int cmd); * Used by the bonding driver for its device specific ioctls: * SIOCBONDENSLAVE, SIOCBONDRELEASE, SIOCBONDSETHWADDR, SIOCBONDCHANGEACTIVE, @@ -1477,8 +1478,6 @@ struct net_device_ops { int (*ndo_set_mac_address)(struct net_device *dev, void *addr); int (*ndo_validate_addr)(struct net_device *dev); - int (*ndo_do_ioctl)(struct net_device *dev, - struct ifreq *ifr, int cmd); int (*ndo_eth_ioctl)(struct net_device *dev, struct ifreq *ifr, int cmd); int (*ndo_siocbond)(struct net_device *dev, @@ -1837,6 +1836,7 @@ enum netdev_reg_state { * drivers. Mainly used by logical interfaces, such as * bonding and tunnels * @netmem_tx: device netmem TX mode + * @pacing_offload: enable EDT pacing offload. * * @name: This is the first field of the "visible" part of this structure * (i.e. as seen by users in the "Space.c" file). It is the name @@ -2167,6 +2167,7 @@ struct net_device { unsigned long priv_flags:32; unsigned long lltx:1; unsigned long netmem_tx:2; + unsigned long pacing_offload:1; ); const struct net_device_ops *netdev_ops; const struct header_ops *header_ops; @@ -3669,7 +3670,11 @@ struct page_pool_bh { }; DECLARE_PER_CPU(struct page_pool_bh, system_page_pool); +#ifdef CONFIG_KASAN +#define XMIT_RECURSION_LIMIT 4 +#else #define XMIT_RECURSION_LIMIT 8 +#endif #ifndef CONFIG_PREEMPT_RT static inline int dev_recursion_level(void) @@ -5321,7 +5326,6 @@ void netdev_lower_state_changed(struct net_device *lower_dev, void *lower_state_info); #define NETDEV_RSS_KEY_LEN 256 -extern u8 netdev_rss_key[NETDEV_RSS_KEY_LEN] __read_mostly; void netdev_rss_key_fill(void *buffer, size_t len); int skb_checksum_help(struct sk_buff *skb); @@ -5606,12 +5610,12 @@ static inline bool netif_has_l3_rx_handler(const struct net_device *dev) static inline bool netif_is_l3_master(const struct net_device *dev) { - return dev->priv_flags & IFF_L3MDEV_MASTER; + return IS_ENABLED(CONFIG_NET_VRF) && (dev->priv_flags & IFF_L3MDEV_MASTER); } static inline bool netif_is_l3_slave(const struct net_device *dev) { - return dev->priv_flags & IFF_L3MDEV_SLAVE; + return IS_ENABLED(CONFIG_NET_VRF) && (dev->priv_flags & IFF_L3MDEV_SLAVE); } static inline int dev_sdif(const struct net_device *dev) diff --git a/include/linux/netfilter_arp/arp_tables.h b/include/linux/netfilter_arp/arp_tables.h index 05631a25e622..8b8d472eff34 100644 --- a/include/linux/netfilter_arp/arp_tables.h +++ b/include/linux/netfilter_arp/arp_tables.h @@ -56,23 +56,4 @@ void arpt_unregister_table(struct net *net, const char *name); extern unsigned int arpt_do_table(void *priv, struct sk_buff *skb, const struct nf_hook_state *state); -#ifdef CONFIG_NETFILTER_XTABLES_COMPAT -#include <net/compat.h> - -struct compat_arpt_entry { - struct arpt_arp arp; - __u16 target_offset; - __u16 next_offset; - compat_uint_t comefrom; - struct compat_xt_counters counters; - unsigned char elems[]; -}; - -static inline struct xt_entry_target * -compat_arpt_get_target(struct compat_arpt_entry *e) -{ - return (void *)e + e->target_offset; -} - -#endif /* CONFIG_COMPAT */ #endif /* _ARPTABLES_H */ diff --git a/include/linux/netfilter_netdev.h b/include/linux/netfilter_netdev.h index 3175073a66ba..2a854a0bbd4b 100644 --- a/include/linux/netfilter_netdev.h +++ b/include/linux/netfilter_netdev.h @@ -133,7 +133,7 @@ static inline struct sk_buff *nf_hook_egress(struct sk_buff *skb, int *rc, static inline void nf_skip_egress(struct sk_buff *skb, bool skip) { -#ifdef CONFIG_NETFILTER_SKIP_EGRESS +#ifdef CONFIG_NET_EGRESS skb->nf_skip_egress = skip; #endif } diff --git a/include/linux/netfs.h b/include/linux/netfs.h index f837a501008c..e5961450f3f1 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -22,6 +22,7 @@ enum netfs_sreq_ref_trace; typedef struct mempool mempool_t; +struct fscache_occupancy; struct folio_queue; /** @@ -62,8 +63,8 @@ struct netfs_inode { struct fscache_cookie *cache; #endif struct list_head wb_queue; /* Queue of processes wanting to do writeback */ - loff_t _remote_i_size; /* Size of the remote file */ - loff_t _zero_point; /* Size after which we assume there's no data + uoff_t _remote_i_size; /* Size of the remote file */ + uoff_t _zero_point; /* Size after which we assume there's no data * on the server */ spinlock_t lock; /* Lock covering wb_queue */ atomic_t io_count; /* Number of outstanding reqs */ @@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio) return priv; } +enum netfs_cache_collect { + NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */ + NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */ + NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */ +}; + /* * Stream of I/O subrequests going to a particular destination, such as the * server or the local cache. This is mainly intended for writing where we may @@ -142,7 +149,7 @@ struct netfs_io_stream { void (*issue_write)(struct netfs_io_subrequest *subreq); /* Collection tracking */ struct list_head subrequests; /* Contributory I/O operations */ - unsigned long long collected_to; /* Position we've collected results to */ + uoff_t collected_to; /* Position we've collected results to */ size_t transferred; /* The amount transferred from this stream */ unsigned short error; /* Aggregate error for the stream */ enum netfs_io_source source; /* Where to read from/write to */ @@ -152,6 +159,7 @@ struct netfs_io_stream { bool need_retry; /* T if this stream needs retrying */ bool failed; /* T if this stream failed */ bool transferred_valid; /* T is ->transferred is valid */ + enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */ }; /* @@ -161,8 +169,11 @@ struct netfs_cache_resources { const struct netfs_cache_ops *ops; void *cache_priv; void *cache_priv2; - unsigned int debug_id; /* Cookie debug ID */ + uoff_t cache_i_size; /* Initial size of cache file */ + unsigned int cookie_id; /* Cache cookie debug ID */ + unsigned int object_id; /* Cache object debug ID */ unsigned int inval_counter; /* object->inval_counter at begin_op */ + unsigned int dio_size; /* DIO block size */ }; /* @@ -177,7 +188,7 @@ struct netfs_io_subrequest { struct work_struct work; struct list_head rreq_link; /* Link in rreq->subrequests */ struct iov_iter io_iter; /* Iterator for this subrequest */ - unsigned long long start; /* Where to start the I/O */ + uoff_t start; /* Where to start the I/O */ size_t len; /* Size of the I/O */ size_t transferred; /* Amount of data transferred */ refcount_t ref; @@ -196,6 +207,7 @@ struct netfs_io_subrequest { #define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */ #define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */ #define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */ +#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */ }; enum netfs_io_origin { @@ -208,7 +220,6 @@ enum netfs_io_origin { NETFS_DIO_READ, /* This is a direct I/O read */ NETFS_WRITEBACK, /* This write was triggered by writepages */ NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */ - NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */ NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */ NETFS_DIO_WRITE, /* This is a direct I/O write */ NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */ @@ -243,16 +254,18 @@ struct netfs_io_request { void *netfs_priv; /* Private data for the netfs */ void *netfs_priv2; /* Private data for the netfs */ struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */ - unsigned long long submitted; /* Amount submitted for I/O so far */ - unsigned long long len; /* Length of the request */ + uoff_t submitted; /* Amount submitted for I/O so far */ + uoff_t len; /* Length of the request */ size_t transferred; /* Amount to be indicated as transferred */ + size_t progress_at; /* Report read progress when hit this much read */ long error; /* 0 or error that occurred */ - unsigned long long i_size; /* Size of the file */ - unsigned long long start; /* Start position */ + uoff_t i_size; /* Size of the file */ + uoff_t start; /* Start position */ atomic64_t issued_to; /* Write issuer folio cursor */ - unsigned long long collected_to; /* Point we've collected to */ - unsigned long long cleaned_to; /* Position we've cleaned folios to */ - unsigned long long abandon_to; /* Position to abandon folios to */ + uoff_t collected_to; /* Point we've collected to */ + uoff_t cache_coll_to; /* Point the cache has collected to */ + uoff_t cleaned_to; /* Position we've cleaned folios to */ + uoff_t abandon_to; /* Position to abandon folios to */ const struct folio *no_unlock_folio; /* Don't unlock this folio after read */ gfp_t gfp; /* GFP flags to use */ unsigned int direct_bv_count; /* Number of elements in direct_bv[] */ @@ -262,7 +275,6 @@ struct netfs_io_request { atomic_t subreq_counter; /* Next subreq->debug_index */ unsigned int nr_group_rel; /* Number of refs to release on ->group */ spinlock_t lock; /* Lock for queuing subreqs */ - unsigned char front_folio_order; /* Order (size) of front folio */ enum netfs_io_origin origin; /* Origin of the request */ bool direct_bv_unpin; /* T if direct_bv[] must be unpinned */ refcount_t ref; @@ -273,13 +285,18 @@ struct netfs_io_request { #define NETFS_RREQ_FAILED 3 /* The request failed */ #define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */ #define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */ -#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */ -#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */ -#define NETFS_RREQ_FOLIO_COPY_TO_CACHE 10 /* Copy current folio to cache from read */ -#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */ -#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */ +#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */ +#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */ +#define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */ +#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */ +#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */ +#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */ +#define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */ +#ifdef CONFIG_NETFS_PGPRIV2 #define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark * write to cache on read */ +#endif const struct netfs_request_ops *netfs_ops; }; @@ -298,12 +315,12 @@ struct netfs_request_ops { int (*prepare_read)(struct netfs_io_subrequest *subreq); void (*issue_read)(struct netfs_io_subrequest *subreq); bool (*is_still_valid)(struct netfs_io_request *rreq); - int (*check_write_begin)(struct file *file, loff_t pos, unsigned len, + int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata); void (*done)(struct netfs_io_request *rreq); /* Modification handling */ - void (*update_i_size)(struct inode *inode, loff_t i_size); + void (*update_i_size)(struct inode *inode, uoff_t i_size); void (*post_modify)(struct inode *inode); /* Write request handling */ @@ -331,7 +348,7 @@ struct netfs_cache_ops { /* Read data from the cache */ int (*read)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -339,7 +356,7 @@ struct netfs_cache_ops { /* Write data to the cache */ int (*write)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -349,15 +366,14 @@ struct netfs_cache_ops { /* Expand readahead request */ void (*expand_readahead)(struct netfs_cache_resources *cres, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size); + uoff_t *_start, + uoff_t *_len, + uoff_t i_size); /* Prepare a read operation, shortening it to a cached/uncached * boundary as appropriate. */ - enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq, - unsigned long long i_size); + int (*prepare_read)(struct netfs_io_subrequest *subreq); /* Prepare a write subrequest, working out if we're allowed to do it * and finding out the maximum amount of data to gather before @@ -370,15 +386,24 @@ struct netfs_cache_ops { * actually do. */ int (*prepare_write)(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet); + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet); /* Query the occupancy of the cache in a region, returning where the * next chunk of data starts and how long it is. */ int (*query_occupancy)(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len); + struct fscache_occupancy *occ); + + /* Collect the result of buffered writeback to the cache. This + * includes copying a read to the cache. block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap + */ + void (*collect_write)(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type); }; /* High-level read API. */ @@ -396,6 +421,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from); ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *iter, struct netfs_group *netfs_group); ssize_t netfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from); +void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from, + uoff_t to); /* Single, monolithic object read/write API. */ void netfs_single_mark_inode_dirty(struct inode *inode); @@ -409,7 +436,7 @@ struct readahead_control; void netfs_readahead(struct readahead_control *); int netfs_read_folio(struct file *, struct folio *); int netfs_write_begin(struct netfs_inode *, struct file *, - struct address_space *, loff_t pos, unsigned int len, + struct address_space *, uoff_t pos, unsigned int len, struct folio **, void **fsdata); int netfs_writepages(struct address_space *mapping, struct writeback_control *wbc); @@ -487,10 +514,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode) * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode) +static inline uoff_t netfs_read_remote_i_size(const struct inode *inode) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long remote_i_size; + uoff_t remote_i_size; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -525,7 +552,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in * spinning forever. */ static inline void netfs_write_remote_i_size(struct inode *inode, - unsigned long long remote_i_size) + uoff_t remote_i_size) { struct netfs_inode *ictx = netfs_inode(inode); @@ -562,10 +589,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode, * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_zero_point(const struct inode *inode) +static inline uoff_t netfs_read_zero_point(const struct inode *inode) { struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long zero_point; + uoff_t zero_point; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -600,7 +627,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode * forever. */ static inline void netfs_write_zero_point(struct inode *inode, - unsigned long long zero_point) + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -641,9 +668,9 @@ static inline void netfs_write_zero_point(struct inode *inode, * archs it makes no difference if preempt is enabled or not. */ static inline void netfs_read_sizes(const struct inode *inode, - unsigned long long *i_size, - unsigned long long *remote_i_size, - unsigned long long *zero_point) + uoff_t *i_size, + uoff_t *remote_i_size, + uoff_t *zero_point) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); #if BITS_PER_LONG==32 && defined(CONFIG_SMP) @@ -689,9 +716,9 @@ static inline void netfs_read_sizes(const struct inode *inode, * forever. */ static inline void netfs_write_sizes(struct inode *inode, - unsigned long long i_size, - unsigned long long remote_i_size, - unsigned long long zero_point) + uoff_t i_size, + uoff_t remote_i_size, + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -759,7 +786,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx, * Inform the netfs lib that a file got resized so that it can adjust its state. */ static inline void netfs_resize_file(struct netfs_inode *ictx, - unsigned long long new_i_size, + uoff_t new_i_size, bool changed_on_server) { #if BITS_PER_LONG==32 && defined(CONFIG_SMP) diff --git a/include/linux/netpoll.h b/include/linux/netpoll.h index 1c6b1eec5efd..ec0821a5b02d 100644 --- a/include/linux/netpoll.h +++ b/include/linux/netpoll.h @@ -16,11 +16,6 @@ #include <linux/ip.h> #include <linux/udp.h> -union inet_addr { - __be32 ip; - struct in6_addr in6; -}; - struct netpoll { struct net_device *dev; netdevice_tracker dev_tracker; diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0906a0b40c6a..8c2818db43c5 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -11,59 +11,8 @@ #include <linux/cred.h> #include <linux/sunrpc/auth.h> #include <linux/sunrpc/msg_prot.h> -#include <linux/string.h> -#include <linux/crc32.h> -#include <uapi/linux/nfs.h> - -/* The LOCALIO program is entirely private to Linux and is - * NOT part of the uapi. - */ -#define NFS_LOCALIO_PROGRAM 400122 -#define LOCALIOPROC_NULL 0 -#define LOCALIOPROC_UUID_IS_LOCAL 1 - -/* - * This is the kernel NFS client file handle representation - */ -#define NFS_MAXFHSIZE 128 -struct nfs_fh { - unsigned short size; - unsigned char data[NFS_MAXFHSIZE]; -}; - -/* - * Returns a zero iff the size and data fields match. - * Checks only "size" bytes in the data field. - */ -static inline int nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) -{ - return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; -} - -static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) -{ - target->size = source->size; - memcpy(target->data, source->data, source->size); -} - -enum nfs3_stable_how { - NFS_UNSTABLE = 0, - NFS_DATA_SYNC = 1, - NFS_FILE_SYNC = 2, +#include <linux/nfs_fh.h> - /* used by direct.c to mark verf as invalid */ - NFS_INVALID_STABLE_HOW = -1 -}; +#include <uapi/linux/nfs.h> -/** - * nfs_fhandle_hash - calculate the crc32 hash for the filehandle - * @fh - pointer to filehandle - * - * returns a crc32 hash for the filehandle that is compatible with - * the one displayed by "wireshark". - */ -static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) -{ - return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); -} #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfs3.h b/include/linux/nfs3.h index 404b8f724fc9..b6539a75edea 100644 --- a/include/linux/nfs3.h +++ b/include/linux/nfs3.h @@ -7,6 +7,49 @@ #include <uapi/linux/nfs3.h> +/* + * NFSv3 error status values. + * See RFC 1813 Section 2.5 + */ +enum { + NFS3ERR_PERM = 1, + NFS3ERR_NOENT = 2, + NFS3ERR_IO = 5, + NFS3ERR_NXIO = 6, + NFS3ERR_ACCES = 13, + NFS3ERR_EXIST = 17, + NFS3ERR_XDEV = 18, + NFS3ERR_NODEV = 19, + NFS3ERR_NOTDIR = 20, + NFS3ERR_ISDIR = 21, + NFS3ERR_INVAL = 22, + NFS3ERR_FBIG = 27, + NFS3ERR_NOSPC = 28, + NFS3ERR_ROFS = 30, + NFS3ERR_MLINK = 31, + NFS3ERR_NAMETOOLONG = 63, + NFS3ERR_NOTEMPTY = 66, + NFS3ERR_DQUOT = 69, + NFS3ERR_STALE = 70, + NFS3ERR_REMOTE = 71, + NFS3ERR_BADHANDLE = 10001, + NFS3ERR_NOT_SYNC = 10002, + NFS3ERR_BAD_COOKIE = 10003, + NFS3ERR_NOTSUPP = 10004, + NFS3ERR_TOOSMALL = 10005, + NFS3ERR_SERVERFAULT = 10006, + NFS3ERR_BADTYPE = 10007, + NFS3ERR_JUKEBOX = 10008, +}; + +enum nfs3_stable_how { + NFS_UNSTABLE = 0, + NFS_DATA_SYNC = 1, + NFS_FILE_SYNC = 2, + + /* used to mark verf as invalid */ + NFS_INVALID_STABLE_HOW = -1 +}; /* Number of 32bit words in post_op_attr */ #define NFS3_POST_OP_ATTR_WORDS 22 diff --git a/include/linux/nfs4.h b/include/linux/nfs4.h index 1a3981c26b23..41b7cdcc674f 100644 --- a/include/linux/nfs4.h +++ b/include/linux/nfs4.h @@ -263,6 +263,12 @@ enum why_no_delegation4 { /* new to v4.1 */ WND4_IS_DIR = 8, }; +enum stable_how4 { + UNSTABLE4 = 0, + DATA_SYNC4 = 1, + FILE_SYNC4 = 2, +}; + enum lock_type4 { NFS4_UNLOCK_LT = 0, NFS4_READ_LT = 1, diff --git a/include/linux/nfs_fh.h b/include/linux/nfs_fh.h new file mode 100644 index 000000000000..49dfc5ec60fe --- /dev/null +++ b/include/linux/nfs_fh.h @@ -0,0 +1,63 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * struct nfs_fh is an NFS version-agnostic data structure that + * stores an NFS file handle. It is also commonly used in NFS + * related APIs. + */ +#ifndef _LINUX_NFS_FH_H +#define _LINUX_NFS_FH_H + +#include <linux/types.h> +#include <linux/string.h> +#include <linux/crc32.h> + +/* + * The largest file handle size today is an NFSv4 file handle, + * which can be up to 128 octets long. + */ +#define NFS_MAXFHSIZE 128 +struct nfs_fh { + unsigned short size; + unsigned char data[NFS_MAXFHSIZE]; +}; + +/** + * nfs_compare_fh - Compare two NFS file handles + * @a: An NFS file handle to be compared + * @b: An NFS file handle to be compared + * + * Checks only "size" bytes in each data field. + * + * Return: %false if the two file handles are equal, otherwise %true + */ +static inline bool nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) +{ + return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; +} + +/** + * nfs_copy_fh - Copy an NFS file handle + * @target: Destination file handle + * @source: Source file handle + * + * Copies source->size bytes of file handle data into target. + */ +static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) +{ + target->size = source->size; + memcpy(target->data, source->data, source->size); +} + +/** + * nfs_fhandle_hash - Calculate the crc32 hash for the filehandle + * @fh: An NFS file handle to hash + * + * Return: a crc32 hash for the filehandle that is compatible with + * the one displayed by "wireshark" + */ +static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) +{ + return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); +} + +#endif /* _LINUX_NFS_FH_H */ diff --git a/include/linux/nfs_fs.h b/include/linux/nfs_fs.h index b85a73ae7919..d2c716322c6f 100644 --- a/include/linux/nfs_fs.h +++ b/include/linux/nfs_fs.h @@ -437,11 +437,11 @@ extern int nfs_refresh_inode(struct inode *, struct nfs_fattr *); extern int nfs_post_op_update_inode(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc_locked(struct inode *inode, struct nfs_fattr *fattr); -extern int nfs_getattr(struct mnt_idmap *, const struct path *, +extern int nfs_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void nfs_access_add_cache(struct inode *, struct nfs_access_entry *, const struct cred *); extern void nfs_access_set_mask(struct nfs_access_entry *, u32); -extern int nfs_permission(struct mnt_idmap *, struct inode *, int); +extern int nfs_permission(const struct mnt_idmap *, struct inode *, int); extern int nfs_open(struct inode *, struct file *); extern int nfs_attribute_cache_expired(struct inode *inode); extern int nfs_revalidate_inode(struct inode *inode, unsigned long flags); @@ -450,7 +450,7 @@ extern int nfs_clear_invalid_mapping(struct address_space *mapping); extern bool nfs_mapping_need_revalidate_inode(struct inode *inode); extern int nfs_revalidate_mapping(struct inode *inode, struct address_space *mapping); extern int nfs_revalidate_mapping_rcu(struct inode *inode); -extern int nfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +extern int nfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern void nfs_setattr_update_inode(struct inode *inode, struct iattr *attr, struct nfs_fattr *); extern void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr); extern struct nfs_open_context *get_nfs_open_context(struct nfs_open_context *ctx); diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 34d294774f8c..416c6f39f31d 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -74,6 +74,8 @@ struct nfs_client { u64 cl_clientid; /* constant */ nfs4_verifier cl_confirm; /* Clientid verifier */ unsigned long cl_state; + /* bumped on each CB_NOTIFY_DEVICEID CHANGE for this client */ + atomic_t cl_deviceid_change_epoch; spinlock_t cl_lock; @@ -101,6 +103,8 @@ struct nfs_client { /* The flags used for obtaining the clientid during EXCHANGE_ID */ u32 cl_exchange_flags; struct nfs4_session *cl_session; /* shared session */ + /* CB_NOTIFY_DEVICEID DELETE suspects, protected by cl_lock */ + struct list_head cl_deviceid_deletes; bool cl_preserve_clid; struct nfs41_server_owner *cl_serverowner; struct nfs41_server_scope *cl_serverscope; @@ -248,6 +252,10 @@ struct nfs_server { that are supported on this filesystem */ struct pnfs_layoutdriver_type *pnfs_curr_ld; /* Active layout driver */ + unsigned int lg_reply_sz; /* Learned LAYOUTGET reply + buffer size, when the layout + driver's default has proved + too small */ struct rpc_wait_queue roc_rpcwaitq; /* the following fields are protected by nfs_client->cl_lock */ diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index 4b9a35dbc062..c38e4b380be5 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -38,6 +38,7 @@ enum { PG_REMOVE, /* page group sync bit in write path */ PG_CONTENDED1, /* Is someone waiting for a lock? */ PG_CONTENDED2, /* Is someone waiting for a lock? */ + PG_PINNED, /* page is pinned by GUP */ }; struct nfs_inode; @@ -58,6 +59,7 @@ struct nfs_page { struct nfs_page *wb_this_page; /* list of reqs for this page */ struct nfs_page *wb_head; /* head pointer for req list */ unsigned short wb_nio; /* Number of I/O attempts */ + unsigned int wb_nr_pinned; /* Number of pinned pages */ }; struct nfs_pgio_mirror; @@ -125,15 +127,17 @@ struct nfs_pageio_descriptor { extern struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, struct page *page, + bool pinned, unsigned int pgbase, loff_t offset, unsigned int count); extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, struct folio *folio, + bool pinned, unsigned int offset, unsigned int count); -extern void nfs_release_request(struct nfs_page *); - +void nfs_release_request(struct nfs_page *req); +void nfs_release_request_list(struct list_head *head); extern void nfs_pageio_init(struct nfs_pageio_descriptor *desc, struct inode *inode, diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index 22265b1ff080..c199ea23e7eb 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -2,80 +2,33 @@ /* * include/linux/nfs_ssc.h * + * NFSv4.2 server-to-server copy, NFS client side APIs + * * Author: Dai Ngo <dai.ngo@oracle.com> * * Copyright (c) 2020, Oracle and/or its affiliates. */ -#include <linux/nfs_fs.h> -#include <linux/sunrpc/svc.h> +#ifndef _LINUX_NFS_SSC_H +#define _LINUX_NFS_SSC_H -extern struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl; +#include <linux/nfs_fh.h> +#include <linux/nfs4.h> + +struct file; +struct vfsmount; -/* - * NFS_V4 - */ struct nfs4_ssc_client_ops { + struct module *owner; struct file *(*sco_open)(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, nfs4_stateid *stateid); void (*sco_close)(struct file *filep); }; -/* - * NFS_FS - */ -struct nfs_ssc_client_ops { - void (*sco_sb_deactive)(struct super_block *sb); -}; - -struct nfs_ssc_client_ops_tbl { - const struct nfs4_ssc_client_ops *ssc_nfs4_ops; - const struct nfs_ssc_client_ops *ssc_nfs_ops; -}; - extern void nfs42_ssc_register_ops(void); extern void nfs42_ssc_unregister_ops(void); extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops); extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops); -#ifdef CONFIG_NFSD_V4_2_INTER_SSC -static inline struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, nfs4_stateid *stateid) -{ - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - return (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_open)(ss_mnt, src_fh, stateid); - return ERR_PTR(-EIO); -} - -static inline void nfs42_ssc_close(struct file *filep) -{ - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_close)(filep); -} -#endif - -struct nfsd4_ssc_umount_item { - struct list_head nsui_list; - bool nsui_busy; - /* - * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry - * is removed when refcnt drops to 1 and nsui_expire expires. - */ - refcount_t nsui_refcnt; - unsigned long nsui_expire; - struct vfsmount *nsui_vfsmount; - char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; -}; - -/* - * NFS_FS - */ -extern void nfs_ssc_register(const struct nfs_ssc_client_ops *ops); -extern void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops); - -static inline void nfs_do_sb_deactive(struct super_block *sb) -{ - if (nfs_ssc_client_tbl.ssc_nfs_ops) - (*nfs_ssc_client_tbl.ssc_nfs_ops->sco_sb_deactive)(sb); -} +#endif /* _LINUX_NFS_SSC_H */ diff --git a/include/linux/nfs_xdr.h b/include/linux/nfs_xdr.h index 7ed8fdb930d6..c0e29b4dfa62 100644 --- a/include/linux/nfs_xdr.h +++ b/include/linux/nfs_xdr.h @@ -1693,6 +1693,7 @@ struct nfs_pgio_header { struct nfs_client *ds_clp; /* pNFS data server */ u32 ds_commit_idx; /* ds index if ds_clp is set */ u32 pgio_mirror_idx;/* mirror index in pgio layer */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the I/O */ }; struct nfs_mds_commit_info { @@ -1731,6 +1732,7 @@ struct nfs_commit_data { struct nfs_open_context *context; struct pnfs_layout_segment *lseg; struct nfs_client *ds_clp; /* pNFS data server */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the commit */ int ds_commit_index; loff_t lwb; const struct rpc_call_ops *mds_ops; diff --git a/include/linux/nfsd_ssc.h b/include/linux/nfsd_ssc.h new file mode 100644 index 000000000000..7001410f01c2 --- /dev/null +++ b/include/linux/nfsd_ssc.h @@ -0,0 +1,38 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * include/linux/nfsd_ssc.h + * + * NFSv4.2 server-to-server copy, NFS server side APIs + * + * Author: Dai Ngo <dai.ngo@oracle.com> + * + * Copyright (c) 2020, Oracle and/or its affiliates. + */ + +#ifndef _LINUX_NFSD_SSC_H +#define _LINUX_NFSD_SSC_H + +#include <linux/nfs_fh.h> +#include <linux/nfs4.h> + +struct file; +struct vfsmount; + +#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid); +void nfsd42_ssc_close(struct file *filp); +#else +static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, + struct nfs_fh *src_fh, + nfs4_stateid *stateid) +{ + return ERR_PTR(-EIO); +} + +static inline void nfsd42_ssc_close(struct file *filp) +{ +} +#endif + +#endif /* _LINUX_NFSD_SSC_H */ diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h index 3d91043254e6..8ce4d978a636 100644 --- a/include/linux/nfslocalio.h +++ b/include/linux/nfslocalio.h @@ -13,9 +13,18 @@ #include <linux/uuid.h> #include <linux/sunrpc/clnt.h> #include <linux/sunrpc/svcauth.h> -#include <linux/nfs.h> +#include <linux/nfs_fh.h> + #include <net/net_namespace.h> +/* + * The LOCALIO program is entirely private to Linux and is NOT part of + * the uapi. + */ +#define NFS_LOCALIO_PROGRAM 400122 +#define LOCALIOPROC_NULL 0 +#define LOCALIOPROC_UUID_IS_LOCAL 1 + struct nfs_client; struct nfs_file_localio; diff --git a/include/linux/ns/ns_common_types.h b/include/linux/ns/ns_common_types.h index ea45c54e4435..fddc62be3b62 100644 --- a/include/linux/ns/ns_common_types.h +++ b/include/linux/ns/ns_common_types.h @@ -116,10 +116,11 @@ struct ns_common { struct dentry *stashed; const struct proc_ns_operations *ops; unsigned int inum; - union { - struct ns_tree; - struct rcu_head ns_rcu; - }; + struct ns_tree; + struct rcu_head ns_rcu; +#ifdef CONFIG_SECURITY + void *ns_security; +#endif }; #define to_ns_common(__ns) \ diff --git a/include/linux/nvme-tcp.h b/include/linux/nvme-tcp.h index e435250fcb4d..859338da8573 100644 --- a/include/linux/nvme-tcp.h +++ b/include/linux/nvme-tcp.h @@ -77,7 +77,7 @@ struct nvme_tcp_hdr { __le32 plen; }; -/** +/* * struct nvme_tcp_icreq_pdu - nvme tcp initialize connection request pdu * * @hdr: pdu generic header @@ -95,7 +95,7 @@ struct nvme_tcp_icreq_pdu { __u8 rsvd2[112]; }; -/** +/* * struct nvme_tcp_icresp_pdu - nvme tcp initialize connection response pdu * * @hdr: pdu common header @@ -113,12 +113,13 @@ struct nvme_tcp_icresp_pdu { __u8 rsvd[112]; }; -/** +/* * struct nvme_tcp_term_pdu - nvme tcp terminate connection pdu * * @hdr: pdu common header * @fes: fatal error status - * @fei: fatal error information + * @feil: fatal error information (low 16 bits) + * @feih: fatal error information (high 16 bits) */ struct nvme_tcp_term_pdu { struct nvme_tcp_hdr hdr; @@ -128,7 +129,7 @@ struct nvme_tcp_term_pdu { __u8 rsvd[10]; }; -/** +/* * struct nvme_tcp_cmd_pdu - nvme tcp command capsule pdu * * @hdr: pdu common header @@ -139,10 +140,9 @@ struct nvme_tcp_cmd_pdu { struct nvme_command cmd; }; -/** +/* * struct nvme_tcp_rsp_pdu - nvme tcp response capsule pdu * - * @hdr: pdu common header * @hdr: nvme-tcp generic header * @cqe: nvme completion queue entry */ @@ -151,7 +151,7 @@ struct nvme_tcp_rsp_pdu { struct nvme_completion cqe; }; -/** +/* * struct nvme_tcp_r2t_pdu - nvme tcp ready-to-transfer pdu * * @hdr: pdu common header @@ -169,7 +169,7 @@ struct nvme_tcp_r2t_pdu { __u8 rsvd[4]; }; -/** +/* * struct nvme_tcp_data_pdu - nvme tcp data pdu * * @hdr: pdu common header diff --git a/include/linux/once_lite.h b/include/linux/once_lite.h index 236592c4eeb1..5e2b67039aaf 100644 --- a/include/linux/once_lite.h +++ b/include/linux/once_lite.h @@ -10,19 +10,21 @@ #define DO_ONCE_LITE(func, ...) \ DO_ONCE_LITE_IF(true, func, ##__VA_ARGS__) -#define __ONCE_LITE_IF(condition) \ +#define __ONCE_LITE() \ ({ \ static bool __section(".data..once") __already_done; \ - bool __ret_cond = !!(condition); \ bool __ret_once = false; \ \ - if (unlikely(__ret_cond) && unlikely(!__already_done)) {\ + if (unlikely(!__already_done)) { \ __already_done = true; \ __ret_once = true; \ } \ unlikely(__ret_once); \ }) +#define __ONCE_LITE_IF(condition) \ + (unlikely(condition) && __ONCE_LITE()) + #define DO_ONCE_LITE_IF(condition, func, ...) \ ({ \ bool __ret_do_once = !!(condition); \ diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 7a863572adce..b0ddc652e76c 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -44,10 +44,6 @@ * Consequently, PG_reserved for a page mapped into user space can indicate * the zero page, the vDSO, MMIO pages or device memory. * - * The PG_private bitflag is set on pagecache pages if they contain filesystem - * specific data (which is normally at page->private). It can be used by - * private allocations for its own usage. - * * During initiation of disk I/O, PG_locked is set. This bit is set before I/O * and cleared when writeback _starts_ or when read _completes_. PG_writeback * is set before writeback starts and cleared when it finishes. @@ -105,7 +101,7 @@ enum pageflags { PG_owner_2, /* Owner use. If pagecache, fs may use */ PG_arch_1, PG_reserved, - PG_private, /* If pagecache, has fs-private data */ + PG_folio, /* Do not use: reserved for folio identification */ PG_private_2, /* If pagecache, has fs aux data */ PG_reclaim, /* To be reclaimed asap */ PG_swapbacked, /* Page is backed by RAM/swap */ @@ -208,14 +204,13 @@ enum pageflags { static __always_inline bool compound_info_has_mask(void) { /* - * Limit mask usage to HugeTLB vmemmap optimization (HVO) where it - * makes a difference. + * Limit mask usage to HVO where it makes a difference. * * The approach with mask would work in the wider set of conditions, * but it requires validating that struct pages are naturally aligned * for all orders up to the MAX_FOLIO_ORDER, which can be tricky. */ - if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) return false; return is_power_of_2(sizeof(struct page)); @@ -576,9 +571,19 @@ FOLIO_FLAG(swapbacked, FOLIO_HEAD_PAGE) /* * Private page markings that may be used by the filesystem that owns the page * for its own purposes. - * - PG_private and PG_private_2 cause release_folio() and co to be invoked + * - folio->private and PG_private_2 cause release_folio() and co to be invoked */ -PAGEFLAG(Private, private, PF_ANY) + +static __always_inline bool folio_test_private(const struct folio *folio) +{ + /* + * data_race() is added for readers without holding the folio lock. + * Only the NULL/non-NULL answer is used and both are valid while + * private is being attached or detached, so the race is benign. + */ + return data_race(folio->private); +} + FOLIO_FLAG(private_2, FOLIO_HEAD_PAGE) /* owner_2 can be set on tail pages for anon memory */ @@ -588,13 +593,12 @@ FOLIO_FLAG(owner_2, FOLIO_HEAD_PAGE) * Only test-and-set exist for PG_writeback. The unconditional operators are * risky: they bypass page accounting. */ -TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL) - TESTSCFLAG(Writeback, writeback, PF_NO_TAIL) +FOLIO_TEST_FLAG(writeback, FOLIO_HEAD_PAGE) + FOLIO_TEST_SET_FLAG(writeback, FOLIO_HEAD_PAGE) FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE) /* PG_readahead is only used for reads; PG_reclaim is only for writes */ -PAGEFLAG(Reclaim, reclaim, PF_NO_TAIL) - TESTCLEARFLAG(Reclaim, reclaim, PF_NO_TAIL) +FOLIO_FLAG(reclaim, FOLIO_HEAD_PAGE) FOLIO_FLAG(readahead, FOLIO_HEAD_PAGE) FOLIO_TEST_CLEAR_FLAG(readahead, FOLIO_HEAD_PAGE) @@ -656,6 +660,7 @@ TESTSCFLAG(HWPoison, hwpoison, PF_ANY) #define __PG_HWPOISON (1UL << PG_hwpoison) #else PAGEFLAG_FALSE(HWPoison, hwpoison) +TESTSCFLAG_FALSE(HWPoison, hwpoison) #define __PG_HWPOISON 0 #endif @@ -1170,7 +1175,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) */ #define PAGE_FLAGS_CHECK_AT_FREE \ (1UL << PG_lru | 1UL << PG_locked | \ - 1UL << PG_private | 1UL << PG_private_2 | \ + 1UL << PG_private_2 | \ 1UL << PG_writeback | 1UL << PG_reserved | \ 1UL << PG_active | \ 1UL << PG_unevictable | __PG_MLOCKED | LRU_GEN_MASK) @@ -1194,8 +1199,31 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) (0xffUL /* order */ | 1UL << PG_has_hwpoisoned | \ 1UL << PG_large_rmappable | 1UL << PG_partially_mapped) -#define PAGE_FLAGS_PRIVATE \ - (1UL << PG_private | 1UL << PG_private_2) +/** + * folio_has_attached_private - check if the folio has private data attached + * @folio: The folio to check. + * + * Use this in code that may encounter swapcache or hugetlb folios but only + * wants to detect attached private data. + * + * Return: true if the folio has private data attached. + */ +static inline bool folio_has_attached_private(const struct folio *folio) +{ + /* + * Swapcache stores swp_entry_t in folio->swap, a union with + * folio->private, and hugetlb stores its own flags in folio->private; + * both are excluded. + * + * NOTE: For swapcache, folio->swap.val PG_swapcache are not set as + * a whole, so folio_test_swapcache() is not reliable to exclude + * swapcache. Use folio_test_swapbacked() instead, since it remains set + * when a folio is added to/removed from swapcache. + */ + + return folio_test_private(folio) && !folio_test_swapbacked(folio) && + !folio_test_hugetlb(folio); +} /** * folio_has_private - Determine if folio has private stuff * @folio: The folio to be checked @@ -1205,7 +1233,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) */ static inline int folio_has_private(const struct folio *folio) { - return !!(folio->flags.f & PAGE_FLAGS_PRIVATE); + return folio_has_attached_private(folio) || folio_test_private_2(folio); } #undef PF_ANY diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h index d649b6bbbc87..2baf7a2b29b2 100644 --- a/include/linux/page_counter.h +++ b/include/linux/page_counter.h @@ -63,11 +63,12 @@ static inline void page_counter_init(struct page_counter *counter, counter->track_failcnt = false; } -static inline unsigned long page_counter_read(struct page_counter *counter) +static inline unsigned long page_counter_read(const struct page_counter *counter) { return atomic_long_read(&counter->usage); } +long page_counter_margin(const struct page_counter *counter); void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages); void page_counter_charge(struct page_counter *counter, unsigned long nr_pages); bool page_counter_try_charge(struct page_counter *counter, diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 0adfa6605653..73af18a37367 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -14,7 +14,6 @@ #include <linux/gfp.h> #include <linux/bitops.h> #include <linux/hardirq.h> /* for in_interrupt() */ -#include <linux/hugetlb_inline.h> struct folio_batch; @@ -594,7 +593,6 @@ static inline void folio_attach_private(struct folio *folio, void *data) { folio_get(folio); folio->private = data; - folio_set_private(folio); } /** @@ -629,9 +627,8 @@ static inline void *folio_detach_private(struct folio *folio) { void *data = folio_get_private(folio); - if (!folio_test_private(folio)) + if (!data) return NULL; - folio_clear_private(folio); folio->private = NULL; folio_put(folio); @@ -1128,8 +1125,7 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, const pgoff_t pgoff = __linear_anon_page_index(vma, address); VM_WARN_ON_ONCE(!vma_is_cow_mapping(vma)); - /* Account for MAP_PRIVATE-/dev/zero which is only semi-anonymous. */ - if (vma_is_anonymous(vma) && !vma->vm_file) + if (vma_is_anonymous(vma)) VM_WARN_ON_ONCE(pgoff != linear_page_index(vma, address)); return pgoff; @@ -1416,6 +1412,7 @@ struct readahead_control { bool dropbehind; bool _workingset; unsigned long _pflags; + bool _forward; }; #define DEFINE_READAHEAD(ractl, f, r, m, i) \ @@ -1480,18 +1477,29 @@ void page_cache_async_readahead(struct address_space *mapping, page_cache_async_ra(&ractl, folio, req_count); } +/* + * Adjust readahead_control to ensure next folio comes from + * [_index, _index + _nr_pages) afterwards and reset _batch_count. + */ +static inline void __readahead_advance(struct readahead_control *rac) +{ + if (rac->_forward) + rac->_index += rac->_batch_count; + + rac->_nr_pages -= rac->_batch_count; + rac->_batch_count = 0; +} + static inline struct folio *__readahead_folio(struct readahead_control *ractl) { struct folio *folio; BUG_ON(ractl->_batch_count > ractl->_nr_pages); - ractl->_nr_pages -= ractl->_batch_count; - ractl->_index += ractl->_batch_count; + __readahead_advance(ractl); + ractl->_forward = true; - if (!ractl->_nr_pages) { - ractl->_batch_count = 0; + if (!ractl->_nr_pages) return NULL; - } folio = xa_load(&ractl->mapping->i_pages, ractl->_index); VM_BUG_ON_FOLIO(!folio_test_locked(folio), folio); @@ -1517,6 +1525,39 @@ static inline struct folio *readahead_folio(struct readahead_control *ractl) return folio; } +/** + * readahead_folio_last - Get the next folio to read, from the tail. + * @ractl: The current readahead request. + * + * Like readahead_folio(), but walks the range back-to-front. The folio is + * returned locked with its refcount dropped; the caller unlocks it once I/O + * completes. Compound folios are returned once, at their head index. + * + * Context: The folio is locked. + * Return: A pointer to the next folio, or %NULL when done. + */ +static inline struct folio *readahead_folio_last(struct readahead_control *ractl) +{ + struct folio *folio; + + /* Drop the previously returned batch from the remaining range. */ + __readahead_advance(ractl); + ractl->_forward = false; + + if (!ractl->_nr_pages) + return NULL; + + /* xa_load() follows sibling entries, so a tail index returns the head */ + folio = xa_load(&ractl->mapping->i_pages, + ractl->_index + ractl->_nr_pages - 1); + VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); + + ractl->_batch_count = folio_nr_pages(folio); + + folio_put(folio); + return folio; +} + static inline unsigned int __readahead_batch(struct readahead_control *rac, struct page **array, unsigned int array_sz) { @@ -1525,9 +1566,8 @@ static inline unsigned int __readahead_batch(struct readahead_control *rac, struct folio *folio; BUG_ON(rac->_batch_count > rac->_nr_pages); - rac->_nr_pages -= rac->_batch_count; - rac->_index += rac->_batch_count; - rac->_batch_count = 0; + __readahead_advance(rac); + rac->_forward = true; xas_set(&xas, rac->_index); rcu_read_lock(); diff --git a/include/linux/panic.h b/include/linux/panic.h index f1dd417e54b2..17e61b61c45f 100644 --- a/include/linux/panic.h +++ b/include/linux/panic.h @@ -13,7 +13,8 @@ __printf(1, 2) void panic(const char *fmt, ...) __noreturn __cold; __printf(1, 0) void vpanic(const char *fmt, va_list args) __noreturn __cold; -void nmi_panic(struct pt_regs *regs, const char *msg); +__printf(2, 3) +void nmi_panic(struct pt_regs *regs, const char *fmt, ...); void check_panic_on_warn(const char *origin); extern void oops_enter(void); extern void oops_exit(void); @@ -110,4 +111,6 @@ extern void add_taint(unsigned flag, enum lockdep_ok); extern int test_taint(unsigned flag); extern unsigned long get_taint(void); +void arch_do_panic(void); + #endif /* _LINUX_PANIC_H */ diff --git a/include/linux/pci-epf.h b/include/linux/pci-epf.h index 704e1dc8b30a..226f3e59cf60 100644 --- a/include/linux/pci-epf.h +++ b/include/linux/pci-epf.h @@ -163,7 +163,8 @@ enum pci_epf_doorbell_type { * For MSI-backed doorbells this is the MSI message, while for * "embedded" doorbells this represents an MMIO write that asserts * an interrupt on the EP side. - * @virq: IRQ number of this doorbell message + * @virq: IRQ number of this doorbell message. Multiple messages may use the + * same IRQ; consumers must request each distinct IRQ only once. * @irq_flags: Required flags for request_irq()/request_threaded_irq(). * Callers may OR-in additional flags (e.g. IRQF_ONESHOT). * @type: Doorbell type. diff --git a/include/linux/pci.h b/include/linux/pci.h index d31a8d107b1e..d50cf9ae142d 100644 --- a/include/linux/pci.h +++ b/include/linux/pci.h @@ -43,6 +43,7 @@ #include <uapi/linux/pci.h> #include <linux/pci_ids.h> +#include <linux/pci_liveupdate.h> #define PCI_STATUS_ERROR_BITS (PCI_STATUS_DETECTED_PARITY | \ PCI_STATUS_SIG_SYSTEM_ERROR | \ @@ -598,6 +599,9 @@ struct pci_dev { u8 tph_mode; /* TPH mode */ u8 tph_req_type; /* TPH requester type */ #endif +#ifdef CONFIG_PCI_LIVEUPDATE + struct pci_liveupdate liveupdate; +#endif }; static inline struct pci_dev *pci_physfn(struct pci_dev *dev) @@ -832,6 +836,9 @@ static inline struct pci_dev *pci_upstream_bridge(struct pci_dev *dev) return dev->bus->self; } +#define for_each_pci_dev_in_path(dev) \ + for (; dev; dev = pci_upstream_bridge(dev)) + #ifdef CONFIG_PCI_MSI static inline bool pci_dev_msi_enabled(struct pci_dev *pci_dev) { @@ -1951,9 +1958,10 @@ int pci_disable_link_state(struct pci_dev *pdev, int state); int pci_disable_link_state_locked(struct pci_dev *pdev, int state); int pci_enable_link_state(struct pci_dev *pdev, int state); int pci_enable_link_state_locked(struct pci_dev *pdev, int state); +int pci_force_enable_link_state(struct pci_dev *pdev, int state); void pcie_no_aspm(void); bool pcie_aspm_support_enabled(void); -bool pcie_aspm_enabled(struct pci_dev *pdev); +u32 pcie_aspm_enabled(struct pci_dev *pdev); #else static inline int pci_disable_link_state(struct pci_dev *pdev, int state) { return 0; } @@ -1963,9 +1971,11 @@ static inline int pci_enable_link_state(struct pci_dev *pdev, int state) { return 0; } static inline int pci_enable_link_state_locked(struct pci_dev *pdev, int state) { return 0; } +static inline int pci_force_enable_link_state(struct pci_dev *pdev, int state) +{ return 0; } static inline void pcie_no_aspm(void) { } static inline bool pcie_aspm_support_enabled(void) { return false; } -static inline bool pcie_aspm_enabled(struct pci_dev *pdev) { return false; } +static inline u32 pcie_aspm_enabled(struct pci_dev *pdev) { return 0; } #endif #ifdef CONFIG_HOTPLUG_PCI diff --git a/include/linux/pci_liveupdate.h b/include/linux/pci_liveupdate.h new file mode 100644 index 000000000000..bb79348769b7 --- /dev/null +++ b/include/linux/pci_liveupdate.h @@ -0,0 +1,72 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * PCI Live Update support (Public/Driver API) + * + * Copyright (c) 2026, Google LLC. + * David Matlack <dmatlack@google.com> + */ +#ifndef LINUX_PCI_LIVEUPDATE_H +#define LINUX_PCI_LIVEUPDATE_H + +#include <linux/kho/abi/pci.h> +#include <linux/liveupdate.h> +#include <linux/spinlock_types.h> +#include <linux/types.h> + +/** + * struct pci_liveupdate - PCI Live Update state for a struct pci_dev + * @outgoing: State preserved for the next kernel. + * @incoming: State preserved by the previous kernel. + * @was_incoming: True if this struct pci_dev was incoming-preserved when it was + * set up, i.e. it was matched to state preserved by the previous + * kernel. Unlike @incoming, this is never cleared, so it stays + * true after the device finishes participating in Live Update. + * @frozen: True if the outgoing preservation status of this device is frozen + * and thus cannot be changed. + */ +struct pci_liveupdate { + struct pci_dev_ser *outgoing; + struct pci_dev_ser *incoming; + bool was_incoming; + bool frozen; +}; + +struct pci_dev; + +#ifdef CONFIG_PCI_LIVEUPDATE +int pci_liveupdate_register_flb(struct liveupdate_file_handler *fh); +void pci_liveupdate_unregister_flb(struct liveupdate_file_handler *fh); +int pci_liveupdate_preserve(struct pci_dev *dev); +void pci_liveupdate_unpreserve(struct pci_dev *dev); +void pci_liveupdate_finish(struct pci_dev *dev); +bool pci_liveupdate_is_incoming(struct pci_dev *dev); +#else +static inline int pci_liveupdate_register_flb(struct liveupdate_file_handler *fh) +{ + return -EOPNOTSUPP; +} + +static inline void pci_liveupdate_unregister_flb(struct liveupdate_file_handler *fh) +{ +} + +static inline int pci_liveupdate_preserve(struct pci_dev *dev) +{ + return -EOPNOTSUPP; +} + +static inline void pci_liveupdate_unpreserve(struct pci_dev *dev) +{ +} + +static inline void pci_liveupdate_finish(struct pci_dev *dev) +{ +} + +static inline bool pci_liveupdate_is_incoming(struct pci_dev *dev) +{ + return false; +} +#endif + +#endif /* LINUX_PCI_LIVEUPDATE_H */ diff --git a/include/linux/perf/riscv_pmu.h b/include/linux/perf/riscv_pmu.h index f82a28040594..ecaa40370830 100644 --- a/include/linux/perf/riscv_pmu.h +++ b/include/linux/perf/riscv_pmu.h @@ -55,7 +55,7 @@ struct riscv_pmu { irqreturn_t (*handle_irq)(int irq_num, void *dev); - unsigned long cmask; + DECLARE_BITMAP(cmask, RISCV_MAX_COUNTERS); u64 (*ctr_read)(struct perf_event *event); int (*ctr_get_idx)(struct perf_event *event); int (*ctr_get_width)(int idx); diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h index 5842552294c1..7797ce207555 100644 --- a/include/linux/perf_event.h +++ b/include/linux/perf_event.h @@ -306,6 +306,7 @@ struct perf_event_pmu_context; #define PERF_PMU_CAP_AUX_PAUSE 0x0200 #define PERF_PMU_CAP_AUX_PREFER_LARGE 0x0400 #define PERF_PMU_CAP_MEDIATED_VPMU 0x0800 +#define PERF_PMU_CAP_SIMD_REGS 0x1000 /** * pmu::scope @@ -1467,23 +1468,7 @@ static inline u32 perf_sample_data_size(struct perf_sample_data *data, return size; } -/* - * Clear all bitfields in the perf_branch_entry. - * The to and from fields are not cleared because they are - * systematically modified by caller. - */ -static inline void perf_clear_branch_entry_bitfields(struct perf_branch_entry *br) -{ - br->mispred = 0; - br->predicted = 0; - br->in_tx = 0; - br->abort = 0; - br->cycles = 0; - br->type = 0; - br->spec = PERF_BR_SPEC_NA; - br->reserved = 0; -} - +extern u64 perf_update_xregs_size(struct perf_event *event, bool intr); extern void perf_output_sample(struct perf_output_handle *handle, struct perf_event_header *header, struct perf_sample_data *data, @@ -1534,6 +1519,27 @@ perf_event__output_id_sample(struct perf_event *event, extern void perf_log_lost_samples(struct perf_event *event, u64 lost); +static inline bool event_has_simd_regs(struct perf_event *event) +{ + struct perf_event_attr *attr = &event->attr; + + if (!(event->attr.sample_type & + (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER))) + return false; + + return attr->sample_simd_regs_enabled != 0; +} + +static inline bool event_has_extended_regs(struct perf_event *event) +{ + struct perf_event_attr *attr = &event->attr; + + return ((attr->sample_type & PERF_SAMPLE_REGS_USER) && + (attr->sample_regs_user & PERF_REG_EXTENDED_MASK)) || + ((attr->sample_type & PERF_SAMPLE_REGS_INTR) && + (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK)); +} + static inline bool event_has_any_exclude_flag(struct perf_event *event) { struct perf_event_attr *attr = &event->attr; diff --git a/include/linux/perf_regs.h b/include/linux/perf_regs.h index f632c5725f16..09dbc2fc3859 100644 --- a/include/linux/perf_regs.h +++ b/include/linux/perf_regs.h @@ -9,6 +9,16 @@ struct perf_regs { struct pt_regs *regs; }; +u64 perf_reg_value(struct pt_regs *regs, int idx); +int perf_reg_validate(u64 mask, bool simd_enabled); +u64 perf_reg_abi(struct task_struct *task); +void perf_get_regs_user(struct perf_regs *regs_user, + struct pt_regs *regs); +int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask, + u16 pred_qwords, u32 pred_mask); +u64 perf_simd_reg_value(struct pt_regs *regs, int idx, + u16 qwords_idx, bool pred); + #ifdef CONFIG_HAVE_PERF_REGS #include <asm/perf_regs.h> @@ -16,35 +26,9 @@ struct perf_regs { #define PERF_REG_EXTENDED_MASK 0 #endif -u64 perf_reg_value(struct pt_regs *regs, int idx); -int perf_reg_validate(u64 mask); -u64 perf_reg_abi(struct task_struct *task); -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs); #else #define PERF_REG_EXTENDED_MASK 0 -static inline u64 perf_reg_value(struct pt_regs *regs, int idx) -{ - return 0; -} - -static inline int perf_reg_validate(u64 mask) -{ - return mask ? -ENOSYS : 0; -} - -static inline u64 perf_reg_abi(struct task_struct *task) -{ - return PERF_SAMPLE_REGS_ABI_NONE; -} - -static inline void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} #endif /* CONFIG_HAVE_PERF_REGS */ #endif /* _LINUX_PERF_REGS_H */ diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h index 8c093c119e5a..780fe849ff8b 100644 --- a/include/linux/pgtable.h +++ b/include/linux/pgtable.h @@ -490,35 +490,35 @@ static inline int pudp_set_access_flags(struct vm_area_struct *vma, #endif #ifndef ptep_get -static inline pte_t ptep_get(pte_t *ptep) +static inline pte_t ptep_get(const pte_t *ptep) { return READ_ONCE(*ptep); } #endif #ifndef pmdp_get -static inline pmd_t pmdp_get(pmd_t *pmdp) +static inline pmd_t pmdp_get(const pmd_t *pmdp) { return READ_ONCE(*pmdp); } #endif #ifndef pudp_get -static inline pud_t pudp_get(pud_t *pudp) +static inline pud_t pudp_get(const pud_t *pudp) { return READ_ONCE(*pudp); } #endif #ifndef p4dp_get -static inline p4d_t p4dp_get(p4d_t *p4dp) +static inline p4d_t p4dp_get(const p4d_t *p4dp) { return READ_ONCE(*p4dp); } #endif #ifndef pgdp_get -static inline pgd_t pgdp_get(pgd_t *pgdp) +static inline pgd_t pgdp_get(const pgd_t *pgdp) { return READ_ONCE(*pgdp); } @@ -2313,6 +2313,20 @@ static inline const char *pgtable_level_to_str(enum pgtable_level level) } } +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size); + +#define ptval_to_str(buf, val) \ + do { \ + auto __val = (val); \ + \ + ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ + } while (0) + +#if defined(__SIZEOF_INT128__) +#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ +#else +#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ +#endif #endif /* !__ASSEMBLER__ */ #if !defined(MAX_POSSIBLE_PHYSMEM_BITS) && !defined(CONFIG_64BIT) diff --git a/include/linux/phy.h b/include/linux/phy.h index 5f8d65868e0f..7c5098a0dd6c 100644 --- a/include/linux/phy.h +++ b/include/linux/phy.h @@ -376,6 +376,24 @@ struct mii_bus { int regnum, u16 val); /** @reset: Perform a reset of the bus */ int (*reset)(struct mii_bus *bus); + /** + * @notify_phy_attach: Perform post-attach handling for MDIO bus + * drivers. Optional and independent of @notify_phy_detach. Called + * in phy_attach_direct() right before phy_resume(). Runs in process + * context, may sleep and may be called with RTNL held. Must not + * acquire or rely on RTNL. Returns 0 on success or negative errno + * on failure. Must unwind its own state on error as attachment is + * aborted. + */ + int (*notify_phy_attach)(struct phy_device *phydev); + /** + * @notify_phy_detach: Perform pre-detach handling for MDIO bus + * drivers. Optional and independent of @notify_phy_attach. Called + * in phy_detach() right after phy_suspend(). Runs in process context, + * may sleep and may be called with RTNL held. Must not acquire or + * rely on RTNL. + */ + void (*notify_phy_detach)(struct phy_device *phydev); /** @stats: Statistic counters per device on the bus */ struct mdio_bus_stats stats[PHY_MAX_ADDR]; @@ -1697,7 +1715,7 @@ static inline bool phy_can_wakeup(struct phy_device *phydev) * phy_may_wakeup() - indicate whether PHY has wakeup enabled * @phydev: The phy_device struct * - * Returns: true/false depending on the PHY driver's device_set_wakeup_enabled() + * Returns: true/false depending on the PHY driver's device_set_wakeup_enable() * setting if using the driver model, otherwise the legacy determination. */ bool phy_may_wakeup(struct phy_device *phydev); @@ -2422,10 +2440,10 @@ int phy_get_mac_termination(struct phy_device *phydev, struct device *dev, void phy_resolve_pause(unsigned long *local_adv, unsigned long *partner_adv, bool *tx_pause, bool *rx_pause); -int phy_register_fixup_for_id(const char *bus_id, - int (*run)(struct phy_device *)); -int phy_register_fixup_for_uid(u32 phy_uid, u32 phy_uid_mask, - int (*run)(struct phy_device *)); +void __init phy_register_fixup_for_id(const char *bus_id, + int (*run)(struct phy_device *)); +void __init phy_register_fixup_for_uid(u32 phy_uid, u32 phy_uid_mask, + int (*run)(struct phy_device *)); int phy_eee_tx_clock_stop_capable(struct phy_device *phydev); int phy_eee_rx_clock_stop(struct phy_device *phydev, bool clk_stop_enable); diff --git a/include/linux/phy/phy.h b/include/linux/phy/phy.h index ea47975e288a..14b924a88411 100644 --- a/include/linux/phy/phy.h +++ b/include/linux/phy/phy.h @@ -284,6 +284,8 @@ struct phy *devm_of_phy_optional_get(struct device *dev, struct device_node *np, const char *con_id); struct phy *devm_of_phy_get_by_index(struct device *dev, struct device_node *np, int index); +struct phy *phy_get_by_of_node(struct device_node *np); +struct phy *devm_phy_get_by_of_node(struct device *dev, struct device_node *np); void of_phy_put(struct phy *phy); void phy_put(struct device *dev, struct phy *phy); void devm_phy_put(struct device *dev, struct phy *phy); @@ -493,6 +495,17 @@ static inline struct phy *devm_of_phy_get_by_index(struct device *dev, return ERR_PTR(-ENOSYS); } +static inline struct phy *phy_get_by_of_node(struct device_node *np) +{ + return ERR_PTR(-ENOSYS); +} + +static inline struct phy *devm_phy_get_by_of_node(struct device *dev, + struct device_node *np) +{ + return ERR_PTR(-ENOSYS); +} + static inline void of_phy_put(struct phy *phy) { } diff --git a/include/linux/phylink.h b/include/linux/phylink.h index 1dda5c7ed5f1..3a88a69882a6 100644 --- a/include/linux/phylink.h +++ b/include/linux/phylink.h @@ -843,4 +843,6 @@ void phylink_replay_link_begin(struct phylink *pl); void phylink_replay_link_end(struct phylink *pl); +void phylink_update_mac_pause_capabilities(struct phylink *pl, unsigned long mac_pause); + #endif diff --git a/include/linux/platform_data/microchip-ksz.h b/include/linux/platform_data/microchip-ksz.h index 028781ad4059..411d5e164383 100644 --- a/include/linux/platform_data/microchip-ksz.h +++ b/include/linux/platform_data/microchip-ksz.h @@ -31,9 +31,11 @@ enum ksz_chip_id { KSZ88X3_CHIP_ID = 0x8830, KSZ8864_CHIP_ID = 0x8864, KSZ8895_CHIP_ID = 0x8895, + KSZ8995XA_CHIP_ID = 0x8995, KSZ9477_CHIP_ID = 0x00947700, KSZ9896_CHIP_ID = 0x00989600, KSZ9897_CHIP_ID = 0x00989700, + KSZ9897S_CHIP_ID = 0x00989701, KSZ9893_CHIP_ID = 0x00989300, KSZ9563_CHIP_ID = 0x00956300, KSZ8567_CHIP_ID = 0x00856700, diff --git a/include/linux/platform_data/mipi-i3c-hci.h b/include/linux/platform_data/mipi-i3c-hci.h index ab7395f455f9..6f61d8a02378 100644 --- a/include/linux/platform_data/mipi-i3c-hci.h +++ b/include/linux/platform_data/mipi-i3c-hci.h @@ -3,13 +3,17 @@ #define INCLUDE_PLATFORM_DATA_MIPI_I3C_HCI_H #include <linux/compiler_types.h> +#include <linux/types.h> /** * struct mipi_i3c_hci_platform_data - Platform-dependent data for mipi_i3c_hci * @base_regs: Register set base address (to support multi-bus instances) + * @instance: Zero-based instance number of the Bus Controller as defined by the + * DisCo specification I3C Target Address (_ADR) Encoding */ struct mipi_i3c_hci_platform_data { void __iomem *base_regs; + u8 instance; }; #endif diff --git a/include/linux/platform_data/x86/asus-wmi.h b/include/linux/platform_data/x86/asus-wmi.h index b5ed8c83ace1..be4d2873ffc9 100644 --- a/include/linux/platform_data/x86/asus-wmi.h +++ b/include/linux/platform_data/x86/asus-wmi.h @@ -54,6 +54,7 @@ #define ASUS_WMI_DEVID_LED5 0x00020015 #define ASUS_WMI_DEVID_LED6 0x00020016 #define ASUS_WMI_DEVID_MICMUTE_LED 0x00040017 +#define ASUS_WMI_DEVID_MUTE_LED 0x0004001C /* Disable Camera LED */ #define ASUS_WMI_DEVID_CAMERA_LED_NEG 0x00060078 /* 0 = on (unused) */ @@ -140,8 +141,9 @@ #define ASUS_WMI_DEVID_APU_MEM 0x000600C1 -#define ASUS_WMI_DEVID_DGPU_BASE_TGP 0x00120099 +#define ASUS_WMI_DEVID_DGPU_POWER_STATE 0x00120097 #define ASUS_WMI_DEVID_DGPU_SET_TGP 0x00120098 +#define ASUS_WMI_DEVID_DGPU_BASE_TGP 0x00120099 /* gpu mux switch, 0 = dGPU, 1 = Optimus */ #define ASUS_WMI_DEVID_GPU_MUX 0x00090016 diff --git a/include/linux/platform_data/x86/int3472.h b/include/linux/platform_data/x86/int3472.h index a73841dfae27..b1040e36deb8 100644 --- a/include/linux/platform_data/x86/int3472.h +++ b/include/linux/platform_data/x86/int3472.h @@ -25,6 +25,8 @@ #define INT3472_GPIO_TYPE_RESET 0x00 #define INT3472_GPIO_TYPE_POWERDOWN 0x01 #define INT3472_GPIO_TYPE_STROBE 0x02 +#define INT3472_GPIO_TYPE_POWER0 0x07 +#define INT3472_GPIO_TYPE_POWER1 0x08 #define INT3472_GPIO_TYPE_POWER_ENABLE 0x0b #define INT3472_GPIO_TYPE_CLK_ENABLE 0x0c #define INT3472_GPIO_TYPE_PRIVACY_LED 0x0d diff --git a/include/linux/platform_device.h b/include/linux/platform_device.h index 3d5bbcbae730..3bd0f1f0d870 100644 --- a/include/linux/platform_device.h +++ b/include/linux/platform_device.h @@ -344,6 +344,19 @@ static inline void platform_set_drvdata(struct platform_device *pdev, #define builtin_platform_driver(__platform_driver) \ builtin_driver(__platform_driver, platform_driver_register) +/* + * subsys_platform_driver() - Helper macro for drivers that don't do anything + * special in module init/exit but need to register earlier, at + * subsys_initcall level, when built in. This eliminates a lot of + * boilerplate. Each driver may only use this macro once, and calling it + * replaces module_init() and module_exit(). This is meant to be a parallel + * of module_platform_driver() above, but with the init call promoted to + * subsys_initcall() so built-in providers are available earlier during boot. + */ +#define subsys_platform_driver(__platform_driver) \ + subsys_driver(__platform_driver, platform_driver_register, \ + platform_driver_unregister) + /* module_platform_driver_probe() - Helper macro for drivers that don't do * anything special in module init/exit. This eliminates a lot of * boilerplate. Each module may only use this macro once, and @@ -376,6 +389,23 @@ static int __init __platform_driver##_init(void) \ } \ device_initcall(__platform_driver##_init); \ +/* + * subsys_platform_driver_probe() - Helper macro for drivers that don't do + * anything special in device init and have no exit, but need to register + * earlier, at subsys_initcall level. This eliminates some boilerplate. Each + * driver may only use this macro once, and using it replaces subsys_initcall. + * This is meant to be a parallel of builtin_platform_driver_probe above, but + * with the init call promoted to subsys_initcall so the provider is available + * earlier during boot. + */ +#define subsys_platform_driver_probe(__platform_driver, __platform_probe) \ +static int __init __platform_driver##_init(void) \ +{ \ + return platform_driver_probe(&(__platform_driver), \ + __platform_probe); \ +} \ +subsys_initcall(__platform_driver##_init) \ + #define platform_create_bundle(driver, probe, res, n_res, data, size) \ __platform_create_bundle(driver, probe, res, n_res, data, size, THIS_MODULE, KBUILD_MODNAME) extern struct platform_device *__platform_create_bundle( diff --git a/include/linux/pm.h b/include/linux/pm.h index afcaaa37a812..ef3f1310e749 100644 --- a/include/linux/pm.h +++ b/include/linux/pm.h @@ -663,6 +663,99 @@ struct pm_subsys_data { #define DPM_FLAG_SMART_SUSPEND BIT(2) #define DPM_FLAG_MAY_SKIP_RESUME BIT(3) +/** + * struct dev_pm_info - Device power management information. + * + * @power_state: Legacy power state (mostly unused in modern kernels). + * @can_wakeup: Device is capable of generating wakeup signals. + * @async_suspend: Device can be suspended and resumed asynchronously. + * @in_dpm_list: Device is on the dpm_list. + * @is_prepared: Device's ->prepare() callback has run successfully. + * @is_suspended: Device is suspended during a system sleep transition. + * @is_noirq_suspended: Device's noirq suspend callback has run successfully. + * @is_late_suspended: Device's late suspend callback has run successfully. + * @no_pm: Device does not participate in power management transitions. + * @early_init: Device was initialized before standard PM initialization. + * @direct_complete: Device can skip suspend/resume callbacks and remain + * runtime-suspended during system sleep. + * @driver_flags: Driver flags (e.g. %DPM_FLAG_SMART_SUSPEND) set at probe time. + * @lock: Spinlock used for synchronizing PM state transitions and runtime PM + * operations. + * @entry: List head for device power management lists. + * @completion: Completion for synchronization during asynchronous system + * suspend/resume. + * @wakeup: Wakeup source object associated with the device. + * @work_in_progress: Asynchronous PM operation in progress. + * @wakeup_path: Device is in the wakeup path or can wake the system up. + * @syscore: Device participates in syscore power management operations. + * @no_pm_callbacks: Device has no PM callbacks; handled by parent or subsystem. + * @smart_suspend: Driver requested smart-suspend behavior. + * @must_resume: Device must be resumed during system resume. + * @may_skip_resume: Set by subsystems to indicate driver resume callbacks may + * be skipped. + * @out_band_wakeup: Out-of-band wakeup is supported. + * @strict_midlayer: Middle layer code does not want callbacks invoked via + * pm_runtime_force_suspend() / pm_runtime_force_resume(). + * @should_wakeup: Wakeup flag when system sleep is not enabled. + * @suspend_timer: High-resolution timer used for scheduling delayed runtime + * suspend and autosuspend requests. + * @timer_expires: Timer expiration time in nanoseconds monotonic time + * (runtime PM). + * @work: Work structure used for queuing up requests into pm_wq (runtime PM). + * @wait_queue: Wait queue used if any helper functions need to wait for another + * state change to complete (runtime PM). + * @wakeirq: Dedicated wakeup interrupt for the device. + * @usage_count: Device runtime PM usage counter. + * @child_count: Count of active children of the device (runtime PM). + * @disable_depth: Disable counter for runtime PM (runtime PM is enabled when + * this is 0; initial value is 1). + * @idle_notification: Set if ->runtime_idle() is being executed. + * @request_pending: Set if a work item is queued into pm_wq (runtime PM). + * @deferred_resume: Set if ->runtime_resume() should run as soon as + * ->runtime_suspend() completes. + * @needs_force_resume: Indicates the device was forced into suspend by + * pm_runtime_force_suspend() and must be resumed by + * pm_runtime_force_resume(). + * @runtime_auto: User space has allowed the driver to power manage the device + * at runtime via sysfs control attribute; also can be set by + * pm_runtime_allow() or pm_runtime_forbid(). + * @ignore_children: If set, the value of child_count is ignored for runtime + * suspend and idle decisions. + * @no_callbacks: Indicates the device does not use runtime PM callbacks. + * @irq_safe: Indicates runtime PM callbacks will be invoked with the spinlock + * held and interrupts disabled. + * @use_autosuspend: Indicates the device driver supports delayed runtime + * autosuspend. + * @timer_autosuspends: Indicates the runtime PM core should attempt an + * autosuspend rather than a normal suspend when the timer expires. + * @memalloc_noio: Indicates memory allocation during runtime PM transitions + * must avoid I/O (GFP_NOIO). + * @links_count: Number of device links that require runtime PM coordination. + * @request: Type of pending runtime PM request (valid if request_pending is + * set). + * @runtime_status: Runtime PM status of the device. + * @last_status: Last status captured before disabling runtime PM, or + * %RPM_BLOCKED / %RPM_INVALID. + * @runtime_error: Fatal error code returned by a failing callback, blocking + * helpers until cleared. + * @autosuspend_delay: Delay time in milliseconds to be used for runtime + * autosuspend. + * @last_busy: Timestamp in nanoseconds when pm_runtime_mark_last_busy() was + * last called. Used in calculating inactivity periods for autosuspend. + * @active_time: Accumulated time in nanoseconds spent in %RPM_ACTIVE state. + * @suspended_time: Accumulated time in nanoseconds spent in %RPM_SUSPENDED + * state. + * @accounting_timestamp: Timestamp in nanoseconds of the last runtime PM state + * accounting update. + * @subsys_data: Subsystem-specific power management data. + * @set_latency_tolerance: Callback for setting latency tolerance. + * @qos: Per-device PM Quality of Service (QoS) constraints. + * @detach_power_off: Indicates device should be detached from PM domain on + * power off. + * + * Device power management information stored in the "power" member of struct + * device. + */ struct dev_pm_info { pm_message_t power_state; bool can_wakeup:1; diff --git a/include/linux/pm_domain.h b/include/linux/pm_domain.h index f925614aebdb..14e0e346c610 100644 --- a/include/linux/pm_domain.h +++ b/include/linux/pm_domain.h @@ -121,6 +121,14 @@ struct dev_pm_domain_list { * powered-off until the ->sync_state() callback is * invoked. This flag informs genpd to allow a * power-off without waiting for ->sync_state(). + * + * GENPD_FLAG_POWER_UNKNOWN: Use this flag to inform genpd that its initial + * status for the PM domain is set to powered off, + * which may not correctly reflect the state of the + * HW, as it's unknown. If the PM domain becomes + * powered on during boot, genpd will prevent it + * from being powered off until the ->sync_state + * callback is invoked for it. */ #define GENPD_FLAG_PM_CLK (1U << 0) #define GENPD_FLAG_IRQ_SAFE (1U << 1) @@ -133,6 +141,7 @@ struct dev_pm_domain_list { #define GENPD_FLAG_DEV_NAME_FW (1U << 8) #define GENPD_FLAG_NO_SYNC_STATE (1U << 9) #define GENPD_FLAG_NO_STAY_ON (1U << 10) +#define GENPD_FLAG_POWER_UNKNOWN (1U << 11) enum gpd_status { GENPD_STATE_ON = 0, /* PM domain is on */ diff --git a/include/linux/pm_qos.h b/include/linux/pm_qos.h index 6cea4455f867..439a9e779d81 100644 --- a/include/linux/pm_qos.h +++ b/include/linux/pm_qos.h @@ -37,6 +37,8 @@ enum pm_qos_flags_status { #define PM_QOS_LATENCY_TOLERANCE_NO_CONSTRAINT (-1) #define PM_QOS_FLAG_NO_POWER_OFF (1 << 0) +/* latency value applies to system-wide suspend/s2idle */ +#define PM_QOS_FLAG_LATENCY_SYS (2 << 0) enum pm_qos_type { PM_QOS_UNITIALIZED, @@ -217,6 +219,12 @@ static inline s32 dev_pm_qos_raw_resume_latency(struct device *dev) PM_QOS_RESUME_LATENCY_NO_CONSTRAINT : pm_qos_read_value(&dev->power.qos->resume_latency); } + +static inline s32 dev_pm_qos_raw_flags(struct device *dev) +{ + return IS_ERR_OR_NULL(dev->power.qos) ? + 0 : READ_ONCE(dev->power.qos->flags.effective_flags); +} #else static inline enum pm_qos_flags_status __dev_pm_qos_flags(struct device *dev, s32 mask) @@ -298,6 +306,7 @@ static inline s32 dev_pm_qos_raw_resume_latency(struct device *dev) { return PM_QOS_RESUME_LATENCY_NO_CONSTRAINT; } +static inline s32 dev_pm_qos_raw_flags(struct device *dev) { return 0; } #endif static inline int freq_qos_request_active(struct freq_qos_request *req) diff --git a/include/linux/pm_runtime.h b/include/linux/pm_runtime.h index 64921b10ac74..322e3b17f987 100644 --- a/include/linux/pm_runtime.h +++ b/include/linux/pm_runtime.h @@ -137,13 +137,14 @@ static inline void pm_runtime_put_noidle(struct device *dev) * pm_runtime_suspended - Check whether or not a device is runtime-suspended. * @dev: Target device. * - * Return %true if runtime PM is enabled for @dev and its runtime PM status is - * %RPM_SUSPENDED, or %false otherwise. - * * Note that the return value of this function can only be trusted if it is * called under the runtime PM lock of @dev or under conditions in which * runtime PM cannot be either disabled or enabled for @dev and its runtime PM * status cannot change. + * + * Return: + * * %true: @dev has runtime PM enabled and its status is %RPM_SUSPENDED. + * * %false: Otherwise. */ static inline bool pm_runtime_suspended(struct device *dev) { @@ -155,13 +156,14 @@ static inline bool pm_runtime_suspended(struct device *dev) * pm_runtime_active - Check whether or not a device is runtime-active. * @dev: Target device. * - * Return %true if runtime PM is disabled for @dev or its runtime PM status is - * %RPM_ACTIVE, or %false otherwise. - * * Note that the return value of this function can only be trusted if it is * called under the runtime PM lock of @dev or under conditions in which * runtime PM cannot be either disabled or enabled for @dev and its runtime PM * status cannot change. + * + * Return: + * * %true: Runtime PM is disabled for @dev or its status is %RPM_ACTIVE. + * * %false: Otherwise. */ static inline bool pm_runtime_active(struct device *dev) { @@ -173,12 +175,13 @@ static inline bool pm_runtime_active(struct device *dev) * pm_runtime_status_suspended - Check if runtime PM status is "suspended". * @dev: Target device. * - * Return %true if the runtime PM status of @dev is %RPM_SUSPENDED, or %false - * otherwise, regardless of whether or not runtime PM has been enabled for @dev. - * * Note that the return value of this function can only be trusted if it is * called under the runtime PM lock of @dev or under conditions in which the * runtime PM status of @dev cannot change. + * + * Return: + * * %true: Runtime PM status of @dev is %RPM_SUSPENDED. + * * %false: Otherwise. */ static inline bool pm_runtime_status_suspended(struct device *dev) { @@ -189,11 +192,13 @@ static inline bool pm_runtime_status_suspended(struct device *dev) * pm_runtime_enabled - Check if runtime PM is enabled. * @dev: Target device. * - * Return %true if runtime PM is enabled for @dev or %false otherwise. - * * Note that the return value of this function can only be trusted if it is * called under the runtime PM lock of @dev or under conditions in which * runtime PM cannot be either disabled or enabled for @dev. + * + * Return: + * * %true: Runtime PM is enabled for @dev. + * * %false: Otherwise. */ static inline bool pm_runtime_enabled(struct device *dev) { @@ -205,6 +210,10 @@ static inline bool pm_runtime_enabled(struct device *dev) * @dev: Target device. * * Do not call this function outside system suspend/resume code paths. + * + * Return: + * * %true: Runtime PM enabling is blocked for @dev. + * * %false: Otherwise. */ static inline bool pm_runtime_blocked(struct device *dev) { @@ -215,8 +224,9 @@ static inline bool pm_runtime_blocked(struct device *dev) * pm_runtime_has_no_callbacks - Check if runtime PM callbacks may be present. * @dev: Target device. * - * Return %true if @dev is a special device without runtime PM callbacks or - * %false otherwise. + * Return: + * * %true: @dev is marked as having no runtime PM callbacks. + * * %false: Otherwise. */ static inline bool pm_runtime_has_no_callbacks(struct device *dev) { @@ -239,9 +249,11 @@ static inline void pm_runtime_mark_last_busy(struct device *dev) * pm_runtime_is_irq_safe - Check if runtime PM can work in interrupt context. * @dev: Target device. * - * Return %true if @dev has been marked as an "IRQ-safe" device (with respect - * to runtime PM), in which case its runtime PM callabcks can be expected to - * work correctly when invoked from interrupt handlers. + * Return: + * * %true: @dev has been marked as an "IRQ-safe" device, in which case its + * runtime PM callbacks can be expected to work correctly from interrupt + * handlers. + * * %false: Otherwise. */ static inline bool pm_runtime_is_irq_safe(struct device *dev) { @@ -340,25 +352,25 @@ static inline int pm_runtime_force_resume(struct device *dev) { return -ENXIO; } #endif /* CONFIG_PM_SLEEP */ /** - * pm_runtime_idle - Conditionally set up autosuspend of a device or suspend it. + * pm_runtime_idle - Conditionally initiate autosuspend of a device or suspend it. * @dev: Target device. * * Invoke the "idle check" callback of @dev and, depending on its return value, - * set up autosuspend of @dev or suspend it (depending on whether or not + * initiate autosuspend of @dev or suspend it (depending on whether or not * autosuspend has been enabled for it). * * Return: - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change - * ongoing or device not in %RPM_ACTIVE state. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -EINPROGRESS: Suspend already in progress. - * * -ENOSYS: CONFIG_PM not enabled. - * Other values and conditions for the above values are possible as returned by - * Runtime PM idle and suspend callbacks. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change + * ongoing or device not in %RPM_ACTIVE state. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-EINPROGRESS: Suspend already in progress. + * * %-ENOSYS: %CONFIG_PM not enabled. + * * Other values and conditions for the above values are possible as returned + * by Runtime PM idle and suspend callbacks. */ static inline int pm_runtime_idle(struct device *dev) { @@ -370,17 +382,17 @@ static inline int pm_runtime_idle(struct device *dev) * @dev: Target device. * * Return: - * * 1: Success; device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change - * ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -ENOSYS: CONFIG_PM not enabled. - * Other values and conditions for the above values are possible as returned by - * Runtime PM suspend callbacks. + * * %1: Success; device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change + * ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-ENOSYS: %CONFIG_PM not enabled. + * * Other values and conditions for the above values are possible as returned + * by Runtime PM suspend callbacks. */ static inline int pm_runtime_suspend(struct device *dev) { @@ -388,26 +400,26 @@ static inline int pm_runtime_suspend(struct device *dev) } /** - * pm_runtime_autosuspend - Update the last access time and set up autosuspend + * pm_runtime_autosuspend - Update the last access time and initiate autosuspend * of a device. * @dev: Target device. * - * First update the last access time, then set up autosuspend of @dev or suspend - * it (depending on whether or not autosuspend is enabled for it) without - * engaging its "idle check" callback. + * First update the last access time, then initiate autosuspend of @dev or + * suspend it (depending on whether or not autosuspend is enabled for it) + * without engaging its "idle check" callback. * * Return: - * * 1: Success; device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change - * ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -ENOSYS: CONFIG_PM not enabled. - * Other values and conditions for the above values are possible as returned by - * Runtime PM suspend callbacks. + * * %1: Success; device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change + * ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-ENOSYS: %CONFIG_PM not enabled. + * * Other values and conditions for the above values are possible as returned + * by Runtime PM suspend callbacks. */ static inline int pm_runtime_autosuspend(struct device *dev) { @@ -418,6 +430,11 @@ static inline int pm_runtime_autosuspend(struct device *dev) /** * pm_runtime_resume - Resume a device synchronously. * @dev: Target device. + * + * Return: + * * %1: Success; @dev is already %RPM_ACTIVE. + * * %0: Success. + * * Error code on failure. */ static inline int pm_runtime_resume(struct device *dev) { @@ -425,22 +442,22 @@ static inline int pm_runtime_resume(struct device *dev) } /** - * pm_request_idle - Queue up "idle check" execution for a device. + * pm_request_idle - Request an asynchronous idle check for a device. * @dev: Target device. * - * Queue up a work item to run an equivalent of pm_runtime_idle() for @dev - * asynchronously. + * Asynchronously request the PM core to evaluate whether @dev can be idled + * or suspended, invoking its ->runtime_idle() callback if provided. * * Return: - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change - * ongoing or device not in %RPM_ACTIVE state. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -EINPROGRESS: Suspend already in progress. - * * -ENOSYS: CONFIG_PM not enabled. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter non-zero, Runtime PM status change + * ongoing or device not in %RPM_ACTIVE state. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-EINPROGRESS: Suspend already in progress. + * * %-ENOSYS: %CONFIG_PM not enabled. */ static inline int pm_request_idle(struct device *dev) { @@ -448,8 +465,16 @@ static inline int pm_request_idle(struct device *dev) } /** - * pm_request_resume - Queue up runtime-resume of a device. + * pm_request_resume - Request an asynchronous runtime resume for a device. * @dev: Target device. + * + * Asynchronously request the PM core to resume @dev to %RPM_ACTIVE state + * without modifying its usage counter. + * + * Return: + * * %1: Success; @dev is already %RPM_ACTIVE. + * * %0: Success. + * * Error code on failure. */ static inline int pm_request_resume(struct device *dev) { @@ -457,24 +482,23 @@ static inline int pm_request_resume(struct device *dev) } /** - * pm_request_autosuspend - Update the last access time and queue up autosuspend - * of a device. + * pm_request_autosuspend - Update access time and request delayed suspension. * @dev: Target device. * - * Update the last access time of a device and queue up a work item to run an - * equivalent pm_runtime_autosuspend() for @dev asynchronously. + * Update the last access time of @dev and asynchronously request the PM core + * to suspend it after the autosuspend delay has elapsed. * * Return: - * * 1: Success; device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change - * ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -EINPROGRESS: Suspend already in progress. - * * -ENOSYS: CONFIG_PM not enabled. + * * %1: Success; device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter non-zero or Runtime PM status change + * ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-EINPROGRESS: Suspend already in progress. + * * %-ENOSYS: %CONFIG_PM not enabled. */ static inline int pm_request_autosuspend(struct device *dev) { @@ -483,11 +507,16 @@ static inline int pm_request_autosuspend(struct device *dev) } /** - * pm_runtime_get - Bump up usage counter and queue up resume of a device. + * pm_runtime_get - Increment usage counter and request asynchronous resume. * @dev: Target device. * - * Bump up the runtime PM usage counter of @dev and queue up a work item to - * carry out runtime-resume of it. + * Increment the runtime PM usage counter of @dev and, if the device is + * currently suspended, asynchronously request the PM core to resume it. + * + * Return: + * * %1: Success; @dev is already %RPM_ACTIVE. + * * %0: Success; runtime-resume was queued. + * * Error code on failure. */ static inline int pm_runtime_get(struct device *dev) { @@ -501,12 +530,15 @@ static inline int pm_runtime_get(struct device *dev) * Bump up the runtime PM usage counter of @dev and carry out runtime-resume of * it synchronously. * - * The possible return values of this function are the same as for - * pm_runtime_resume() and the runtime PM usage counter of @dev remains - * incremented in all cases, even if it returns an error code. - * Consider using pm_runtime_resume_and_get() instead of it, especially - * if its return value is checked by the caller, as this is likely to result - * in cleaner code. + * Note that the runtime PM usage counter of @dev remains incremented in all + * cases, even if it returns an error code. Consider using + * pm_runtime_resume_and_get() instead, especially if the return value is + * checked by the caller, as this is likely to result in cleaner code. + * + * Return: + * * %1: Success; @dev is already %RPM_ACTIVE. + * * %0: Success. + * * Error code on failure. */ static inline int pm_runtime_get_sync(struct device *dev) { @@ -531,8 +563,11 @@ static inline int pm_runtime_get_active(struct device *dev, int rpmflags) * @dev: Target device. * * Resume @dev synchronously and if that is successful, increment its runtime - * PM usage counter. Return 0 if the runtime PM usage counter of @dev has been - * incremented or a negative error code otherwise. + * PM usage counter. + * + * Return: + * * %0: Success; @dev is active and its usage counter has been incremented. + * * Negative error code on failure; usage counter is unchanged. */ static inline int pm_runtime_resume_and_get(struct device *dev) { @@ -540,11 +575,12 @@ static inline int pm_runtime_resume_and_get(struct device *dev) } /** - * pm_runtime_put - Drop device usage counter and queue up "idle check" if 0. + * pm_runtime_put - Drop device usage counter and request asynchronous idle check. * @dev: Target device. * - * Decrement the runtime PM usage counter of @dev and if it turns out to be - * equal to 0, queue up a work item for @dev like in pm_request_idle(). + * Decrement the runtime PM usage counter of @dev. If the counter reaches zero + * and the device has no active child dependencies, asynchronously request the + * PM core to idle or suspend the device. */ static inline void pm_runtime_put(struct device *dev) { @@ -559,16 +595,16 @@ static inline void pm_runtime_put(struct device *dev) * equal to 0, queue up a work item for @dev like in pm_request_autosuspend(). * * Return: - * * 1: Success. Usage counter dropped to zero, but device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status - * change ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -EINPROGRESS: Suspend already in progress. - * * -ENOSYS: CONFIG_PM not enabled. + * * %1: Success. Usage counter dropped to zero, but device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status + * change ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-EINPROGRESS: Suspend already in progress. + * * %-ENOSYS: %CONFIG_PM not enabled. */ static inline int __pm_runtime_put_autosuspend(struct device *dev) { @@ -576,25 +612,25 @@ static inline int __pm_runtime_put_autosuspend(struct device *dev) } /** - * pm_runtime_put_autosuspend - Update the last access time of a device, drop - * its usage counter and queue autosuspend if the usage counter becomes 0. + * pm_runtime_put_autosuspend - Update the last access time, drop usage counter + * and request autosuspend. * @dev: Target device. * - * Update the last access time of @dev, decrement runtime PM usage counter of - * @dev and if it turns out to be equal to 0, queue up a work item for @dev like - * in pm_request_autosuspend(). + * Update the last access time of @dev and decrement its runtime PM usage + * counter. If the counter drops to zero, asynchronously request the PM core to + * suspend the device once its autosuspend delay has elapsed. * * Return: - * * 1: Success. Usage counter dropped to zero, but device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status - * change ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -EINPROGRESS: Suspend already in progress. - * * -ENOSYS: CONFIG_PM not enabled. + * * %1: Success. Usage counter dropped to zero, but device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status + * change ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-EINPROGRESS: Suspend already in progress. + * * %-ENOSYS: %CONFIG_PM not enabled. */ static inline int pm_runtime_put_autosuspend(struct device *dev) { @@ -653,26 +689,28 @@ DEFINE_GUARD_COND(pm_runtime_active_auto, _try_enabled, * pm_runtime_put_sync - Drop device usage counter and run "idle check" if 0. * @dev: Target device. * - * Decrement the runtime PM usage counter of @dev and if it turns out to be - * equal to 0, invoke the "idle check" callback of @dev and, depending on its - * return value, set up autosuspend of @dev or suspend it (depending on whether - * or not autosuspend has been enabled for it). + * Decrement the runtime PM usage counter of @dev. If the counter drops to zero, + * synchronously evaluate and trigger idle/suspend handling. + * + * Note that this does not update the last access time, but it does respect + * existing autosuspend timers. If @dev uses autosuspend, consider using + * pm_runtime_put_sync_autosuspend() or pm_runtime_put_sync_suspend() instead. * * The runtime PM usage counter of @dev remains decremented in all cases, even * if it returns an error code. * * Return: - * * 1: Success. Usage counter dropped to zero, but device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status - * change ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -ENOSYS: CONFIG_PM not enabled. - * Other values and conditions for the above values are possible as returned by - * Runtime PM suspend callbacks. + * * %1: Success. Usage counter dropped to zero, but device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status + * change ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-ENOSYS: %CONFIG_PM not enabled. + * * Other values and conditions for the above values are possible as returned + * by Runtime PM suspend callbacks. */ static inline int pm_runtime_put_sync(struct device *dev) { @@ -683,24 +721,28 @@ static inline int pm_runtime_put_sync(struct device *dev) * pm_runtime_put_sync_suspend - Drop device usage counter and suspend if 0. * @dev: Target device. * - * Decrement the runtime PM usage counter of @dev and if it turns out to be - * equal to 0, carry out runtime-suspend of @dev synchronously. + * Decrement the runtime PM usage counter of @dev. If the counter drops to zero, + * suspend the device synchronously. + * + * This API differs from pm_runtime_put_sync() and + * pm_runtime_put_sync_autosuspend() in that it ignores any outstanding + * autosuspend delays. * * The runtime PM usage counter of @dev remains decremented in all cases, even * if it returns an error code. * * Return: - * * 1: Success. Usage counter dropped to zero, but device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status - * change ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -ENOSYS: CONFIG_PM not enabled. - * Other values and conditions for the above values are possible as returned by - * Runtime PM suspend callbacks. + * * %1: Success. Usage counter dropped to zero, but device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status + * change ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-ENOSYS: %CONFIG_PM not enabled. + * * Other values and conditions for the above values are possible as returned + * by Runtime PM suspend callbacks. */ static inline int pm_runtime_put_sync_suspend(struct device *dev) { @@ -712,27 +754,28 @@ static inline int pm_runtime_put_sync_suspend(struct device *dev) * drop device usage counter and autosuspend if 0. * @dev: Target device. * - * Update the last access time of @dev, decrement the runtime PM usage counter - * of @dev and if it turns out to be equal to 0, set up autosuspend of @dev or - * suspend it synchronously (depending on whether or not autosuspend has been - * enabled for it). + * Update the last access time of @dev and decrement its runtime PM usage + * counter. If the counter drops to zero, synchronously suspend the device (or + * schedule autosuspend if the delay has not elapsed). + * + * Prefer this API over pm_runtime_put_sync() for devices that use autosuspend. * * The runtime PM usage counter of @dev remains decremented in all cases, even * if it returns an error code. * * Return: - * * 1: Success. Usage counter dropped to zero, but device was already suspended. - * * 0: Success. - * * -EINVAL: Runtime PM error. - * * -EACCES: Runtime PM disabled. - * * -EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status - * change ongoing. - * * -EBUSY: Runtime PM child_count non-zero. - * * -EPERM: Device PM QoS resume latency 0. - * * -EINPROGRESS: Suspend already in progress. - * * -ENOSYS: CONFIG_PM not enabled. - * Other values and conditions for the above values are possible as returned by - * Runtime PM suspend callbacks. + * * %1: Success. Usage counter dropped to zero, but device was already suspended. + * * %0: Success. + * * %-EINVAL: Runtime PM error. + * * %-EACCES: Runtime PM disabled. + * * %-EAGAIN: Runtime PM usage counter became non-zero or Runtime PM status + * change ongoing. + * * %-EBUSY: Runtime PM child_count non-zero. + * * %-EPERM: Device PM QoS resume latency 0. + * * %-EINPROGRESS: Suspend already in progress. + * * %-ENOSYS: %CONFIG_PM not enabled. + * * Other values and conditions for the above values are possible as returned + * by Runtime PM suspend callbacks. */ static inline int pm_runtime_put_sync_autosuspend(struct device *dev) { @@ -741,13 +784,22 @@ static inline int pm_runtime_put_sync_autosuspend(struct device *dev) } /** - * pm_runtime_set_active - Set runtime PM status to "active". + * pm_runtime_set_active - Set runtime PM status to "active" and clear errors. * @dev: Target device. * - * Set the runtime PM status of @dev to %RPM_ACTIVE and ensure that dependencies - * of it will be taken into account. + * Set the runtime PM status of @dev to %RPM_ACTIVE and ensure that its + * dependencies will be taken into account. Also clear the device's error + * status (@dev->power.runtime_error). + * + * It is only valid to call this function if runtime PM is disabled or if + * @dev->power.runtime_error is set. * - * It is not valid to call this function for devices with runtime PM enabled. + * This will fail if suppliers cannot be resumed, or if the parent is not in + * the correct state. + * + * Return: + * * %0: Success. + * * Error code on failure. */ static inline int pm_runtime_set_active(struct device *dev) { @@ -755,13 +807,19 @@ static inline int pm_runtime_set_active(struct device *dev) } /** - * pm_runtime_set_suspended - Set runtime PM status to "suspended". + * pm_runtime_set_suspended - Set runtime PM status to "suspended" and clear errors. * @dev: Target device. * - * Set the runtime PM status of @dev to %RPM_SUSPENDED and ensure that - * dependencies of it will be taken into account. + * Set the runtime PM status of @dev to %RPM_SUSPENDED and ensure that its + * dependencies will be taken into account. Also clear the device's error + * status (@dev->power.runtime_error). * - * It is not valid to call this function for devices with runtime PM enabled. + * It is only valid to call this function if runtime PM is disabled or if + * @dev->power.runtime_error is set. + * + * Return: + * * %0: Success. + * * Error code on failure. */ static inline int pm_runtime_set_suspended(struct device *dev) { @@ -777,9 +835,8 @@ static inline int pm_runtime_set_suspended(struct device *dev) * * If the counter is zero when this function runs and there is a pending runtime * resume request for @dev, it will be resumed. If the counter is still zero at - * that point, all of the pending runtime PM requests for @dev will be canceled - * and all runtime PM operations in progress involving it will be waited for to - * complete. + * that point, this function cancels all pending runtime PM requests for @dev + * and waits for its runtime PM operations to complete (if any). * * For each invocation of this function for @dev, there must be a matching * pm_runtime_enable() call, so that runtime PM is eventually enabled for it diff --git a/include/linux/posix-timers.h b/include/linux/posix-timers.h index 9a1a0c61361c..00767acbc111 100644 --- a/include/linux/posix-timers.h +++ b/include/linux/posix-timers.h @@ -66,38 +66,6 @@ struct cpu_timer { struct task_struct __rcu *handling; }; -static inline bool cpu_timer_enqueue(struct timerqueue_head *head, - struct cpu_timer *ctmr) -{ - ctmr->head = head; - return timerqueue_add(head, &ctmr->node); -} - -static inline bool cpu_timer_queued(struct cpu_timer *ctmr) -{ - return !!ctmr->head; -} - -static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr) -{ - if (cpu_timer_queued(ctmr)) { - timerqueue_del(ctmr->head, &ctmr->node); - ctmr->head = NULL; - return true; - } - return false; -} - -static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr) -{ - return ctmr->node.expires; -} - -static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp) -{ - ctmr->node.expires = exp; -} - static inline void posix_cputimers_init(struct posix_cputimers *pct) { memset(pct, 0, sizeof(*pct)); @@ -224,14 +192,15 @@ struct k_itimer { } ____cacheline_aligned_in_smp; void run_posix_cpu_timers(void); -void posix_cpu_timers_exit(struct task_struct *task); -void posix_cpu_timers_exit_group(struct task_struct *task); void set_process_cpu_timer(struct task_struct *task, unsigned int clock_idx, u64 *newval, u64 *oldval); int update_rlimit_cpu(struct task_struct *task, unsigned long rlim_new); #ifdef CONFIG_POSIX_TIMERS +void posixtimer_exec(void); +void posixtimer_exit(bool group_dead); + static inline void posixtimer_putref(struct k_itimer *tmr) { if (rcuref_put(&tmr->rcuref)) @@ -259,6 +228,8 @@ static inline bool posixtimer_valid(const struct k_itimer *timer) return !(val & 0x1UL); } #else /* CONFIG_POSIX_TIMERS */ +static inline void posixtimer_exec(void) { } +static inline void posixtimer_exit(bool group_dead) { } static inline void posixtimer_sigqueue_getref(struct sigqueue *q) { } static inline void posixtimer_sigqueue_putref(struct sigqueue *q) { } #endif /* !CONFIG_POSIX_TIMERS */ diff --git a/include/linux/posix_acl.h b/include/linux/posix_acl.h index 62d497763e25..caf500bed993 100644 --- a/include/linux/posix_acl.h +++ b/include/linux/posix_acl.h @@ -74,20 +74,20 @@ extern int __posix_acl_create(struct posix_acl **, gfp_t, umode_t *); extern int __posix_acl_chmod(struct posix_acl **, gfp_t, umode_t); extern struct posix_acl *get_posix_acl(struct inode *, int); -int set_posix_acl(struct mnt_idmap *, struct dentry *, int, +int set_posix_acl(const struct mnt_idmap *, struct dentry *, int, struct posix_acl *); struct posix_acl *get_cached_acl_rcu(struct inode *inode, int type); struct posix_acl *posix_acl_clone(const struct posix_acl *acl, gfp_t flags); #ifdef CONFIG_FS_POSIX_ACL -int posix_acl_chmod(struct mnt_idmap *, struct dentry *, umode_t); +int posix_acl_chmod(const struct mnt_idmap *, struct dentry *, umode_t); extern int posix_acl_create(struct inode *, umode_t *, struct posix_acl **, struct posix_acl **); -int posix_acl_update_mode(struct mnt_idmap *, struct inode *, umode_t *, +int posix_acl_update_mode(const struct mnt_idmap *, struct inode *, umode_t *, struct posix_acl **); -int simple_set_acl(struct mnt_idmap *, struct dentry *, +int simple_set_acl(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); extern int simple_acl_create(struct inode *, struct inode *); @@ -96,7 +96,7 @@ void set_cached_acl(struct inode *inode, int type, struct posix_acl *acl); void forget_cached_acl(struct inode *inode, int type); void forget_all_cached_acls(struct inode *inode); int posix_acl_valid(struct user_namespace *, const struct posix_acl *); -int posix_acl_permission(struct mnt_idmap *, struct inode *, +int posix_acl_permission(const struct mnt_idmap *, struct inode *, const struct posix_acl *, int); static inline void cache_no_acl(struct inode *inode) @@ -105,16 +105,16 @@ static inline void cache_no_acl(struct inode *inode) inode->i_default_acl = NULL; } -int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); int posix_acl_listxattr(struct inode *inode, char **buffer, ssize_t *remaining_size); #else -static inline int posix_acl_chmod(struct mnt_idmap *idmap, +static inline int posix_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode) { return 0; @@ -141,21 +141,21 @@ static inline void forget_all_cached_acls(struct inode *inode) { } -static inline int vfs_set_acl(struct mnt_idmap *idmap, +static inline int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct posix_acl *acl) { return -EOPNOTSUPP; } -static inline struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +static inline struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return ERR_PTR(-EOPNOTSUPP); } -static inline int vfs_remove_acl(struct mnt_idmap *idmap, +static inline int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return -EOPNOTSUPP; diff --git a/include/linux/power/bq27xxx_battery.h b/include/linux/power/bq27xxx_battery.h index d56e1276aafe..0b833c4bb583 100644 --- a/include/linux/power/bq27xxx_battery.h +++ b/include/linux/power/bq27xxx_battery.h @@ -33,6 +33,7 @@ enum bq27xxx_chip { BQ27441, BQ27621, BQ27Z561, + BQ27Z746, BQ28Z610, BQ34Z100, BQ78Z100, diff --git a/include/linux/power_supply.h b/include/linux/power_supply.h index e749d2189335..bee659427166 100644 --- a/include/linux/power_supply.h +++ b/include/linux/power_supply.h @@ -104,6 +104,14 @@ enum { POWER_SUPPLY_SCOPE_DEVICE, }; +enum { + POWER_SUPPLY_LOAD_SWITCH_UNKNOWN = 0, + POWER_SUPPLY_LOAD_SWITCH_ON, + POWER_SUPPLY_LOAD_SWITCH_OFF, + POWER_SUPPLY_LOAD_SWITCH_STANDBY, + POWER_SUPPLY_LOAD_SWITCH_SHIP, +}; + enum power_supply_property { /* Properties of type `int' */ POWER_SUPPLY_PROP_STATUS = 0, @@ -182,6 +190,7 @@ enum power_supply_property { POWER_SUPPLY_PROP_MANUFACTURE_DAY, POWER_SUPPLY_PROP_INTERNAL_RESISTANCE, POWER_SUPPLY_PROP_STATE_OF_HEALTH, + POWER_SUPPLY_PROP_LOAD_SWITCH, /* Properties of type `const char *' */ POWER_SUPPLY_PROP_MODEL_NAME, POWER_SUPPLY_PROP_MANUFACTURER, @@ -259,9 +268,9 @@ struct power_supply_config { struct power_supply_desc { const char *name; enum power_supply_type type; - u8 charge_behaviours; u32 charge_types; u32 usb_types; + u32 load_switches; const enum power_supply_property *properties; size_t num_properties; @@ -295,14 +304,17 @@ struct power_supply_desc { */ int (*init)(struct power_supply *psy); + /* For APM emulation, think legacy userspace. */ + int use_for_apm; + /* * Set if thermal zone should not be created for this power supply. * For example for virtual supplies forwarding calls to actual * sensors or other supplies. */ bool no_thermal; - /* For APM emulation, think legacy userspace. */ - int use_for_apm; + + u8 charge_behaviours; }; struct power_supply_ext { @@ -414,9 +426,9 @@ struct power_supply_vbat_ri_table { * @charge_voltage_max_uv: maintenance charging voltage that is usually a bit * lower than the constant_charge_voltage_max_uv. We can apply this settings * charge_current_max_ua until we get back up to this voltage. - * @safety_timer_minutes: maintenance charging safety timer, with an expiry - * time in minutes. We will only use maintenance charging in this setting - * for a certain amount of time, then we will first move to the next + * @charge_safety_timer_minutes: maintenance charging safety timer, with an + * expiry time in minutes. We will only use maintenance charging in this + * setting for a certain amount of time, then we will first move to the next * maintenance charge current and voltage pair in respective array and wait * for the next safety timer timeout, or, if we reached the last maintencance * charging setting, disable charging until we reach diff --git a/include/linux/preempt.h b/include/linux/preempt.h index 8299657f0f86..06ff44a4b8b6 100644 --- a/include/linux/preempt.h +++ b/include/linux/preempt.h @@ -168,10 +168,6 @@ static __always_inline unsigned char interrupt_context_level(void) #define in_softirq() (softirq_count()) #define in_interrupt() (irq_count()) -#define hardirq_disable_count() ((preempt_count() & HARDIRQ_DISABLE_MASK) >> HARDIRQ_DISABLE_SHIFT) -#define hardirq_disable_enter() __preempt_count_add_return(HARDIRQ_DISABLE_OFFSET) -#define hardirq_disable_exit() __preempt_count_sub_return(HARDIRQ_DISABLE_OFFSET) - /* * The preempt_count offset after preempt_disable(); */ @@ -502,21 +498,11 @@ DEFINE_LOCK_GUARD_0(preempt_notrace, preempt_disable_notrace(), preempt_enable_n #ifdef CONFIG_PREEMPT_DYNAMIC -extern bool preempt_model_none(void); -extern bool preempt_model_voluntary(void); extern bool preempt_model_full(void); extern bool preempt_model_lazy(void); #else -static inline bool preempt_model_none(void) -{ - return IS_ENABLED(CONFIG_PREEMPT_NONE); -} -static inline bool preempt_model_voluntary(void) -{ - return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY); -} static inline bool preempt_model_full(void) { return IS_ENABLED(CONFIG_PREEMPT); @@ -529,6 +515,16 @@ static inline bool preempt_model_lazy(void) #endif +static inline bool preempt_model_none(void) +{ + return IS_ENABLED(CONFIG_PREEMPT_NONE); +} + +static inline bool preempt_model_voluntary(void) +{ + return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY); +} + static inline bool preempt_model_rt(void) { return IS_ENABLED(CONFIG_PREEMPT_RT); diff --git a/include/linux/property.h b/include/linux/property.h index 907c790a3f01..9143fe4f5c05 100644 --- a/include/linux/property.h +++ b/include/linux/property.h @@ -547,6 +547,11 @@ unsigned int fwnode_graph_get_endpoint_count(const struct fwnode_handle *fwnode, for (child = fwnode_graph_get_next_endpoint(fwnode, NULL); child; \ child = fwnode_graph_get_next_endpoint(fwnode, child)) +#define fwnode_graph_for_each_endpoint_scoped(fwnode, child) \ + for (struct fwnode_handle *child __free(fwnode_handle) = \ + fwnode_graph_get_next_endpoint(fwnode, NULL); \ + child; child = fwnode_graph_get_next_endpoint(fwnode, child)) + int fwnode_graph_parse_endpoint(const struct fwnode_handle *fwnode, struct fwnode_endpoint *endpoint); diff --git a/include/linux/psp-sev.h b/include/linux/psp-sev.h index 03a79786df1d..fab62228f981 100644 --- a/include/linux/psp-sev.h +++ b/include/linux/psp-sev.h @@ -49,7 +49,7 @@ #define SEV_FW_BLOB_MAX_SIZE 0x4000 /* 16KB */ -/** +/* * SEV platform state */ enum sev_state { @@ -60,7 +60,7 @@ enum sev_state { SEV_STATE_MAX }; -/** +/* * SEV platform and guest management commands */ enum sev_cmd { @@ -157,6 +157,7 @@ enum sev_cmd { * struct sev_data_init - INIT command parameters * * @flags: processing flags + * @reserved: reserved * @tmr_address: system physical address used for SEV-ES * @tmr_len: len of tmr_address */ @@ -174,6 +175,7 @@ struct sev_data_init { * @flags: processing flags * @tmr_address: system physical address used for SEV-ES * @tmr_len: len of tmr_address + * @reserved: reserved * @nv_address: system physical address used for PSP NV storage * @nv_len: len of nv_address */ @@ -201,12 +203,13 @@ struct sev_data_pek_csr { } __packed; /** - * struct sev_data_cert_import - PEK_CERT_IMPORT command parameters + * struct sev_data_pek_cert_import - PEK_CERT_IMPORT command parameters * - * @pek_address: PEK certificate chain - * @pek_len: len of PEK certificate - * @oca_address: OCA certificate chain - * @oca_len: len of OCA certificate + * @pek_cert_address: PEK certificate chain + * @pek_cert_len: len of PEK certificate + * @reserved: reserved + * @oca_cert_address: OCA certificate chain + * @oca_cert_len: len of OCA certificate */ struct sev_data_pek_cert_import { u64 pek_cert_address; /* In */ @@ -240,8 +243,9 @@ struct sev_data_get_id { /** * struct sev_data_pdh_cert_export - PDH_CERT_EXPORT command parameters * - * @pdh_address: PDH certificate address - * @pdh_len: len of PDH certificate + * @pdh_cert_address: PDH certificate address + * @pdh_cert_len: len of PDH certificate + * @reserved: reserved * @cert_chain_address: PDH certificate chain * @cert_chain_len: len of PDH certificate chain */ @@ -304,6 +308,7 @@ struct sev_data_guest_status { * @policy: guest launch policy * @dh_cert_address: physical address of DH certificate blob * @dh_cert_len: len of DH certificate blob + * @reserved: reserved * @session_address: physical address of session parameters * @session_len: len of session parameters */ @@ -321,6 +326,7 @@ struct sev_data_launch_start { * struct sev_data_launch_update_data - LAUNCH_UPDATE_DATA command parameter * * @handle: handle of the VM to update + * @reserved: reserved * @len: len of memory to be encrypted * @address: physical address of memory region to encrypt */ @@ -335,6 +341,7 @@ struct sev_data_launch_update_data { * struct sev_data_launch_update_vmsa - LAUNCH_UPDATE_VMSA command * * @handle: handle of the VM + * @reserved: reserved * @address: physical address of memory region to encrypt * @len: len of memory region to encrypt */ @@ -349,6 +356,7 @@ struct sev_data_launch_update_vmsa { * struct sev_data_launch_measure - LAUNCH_MEASURE command parameters * * @handle: handle of the VM to process + * @reserved: reserved * @address: physical address containing the measurement blob * @len: len of measurement blob */ @@ -363,10 +371,13 @@ struct sev_data_launch_measure { * struct sev_data_launch_secret - LAUNCH_SECRET command parameters * * @handle: handle of the VM to process + * @reserved1: reserved * @hdr_address: physical address containing the packet header * @hdr_len: len of packet header + * @reserved2: reserved * @guest_address: system physical address of guest memory region * @guest_len: len of guest_paddr + * @reserved3: reserved * @trans_address: physical address of transport memory buffer * @trans_len: len of transport memory buffer */ @@ -399,10 +410,13 @@ struct sev_data_launch_finish { * @policy: policy information for the VM * @pdh_cert_address: physical address containing PDH certificate * @pdh_cert_len: len of PDH certificate + * @reserved1: reserved * @plat_certs_address: physical address containing platform certificate * @plat_certs_len: len of platform certificate + * @reserved2: reserved * @amd_certs_address: physical address containing AMD certificate * @amd_certs_len: len of AMD certificate + * @reserved3: reserved * @session_address: physical address containing Session data * @session_len: len of session data */ @@ -423,13 +437,16 @@ struct sev_data_send_start { } __packed; /** - * struct sev_data_send_update - SEND_UPDATE_DATA command + * struct sev_data_send_update_data - SEND_UPDATE_DATA command * * @handle: handle of the VM to process + * @reserved1: reserved * @hdr_address: physical address containing packet header * @hdr_len: len of packet header + * @reserved2: reserved * @guest_address: physical address of guest memory region to send * @guest_len: len of guest memory region to send + * @reserved3: reserved * @trans_address: physical address of host memory region * @trans_len: len of host memory region */ @@ -447,13 +464,15 @@ struct sev_data_send_update_data { } __packed; /** - * struct sev_data_send_update - SEND_UPDATE_VMSA command + * struct sev_data_send_update_vmsa - SEND_UPDATE_VMSA command * * @handle: handle of the VM to process * @hdr_address: physical address containing packet header * @hdr_len: len of packet header + * @reserved2: reserved * @guest_address: physical address of guest memory region to send * @guest_len: len of guest memory region to send + * @reserved3: reserved * @trans_address: physical address of host memory region * @trans_len: len of host memory region */ @@ -491,8 +510,10 @@ struct sev_data_send_cancel { * struct sev_data_receive_start - RECEIVE_START command parameters * * @handle: handle of the VM to perform receive operation + * @policy: policy information for the VM * @pdh_cert_address: system physical address containing PDH certificate blob * @pdh_cert_len: len of PDH certificate blob + * @reserved1: reserved * @session_address: system physical address containing session blob * @session_len: len of session blob */ @@ -510,10 +531,13 @@ struct sev_data_receive_start { * struct sev_data_receive_update_data - RECEIVE_UPDATE_DATA command parameters * * @handle: handle of the VM to update + * @reserved1: reserved * @hdr_address: physical address containing packet header blob * @hdr_len: len of packet header + * @reserved2: reserved * @guest_address: system physical address of guest memory region * @guest_len: len of guest memory region + * @reserved3: reserved * @trans_address: system physical address of transport buffer * @trans_len: len of transport buffer */ @@ -534,10 +558,13 @@ struct sev_data_receive_update_data { * struct sev_data_receive_update_vmsa - RECEIVE_UPDATE_VMSA command parameters * * @handle: handle of the VM to update + * @reserved1: reserved * @hdr_address: physical address containing packet header blob * @hdr_len: len of packet header + * @reserved2: reserved * @guest_address: system physical address of guest memory region * @guest_len: len of guest memory region + * @reserved3: reserved * @trans_address: system physical address of transport buffer * @trans_len: len of transport buffer */ @@ -567,6 +594,7 @@ struct sev_data_receive_finish { * struct sev_data_dbg - DBG_ENCRYPT/DBG_DECRYPT command parameters * * @handle: handle of the VM to perform debug operation + * @reserved: reserved * @src_addr: source address of data to operate on * @dst_addr: destination address of data to operate on * @len: len of data to operate on @@ -583,6 +611,7 @@ struct sev_data_dbg { * struct sev_data_attestation_report - SEV_ATTESTATION_REPORT command parameters * * @handle: handle of the VM + * @reserved: reserved * @mnonce: a random nonce that will be included in the report. * @address: physical address where the report will be copied. * @len: length of the physical buffer. @@ -781,9 +810,14 @@ struct sev_data_snp_guest_request { * * @init_rmp: indicate that the RMP should be initialized. * @list_paddr_en: indicate that list_paddr is valid + * @rapl_dis: whether RAPL is disabled + * @ciphertext_hiding_en: whether ciphertext hiding is enabled + * @tio_en: Indicates that SNP_INIT_EX initialized the RMP for SEV-TIO * @rsvd: reserved * @rsvd1: reserved * @list_paddr: system physical address of range list + * @max_snp_asid: When non-zero, enable ciphertext hiding and specify the + * maximum ASID that can be used for an SEV-SNP guest. * @rsvd2: reserved */ struct sev_data_snp_init_ex { @@ -841,7 +875,7 @@ struct sev_data_snp_shutdown_ex { } __packed; /** - * struct sev_platform_init_args + * struct sev_platform_init_args - parameters for sev_platform_init() * * @error: SEV firmware error code * @probe: True if this is being called as part of CCP module probe, which @@ -879,12 +913,12 @@ struct sev_data_snp_feature_info { } __packed; /** - * struct feature_info - FEATURE_INFO structure + * struct snp_feature_info - FEATURE_INFO structure * * @eax: output of SNP_FEATURE_INFO command * @ebx: output of SNP_FEATURE_INFO command * @ecx: output of SNP_FEATURE_INFO command - * #edx: output of SNP_FEATURE_INFO command + * @edx: output of SNP_FEATURE_INFO command */ struct snp_feature_info { u32 eax; @@ -954,7 +988,7 @@ struct sev_data_snp_verify_mitigation_dst { } __packed; /** - * struct sev_snp_tcb_version_genoa_milan + * struct sev_snp_tcb_version_genoa_milan - v1 SVN payload * * @boot_loader: SVN of PSP bootloader * @tee: SVN of PSP operating system @@ -971,7 +1005,7 @@ struct sev_snp_tcb_version_genoa_milan { }; /** - * struct sev_snp_tcb_version_turin + * struct sev_snp_tcb_version_turin - v2 SVN payload * * @fmc: SVN of FMC firmware * @boot_loader: SVN of PSP bootloader @@ -1037,9 +1071,9 @@ int sev_platform_status(struct sev_user_data_status *status, int *error); * behalf of userspace. The caller must pass a valid SEV file descriptor * so that we know that it has access to SEV device. * - * @filep - SEV device file pointer - * @cmd - command to issue - * @data - command buffer + * @filep: SEV device file pointer + * @id: command to issue + * @data: command buffer * @error: SEV command return code * * Returns: @@ -1056,8 +1090,8 @@ int sev_issue_cmd_external_user(struct file *filep, unsigned int id, /** * sev_guest_deactivate - perform SEV DEACTIVATE command * - * @deactivate: sev_data_deactivate structure to be processed - * @sev_ret: sev command return code + * @data: sev_data_deactivate structure to be processed + * @error: sev command return code * * Returns: * 0 if the sev successfully processed the command @@ -1071,8 +1105,8 @@ int sev_guest_deactivate(struct sev_data_deactivate *data, int *error); /** * sev_guest_activate - perform SEV ACTIVATE command * - * @activate: sev_data_activate structure to be processed - * @sev_ret: sev command return code + * @data: sev_data_activate structure to be processed + * @error: sev command return code * * Returns: * 0 if the sev successfully processed the command @@ -1086,7 +1120,7 @@ int sev_guest_activate(struct sev_data_activate *data, int *error); /** * sev_guest_df_flush - perform SEV DF_FLUSH command * - * @sev_ret: sev command return code + * @error: sev command return code * * Returns: * 0 if the sev successfully processed the command @@ -1100,8 +1134,8 @@ int sev_guest_df_flush(int *error); /** * sev_guest_decommission - perform SEV DECOMMISSION command * - * @decommission: sev_data_decommission structure to be processed - * @sev_ret: sev command return code + * @data: sev_data_decommission structure to be processed + * @error: sev command return code * * Returns: * 0 if the sev successfully processed the command diff --git a/include/linux/ptdump.h b/include/linux/ptdump.h index 240bd3bff18d..af18d1459b2f 100644 --- a/include/linux/ptdump.h +++ b/include/linux/ptdump.h @@ -31,9 +31,9 @@ bool ptdump_walk_pgd_level_core(struct seq_file *m, void ptdump_walk_pgd(struct ptdump_state *st, struct mm_struct *mm, pgd_t *pgd); bool ptdump_check_wx(void); -static inline void debug_checkwx(void) +static inline void pgtable_checkwx(void) { - if (IS_ENABLED(CONFIG_DEBUG_WX)) + if (IS_ENABLED(CONFIG_CHECK_WX)) ptdump_check_wx(); } diff --git a/include/linux/quotaops.h b/include/linux/quotaops.h index f9c0f9d7c9d9..0c64ca674e77 100644 --- a/include/linux/quotaops.h +++ b/include/linux/quotaops.h @@ -20,7 +20,7 @@ static inline struct quota_info *sb_dqopt(struct super_block *sb) } /* i_rwsem must being held */ -static inline bool is_quota_modification(struct mnt_idmap *idmap, +static inline bool is_quota_modification(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *ia) { return ((ia->ia_valid & ATTR_SIZE) || @@ -109,7 +109,7 @@ int dquot_set_dqblk(struct super_block *sb, struct kqid id, struct qc_dqblk *di); int __dquot_transfer(struct inode *inode, struct dquot **transfer_to); -int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode, +int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr); static inline struct mem_dqinfo *sb_dqinfo(struct super_block *sb, int type) @@ -229,7 +229,7 @@ static inline void dquot_free_inode(struct inode *inode) { } -static inline int dquot_transfer(struct mnt_idmap *idmap, +static inline int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr) { return 0; diff --git a/include/linux/rbtree_augmented.h b/include/linux/rbtree_augmented.h index 6dbc5a1bf6a8..d2fa1c41bfd2 100644 --- a/include/linux/rbtree_augmented.h +++ b/include/linux/rbtree_augmented.h @@ -87,18 +87,18 @@ rb_add_augmented_cached(struct rb_node *node, struct rb_root_cached *tree, } /* - * Template for declaring augmented rbtree callbacks (generic case) + * Template for declaring augmented rbtree callbacks (generic multi fields) * * RBSTATIC: 'static' or empty * RBNAME: name of the rb_augment_callbacks structure * RBSTRUCT: struct type of the tree nodes * RBFIELD: name of struct rb_node field within RBSTRUCT - * RBAUGMENTED: name of field within RBSTRUCT holding data for subtree - * RBCOMPUTE: name of function that recomputes the RBAUGMENTED data + * RBCOPY: name of function that copies the RBAUGMENTED datas + * RBCOMPUTE: name of function that recomputes the RBAUGMENTED datas */ -#define RB_DECLARE_CALLBACKS(RBSTATIC, RBNAME, \ - RBSTRUCT, RBFIELD, RBAUGMENTED, RBCOMPUTE) \ +#define RB_DECLARE_CALLBACKS_MULTI(RBSTATIC, RBNAME, \ + RBSTRUCT, RBFIELD, RBCOPY, RBCOMPUTE) \ static inline void \ RBNAME ## _propagate(struct rb_node *rb, struct rb_node *stop) \ { \ @@ -114,14 +114,14 @@ RBNAME ## _copy(struct rb_node *rb_old, struct rb_node *rb_new) \ { \ RBSTRUCT *old = rb_entry(rb_old, RBSTRUCT, RBFIELD); \ RBSTRUCT *new = rb_entry(rb_new, RBSTRUCT, RBFIELD); \ - new->RBAUGMENTED = old->RBAUGMENTED; \ + RBCOPY(new, old); \ } \ static void \ RBNAME ## _rotate(struct rb_node *rb_old, struct rb_node *rb_new) \ { \ RBSTRUCT *old = rb_entry(rb_old, RBSTRUCT, RBFIELD); \ RBSTRUCT *new = rb_entry(rb_new, RBSTRUCT, RBFIELD); \ - new->RBAUGMENTED = old->RBAUGMENTED; \ + RBCOPY(new, old); \ RBCOMPUTE(old, false); \ } \ RBSTATIC const struct rb_augment_callbacks RBNAME = { \ @@ -131,6 +131,27 @@ RBSTATIC const struct rb_augment_callbacks RBNAME = { \ }; /* + * Template for declaring augmented rbtree callbacks (generic single field) + * + * RBSTATIC: 'static' or empty + * RBNAME: name of the rb_augment_callbacks structure + * RBSTRUCT: struct type of the tree nodes + * RBFIELD: name of struct rb_node field within RBSTRUCT + * RBAUGMENTED: name of field within RBSTRUCT holding data for subtree + * RBCOMPUTE: name of function that recomputes the RBAUGMENTED data + */ + +#define RB_DECLARE_CALLBACKS(RBSTATIC, RBNAME, \ + RBSTRUCT, RBFIELD, RBAUGMENTED, RBCOMPUTE) \ +static inline void \ +RBNAME ## _copy_single(RBSTRUCT *new, RBSTRUCT *old) \ +{ \ + new->RBAUGMENTED = old->RBAUGMENTED; \ +} \ +RB_DECLARE_CALLBACKS_MULTI(RBSTATIC, RBNAME, \ + RBSTRUCT, RBFIELD, RBNAME ## _copy_single, RBCOMPUTE) + +/* * Template for declaring augmented rbtree callbacks, * computing RBAUGMENTED scalar as max(RBCOMPUTE(node)) for all subtree nodes. * diff --git a/include/linux/rcupdate.h b/include/linux/rcupdate.h index 44c07a66edff..3f74ae6d6e1f 100644 --- a/include/linux/rcupdate.h +++ b/include/linux/rcupdate.h @@ -488,7 +488,7 @@ static __always_inline bool lockdep_assert_rcu_helper(bool c, const struct __ctx context_unsafe( \ typeof(*p) *local = (typeof(*p) *__force)(p); \ rcu_check_sparse(p, __rcu); \ - ((typeof(*p) __force __kernel *)(local)) \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)) \ ) /** * unrcu_pointer - mark a pointer as not being RCU protected @@ -503,7 +503,7 @@ context_unsafe( \ ({ \ typeof(*p) *local = (typeof(*p) *__force)READ_ONCE(p); \ rcu_check_sparse(p, space); \ - ((typeof(*p) __force __kernel *)(local)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \ }) ) #define __rcu_dereference_check(p, local, c, space) \ ({ \ @@ -511,19 +511,19 @@ context_unsafe( \ typeof(*p) *local = (typeof(*p) *__force)READ_ONCE(p); \ RCU_LOCKDEP_WARN(!(c), "suspicious rcu_dereference_check() usage"); \ rcu_check_sparse(p, space); \ - ((typeof(*p) __force __kernel *)(local)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \ }) #define __rcu_dereference_protected(p, local, c, space) \ ({ \ RCU_LOCKDEP_WARN(!(c), "suspicious rcu_dereference_protected() usage"); \ rcu_check_sparse(p, space); \ - ((typeof(*p) __force __kernel *)(p)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(p)); \ }) #define __rcu_dereference_raw(p, local) \ ({ \ /* Dependency order vs. p above. */ \ typeof(p) local = READ_ONCE(p); \ - ((typeof(*p) __force __kernel *)(local)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \ }) #define rcu_dereference_raw(p) __rcu_dereference_raw(p, __UNIQUE_ID(rcu)) diff --git a/include/linux/rcuref.h b/include/linux/rcuref.h index 2fb2af6d9824..01fee161b67c 100644 --- a/include/linux/rcuref.h +++ b/include/linux/rcuref.h @@ -34,7 +34,7 @@ static inline void rcuref_init(rcuref_t *ref, unsigned int cnt) * indicate that it is safe to schedule the object, protected by this reference * counter, for deconstruction. * If you want to know if the reference counter has been marked DEAD (as - * signaled by rcuref_put()) please use rcuread_is_dead(). + * signaled by rcuref_put()) please use rcuref_is_dead(). */ static inline unsigned int rcuref_read(rcuref_t *ref) { diff --git a/include/linux/regulator/driver.h b/include/linux/regulator/driver.h index cc6ce709ec86..55cc519103ed 100644 --- a/include/linux/regulator/driver.h +++ b/include/linux/regulator/driver.h @@ -362,7 +362,7 @@ enum regulator_type { * @off_on_delay: guard time (in uS), before re-enabling a regulator * * @poll_enabled_time: The polling interval (in uS) to use while checking that - * the regulator was actually enabled. Max upto enable_time. + * the regulator was actually enabled. Max up to enable_time. * * @of_map_mode: Maps a hardware mode defined in a DeviceTree to a standard mode */ diff --git a/include/linux/regulator/pca9450.h b/include/linux/regulator/pca9450.h index 0df8b3c48082..bf94df5fafe3 100644 --- a/include/linux/regulator/pca9450.h +++ b/include/linux/regulator/pca9450.h @@ -210,9 +210,11 @@ enum { #define LDO5L_EN_MASK 0xC0 #define LDO5LOUT_MASK 0x0F -#define LDO5H_EN_MASK 0xC0 #define LDO5HOUT_MASK 0x0F +/* LDO ENMODE value: ON in RUN, OFF while PMIC_STBY_REQ is asserted */ +#define LDO_ENMODE_ONREQ_STBYREQ 0x80 + /* PCA9450_REG_IRQ bits */ #define IRQ_PWRON 0x80 #define IRQ_WDOGB 0x40 diff --git a/include/linux/resctrl.h b/include/linux/resctrl.h index dd09c2ce9a0f..10dfdca7f4bf 100644 --- a/include/linux/resctrl.h +++ b/include/linux/resctrl.h @@ -505,6 +505,25 @@ bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r); */ int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable); +/** + * resctrl_arch_preconvert_bw() - Prepare bandwidth control value for arch use. + * @r: Resource whose schema was written. + * @val: Bandwidth control value written to the schemata file by userspace. + * + * Convert the user provided bandwidth control value to an appropriate form for + * consumption by the hardware driver for resource @r. Converted value is stored + * in rdt_ctrl_domain::staged_config[] for later consumption by + * resctrl_arch_update_domains(). Is not called when MBA software controller is + * enabled. + * + * Architectures for which this pre-conversion hook is not useful should supply + * an implementation of this function that just returns @val unmodified. + * + * Return: + * The converted value. + */ +u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val); + /* * Update the ctrl_val and apply this config right now. * Must be called on one of the domain's CPUs. diff --git a/include/linux/rhashtable-types.h b/include/linux/rhashtable-types.h index 57c11ec9dc64..3576b8f08aff 100644 --- a/include/linux/rhashtable-types.h +++ b/include/linux/rhashtable-types.h @@ -70,6 +70,12 @@ struct rhashtable_params { rht_obj_cmpfn_t obj_cmpfn; }; +struct rhashtable_lockdep_keys { + struct lock_class_key lock_key; + struct lock_class_key mutex_key; + struct lock_class_key bucket_key; +}; + /** * struct rhashtable - Hash table handle * @tbl: Bucket table @@ -97,6 +103,9 @@ struct rhashtable { #ifdef CONFIG_MEM_ALLOC_PROFILING struct alloc_tag *alloc_tag; #endif +#ifdef CONFIG_LOCKDEP + struct lock_class_key *lockdep_key; +#endif }; /** @@ -138,24 +147,77 @@ struct rhashtable_iter { int __rhashtable_init_noprof(struct rhashtable *ht, const struct rhashtable_params *params, - struct lock_class_key *key); + struct rhashtable_lockdep_keys *keys); #define rhashtable_init_noprof(ht, params) \ ({ \ - static struct lock_class_key __key; \ + static struct rhashtable_lockdep_keys __keys; \ \ - __rhashtable_init_noprof(ht, params, &__key); \ + __rhashtable_init_noprof(ht, params, &__keys); \ }) + +/** + * rhashtable_init - initialize a new hash table + * @ht: hash table to be initialized + * @params: configuration parameters + * + * Initializes a new hash table based on the provided configuration + * parameters. A table can be configured either with a variable or + * fixed length key: + * + * Configuration Example 1: Fixed length keys + * struct test_obj { + * int key; + * void * my_member; + * struct rhash_head node; + * }; + * + * struct rhashtable_params params = { + * .head_offset = offsetof(struct test_obj, node), + * .key_offset = offsetof(struct test_obj, key), + * .key_len = sizeof(int), + * .hashfn = jhash, + * }; + * + * Configuration Example 2: Variable length keys + * struct test_obj { + * [...] + * struct rhash_head node; + * }; + * + * u32 my_hash_fn(const void *data, u32 len, u32 seed) + * { + * struct test_obj *obj = data; + * + * return [... hash ...]; + * } + * + * struct rhashtable_params params = { + * .head_offset = offsetof(struct test_obj, node), + * .hashfn = jhash, + * .obj_hashfn = my_hash_fn, + * }; + */ #define rhashtable_init(...) alloc_hooks(rhashtable_init_noprof(__VA_ARGS__)) int __rhltable_init_noprof(struct rhltable *hlt, const struct rhashtable_params *params, - struct lock_class_key *key); + struct rhashtable_lockdep_keys *keys); #define rhltable_init_noprof(hlt, params) \ ({ \ - static struct lock_class_key __key; \ + static struct rhashtable_lockdep_keys __keys; \ \ - __rhltable_init_noprof(hlt, params, &__key); \ + __rhltable_init_noprof(hlt, params, &__keys); \ }) + +/** + * rhltable_init - initialize a new hash list table + * @hlt: hash list table to be initialized + * @params: configuration parameters + * + * Initializes a new hash list table. + * + * See documentation for rhashtable_init. + */ #define rhltable_init(...) alloc_hooks(rhltable_init_noprof(__VA_ARGS__)) #endif /* _LINUX_RHASHTABLE_TYPES_H */ diff --git a/include/linux/rhashtable.h b/include/linux/rhashtable.h index 57a2a29bef0e..ec853c1b9af3 100644 --- a/include/linux/rhashtable.h +++ b/include/linux/rhashtable.h @@ -319,18 +319,6 @@ static inline struct rhash_lock_head __rcu **rht_bucket_insert( * When we write to a bucket without unlocking, we use rht_assign_locked(). */ -static inline unsigned long rht_lock(struct bucket_table *tbl, - struct rhash_lock_head __rcu **bkt) - __acquires(__bitlock(0, bkt)) -{ - unsigned long flags; - - local_irq_save(flags); - bit_spin_lock(0, (unsigned long *)bkt); - lock_map_acquire(&tbl->dep_map); - return flags; -} - static inline unsigned long rht_lock_nested(struct bucket_table *tbl, struct rhash_lock_head __rcu **bucket, unsigned int subclass) @@ -344,6 +332,13 @@ static inline unsigned long rht_lock_nested(struct bucket_table *tbl, return flags; } +static inline unsigned long rht_lock(struct bucket_table *tbl, + struct rhash_lock_head __rcu **bkt) + __acquires(__bitlock(0, bkt)) +{ + return rht_lock_nested(tbl, bkt, 0); +} + static inline void rht_unlock(struct bucket_table *tbl, struct rhash_lock_head __rcu **bkt, unsigned long flags) diff --git a/include/linux/ring_buffer.h b/include/linux/ring_buffer.h index 0670742b2d60..eac3e9080c3c 100644 --- a/include/linux/ring_buffer.h +++ b/include/linux/ring_buffer.h @@ -3,8 +3,9 @@ #define _LINUX_RING_BUFFER_H #include <linux/mm.h> -#include <linux/seq_file.h> #include <linux/poll.h> +#include <linux/ring_buffer_types.h> +#include <linux/seq_file.h> #include <uapi/linux/trace_mmap.h> @@ -218,14 +219,15 @@ bool ring_buffer_time_stamp_abs(struct trace_buffer *buffer); size_t ring_buffer_nr_dirty_pages(struct trace_buffer *buffer, int cpu); struct buffer_data_read_page; -struct buffer_data_read_page * -ring_buffer_alloc_read_page(struct trace_buffer *buffer, int cpu); +int ring_buffer_alloc_read_page(struct trace_buffer *buffer, int cpu, + struct buffer_data_read_page **rpage); void ring_buffer_free_read_page(struct trace_buffer *buffer, int cpu, struct buffer_data_read_page *page); int ring_buffer_read_page(struct trace_buffer *buffer, struct buffer_data_read_page *data_page, size_t len, int cpu, int full); void *ring_buffer_read_page_data(struct buffer_data_read_page *page); +unsigned int ring_buffer_read_page_size(struct buffer_data_read_page *rpage); struct trace_seq; @@ -278,11 +280,25 @@ static inline struct ring_buffer_desc *__first_ring_buffer_desc(struct trace_buf return (struct ring_buffer_desc *)(&desc->__data[0]); } +/* + * Returns the number of pages for a ring_buffer_desc. The caller must ensure it + * does not overflow ring_buffer_desc::nr_page_va. + */ +static inline unsigned long __calc_nr_pages_ring_buffer_desc(size_t size) +{ + /* Takes into account the reader page */ + return max(DIV_ROUND_UP(size, PAGE_SIZE - BUF_PAGE_HDR_SIZE), 2UL) + 1; +} + static inline size_t trace_buffer_desc_size(size_t buffer_size, unsigned int nr_cpus) { - unsigned int nr_pages = max(DIV_ROUND_UP(buffer_size, PAGE_SIZE), 2UL) + 1; + unsigned long nr_pages = __calc_nr_pages_ring_buffer_desc(buffer_size); struct ring_buffer_desc *rbdesc; + /* Capped by ring_buffer_desc::nr_page_va */ + if (nr_pages > UINT_MAX) + return SIZE_MAX; + return size_add(offsetof(struct trace_buffer_desc, __data), size_mul(nr_cpus, struct_size(rbdesc, page_va, nr_pages))); } diff --git a/include/linux/rmap.h b/include/linux/rmap.h index 0b332770abee..74cca0e3c726 100644 --- a/include/linux/rmap.h +++ b/include/linux/rmap.h @@ -888,7 +888,7 @@ struct page_vma_mapped_walk { static inline void page_vma_mapped_walk_done(struct page_vma_mapped_walk *pvmw) { /* HugeTLB pte is set to the relevant page table entry without pte_mapped. */ - if (pvmw->pte && !is_vm_hugetlb_page(pvmw->vma)) + if (pvmw->pte && !vma_is_hugetlb(pvmw->vma)) pte_unmap(pvmw->pte); if (pvmw->ptl) spin_unlock(pvmw->ptl); diff --git a/include/linux/rolling_buffer.h b/include/linux/rolling_buffer.h index 9e5dad29669c..a97f7cfaacaa 100644 --- a/include/linux/rolling_buffer.h +++ b/include/linux/rolling_buffer.h @@ -45,9 +45,9 @@ struct rolling_buffer_snapshot { int rolling_buffer_init(struct rolling_buffer *roll, unsigned int rreq_id, unsigned int direction, gfp_t gfp); int rolling_buffer_make_space(struct rolling_buffer *roll, gfp_t gfp); -ssize_t rolling_buffer_load_from_ra(struct rolling_buffer *roll, - struct readahead_control *ractl, - struct folio_batch *put_batch); +ssize_t rolling_buffer_bulk_load_from_ra(struct rolling_buffer *roll, + struct readahead_control *ractl, + unsigned int rreq_id, gfp_t gfp); ssize_t rolling_buffer_append(struct rolling_buffer *roll, struct folio *folio, unsigned int flags, gfp_t gfp); struct folio_queue *rolling_buffer_delete_spent(struct rolling_buffer *roll); diff --git a/include/linux/rtnetlink.h b/include/linux/rtnetlink.h index 95729339e7a5..a54ec40d095c 100644 --- a/include/linux/rtnetlink.h +++ b/include/linux/rtnetlink.h @@ -186,7 +186,12 @@ void net_dec_ingress_queue(void); #ifdef CONFIG_NET_EGRESS void net_inc_egress_queue(void); void net_dec_egress_queue(void); -void netdev_xmit_skip_txqueue(bool skip); +bool netdev_xmit_skip_txqueue(bool skip); +#else +static inline bool netdev_xmit_skip_txqueue(bool skip) +{ + return false; +} #endif void rtnetlink_init(void); diff --git a/include/linux/sched.h b/include/linux/sched.h index 8b3d47a325cc..d828f0fb1896 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -554,6 +554,7 @@ struct sched_statistics { u64 nr_failed_migrations_running; u64 nr_failed_migrations_hot; u64 nr_forced_migrations; + u64 nr_migrations_cpu_non_preferred; u64 nr_wakeups; u64 nr_wakeups_sync; @@ -1172,8 +1173,12 @@ struct task_struct { /* Objective and real subjective task credentials (COW): */ const struct cred __rcu *real_cred; - /* Effective (overridable) subjective task credentials (COW): */ - const struct cred __rcu *cred; + /* + * Effective (overridable) subjective task credentials (COW). + * Only accessible for the current task and during task creation/freeing. + * This pointer is not managed by RCU! + */ + const struct cred *cred; #ifdef CONFIG_KEYS /* Cached requested key. */ @@ -1433,6 +1438,7 @@ struct task_struct { #ifdef CONFIG_SCHED_CACHE struct callback_head cache_work; + struct sched_cache_group __rcu *sched_cache_grp; int preferred_llc; /* 1: task was enqueued to its preferred LLC, 0 otherwise */ int pref_llc_queued; @@ -1787,7 +1793,7 @@ static inline bool is_lazy_mmu_mode_active(void) } #endif -extern struct pid *cad_pid; +extern struct pid __rcu *cad_pid; /* * Per process flags @@ -1808,15 +1814,15 @@ extern struct pid *cad_pid; #define PF_USED_MATH 0x00002000 /* If unset the fpu must be initialized before use */ #define PF_USER_WORKER 0x00004000 /* Kernel thread cloned from userspace thread */ #define PF_NOFREEZE 0x00008000 /* This thread should not be frozen */ -#define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ -#define PF_KSWAPD 0x00020000 /* I am kswapd */ +#define PF__HOLE__00010000 0x00010000 +#define PF__HOLE__00020000 0x00020000 #define PF_MEMALLOC_NOFS 0x00040000 /* All allocations inherit GFP_NOFS. See memalloc_nfs_save() */ #define PF_MEMALLOC_NOIO 0x00080000 /* All allocations inherit GFP_NOIO. See memalloc_noio_save() */ #define PF_LOCAL_THROTTLE 0x00100000 /* Throttle writes only against the bdi I write to, * I am cleaning dirty pages from some other bdi. */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_RANDOMIZE 0x00400000 /* Randomize virtual address space */ -#define PF__HOLE__00800000 0x00800000 +#define PF_NO_NOTIFY_SIGNAL 0x00800000 /* see no_notify_signal_save() */ #define PF__HOLE__01000000 0x01000000 #define PF__HOLE__02000000 0x02000000 #define PF_NO_SETAFFINITY 0x04000000 /* Userland is not allowed to meddle with cpus_mask */ @@ -2134,44 +2140,19 @@ static inline void set_need_resched_current(void) * value indicates whether a reschedule was done in fact. * cond_resched_lock() will drop the spinlock before scheduling, */ -#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC) +#if !defined(CONFIG_PREEMPTION) extern int __cond_resched(void); -#if defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) - -DECLARE_STATIC_CALL(cond_resched, __cond_resched); - -static __always_inline int _cond_resched(void) -{ - return static_call_mod(cond_resched)(); -} - -#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) - -extern int dynamic_cond_resched(void); - -static __always_inline int _cond_resched(void) -{ - return dynamic_cond_resched(); -} - -#else /* !CONFIG_PREEMPTION */ - static inline int _cond_resched(void) { return __cond_resched(); } - -#endif /* PREEMPT_DYNAMIC && CONFIG_HAVE_PREEMPT_DYNAMIC_CALL */ - -#else /* CONFIG_PREEMPTION && !CONFIG_PREEMPT_DYNAMIC */ - +#else static inline int _cond_resched(void) { return 0; } - -#endif /* !CONFIG_PREEMPTION || CONFIG_PREEMPT_DYNAMIC */ +#endif #define cond_resched() ({ \ __might_resched(__FILE__, __LINE__, 0); \ @@ -2405,7 +2386,7 @@ struct sched_cache_time { unsigned long epoch; }; -struct sched_cache_stat { +struct sched_cache_group { struct sched_cache_time __percpu *pcpu_sched; raw_spinlock_t lock; unsigned long epoch; @@ -2413,11 +2394,26 @@ struct sched_cache_stat { unsigned long next_scan; unsigned long footprint; int cpu; + refcount_t refcnt; + struct rcu_head rcu; } ____cacheline_aligned_in_smp; +struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp); +struct sched_cache_group *task_cache_group_get(struct task_struct *p); + +void sched_cache_fork(struct task_struct *p); +void sched_cache_fork_cleanup(struct task_struct *p); +void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm); +void sched_cache_exit_mm(struct task_struct *p); + #else -struct sched_cache_stat { }; +struct sched_cache_group { }; + +static inline void sched_cache_fork(struct task_struct *p) { } +static inline void sched_cache_fork_cleanup(struct task_struct *p) { } +static inline void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm) { } +static inline void sched_cache_exit_mm(struct task_struct *p) { } #endif diff --git a/include/linux/sched/cputime.h b/include/linux/sched/cputime.h index e90efaf6d26e..694126411dfe 100644 --- a/include/linux/sched/cputime.h +++ b/include/linux/sched/cputime.h @@ -182,9 +182,9 @@ extern unsigned long long task_sched_runtime(struct task_struct *task); #ifdef CONFIG_PARAVIRT -struct static_key; -extern struct static_key paravirt_steal_enabled; -extern struct static_key paravirt_steal_rq_enabled; +#include <linux/jump_label.h> +DECLARE_STATIC_KEY_FALSE(paravirt_steal_enabled); +DECLARE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled); #ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN u64 dummy_steal_clock(int cpu); diff --git a/include/linux/sched/ext.h b/include/linux/sched/ext.h index 582d7cd4a983..23f9e178bc5a 100644 --- a/include/linux/sched/ext.h +++ b/include/linux/sched/ext.h @@ -91,7 +91,7 @@ struct scx_dispatch_q { struct rhash_head hash_node; struct llist_node free_node; struct scx_sched *sched; - struct scx_dsq_pcpu __percpu *pcpu; + struct scx_dsq_pcpu __percpu *pcpu_user; struct rcu_head rcu; }; @@ -104,6 +104,7 @@ enum scx_ent_flags { SCX_TASK_SUB_INIT = 1 << 4, /* task being initialized for a sub sched */ SCX_TASK_IMMED = 1 << 5, /* task is on local DSQ with %SCX_ENQ_IMMED */ SCX_TASK_PROTECTED = 1 << 6, /* slice and DSQ head position protected */ + SCX_TASK_RUN_TRACKED = 1 << 7, /* task is in an ops.running()/stopping() session */ /* * Bits 8 to 10 are used to carry task state: @@ -136,6 +137,7 @@ enum scx_ent_flags { * IMMED reenqueued due to failed ENQ_IMMED * PREEMPTED preempted while running * CAP sub-sched cap miss, see p->scx.reenq_reason_* + * PROXY proxy state prevented a remote DSQ transfer */ SCX_TASK_REENQ_REASON_SHIFT = 12, SCX_TASK_REENQ_REASON_BITS = 3, @@ -146,6 +148,7 @@ enum scx_ent_flags { SCX_TASK_REENQ_IMMED = 2 << SCX_TASK_REENQ_REASON_SHIFT, SCX_TASK_REENQ_PREEMPTED = 3 << SCX_TASK_REENQ_REASON_SHIFT, SCX_TASK_REENQ_CAP = 4 << SCX_TASK_REENQ_REASON_SHIFT, + SCX_TASK_REENQ_PROXY = 5 << SCX_TASK_REENQ_REASON_SHIFT, /* iteration cursor, not a task */ SCX_TASK_CURSOR = 1 << 31, @@ -197,9 +200,19 @@ struct sched_ext_entity { u64 ddsp_slice; u64 ddsp_vtime; struct scx_dsq_list_node dsq_list; /* dispatch order */ - struct rb_node dsq_priq; /* p->scx.dsq_vtime order */ u32 dsq_seq; u32 dsq_flags; /* protected by DSQ lock */ + + /* + * Used to order tasks when dispatching to the vtime-ordered priority + * queue of a dsq. This is usually set through + * scx_bpf_dsq_insert_vtime() but can also be modified directly by the + * BPF scheduler. Modifying it while a task is queued on a dsq may + * mangle the ordering and is not recommended. Kept next to @dsq_priq + * as rbtree insertion reads both on every visited node. + */ + u64 dsq_vtime; + struct rb_node dsq_priq; /* p->scx.dsq_vtime order */ u32 flags; /* protected by rq lock */ u32 weight; u32 reenq_cnt; /* reenqueues since last run */ @@ -225,7 +238,7 @@ struct sched_ext_entity { u64 tid; struct rhash_head tid_hash_node; /* see SCX_OPS_TID_TO_TASK */ - /* BPF scheduler modifiable fields */ + /* BPF scheduler modifiable fields, along with @dsq_vtime above */ /* * Runtime budget in nsecs - how long the task may hold its cpu. Owned @@ -241,15 +254,6 @@ struct sched_ext_entity { u64 slice; /* - * Used to order tasks when dispatching to the vtime-ordered priority - * queue of a dsq. This is usually set through - * scx_bpf_dsq_insert_vtime() but can also be modified directly by the - * BPF scheduler. Modifying it while a task is queued on a dsq may - * mangle the ordering and is not recommended. - */ - u64 dsq_vtime; - - /* * Out-of-band slice request from scx_bpf_task_set_slice() when the * caller does not hold the rq lock, applied under the rq lock at the * next slice consideration. One atomic64 packs the pending flag, the @@ -278,6 +282,14 @@ struct sched_ext_entity { */ bool disallow; /* reject switching into SCX */ + /* + * If set, depletion of this task's slice at the scheduler tick requests + * lazy instead of immediate rescheduling. Initialized from + * %SCX_OPS_LAZY_RESCHED immediately before ops.enable() and may be + * modified afterwards with scx_bpf_task_set_lazy_resched(). + */ + bool lazy_resched; + /* cold fields */ #ifdef CONFIG_EXT_GROUP_SCHED struct cgroup *cgrp_moving_from; @@ -323,7 +335,7 @@ struct scx_task_group { u64 bw_period_us; u64 bw_quota_us; u64 bw_burst_us; - bool idle; + bool sched_idle; #endif }; diff --git a/include/linux/sched/rt.h b/include/linux/sched/rt.h index 4e3338103654..922935cc3383 100644 --- a/include/linux/sched/rt.h +++ b/include/linux/sched/rt.h @@ -52,8 +52,10 @@ static inline bool rt_or_dl_task_policy(struct task_struct *tsk) #ifdef CONFIG_RT_MUTEXES extern void rt_mutex_pre_schedule(void); +extern void rt_mutex_futex_pre_schedule(void); extern void rt_mutex_schedule(void); extern void rt_mutex_post_schedule(void); +extern void rt_mutex_futex_post_schedule(void); /* * Must hold either p->pi_lock or task_rq(p)->lock. diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h index 584ae88b435e..d9419dc902f6 100644 --- a/include/linux/sched/signal.h +++ b/include/linux/sched/signal.h @@ -2,6 +2,7 @@ #ifndef _LINUX_SCHED_SIGNAL_H #define _LINUX_SCHED_SIGNAL_H +#include <linux/cleanup.h> #include <linux/rculist.h> #include <linux/signal.h> #include <linux/sched.h> @@ -79,9 +80,9 @@ struct core_thread { }; struct core_state { - atomic_t nr_threads; - struct core_thread dumper; - struct completion startup; + /* Threads the dumper still waits for. */ + atomic_t threads_remaining; + struct core_thread *tasks; }; /* @@ -384,14 +385,36 @@ static inline int task_sigpending(struct task_struct *p) return unlikely(test_tsk_thread_flag(p,TIF_SIGPENDING)); } +/* Prevent TIF_NOTIFY_SIGNAL from interrupting this task. */ +static inline unsigned int no_notify_signal_save(void) +{ + unsigned int flags = current->flags; + + current->flags |= PF_NO_NOTIFY_SIGNAL; + return flags; +} + +/* Restore the previous PF_NO_NOTIFY_SIGNAL state. */ +static inline void no_notify_signal_restore(unsigned int flags) +{ + current_restore_flags(flags, PF_NO_NOTIFY_SIGNAL); +} + +DEFINE_LOCK_GUARD_0(no_notify_signal, + _T->flags = no_notify_signal_save(), + no_notify_signal_restore(_T->flags), + unsigned int flags) + static inline int signal_pending(struct task_struct *p) { /* * TIF_NOTIFY_SIGNAL isn't really a signal, but it requires the same * behavior in terms of ensuring that we break out of wait loops - * so that notify signal callbacks can be processed. + * so that notify signal callbacks can be processed. Not for a task + * that asked not to be interrupted by it, see no_notify_signal_save(). */ - if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL))) + if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)) && + likely(!(READ_ONCE(p->flags) & PF_NO_NOTIFY_SIGNAL))) return 1; return task_sigpending(p); } @@ -562,10 +585,7 @@ static inline sigset_t *sigmask_to_save(void) return res; } -static inline int kill_cad_pid(int sig, int priv) -{ - return kill_pid(cad_pid, sig, priv); -} +int kill_cad_pid(int sig, int priv); /* These can be the second arg to send_sig_info/send_group_sig_info. */ #define SEND_SIG_NOINFO ((struct kernel_siginfo *) 0) diff --git a/include/linux/sched/task.h b/include/linux/sched/task.h index e0c1ca8c6a18..90ed5bc3c7af 100644 --- a/include/linux/sched/task.h +++ b/include/linux/sched/task.h @@ -94,7 +94,6 @@ static inline void exit_thread(struct task_struct *tsk) extern __noreturn void do_group_exit(int); extern void exit_files(struct task_struct *); -extern void exit_itimers(struct task_struct *); extern pid_t kernel_clone(struct kernel_clone_args *kargs); struct task_struct *copy_process(struct pid *pid, int trace, int node, diff --git a/include/linux/sched/topology.h b/include/linux/sched/topology.h index b5d9d7c2b8ad..f96812d71c51 100644 --- a/include/linux/sched/topology.h +++ b/include/linux/sched/topology.h @@ -281,9 +281,9 @@ static inline int task_node(const struct task_struct *p) } #ifdef CONFIG_SCHED_CACHE -extern void sched_update_llc_bytes(unsigned int cpu); +extern void sched_update_llc_bytes(const struct cpumask *cpus); #else -static inline void sched_update_llc_bytes(unsigned int cpu) { } +static inline void sched_update_llc_bytes(const struct cpumask *cpus) { } #endif #endif /* _LINUX_SCHED_TOPOLOGY_H */ diff --git a/include/linux/sched/user.h b/include/linux/sched/user.h index 4cc52698e214..8d7e5521f7cd 100644 --- a/include/linux/sched/user.h +++ b/include/linux/sched/user.h @@ -25,7 +25,8 @@ struct user_struct { #if defined(CONFIG_PERF_EVENTS) || defined(CONFIG_BPF_SYSCALL) || \ defined(CONFIG_NET) || defined(CONFIG_IO_URING) || \ - defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD) + defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD) || \ + defined(CONFIG_SECRETMEM) atomic_long_t locked_vm; #endif #ifdef CONFIG_WATCH_QUEUE diff --git a/include/linux/scmi_protocol.h b/include/linux/scmi_protocol.h index 5ab73b1ab9aa..3f7cc35eb861 100644 --- a/include/linux/scmi_protocol.h +++ b/include/linux/scmi_protocol.h @@ -9,6 +9,7 @@ #define _LINUX_SCMI_PROTOCOL_H #include <linux/bitfield.h> +#include <linux/device-id/scmi.h> #include <linux/device.h> #include <linux/notifier.h> #include <linux/types.h> @@ -528,22 +529,25 @@ struct scmi_sensor_proto_ops { u32 sensor_id, u32 sensor_config); }; +struct scmi_reset_domain_info { + char name[SCMI_MAX_STR_SIZE]; + u32 latency_us; +}; + /** * struct scmi_reset_proto_ops - represents the various operations provided * by SCMI Reset Protocol * * @num_domains_get: get the count of reset domains provided by SCMI - * @name_get: gets the name of a reset domain - * @latency_get: gets the reset latency for the specified reset domain + * @info_get: gets the information of the specified reset domain * @reset: resets the specified reset domain * @assert: explicitly assert reset signal of the specified reset domain * @deassert: explicitly deassert reset signal of the specified reset domain */ struct scmi_reset_proto_ops { int (*num_domains_get)(const struct scmi_protocol_handle *ph); - const char *(*name_get)(const struct scmi_protocol_handle *ph, - u32 domain); - int (*latency_get)(const struct scmi_protocol_handle *ph, u32 domain); + const struct scmi_reset_domain_info __must_check *(*info_get) + (const struct scmi_protocol_handle *ph, u32 domain); int (*reset)(const struct scmi_protocol_handle *ph, u32 domain); int (*assert)(const struct scmi_protocol_handle *ph, u32 domain); int (*deassert)(const struct scmi_protocol_handle *ph, u32 domain); @@ -930,6 +934,7 @@ enum scmi_std_protocol { SCMI_PROTOCOL_VOLTAGE = 0x17, SCMI_PROTOCOL_POWERCAP = 0x18, SCMI_PROTOCOL_PINCTRL = 0x19, + SCMI_PROTOCOL_TELEMETRY = 0x1B, }; enum scmi_system_events { @@ -951,11 +956,6 @@ struct scmi_device { #define to_scmi_dev(d) container_of_const(d, struct scmi_device, dev) -struct scmi_device_id { - u8 protocol_id; - const char *name; -}; - struct scmi_driver { const char *name; int (*probe)(struct scmi_device *sdev); diff --git a/include/linux/security.h b/include/linux/security.h index 153e9043058f..8b8d6b8b802f 100644 --- a/include/linux/security.h +++ b/include/linux/security.h @@ -67,6 +67,7 @@ enum fs_value_type; struct watch; struct watch_notification; struct lsm_ctx; +struct nsset; /* Default (no) options for the capable function */ #define CAP_OPT_NONE 0x0 @@ -80,6 +81,7 @@ struct lsm_ctx; struct ctl_table; struct audit_krule; +struct ns_common; struct user_namespace; struct timezone; @@ -185,11 +187,11 @@ extern int cap_capset(struct cred *new, const struct cred *old, extern int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file); int cap_inode_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int cap_inode_removexattr(struct mnt_idmap *idmap, +int cap_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); int cap_inode_need_killpriv(struct dentry *dentry); -int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int cap_inode_getsecurity(struct mnt_idmap *idmap, +int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int cap_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); extern int cap_mmap_addr(unsigned long addr); @@ -338,6 +340,7 @@ int security_binder_transfer_file(const struct cred *from, const struct cred *to, const struct file *file); int security_ptrace_access_check(struct task_struct *child, unsigned int mode); int security_ptrace_traceme(struct task_struct *parent); +int security_mem_foll_force(const struct cred *subject, bool opened_by_owner); int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -404,49 +407,53 @@ int security_inode_init_security(struct inode *inode, struct inode *dir, int security_inode_init_security_anon(struct inode *inode, const struct qstr *name, const struct inode *context_inode); -int security_inode_create(struct inode *dir, struct dentry *dentry, umode_t mode); -void security_inode_post_create_tmpfile(struct mnt_idmap *idmap, +int security_inode_create(const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode); +void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode); -int security_inode_link(struct dentry *old_dentry, struct inode *dir, - struct dentry *new_dentry); +int security_inode_link(const struct mnt_idmap *idmap, struct dentry *old_dentry, + struct inode *dir, struct dentry *new_dentry); int security_inode_unlink(struct inode *dir, struct dentry *dentry); -int security_inode_symlink(struct inode *dir, struct dentry *dentry, - const char *old_name); -int security_inode_mkdir(struct inode *dir, struct dentry *dentry, umode_t mode); +int security_inode_symlink(const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, const char *old_name); +int security_inode_mkdir(const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode); int security_inode_rmdir(struct inode *dir, struct dentry *dentry); -int security_inode_mknod(struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev); +int security_inode_mknod(const struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode, dev_t dev); int security_inode_rename(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); int security_inode_readlink(struct dentry *dentry); int security_inode_follow_link(struct dentry *dentry, struct inode *inode, bool rcu); -int security_inode_permission(struct inode *inode, int mask); -int security_inode_setattr(struct mnt_idmap *idmap, +int security_inode_permission(const struct mnt_idmap *idmap, struct inode *inode, + int mask); +int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid); int security_inode_getattr(const struct path *path); -int security_inode_setxattr(struct mnt_idmap *idmap, +int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int security_inode_set_acl(struct mnt_idmap *idmap, +int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -int security_inode_get_acl(struct mnt_idmap *idmap, +int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int security_inode_remove_acl(struct mnt_idmap *idmap, +int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -void security_inode_post_remove_acl(struct mnt_idmap *idmap, +void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); void security_inode_post_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); int security_inode_getxattr(struct dentry *dentry, const char *name); int security_inode_listxattr(struct dentry *dentry); -int security_inode_removexattr(struct mnt_idmap *idmap, +int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); void security_inode_post_removexattr(struct dentry *dentry, const char *name); int security_inode_file_setattr(struct dentry *dentry, @@ -454,8 +461,8 @@ int security_inode_file_setattr(struct dentry *dentry, int security_inode_file_getattr(struct dentry *dentry, struct file_kattr *fa); int security_inode_need_killpriv(struct dentry *dentry); -int security_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int security_inode_getsecurity(struct mnt_idmap *idmap, +int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); int security_inode_setsecurity(struct inode *inode, const char *name, const void *value, size_t size, int flags); @@ -468,7 +475,7 @@ int security_inode_setintegrity(const struct inode *inode, size_t size); int security_kernfs_init_security(struct kernfs_node *kn_dir, struct kernfs_node *kn); -int security_file_permission(struct file *file, int mask); +int security_file_permission(const struct file *file, int mask); int security_file_alloc(struct file *file); void security_file_release(struct file *file); void security_file_free(struct file *file); @@ -540,6 +547,9 @@ int security_task_prctl(int option, unsigned long arg2, unsigned long arg3, unsigned long arg4, unsigned long arg5); void security_task_to_inode(struct task_struct *p, struct inode *inode); int security_create_user_ns(const struct cred *cred); +int security_namespace_init(struct ns_common *ns); +void security_namespace_free(struct ns_common *ns); +int security_namespace_install(const struct nsset *nsset, struct ns_common *ns); int security_ipc_permission(struct kern_ipc_perm *ipcp, short flag); void security_ipc_getlsmprop(struct kern_ipc_perm *ipcp, struct lsm_prop *prop); int security_msg_msg_alloc(struct msg_msg *msg); @@ -676,6 +686,12 @@ static inline int security_ptrace_traceme(struct task_struct *parent) return cap_ptrace_traceme(parent); } +static inline int security_mem_foll_force(const struct cred *subject, + bool opened_by_owner) +{ + return 0; +} + static inline int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -902,20 +918,22 @@ static inline int security_inode_init_security_anon(struct inode *inode, return 0; } -static inline int security_inode_create(struct inode *dir, - struct dentry *dentry, - umode_t mode) +static inline int security_inode_create(const struct mnt_idmap *idmap, + struct inode *dir, + struct dentry *dentry, + umode_t mode) { return 0; } static inline void -security_inode_post_create_tmpfile(struct mnt_idmap *idmap, struct inode *inode) +security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode) { } -static inline int security_inode_link(struct dentry *old_dentry, - struct inode *dir, - struct dentry *new_dentry) +static inline int security_inode_link(const struct mnt_idmap *idmap, + struct dentry *old_dentry, + struct inode *dir, + struct dentry *new_dentry) { return 0; } @@ -926,16 +944,18 @@ static inline int security_inode_unlink(struct inode *dir, return 0; } -static inline int security_inode_symlink(struct inode *dir, - struct dentry *dentry, - const char *old_name) +static inline int security_inode_symlink(const struct mnt_idmap *idmap, + struct inode *dir, + struct dentry *dentry, + const char *old_name) { return 0; } -static inline int security_inode_mkdir(struct inode *dir, - struct dentry *dentry, - int mode) +static inline int security_inode_mkdir(const struct mnt_idmap *idmap, + struct inode *dir, + struct dentry *dentry, + int mode) { return 0; } @@ -946,9 +966,10 @@ static inline int security_inode_rmdir(struct inode *dir, return 0; } -static inline int security_inode_mknod(struct inode *dir, - struct dentry *dentry, - int mode, dev_t dev) +static inline int security_inode_mknod(const struct mnt_idmap *idmap, + struct inode *dir, + struct dentry *dentry, + int mode, dev_t dev) { return 0; } @@ -974,12 +995,13 @@ static inline int security_inode_follow_link(struct dentry *dentry, return 0; } -static inline int security_inode_permission(struct inode *inode, int mask) +static inline int security_inode_permission(const struct mnt_idmap *idmap, + struct inode *inode, int mask) { return 0; } -static inline int security_inode_setattr(struct mnt_idmap *idmap, +static inline int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { @@ -987,7 +1009,7 @@ static inline int security_inode_setattr(struct mnt_idmap *idmap, } static inline void -security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { } @@ -996,14 +1018,14 @@ static inline int security_inode_getattr(const struct path *path) return 0; } -static inline int security_inode_setxattr(struct mnt_idmap *idmap, +static inline int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { return cap_inode_setxattr(dentry, name, value, size, flags); } -static inline int security_inode_set_acl(struct mnt_idmap *idmap, +static inline int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) @@ -1016,21 +1038,21 @@ static inline void security_inode_post_set_acl(struct dentry *dentry, struct posix_acl *kacl) { } -static inline int security_inode_get_acl(struct mnt_idmap *idmap, +static inline int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline int security_inode_remove_acl(struct mnt_idmap *idmap, +static inline int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline void security_inode_post_remove_acl(struct mnt_idmap *idmap, +static inline void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { } @@ -1050,7 +1072,7 @@ static inline int security_inode_listxattr(struct dentry *dentry) return 0; } -static inline int security_inode_removexattr(struct mnt_idmap *idmap, +static inline int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { @@ -1078,13 +1100,13 @@ static inline int security_inode_need_killpriv(struct dentry *dentry) return cap_inode_need_killpriv(dentry); } -static inline int security_inode_killpriv(struct mnt_idmap *idmap, +static inline int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry) { return cap_inode_killpriv(idmap, dentry); } -static inline int security_inode_getsecurity(struct mnt_idmap *idmap, +static inline int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) @@ -1132,7 +1154,7 @@ static inline int security_inode_copy_up_xattr(struct dentry *src, const char *n return -EOPNOTSUPP; } -static inline int security_file_permission(struct file *file, int mask) +static inline int security_file_permission(const struct file *file, int mask) { return 0; } @@ -1431,6 +1453,21 @@ static inline int security_create_user_ns(const struct cred *cred) return 0; } +static inline int security_namespace_init(struct ns_common *ns) +{ + return 0; +} + +static inline void security_namespace_free(struct ns_common *ns) +{ +} + +static inline int security_namespace_install(const struct nsset *nsset, + struct ns_common *ns) +{ + return 0; +} + static inline int security_ipc_permission(struct kern_ipc_perm *ipcp, short flag) { @@ -2085,7 +2122,7 @@ int security_path_mkdir(const struct path *dir, struct dentry *dentry, umode_t m int security_path_rmdir(const struct path *dir, struct dentry *dentry); int security_path_mknod(const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev); -void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry); +void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry); int security_path_truncate(const struct path *path); int security_path_symlink(const struct path *dir, struct dentry *dentry, const char *old_name); @@ -2120,7 +2157,7 @@ static inline int security_path_mknod(const struct path *dir, struct dentry *den return 0; } -static inline void security_path_post_mknod(struct mnt_idmap *idmap, +static inline void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { } diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 3030d9245f5a..27c32fbabfec 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -5,16 +5,92 @@ #ifndef _LINUX_SET_MEMORY_H_ #define _LINUX_SET_MEMORY_H_ +/** + * DOC: Kernel page table permissions + * + * The set_memory() and set_direct_map() APIs update permissions of existing + * kernel mappings. + * + * The set_memory() functions operate on a range of kernel virtual addresses, + * the set_direct_map() functions operate on the direct map. + * + * The updates are not atomic: when a call fails, an arbitrary prefix of the + * range may have been updated already and there is no automatic rollback. + * A caller must restore the required permissions before reusing or freeing + * the memory. + * + * When an architecture does not implement these APIs they succeed without + * doing anything, so a return value of 0 does not mean that the permissions + * were actually changed. + * + * Callers that depend on the permissions being applied must ensure that the + * architecture supports the required operation for the target addresses. The + * Kconfig symbols alone do not guarantee this. + * + * See Documentation/mm/kernel-page-tables.rst for the details and for the + * differences between the architecture implementations. + */ + #ifdef CONFIG_ARCH_HAS_SET_MEMORY #include <asm/set_memory.h> #else +/** + * set_memory_ro - make a kernel mapping read-only + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_ro(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_rw - make a kernel mapping writable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_rw(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_x - make a kernel mapping executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_x(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_nx - make a kernel mapping non-executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_nx(unsigned long addr, int numpages) { return 0; } #endif #ifndef set_memory_rox +/** + * set_memory_rox - make a kernel mapping read-only and executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * A failure may leave the range read-only but not executable. + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_memory_rox(unsigned long addr, int numpages) { int ret = set_memory_ro(addr, numpages); @@ -25,17 +101,35 @@ static inline int set_memory_rox(unsigned long addr, int numpages) #endif #ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP -static inline int set_direct_map_invalid_noflush(struct page *page) -{ - return 0; -} -static inline int set_direct_map_default_noflush(struct page *page) +/** + * set_direct_map_invalid_noflush - remove pages from the direct map + * @page: first page to update + * @nr: number of pages to update + * + * Makes the direct mapping of @nr pages starting at @page not present. + * The caller is responsible for any required TLB flushing. + * + * Return: 0 on success, negative error code on failure. + */ +static inline int set_direct_map_invalid_noflush(struct page *page, + unsigned int nr) { return 0; } -static inline int set_direct_map_valid_noflush(struct page *page, - unsigned nr, bool valid) +/** + * set_direct_map_default_noflush - restore the direct map of pages + * @page: first page to update + * @nr: number of pages to update + * + * Restores the default kernel permissions of the direct mapping of @nr + * pages starting at @page. + * The caller is responsible for any required TLB flushing. + * + * Return: 0 on success, negative error code on failure. + */ +static inline int set_direct_map_default_noflush(struct page *page, + unsigned int nr) { return 0; } @@ -50,6 +144,18 @@ static inline bool kernel_page_present(struct page *page) * boot time. Let them overrive this query. */ #ifndef can_set_direct_map +/** + * can_set_direct_map - check if the direct map can be modified + * + * Available with CONFIG_ARCH_HAS_SET_DIRECT_MAP. Architectures may override + * this to report whether direct map updates are enabled at runtime. + * Even though the generic implementation returns true this does not guarantee + * that every address can be updated. + * + * See Documentation/mm/kernel-page-tables.rst for the details + * + * Return: true unless the architecture reports direct map updates disabled. + */ static inline bool can_set_direct_map(void) { return true; diff --git a/include/linux/shmem_fs.h b/include/linux/shmem_fs.h index 5663dff53186..a7c7a96a7cbf 100644 --- a/include/linux/shmem_fs.h +++ b/include/linux/shmem_fs.h @@ -11,6 +11,7 @@ #include <linux/fs_parser.h> #include <linux/userfaultfd_k.h> #include <linux/bits.h> +#include <linux/list_lru.h> /* inode in-kernel data */ @@ -54,6 +55,11 @@ struct shmem_inode_info { struct dquot __rcu *i_dquot[MAXQUOTAS]; #endif struct inode vfs_inode; + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct obj_cgroup *shrinklist_objcg; + int shrinklist_nid; +#endif }; #define SHMEM_FL_USER_VISIBLE (FS_FL_USER_VISIBLE | FS_CASEFOLD_FL) @@ -83,9 +89,9 @@ struct shmem_sb_info { ino_t next_ino; /* The next per-sb inode number to use */ ino_t __percpu *ino_batch; /* The next per-cpu inode number to use */ struct mempolicy *mpol; /* default memory policy for mappings */ - spinlock_t shrinklist_lock; /* Protects shrinklist */ - struct list_head shrinklist; /* List of shinkable inodes */ - unsigned long shrinklist_len; /* Length of shrinklist */ +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct list_lru shrinklist; /* List of shrinkable inodes */ +#endif struct shmem_quota_limits qlimits; /* Default quota limits */ struct simple_xattr_cache xa_cache; }; diff --git a/include/linux/skbuff.h b/include/linux/skbuff.h index 671c13494566..27ec1e38c828 100644 --- a/include/linux/skbuff.h +++ b/include/linux/skbuff.h @@ -1020,7 +1020,7 @@ struct sk_buff { #ifdef CONFIG_NET_REDIRECT __u8 from_ingress:1; #endif -#ifdef CONFIG_NETFILTER_SKIP_EGRESS +#ifdef CONFIG_NET_EGRESS __u8 nf_skip_egress:1; #endif #ifdef CONFIG_SKB_DECRYPTED @@ -1834,22 +1834,6 @@ static inline void skb_zcopy_set(struct sk_buff *skb, struct ubuf_info *uarg, } } -static inline void skb_zcopy_set_nouarg(struct sk_buff *skb, void *val) -{ - skb_shinfo(skb)->destructor_arg = (void *)((uintptr_t) val | 0x1UL); - skb_shinfo(skb)->flags |= SKBFL_ZEROCOPY_FRAG; -} - -static inline bool skb_zcopy_is_nouarg(struct sk_buff *skb) -{ - return (uintptr_t) skb_shinfo(skb)->destructor_arg & 0x1UL; -} - -static inline void *skb_zcopy_get_nouarg(struct sk_buff *skb) -{ - return (void *)((uintptr_t) skb_shinfo(skb)->destructor_arg & ~0x1UL); -} - static inline void net_zcopy_put(struct ubuf_info *uarg) { if (uarg) @@ -1872,8 +1856,7 @@ static inline void skb_zcopy_clear(struct sk_buff *skb, bool zerocopy_success) struct ubuf_info *uarg = skb_zcopy(skb); if (uarg) { - if (!skb_zcopy_is_nouarg(skb)) - uarg->ops->complete(skb, uarg, zerocopy_success); + uarg->ops->complete(skb, uarg, zerocopy_success); skb_shinfo(skb)->flags &= ~SKBFL_ALL_ZEROCOPY; } @@ -3082,6 +3065,11 @@ static inline bool skb_transport_header_was_set(const struct sk_buff *skb) return skb->transport_header != (typeof(skb->transport_header))~0U; } +static inline void skb_unset_transport_header(struct sk_buff *skb) +{ + skb->transport_header = (typeof(skb->transport_header))~0U; +} + static inline unsigned char *skb_transport_header(const struct sk_buff *skb) { DEBUG_NET_WARN_ON_ONCE(!skb_transport_header_was_set(skb)); @@ -3828,9 +3816,12 @@ static inline dma_addr_t __skb_frag_dma_map(struct device *dev, size_t offset, size_t size, enum dma_data_direction dir) { + dma_addr_t addr; + if (skb_frag_is_net_iov(frag)) { - return netmem_to_net_iov(frag->netmem)->desc.dma_addr + - offset + frag->offset; + addr = netmem_dma_addr_decode( + netmem_get_dma_addr(frag->netmem)); + return addr + offset + frag->offset; } return dma_map_page(dev, skb_frag_page(frag), skb_frag_off(frag) + offset, size, dir); @@ -4367,7 +4358,10 @@ skb_header_pointer_careful(const struct sk_buff *skb, int offset, static inline void * __must_check skb_pointer_if_linear(const struct sk_buff *skb, int offset, int len) { - if (likely(skb_headlen(skb) - offset >= len)) + unsigned int uoffset = (unsigned int)offset; + + if (likely(uoffset <= skb_headlen(skb) && + (unsigned int)len <= skb_headlen(skb) - uoffset)) return skb->data + offset; return NULL; } diff --git a/include/linux/soc/qcom/apr.h b/include/linux/soc/qcom/apr.h index 909e84f84e0c..2e4231092088 100644 --- a/include/linux/soc/qcom/apr.h +++ b/include/linux/soc/qcom/apr.h @@ -5,7 +5,6 @@ #include <linux/spinlock.h> #include <linux/device.h> -#include <linux/device-id/apr.h> #include <dt-bindings/soc/qcom,apr.h> #include <dt-bindings/soc/qcom,gpr.h> @@ -135,6 +134,8 @@ struct pkt_router_svc { typedef struct pkt_router_svc gpr_port_t; +#define APR_NAME_SIZE 32 + struct apr_device { struct device dev; uint16_t svc_id; @@ -158,7 +159,6 @@ struct apr_driver { const struct apr_resp_pkt *d); gpr_port_cb gpr_callback; struct device_driver driver; - const struct apr_device_id *id_table; }; typedef struct apr_driver gpr_driver_t; diff --git a/include/linux/soc/qcom/geni-se.h b/include/linux/soc/qcom/geni-se.h index 29a53bbc0dd4..5f18d281e6a4 100644 --- a/include/linux/soc/qcom/geni-se.h +++ b/include/linux/soc/qcom/geni-se.h @@ -347,17 +347,12 @@ struct geni_se { #define QUP_SE_VERSION_2_5 0x20050000 /* - * Define bandwidth thresholds that cause the underlying Core 2X interconnect - * clock to run at the named frequency. These baseline values are recommended - * by the hardware team, and are not dynamically scaled with GENI bandwidth - * beyond basic on/off. + * QUP Core 2X clock votes used by GENI clients through the "qup-core" ICC + * path. Values are in Bps and must be converted with Bps_to_icc() before + * setting avg_bw. */ -#define CORE_2X_19_2_MHZ 960 -#define CORE_2X_50_MHZ 2500 -#define CORE_2X_100_MHZ 5000 -#define CORE_2X_150_MHZ 7500 -#define CORE_2X_200_MHZ 10000 -#define CORE_2X_236_MHZ 16383 +#define CORE_2X_19_2_MHZ 9600000 +#define CORE_2X_50_MHZ 25000000 #define GENI_DEFAULT_BW Bps_to_icc(1000) diff --git a/include/linux/soc/qcom/llcc-qcom.h b/include/linux/soc/qcom/llcc-qcom.h index f3ed63e475ab..713cb0221b14 100644 --- a/include/linux/soc/qcom/llcc-qcom.h +++ b/include/linux/soc/qcom/llcc-qcom.h @@ -74,6 +74,7 @@ #define LLCC_CAMSRTIP 73 #define LLCC_CAMRTRF 74 #define LLCC_CAMSRTRF 75 +#define LLCC_GPU_LITTLE 80 #define LLCC_OOBM_NS 81 #define LLCC_OOBM_S 82 #define LLCC_VIDEO_APV 83 @@ -85,7 +86,12 @@ #define LLCC_CAM_OFE_STROV 93 #define LLCC_CPUSS_HEU 94 #define LLCC_PCIE_TCU 97 +#define LLCC_GPUHTW_LITTLE 98 #define LLCC_MDM_PNG_FIXED 100 +#define LLCC_PPE_RXDESC 101 +#define LLCC_PPE_RXFILL 102 +#define LLCC_WLAN_5G 103 +#define LLCC_WLAN_6G 104 /** * struct llcc_slice_desc - Cache slice descriptor diff --git a/include/linux/soc/qcom/tc9563.h b/include/linux/soc/qcom/tc9563.h new file mode 100644 index 000000000000..086f37a40d80 --- /dev/null +++ b/include/linux/soc/qcom/tc9563.h @@ -0,0 +1,19 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2026 Qualcomm Innovation Center, Inc. All rights reserved. + * Author: Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com> + */ + +#ifndef __QCOM_TC9563_H +#define __QCOM_TC9563_H + +#define TC9563_GPIO_DEV_NAME "tc9563-gpio" + +#define TC9563_GPIO_IN0_OFFSET 0x801200 +#define TC9563_GPIO_EN0_OFFSET 0x801208 +#define TC9563_GPIO_OUT0_OFFSET 0x801210 + +#define TC9563_GPIO_CONFIG TC9563_GPIO_EN0_OFFSET +#define TC9563_RESET_GPIO TC9563_GPIO_OUT0_OFFSET + +#endif /* __QCOM_TC9563_H */ diff --git a/include/linux/soc/qcom/ubwc.h b/include/linux/soc/qcom/ubwc.h index f3a70360b177..b09ae05cbeae 100644 --- a/include/linux/soc/qcom/ubwc.h +++ b/include/linux/soc/qcom/ubwc.h @@ -36,6 +36,7 @@ struct qcom_ubwc_cfg_data { #define UBWC_4_3 0x40030000 #define UBWC_5_0 0x50000000 #define UBWC_6_0 0x60000000 +#define UBWC_7_0 0x70000000 #if IS_ENABLED(CONFIG_QCOM_UBWC_CONFIG) const struct qcom_ubwc_cfg_data *qcom_ubwc_config_get_data(void); @@ -108,6 +109,8 @@ static inline u32 qcom_ubwc_swizzle(const struct qcom_ubwc_cfg_data *cfg) static inline u32 qcom_ubwc_version_tag(const struct qcom_ubwc_cfg_data *cfg) { + if (cfg->ubwc_enc_version >= UBWC_7_0) + return 6; if (cfg->ubwc_enc_version >= UBWC_6_0) return 5; if (cfg->ubwc_enc_version >= UBWC_5_0) diff --git a/include/linux/soc/ti/k3-ringacc.h b/include/linux/soc/ti/k3-ringacc.h index 39b022b92598..7b7fc237c08f 100644 --- a/include/linux/soc/ti/k3-ringacc.h +++ b/include/linux/soc/ti/k3-ringacc.h @@ -57,13 +57,13 @@ struct k3_ringacc; struct k3_ring; /** - * enum k3_ring_cfg - RA ring configuration structure + * struct k3_ring_cfg - RA ring configuration structure * * @size: Ring size, number of elements * @elm_size: Ring element size * @mode: Ring operational mode * @flags: Ring configuration flags. Possible values: - * @K3_RINGACC_RING_SHARED: when set allows to request the same ring + * %K3_RINGACC_RING_SHARED: when set allows to request the same ring * few times. It's usable when the same ring is used as Free Host PD ring * for different flows, for example. * Note: Locking should be done by consumer if required @@ -88,11 +88,10 @@ struct k3_ring_cfg { /** * of_k3_ringacc_get_by_phandle - find a RA by phandle property * @np: device node - * @propname: property name containing phandle on RA node + * @property: property name containing phandle on RA node * - * Returns pointer on the RA - struct k3_ringacc - * or -ENODEV if not found, - * or -EPROBE_DEFER if not yet registered + * Return: Pointer to the RA, or ERR_PTR(-ENODEV) if the phandle cannot + * be resolved, or ERR_PTR(-EPROBE_DEFER) if the RA is not yet registered. */ struct k3_ringacc *of_k3_ringacc_get_by_phandle(struct device_node *np, const char *property); @@ -103,10 +102,8 @@ struct k3_ringacc *of_k3_ringacc_get_by_phandle(struct device_node *np, * k3_ringacc_request_ring - request ring from ringacc * @ringacc: pointer on ringacc * @id: ring id or K3_RINGACC_RING_ID_ANY for any general purpose ring - * @flags: - * @K3_RINGACC_RING_USE_PROXY: if set - proxy will be allocated and - * used to access ring memory. Sopported only for rings in - * Message/Credentials/Queue mode. + * @flags: Set %K3_RINGACC_RING_USE_PROXY to allocate a proxy for + * accessing ring memory in Message mode. * * Returns pointer on the Ring - struct k3_ring * or NULL in case of failure. @@ -126,8 +123,9 @@ int k3_ringacc_request_rings_pair(struct k3_ringacc *ringacc, */ void k3_ringacc_ring_reset(struct k3_ring *ring); /** - * k3_ringacc_ring_reset - ring reset for DMA rings + * k3_ringacc_ring_reset_dma - ring reset for DMA rings * @ring: pointer on Ring + * @occ: occupancy used by the DMA reset quirk, or zero to read it from hardware * * Resets ring internal state ((hw)occ, (hw)idx). Should be used for rings * which are read by K3 UDMA, like TX or Free Host PD rings. @@ -217,8 +215,8 @@ int k3_ringacc_ring_push(struct k3_ring *ring, void *elem); * @ring: pointer on ring * @elem: pointer on ring element buffer * - * Push one ring element from the ring head. Size of the ring element is - * determined by ring configuration &struct k3_ring_cfg elm_size.. + * Pop one ring element from the ring head. Size of the ring element is + * determined by ring configuration &struct k3_ring_cfg elm_size. * * Returns 0 on success, errno otherwise. */ @@ -242,7 +240,7 @@ int k3_ringacc_ring_push_head(struct k3_ring *ring, void *elem); * @ring: pointer on ring * @elem: pointer on ring element buffer * - * Push one ring element from the ring tail. Size of the ring element is + * Pop one ring element from the ring tail. Size of the ring element is * determined by ring configuration &struct k3_ring_cfg elm_size. * * Returns 0 on success, errno otherwise. @@ -256,7 +254,10 @@ u32 k3_ringacc_get_tisci_dev_id(struct k3_ring *ring); struct ti_sci_handle; /** - * struct struct k3_ringacc_init_data - Initialization data for DMA rings + * struct k3_ringacc_init_data - Initialization data for DMA rings + * @tisci: TI SCI firmware handle + * @tisci_dev_id: TI SCI device ID of the DMA controller + * @num_rings: Total number of available rings */ struct k3_ringacc_init_data { const struct ti_sci_handle *tisci; diff --git a/include/linux/spi/spi.h b/include/linux/spi/spi.h index 88d17fce02dc..fb4baa0d3398 100644 --- a/include/linux/spi/spi.h +++ b/include/linux/spi/spi.h @@ -169,6 +169,12 @@ extern void spi_transfer_cs_change_delay_exec(struct spi_message *msg, * @cs_inactive: delay to be introduced by the controller after CS is * deasserted. If @cs_change_delay is used from @spi_transfer, then the * two delays will be added up. + * @rx_sample_delay_ns: Delay in nanoseconds by which the controller should + * postpone sampling the incoming data, relative to the sampling point it + * uses by default. Describes the board rather than the device, namely the + * flight time of the clock and data signals between controller and + * device, and comes from the "rx-sample-delay-ns" property. Zero when the + * property is absent. * @chip_select: Array of physical chipselect, spi->chipselect[i] gives * the corresponding physical CS for logical CS i. * @num_chipselect: Number of physical chipselects used. @@ -235,6 +241,9 @@ struct spi_device { struct spi_delay cs_hold; struct spi_delay cs_inactive; + /* Additional delay before the incoming data is sampled, in ns */ + u32 rx_sample_delay_ns; + u8 chip_select[SPI_DEVICE_CS_CNT_MAX]; u8 num_chipselect; diff --git a/include/linux/splice.h b/include/linux/splice.h index 9dec4861d09f..0e6c955dc6ff 100644 --- a/include/linux/splice.h +++ b/include/linux/splice.h @@ -79,8 +79,8 @@ ssize_t add_to_pipe(struct pipe_inode_info *pipe, struct pipe_buffer *buf); ssize_t vfs_splice_read(struct file *in, loff_t *ppos, struct pipe_inode_info *pipe, size_t len, unsigned int flags); -ssize_t splice_direct_to_actor(struct file *file, struct splice_desc *sd, - splice_direct_actor *actor); +ssize_t vfs_splice_to_actor(struct file *file, loff_t pos, size_t count, + splice_direct_actor *actor, void *private); ssize_t do_splice(struct file *in, loff_t *off_in, struct file *out, loff_t *off_out, size_t len, unsigned int flags); ssize_t do_splice_direct(struct file *in, loff_t *ppos, struct file *out, diff --git a/include/linux/srcu.h b/include/linux/srcu.h index 7d9bc06df98d..1a8a465a5650 100644 --- a/include/linux/srcu.h +++ b/include/linux/srcu.h @@ -36,6 +36,8 @@ static inline int __init_srcu_struct(struct srcu_struct *ssp, const char *name, int __init_srcu_struct_fast(struct srcu_struct *ssp, const char *name, struct lock_class_key *key); int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name, struct lock_class_key *key); +int __init_srcu_struct_atomic(struct srcu_struct *ssp, const char *name, + struct lock_class_key *key); #endif // #ifndef CONFIG_TINY_SRCU #define init_srcu_struct_fast(ssp) \ @@ -52,6 +54,13 @@ int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name, __init_srcu_struct_fast_updown((ssp), #ssp, &__srcu_key); \ }) +#define init_srcu_struct_atomic(ssp) \ +({ \ + static struct lock_class_key __srcu_key; \ + \ + __init_srcu_struct_atomic((ssp), #ssp, &__srcu_key); \ +}) + #define __SRCU_DEP_MAP_INIT(srcu_name) .dep_map = { .name = #srcu_name }, #else /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */ @@ -64,6 +73,7 @@ static inline int __init_srcu_struct(struct srcu_struct *ssp, const char *name, #ifndef CONFIG_TINY_SRCU int init_srcu_struct_fast(struct srcu_struct *ssp); int init_srcu_struct_fast_updown(struct srcu_struct *ssp); +int init_srcu_struct_atomic(struct srcu_struct *ssp); #endif // #ifndef CONFIG_TINY_SRCU #define __SRCU_DEP_MAP_INIT(srcu_name) @@ -77,14 +87,18 @@ int init_srcu_struct_fast_updown(struct srcu_struct *ssp); }) /* Values for SRCU Tree srcu_data ->srcu_reader_flavor, but also used by rcutorture. */ -#define SRCU_READ_FLAVOR_NORMAL 0x1 // srcu_read_lock(). -#define SRCU_READ_FLAVOR_NMI 0x2 // srcu_read_lock_nmisafe(). -// 0x4 // SRCU-lite is no longer with us. -#define SRCU_READ_FLAVOR_FAST 0x4 // srcu_read_lock_fast(), also NMI-safe. -#define SRCU_READ_FLAVOR_FAST_UPDOWN 0x8 // srcu_read_lock_fast_updown(). +#define SRCU_READ_FLAVOR_NORMAL 0x01 // srcu_read_lock(). +#define SRCU_READ_FLAVOR_NMI 0x02 // srcu_read_lock_nmisafe(). +// 0x04 // SRCU-lite is no longer with us. +#define SRCU_READ_FLAVOR_FAST 0x04 // srcu_read_lock_fast(), also NMI-safe. +#define SRCU_READ_FLAVOR_FAST_UPDOWN 0x08 // srcu_read_lock_fast_updown(). +#define SRCU_READ_FLAVOR_ATOMIC 0x10 // srcu_read_lock_atomic(). #define SRCU_READ_FLAVOR_ALL (SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_NMI | \ - SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN) + SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN | \ + SRCU_READ_FLAVOR_ATOMIC) // All of the above. +#define SRCU_READ_FLAVOR_PREDEF (SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_ATOMIC) + // Flavors special DEFINE_SRCU() flavors. #define SRCU_READ_FLAVOR_SLOWGP (SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN) // Flavors requiring synchronize_rcu() // instead of smp_mb(). @@ -102,6 +116,7 @@ void call_srcu(struct srcu_struct *ssp, struct rcu_head *head, void (*func)(struct rcu_head *head)); void cleanup_srcu_struct(struct srcu_struct *ssp); void synchronize_srcu(struct srcu_struct *ssp); +void synchronize_srcu_atomic(struct srcu_struct *ssp); #define SRCU_GET_STATE_COMPLETED 0x1 @@ -307,6 +322,45 @@ static inline int srcu_read_lock(struct srcu_struct *ssp) } /** + * srcu_read_lock_atomic - register a new reader promising an atomic section + * @ssp: srcu_struct in which to register the new reader. + * + * As srcu_read_lock(), but the caller promises that the read-side + * critical section never sleeps and never blocks on anything which + * may itself depend on memory allocation to make progress. Preemption + * is disabled for the duration, which both enforces that promise (any + * sleepable call in the section will splat on every configuration) + * and bounds the section so that the update side may spin rather + * than sleep when waiting for readers: see synchronize_srcu_atomic(). + * + * The lock and matching srcu_read_unlock_atomic() must be invoked on + * the same CPU, from the same context; passing the return value to + * another task is not permitted for this flavor. + */ +static inline int srcu_read_lock_atomic(struct srcu_struct *ssp) + __acquires_shared(ssp) +{ + int retval; + + preempt_disable(); + /* + * Arm might_sleep() to catch even a *potentially* sleeping call + * in the section, not just an actual schedule: the atomic-domain + * promise must hold on every path, contended or not. In hardirq, + * softirq, or NMI the annotation would land on the interrupted + * task, and can also result in data races against that task's + * own non_block_start()/non_block_end() invocations; it is also + * redundant there, so skip it. + */ + if (in_task()) + non_block_start(); + srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); + retval = __srcu_read_lock(ssp); + srcu_lock_acquire(&ssp->dep_map); + return retval; +} + +/** * srcu_read_lock_fast - register a new reader for an SRCU-protected structure. * @ssp: srcu_struct in which to register the new reader. * @@ -499,6 +553,23 @@ static inline void srcu_read_unlock(struct srcu_struct *ssp, int idx) } /** + * srcu_read_unlock_atomic - unregister an atomic-section reader + * @ssp: srcu_struct from which to unregister the old reader. + * @idx: return value from corresponding srcu_read_lock_atomic(). + */ +static inline void srcu_read_unlock_atomic(struct srcu_struct *ssp, int idx) + __releases_shared(ssp) +{ + WARN_ON_ONCE(idx & ~0x1); + srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); + srcu_lock_release(&ssp->dep_map); + __srcu_read_unlock(ssp, idx); + if (in_task()) + non_block_end(); + preempt_enable(); +} + +/** * srcu_read_unlock_fast - unregister a old reader from an SRCU-protected structure. * @ssp: srcu_struct in which to unregister the old reader. * @scp: return value from corresponding srcu_read_lock_fast(). diff --git a/include/linux/srcutiny.h b/include/linux/srcutiny.h index fbcf13bc12d1..47a368f945e3 100644 --- a/include/linux/srcutiny.h +++ b/include/linux/srcutiny.h @@ -12,12 +12,14 @@ #define _LINUX_SRCU_TINY_H #include <linux/irq_work_types.h> +#include <linux/llist.h> #include <linux/swait.h> struct srcu_struct { short srcu_lock_nesting[2]; /* srcu_read_lock() nesting depth. */ u8 srcu_gp_running; /* GP workqueue running? */ u8 srcu_gp_waiting; /* GP waiting for readers? */ + u8 srcu_atomic_gp_flag; /* Serialize atomic GP work.*/ unsigned long srcu_idx; /* Current reader array element in bit 0x2. */ unsigned long srcu_idx_max; /* Furthest future srcu_idx request. */ struct swait_queue_head srcu_wq; @@ -26,6 +28,8 @@ struct srcu_struct { struct rcu_head **srcu_cb_tail; /* Pending callbacks: Tail. */ struct work_struct srcu_work; /* For driving grace periods. */ struct irq_work srcu_irq_work; /* Defer schedule_work() to irq work. */ + struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */ + struct irq_work defer_iw; /* Re-issues defer_cbs later. */ #ifdef CONFIG_DEBUG_LOCK_ALLOC struct lockdep_map dep_map; #endif /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */ @@ -33,6 +37,7 @@ struct srcu_struct { void srcu_drive_gp(struct work_struct *wp); void srcu_tiny_irq_work(struct irq_work *irq_work); +void srcu_defer_drain(struct irq_work *irq_work); #define __SRCU_STRUCT_INIT(name, __ignored, ___ignored, ____ignored) \ { \ @@ -40,6 +45,9 @@ void srcu_tiny_irq_work(struct irq_work *irq_work); .srcu_cb_tail = &name.srcu_cb_head, \ .srcu_work = __WORK_INITIALIZER(name.srcu_work, srcu_drive_gp), \ .srcu_irq_work = { .func = srcu_tiny_irq_work }, \ + .defer_cbs = LLIST_HEAD_INIT(name.defer_cbs), \ + .defer_iw = { .node = { .u_flags = IRQ_WORK_HARD_IRQ }, \ + .func = srcu_defer_drain }, \ __SRCU_DEP_MAP_INIT(name) \ } @@ -57,15 +65,20 @@ void srcu_tiny_irq_work(struct irq_work *irq_work); #define DEFINE_SRCU_FAST_UPDOWN(name) DEFINE_SRCU(name) #define DEFINE_STATIC_SRCU_FAST_UPDOWN(name) \ static struct srcu_struct name = __SRCU_STRUCT_INIT(name, name, name, name) +#define DEFINE_SRCU_ATOMIC(name) DEFINE_SRCU(name) +#define DEFINE_STATIC_SRCU_ATOMIC(name) \ + static struct srcu_struct name = __SRCU_STRUCT_INIT(name, name, name, name) // Dummy structure for srcu_notifier_head. struct srcu_usage { }; #define __SRCU_USAGE_INIT(name) { } #define __init_srcu_struct_fast __init_srcu_struct #define __init_srcu_struct_fast_updown __init_srcu_struct +#define __init_srcu_struct_atomic __init_srcu_struct #ifndef CONFIG_DEBUG_LOCK_ALLOC #define init_srcu_struct_fast init_srcu_struct #define init_srcu_struct_fast_updown init_srcu_struct +#define init_srcu_struct_atomic init_srcu_struct #endif // #ifndef CONFIG_DEBUG_LOCK_ALLOC void synchronize_srcu(struct srcu_struct *ssp); @@ -131,10 +144,7 @@ static inline void synchronize_srcu_expedited(struct srcu_struct *ssp) synchronize_srcu(ssp); } -static inline void srcu_barrier(struct srcu_struct *ssp) -{ - synchronize_srcu(ssp); -} +void srcu_barrier(struct srcu_struct *ssp); static inline void srcu_expedite_current(struct srcu_struct *ssp) { } #define srcu_check_read_flavor(ssp, read_flavor) do { } while (0) diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h index 75e54e4f963f..93b76e543894 100644 --- a/include/linux/srcutree.h +++ b/include/linux/srcutree.h @@ -13,6 +13,8 @@ #include <linux/rcu_node_tree.h> #include <linux/completion.h> +#include <linux/irq_work_types.h> +#include <linux/llist.h> struct srcu_node; struct srcu_struct; @@ -41,6 +43,8 @@ struct srcu_data { bool srcu_cblist_invoking; /* Invoking these CBs? */ struct timer_list delay_work; /* Delay for CB invoking */ struct work_struct work; /* Context for CB invoking. */ + struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */ + struct llist_node defer_link; /* Links onto the per-CPU deferral drain list */ struct rcu_head srcu_barrier_head; /* For srcu_barrier() use. */ struct rcu_head srcu_ec_head; /* For srcu_expedite_current() use. */ int srcu_ec_state; /* State for srcu_expedite_current(). */ @@ -76,6 +80,7 @@ struct srcu_usage { struct mutex srcu_cb_mutex; /* Serialize CB preparation. */ raw_spinlock_t __private lock; /* Protect counters and size state. */ struct mutex srcu_gp_mutex; /* Serialize GP work. */ + atomic_t srcu_atomic_gp_flag; /* Serialize atomic GP work. */ unsigned long srcu_gp_seq; /* Grace-period seq #. */ unsigned long srcu_gp_seq_needed; /* Latest gp_seq needed. */ unsigned long srcu_gp_seq_needed_exp; /* Furthest future exp GP. */ @@ -209,6 +214,12 @@ struct srcu_struct { * instead of smp_mb(), and given that the first (for example) * srcu_read_lock_fast() might race with the first synchronize_srcu(), * this different must be specified at initialization time. + * + * If you use any of the DEFINE_SRCU() functions within a module, the + * module-entry code will invoke init_srcu_struct() and the module-exit + * code will invoke cleanup_srcu_struct(). This means that if your module + * passes the resulting srcu_struct structure to call_srcu(), you will + * need to also pass this structure to srcu_barrier() prior to module exit. */ #ifdef MODULE # define __DEFINE_SRCU(name, fast, is_static) \ @@ -225,14 +236,16 @@ struct srcu_struct { is_static struct srcu_struct name = \ __SRCU_STRUCT_INIT(name, name##_srcu_usage, name##_srcu_data, fast) #endif -#define DEFINE_SRCU(name) __DEFINE_SRCU(name, 0, /* not static */) +#define DEFINE_SRCU(name) __DEFINE_SRCU(name, 0, /* !static */) #define DEFINE_STATIC_SRCU(name) __DEFINE_SRCU(name, 0, static) -#define DEFINE_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, /* not static */) +#define DEFINE_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, /* !static */) #define DEFINE_STATIC_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, static) #define DEFINE_SRCU_FAST_UPDOWN(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST_UPDOWN, \ - /* not static */) + /* !static */) #define DEFINE_STATIC_SRCU_FAST_UPDOWN(name) \ __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST_UPDOWN, static) +#define DEFINE_SRCU_ATOMIC(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_ATOMIC, /* !static */) +#define DEFINE_STATIC_SRCU_ATOMIC(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_ATOMIC, static) int __srcu_read_lock(struct srcu_struct *ssp) __acquires_shared(ssp); void synchronize_srcu_expedited(struct srcu_struct *ssp); diff --git a/include/linux/stmmac.h b/include/linux/stmmac.h index 4430b967abde..00be2df63d22 100644 --- a/include/linux/stmmac.h +++ b/include/linux/stmmac.h @@ -242,6 +242,8 @@ struct plat_stmmacenet_data { * that phylink uses. */ phy_interface_t phy_interface; + bool has_internal_tx_delay; + bool has_internal_rx_delay; struct stmmac_mdio_bus_data *mdio_bus_data; struct device_node *phy_node; struct device_node *mdio_node; diff --git a/include/linux/string.h b/include/linux/string.h index 5702daca4326..6cb5cdd01158 100644 --- a/include/linux/string.h +++ b/include/linux/string.h @@ -278,6 +278,19 @@ static inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) } #endif +#ifndef memcpy_nontemporal +/* + * memcpy_nontemporal() requests a non-temporal copy when the + * architecture has a suitable backend. Architectures without a + * specialized backend fall back to memcpy(). Keep this as a + * function-like macro so the compiler can still see the original + * memcpy() call site and preserve the usual FORTIFY coverage when + * object sizes remain visible there, while keeping the API void. + */ +#define memcpy_nontemporal(dst, src, len) \ + ((void)memcpy(dst, src, len)) +#endif + void *memchr_inv(const void *s, int c, size_t n); char *strreplace(char *str, char old, char new); diff --git a/include/linux/string_choices.h b/include/linux/string_choices.h index ee84087d4b26..a2eba454004f 100644 --- a/include/linux/string_choices.h +++ b/include/linux/string_choices.h @@ -65,6 +65,12 @@ static inline const char *str_read_write(bool v) } #define str_write_read(v) str_read_write(!(v)) +static inline const char *str_supported_unsupported(bool v) +{ + return v ? "supported" : "unsupported"; +} +#define str_unsupported_supported(v) str_supported_unsupported(!(v)) + static inline const char *str_true_false(bool v) { return v ? "true" : "false"; diff --git a/include/linux/sunrpc/svc_xprt.h b/include/linux/sunrpc/svc_xprt.h index da2a2531e110..2af222f3ea2c 100644 --- a/include/linux/sunrpc/svc_xprt.h +++ b/include/linux/sunrpc/svc_xprt.h @@ -37,6 +37,9 @@ struct svc_xprt_class { struct list_head xcl_list; u32 xcl_max_payload; int xcl_ident; + u32 xcl_flags; +/* Set only on classes whose xpo_has_wspace() reads xpt_reserved */ +#define SVC_XPRT_FLAG_WSPACE_RESERVE BIT(0) }; /* @@ -59,7 +62,7 @@ struct svc_xprt { unsigned long xpt_flags; struct svc_serv *xpt_server; /* service for transport */ - atomic_t xpt_reserved; /* space on outq that is rsvd */ + atomic_t xpt_reserved; /* outq space rsvd, UDP only */ atomic_t xpt_nr_rqsts; /* Number of requests */ struct mutex xpt_mutex; /* to serialize sending data */ spinlock_t xpt_lock; /* protects sk_deferred diff --git a/include/linux/swap.h b/include/linux/swap.h index 5658a1634b85..cb434cccd653 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -25,12 +25,6 @@ #define SWAP_FLAGS_VALID (SWAP_FLAG_PRIO_MASK | SWAP_FLAG_PREFER | \ SWAP_FLAG_DISCARD | SWAP_FLAG_DISCARD_ONCE | \ SWAP_FLAG_DISCARD_PAGES) - -static inline int current_is_kswapd(void) -{ - return current->flags & PF_KSWAPD; -} - /* * MAX_SWAPFILES defines the maximum number of swaptypes: things which can * be swapped to. The swap type and the offset into that swap type are @@ -246,7 +240,7 @@ struct swap_info_struct { struct plist_node list; /* entry in swap_active_head */ signed char type; /* strange name for an index */ unsigned int max; /* size of this swap device */ - struct swap_cluster_info *cluster_info; /* cluster info. Only for SSD */ + struct swap_cluster_info *cluster_info; /* array, one entry per cluster */ struct list_head free_clusters; /* free clusters list */ struct list_head full_clusters; /* full clusters list */ struct list_head nonfull_clusters[SWAP_NR_ORDERS]; @@ -278,15 +272,39 @@ struct swap_info_struct { const struct swap_ops *ops; }; -static inline swp_entry_t page_swap_entry(struct page *page) +/** + * folio_swap_entry - Return the swap entry at a page index within a folio. + * @folio: The folio. + * @idx: The index of the page within the folio. + * + * A folio in the swap cache occupies folio_nr_pages() contiguous swap + * entries starting at folio->swap. The caller must ensure the folio is + * in the swap cache and that @idx is within the folio. + */ +static inline +swp_entry_t folio_swap_entry(const struct folio *folio, unsigned long idx) { - struct folio *folio = page_folio(page); swp_entry_t entry = folio->swap; - entry.val += folio_page_idx(folio, page); + VM_WARN_ON_ONCE_FOLIO(idx >= folio_nr_pages(folio), folio); + entry.val += idx; return entry; } +/** + * folio_page_swap_entry - Return the swap entry of a page in a folio. + * @folio: The folio containing @page. + * @page: A page within @folio. + * + * The caller must ensure the folio is in the swap cache and that @page + * is part of @folio. + */ +static inline swp_entry_t folio_page_swap_entry(const struct folio *folio, + const struct page *page) +{ + return folio_swap_entry(folio, folio_page_idx(folio, page)); +} + /* linux/mm/page_alloc.c */ extern unsigned long totalreserve_pages; @@ -318,6 +336,9 @@ static inline bool lru_cache_disabled(void) extern unsigned long shrink_all_memory(unsigned long nr_pages); long remove_mapping(struct address_space *mapping, struct folio *folio); +long remove_mapping_set_shadow(struct address_space *mapping, + struct folio *folio, + struct mem_cgroup *target_memcg); #if defined(CONFIG_SYSFS) && defined(CONFIG_NUMA) extern int reclaim_register_node(struct node *node); @@ -339,6 +360,7 @@ void check_move_unevictable_folios(struct folio_batch *fbatch); extern void __meminit kswapd_run(int nid); extern void __meminit kswapd_stop(int nid); +bool current_is_kswapd(void); #ifdef CONFIG_SWAP int add_swap_extent(struct swap_info_struct *sis, unsigned long start_page, @@ -402,6 +424,8 @@ void swap_put_entries_direct(swp_entry_t entry, int nr); */ bool folio_free_swap(struct folio *folio); +void swap_writeback_dropbehind_folio(struct folio *folio); + /* Allocate / free (hibernation) exclusive entries */ swp_entry_t swap_alloc_hibernation_slot(int type); void swap_free_hibernation_slot(swp_entry_t entry); @@ -412,6 +436,7 @@ static inline void put_swap_device(struct swap_info_struct *si) } #else /* CONFIG_SWAP */ +static inline void swap_writeback_dropbehind_folio(struct folio *folio) {} static inline struct swap_info_struct *get_swap_device(swp_entry_t entry) { return NULL; @@ -508,8 +533,9 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p __mem_cgroup_uncharge_swap(id, nr_pages); } -extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg); -extern bool mem_cgroup_swap_full(struct folio *folio); +long mem_cgroup_get_folio_swap_margin(const struct folio *folio); +long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg); +bool mem_cgroup_swap_full(const struct folio *folio); #else static inline int mem_cgroup_try_charge_swap(struct folio *folio) { @@ -521,12 +547,17 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, { } -static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) +static inline long mem_cgroup_get_folio_swap_margin(const struct folio *folio) +{ + return PAGE_COUNTER_MAX; +} + +static inline long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg) { return get_nr_swap_pages(); } -static inline bool mem_cgroup_swap_full(struct folio *folio) +static inline bool mem_cgroup_swap_full(const struct folio *folio) { return vm_swap_full(); } diff --git a/include/linux/swapops.h b/include/linux/swapops.h index e7d0d529f3e0..603f9e3f909c 100644 --- a/include/linux/swapops.h +++ b/include/linux/swapops.h @@ -261,11 +261,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page) return swp_entry(SWP_HWPOISON, page_to_pfn(page)); } -static inline int is_hwpoison_entry(swp_entry_t entry) -{ - return swp_type(entry) == SWP_HWPOISON; -} - #else static inline swp_entry_t make_hwpoison_entry(struct page *page) @@ -273,10 +268,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page) return swp_entry(0, 0); } -static inline int is_hwpoison_entry(swp_entry_t swp) -{ - return 0; -} #endif typedef unsigned long pte_marker; diff --git a/include/linux/thermal.h b/include/linux/thermal.h index 083b4f533933..306ad17aed89 100644 --- a/include/linux/thermal.h +++ b/include/linux/thermal.h @@ -293,8 +293,9 @@ struct device *thermal_zone_device(struct thermal_zone_device *tzd); void thermal_zone_device_update(struct thermal_zone_device *, enum thermal_notify_event); -struct thermal_cooling_device *thermal_cooling_device_register(const char *, - void *, const struct thermal_cooling_device_ops *); +struct thermal_cooling_device *thermal_cooling_device_create( + struct device *parent, const char *type, void *devdata, + const struct thermal_cooling_device_ops *ops); struct thermal_cooling_device * devm_thermal_cooling_device_register(struct device *dev, const char *type, void *devdata, @@ -340,9 +341,9 @@ static inline void thermal_zone_device_update(struct thermal_zone_device *tz, enum thermal_notify_event event) { } -static inline struct thermal_cooling_device * -thermal_cooling_device_register(const char *type, void *devdata, - const struct thermal_cooling_device_ops *ops) +static inline struct thermal_cooling_device *thermal_cooling_device_create( + struct device *parent, const char *type, void *devdata, + const struct thermal_cooling_device_ops *ops) { return ERR_PTR(-ENODEV); } static inline struct thermal_cooling_device * @@ -391,4 +392,12 @@ static inline void thermal_pm_prepare(void) {} static inline void thermal_pm_complete(void) {} #endif /* CONFIG_THERMAL */ +static inline struct thermal_cooling_device *thermal_cooling_device_register( + const char *type, void *devdata, + const struct thermal_cooling_device_ops *ops) +{ + return thermal_cooling_device_create(NULL, type, devdata, ops); +} + + #endif /* __THERMAL_H__ */ diff --git a/include/linux/thunderbolt.h b/include/linux/thunderbolt.h index d48623fda79b..b62dfa52b149 100644 --- a/include/linux/thunderbolt.h +++ b/include/linux/thunderbolt.h @@ -22,6 +22,7 @@ struct device; #include <linux/device.h> #include <linux/idr.h> #include <linux/list.h> +#include <linux/lockdep.h> #include <linux/mutex.h> #include <linux/device-id/tb.h> #include <linux/pci.h> @@ -565,6 +566,7 @@ struct tb_nhi { * @interval_nsec: Interval counter if interrupt throttling is to be * used with this ring (in ns) * @wait: Used to signal that the ring may be empty now + * @lock_key: Lock validator class key per-ring */ struct tb_ring { spinlock_t lock; @@ -590,6 +592,7 @@ struct tb_ring { void *poll_data; unsigned int interval_nsec; wait_queue_head_t wait; + struct lock_class_key lock_key; }; /* Leave ring interrupt enabled on suspend */ diff --git a/include/linux/torture.h b/include/linux/torture.h index c9b47d138302..1b6d5be641f8 100644 --- a/include/linux/torture.h +++ b/include/linux/torture.h @@ -98,6 +98,7 @@ void torture_shutdown_absorb(const char *title); int torture_shutdown_init(int ssecs, void (*cleanup)(void)); /* Task stuttering, which forces load/no-load transitions. */ +bool stutter_will_wait(void); bool stutter_wait(const char *title); int torture_stutter_init(int s, int sgap); @@ -131,8 +132,11 @@ void _torture_stop_kthread(char *m, struct task_struct **tp); #endif void torture_sched_set_normal(struct task_struct *t, int nice); -#if IS_ENABLED(CONFIG_RCU_TORTURE_TEST) || IS_MODULE(CONFIG_RCU_TORTURE_TEST) || IS_ENABLED(CONFIG_LOCK_TORTURE_TEST) || IS_MODULE(CONFIG_LOCK_TORTURE_TEST) +#if IS_ENABLED(CONFIG_RCU_TORTURE_TEST) || IS_ENABLED(CONFIG_LOCK_TORTURE_TEST) || IS_ENABLED(CONFIG_HAZPTR_TORTURE_TEST) long torture_sched_setaffinity(pid_t pid, const struct cpumask *in_mask, bool dowarn); #endif +/* Atomic per-CPU counters. */ +s64 torture_sum_pcpu_atomic_long(atomic_long_t __percpu *pcp); + #endif /* __LINUX_TORTURE_H */ diff --git a/include/linux/trace_remote_event.h b/include/linux/trace_remote_event.h index c8ae1e1f5e72..e4cc2d4497bc 100644 --- a/include/linux/trace_remote_event.h +++ b/include/linux/trace_remote_event.h @@ -3,6 +3,8 @@ #ifndef _LINUX_TRACE_REMOTE_EVENTS_H #define _LINUX_TRACE_REMOTE_EVENTS_H +#include <linux/types.h> + struct trace_remote; struct trace_event_fields; struct trace_seq; diff --git a/include/linux/tty.h b/include/linux/tty.h index 0a46e4054dec..833bb7b97ceb 100644 --- a/include/linux/tty.h +++ b/include/linux/tty.h @@ -168,6 +168,7 @@ struct tty_operations; * @write_wait: concurrent writers are waiting in this queue until they are * allowed to write * @read_wait: readers wait for data in this queue + * @break_wait: wait queue for timed breaks * @hangup_work: normally a work to perform a hangup (do_tty_hangup()); while * freeing the tty, (re)used to release_one_tty() * @disc_data: pointer to @ldisc's private data (e.g. to &struct n_tty_data) @@ -230,6 +231,7 @@ struct tty_struct { struct fasync_struct *fasync; wait_queue_head_t write_wait; wait_queue_head_t read_wait; + wait_queue_head_t break_wait; struct work_struct hangup_work; void *disc_data; void *driver_data; diff --git a/include/linux/tty_driver.h b/include/linux/tty_driver.h index 1f2896e56e77..62e459c5d3ae 100644 --- a/include/linux/tty_driver.h +++ b/include/linux/tty_driver.h @@ -73,6 +73,11 @@ struct serial_struct; * @TTY_DRIVER_NO_WORKQUEUE: * Do not create workqueue when tty_register_driver(). Whenever set, flip * buffer workqueue can be set by tty_port_link_wq() for every port. + * + * @TTY_DRIVER_RESET_SAVED_TERMIOS + * Reset any saved termios settings on device registration when reusing a + * minor number. Must only be set by drivers that guarantee that the minor + * number is no longer in use. */ enum tty_driver_flag { TTY_DRIVER_INSTALLED = BIT(0), @@ -84,6 +89,7 @@ enum tty_driver_flag { TTY_DRIVER_DYNAMIC_ALLOC = BIT(6), TTY_DRIVER_UNNUMBERED_NODE = BIT(7), TTY_DRIVER_NO_WORKQUEUE = BIT(8), + TTY_DRIVER_RESET_SAVED_TERMIOS = BIT(9), }; enum tty_driver_type { diff --git a/include/linux/tty_port.h b/include/linux/tty_port.h index 23cad403bb8f..8d22c59c6f15 100644 --- a/include/linux/tty_port.h +++ b/include/linux/tty_port.h @@ -245,7 +245,8 @@ bool tty_port_carrier_raised(struct tty_port *port); void tty_port_raise_dtr_rts(struct tty_port *port); void tty_port_lower_dtr_rts(struct tty_port *port); void tty_port_hangup(struct tty_port *port); -void __tty_port_tty_hangup(struct tty_port *port, bool check_clocal, bool async); +void tty_port_tty_hangup(struct tty_port *port, bool check_clocal); +void tty_port_tty_vhangup(struct tty_port *port); void tty_port_tty_wakeup(struct tty_port *port); int tty_port_block_til_ready(struct tty_port *port, struct tty_struct *tty, struct file *filp); @@ -264,25 +265,6 @@ static inline int tty_port_users(struct tty_port *port) return port->count + port->blocked_open; } -/** - * tty_port_tty_hangup - helper to hang up a tty asynchronously - * @port: tty port - * @check_clocal: hang only ttys with %CLOCAL unset? - */ -static inline void tty_port_tty_hangup(struct tty_port *port, bool check_clocal) -{ - __tty_port_tty_hangup(port, check_clocal, true); -} - -/** - * tty_port_tty_vhangup - helper to hang up a tty synchronously - * @port: tty port - */ -static inline void tty_port_tty_vhangup(struct tty_port *port) -{ - __tty_port_tty_hangup(port, false, false); -} - #ifdef CONFIG_TTY void tty_kref_put(struct tty_struct *tty); __DEFINE_CLASS_IS_CONDITIONAL(tty_port_tty, true); diff --git a/include/linux/uidgid.h b/include/linux/uidgid.h index 2dc767e08f54..02403629b49f 100644 --- a/include/linux/uidgid.h +++ b/include/linux/uidgid.h @@ -130,9 +130,9 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return from_kgid(ns, gid) != (gid_t) -1; } -u32 map_id_down(struct uid_gid_map *map, u32 id); -u32 map_id_up(struct uid_gid_map *map, u32 id); -u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count); +u32 map_id_down(const struct uid_gid_map *map, u32 id); +u32 map_id_up(const struct uid_gid_map *map, u32 id); +u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count); #else @@ -182,17 +182,17 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return gid_valid(gid); } -static inline u32 map_id_down(struct uid_gid_map *map, u32 id) +static inline u32 map_id_down(const struct uid_gid_map *map, u32 id) { return id; } -static inline u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count) +static inline u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count) { return id; } -static inline u32 map_id_up(struct uid_gid_map *map, u32 id) +static inline u32 map_id_up(const struct uid_gid_map *map, u32 id) { return id; } diff --git a/include/linux/usb/serial.h b/include/linux/usb/serial.h index 534e6650e2aa..ab0a32a90629 100644 --- a/include/linux/usb/serial.h +++ b/include/linux/usb/serial.h @@ -381,6 +381,7 @@ void usb_serial_handle_dcd_change(struct usb_serial_port *usb_port, int usb_serial_bus_register(struct usb_serial_driver *device); void usb_serial_bus_deregister(struct usb_serial_driver *device); +void usb_serial_bus_remove_new_id(struct usb_serial_driver *driver); extern const struct bus_type usb_serial_bus_type; extern struct tty_driver *usb_serial_tty_driver; diff --git a/include/linux/user_namespace.h b/include/linux/user_namespace.h index e38d9e60569f..91232053775d 100644 --- a/include/linux/user_namespace.h +++ b/include/linux/user_namespace.h @@ -29,8 +29,8 @@ struct uid_gid_map { /* 64 bytes -- 1 cache line */ u32 nr_extents; }; struct { - struct uid_gid_extent *forward; - struct uid_gid_extent *reverse; + struct uid_gid_extent *forward __counted_by_ptr(nr_extents); + struct uid_gid_extent *reverse __counted_by_ptr(nr_extents); }; }; }; @@ -207,6 +207,13 @@ extern bool in_userns(const struct user_namespace *ancestor, const struct user_namespace *child); extern bool current_in_userns(const struct user_namespace *target_ns); struct ns_common *ns_get_owner(struct ns_common *ns); + +#if IS_ENABLED(CONFIG_KUNIT) +extern int uid_gid_map_insert_extent(struct uid_gid_map *map, + struct uid_gid_extent *extent); +extern int uid_gid_map_sort(struct uid_gid_map *map); +#endif /* CONFIG_KUNIT */ + #else static inline struct user_namespace *get_user_ns(struct user_namespace *ns) diff --git a/include/linux/userfaultfd_k.h b/include/linux/userfaultfd_k.h index a4351cffc60c..a14b8a9ffb7b 100644 --- a/include/linux/userfaultfd_k.h +++ b/include/linux/userfaultfd_k.h @@ -18,7 +18,6 @@ #include <linux/swap.h> #include <linux/leafops.h> #include <asm-generic/pgtable_uffd.h> -#include <linux/hugetlb_inline.h> /* The set of all possible UFFD-related VM flags. */ #define __VM_UFFD_FLAGS (VM_UFFD_MISSING | VM_UFFD_MINOR | \ diff --git a/include/linux/vga_switcheroo.h b/include/linux/vga_switcheroo.h index 7e6ac0114d55..d7eee65664ab 100644 --- a/include/linux/vga_switcheroo.h +++ b/include/linux/vga_switcheroo.h @@ -31,8 +31,12 @@ #ifndef _LINUX_VGA_SWITCHEROO_H_ #define _LINUX_VGA_SWITCHEROO_H_ -#include <linux/fb.h> +#include <linux/errno.h> +#include <linux/types.h> +struct device; +struct dev_pm_domain; +struct fb_info; struct pci_dev; /** @@ -127,23 +131,28 @@ struct vga_switcheroo_handler { * @set_gpu_state: do the equivalent of suspend/resume for the card. * Mandatory. This should not cut power to the discrete GPU, * which is the job of the handler - * @reprobe: poll outputs. - * Optional. This gets called after waking the GPU and switching - * the outputs to it * @can_switch: check if the device is in a position to switch now. * Mandatory. The client should return false if a user space process * has one of its device files open + * @pre_switch: prepare switch + * Optional. This gets called before switching the outputs to the + * GPU. Allows drivers to prepare for the switch. + * @post_switch: completes switch + * Optional. This gets called after waking the GPU and switching + * the outputs to it. Allows drivers to poll the switched outputs. * @gpu_bound: notify the client id to audio client when the GPU is bound. * * Client callbacks. A client can be either a GPU or an audio device on a GPU. - * The @set_gpu_state and @can_switch methods are mandatory, @reprobe may be - * set to NULL. For audio clients, the @reprobe member is bogus. - * OTOH, @gpu_bound is only for audio clients, and not used for GPU clients. + * The @set_gpu_state and @can_switch methods are mandatory, @pre_switch and + * @post_switch may be set to NULL. For audio clients, the @pre_switch and + * @post_switch members are bogus. OTOH, @gpu_bound is only for audio clients, + * and not used for GPU clients. */ struct vga_switcheroo_client_ops { void (*set_gpu_state)(struct pci_dev *dev, enum vga_switcheroo_state); - void (*reprobe)(struct pci_dev *dev); bool (*can_switch)(struct pci_dev *dev); + void (*pre_switch)(struct pci_dev *dev); + void (*post_switch)(struct pci_dev *dev); void (*gpu_bound)(struct pci_dev *dev, enum vga_switcheroo_client_id); }; @@ -156,9 +165,6 @@ int vga_switcheroo_register_audio_client(struct pci_dev *pdev, const struct vga_switcheroo_client_ops *ops, struct pci_dev *vga_dev); -void vga_switcheroo_client_fb_set(struct pci_dev *dev, - struct fb_info *info); - int vga_switcheroo_register_handler(const struct vga_switcheroo_handler *handler, enum vga_switcheroo_handler_flags_t handler_flags); void vga_switcheroo_unregister_handler(void); @@ -178,7 +184,6 @@ void vga_switcheroo_fini_domain_pm_ops(struct device *dev); static inline void vga_switcheroo_unregister_client(struct pci_dev *dev) {} static inline int vga_switcheroo_register_client(struct pci_dev *dev, const struct vga_switcheroo_client_ops *ops, bool driver_power_control) { return 0; } -static inline void vga_switcheroo_client_fb_set(struct pci_dev *dev, struct fb_info *info) {} static inline int vga_switcheroo_register_handler(const struct vga_switcheroo_handler *handler, enum vga_switcheroo_handler_flags_t handler_flags) { return 0; } static inline int vga_switcheroo_register_audio_client(struct pci_dev *pdev, diff --git a/include/linux/vgaarb.h b/include/linux/vgaarb.h index 97129a1bbb7d..71a364669eaf 100644 --- a/include/linux/vgaarb.h +++ b/include/linux/vgaarb.h @@ -33,7 +33,8 @@ struct pci_dev *vga_default_device(void); void vga_set_default_device(struct pci_dev *pdev); int vga_remove_vgacon(struct pci_dev *pdev); int vga_client_register(struct pci_dev *pdev, - unsigned int (*set_decode)(struct pci_dev *pdev, bool state)); + unsigned int (*set_decode)(void *data, bool state), + void *data); #else /* CONFIG_VGA_ARB */ static inline void vga_set_legacy_decoding(struct pci_dev *pdev, unsigned int decodes) @@ -59,7 +60,8 @@ static inline int vga_remove_vgacon(struct pci_dev *pdev) return 0; } static inline int vga_client_register(struct pci_dev *pdev, - unsigned int (*set_decode)(struct pci_dev *pdev, bool state)) + unsigned int (*set_decode)(void *data, bool state), + void *data) { return 0; } @@ -97,7 +99,7 @@ static inline int vga_get_uninterruptible(struct pci_dev *pdev, static inline void vga_client_unregister(struct pci_dev *pdev) { - vga_client_register(pdev, NULL); + vga_client_register(pdev, NULL, NULL); } #endif /* LINUX_VGA_H */ diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h index aed121d729b0..034a693777ca 100644 --- a/include/linux/vmalloc.h +++ b/include/linux/vmalloc.h @@ -3,6 +3,7 @@ #define _LINUX_VMALLOC_H #include <linux/alloc_tag.h> +#include <linux/cleanup.h> #include <linux/sched.h> #include <linux/spinlock.h> #include <linux/init.h> @@ -38,6 +39,7 @@ struct iov_iter; /* in uio.h */ #define VM_DEFER_KMEMLEAK 0 #endif #define VM_SPARSE 0x00001000 /* sparse vm_area. not all pages are present. */ +#define VM_REQUIRE_HUGE_VMAP 0x00002000 /* huge page mapping or nothing */ /* bits [20..32] reserved for arch specific ioremap internals */ @@ -214,6 +216,8 @@ void *__must_check vrealloc_node_align_noprof(const void *p, size_t size, extern void vfree(const void *addr); extern void vfree_atomic(const void *addr); +DEFINE_FREE(vfree, void *, if (!IS_ERR_OR_NULL(_T)) vfree(_T)) + extern void *vmap(struct page **pages, unsigned int count, unsigned long flags, pgprot_t prot); void *vmap_pfn(unsigned long *pfns, unsigned int count, pgprot_t prot); diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h new file mode 100644 index 000000000000..fa9e9abd6656 --- /dev/null +++ b/include/linux/vmemmap-optimization.h @@ -0,0 +1,115 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * vmemmap-optimization.h + * + * Generic vmemmap optimization declarations. + * + * Author: Muchun Song <songmuchun@bytedance.com> + */ +#ifndef _LINUX_VMEMMAP_OPTIMIZATION_H +#define _LINUX_VMEMMAP_OPTIMIZATION_H + +#include <linux/align.h> +#include <linux/log2.h> +#include <linux/mmdebug.h> +#include <linux/mmzone.h> + +/* + * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to + * be naturally aligned with regard to the folio size. + * + * HVO which is only active if the size of struct page is a power of 2. + */ +#define MAX_FOLIO_VMEMMAP_ALIGN \ + (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ + is_power_of_2(sizeof(struct page)) ? \ + MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) + +/* The number of retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_PAGES 1 +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ + (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) +#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) + +#ifdef CONFIG_VMEMMAP_OPTIMIZATION +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return section->compound_page_order; +} + +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ + VM_WARN_ON(section_compound_order(section) && order && + section_compound_order(section) != order); + section->compound_page_order = order; +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_compound_order(__nr_to_section(section_nr + i), order); +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return section_compound_order(__pfn_to_section(pfn)); +} + +struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); +#else +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return 0; +} + +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return 0; +} + +static inline struct page *vmemmap_shared_tail_page(unsigned int order, + struct zone *zone) +{ + return NULL; +} +#endif /* CONFIG_VMEMMAP_OPTIMIZATION */ + +static inline bool vmemmap_optimizable_pfn(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_compound_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + +static inline bool vmemmap_optimizable_order(unsigned int order) +{ + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) + return false; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; +} +#endif /* _LINUX_VMEMMAP_OPTIMIZATION_H */ diff --git a/include/linux/wait.h b/include/linux/wait.h index 7e215330199c..c2af98b0074d 100644 --- a/include/linux/wait.h +++ b/include/linux/wait.h @@ -556,7 +556,7 @@ do { \ } \ \ __ret = ___wait_event(wq_head, condition, state, 0, 0, \ - if (!__t.task) { \ + if (!hrtimer_sleeper_task_get(&__t)) { \ __ret = -ETIME; \ break; \ } \ diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h index 553d7b23e3ad..af077ed4caf6 100644 --- a/include/linux/wait_bit.h +++ b/include/linux/wait_bit.h @@ -433,6 +433,32 @@ do { \ }) /** + * wait_var_event_state - wait for a variable to be updated and notified + * @var: the address of variable being waited on + * @condition: the condition to wait for + * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc. + * + * Wait for a @condition to be true, only re-checking when a wake up is + * received for the given @var (an arbitrary kernel address which need + * not be directly related to the given condition, but usually is). + * + * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal + * arrived which @state allows to interrupt. + * + * The condition should normally use smp_load_acquire() or a similarly + * ordered access to ensure that any changes to memory made before the + * condition became true will be visible after the wait completes. + */ +#define wait_var_event_state(var, condition, state) \ +({ \ + int __ret = 0; \ + might_sleep(); \ + if (!(condition)) \ + __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \ + __ret; \ +}) + +/** * wait_var_event_any_lock - wait for a variable to be updated under a lock * @var: the address of the variable being waited on * @condition: condition to wait for diff --git a/include/linux/xattr.h b/include/linux/xattr.h index 54ac3cbc133f..4cc4257de084 100644 --- a/include/linux/xattr.h +++ b/include/linux/xattr.h @@ -47,7 +47,7 @@ struct xattr_handler { struct inode *inode, const char *name, void *buffer, size_t size); int (*set)(const struct xattr_handler *, - struct mnt_idmap *idmap, struct dentry *dentry, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags); }; @@ -77,25 +77,25 @@ struct xattr { }; ssize_t __vfs_getxattr(struct dentry *, struct inode *, const char *, void *, size_t); -ssize_t vfs_getxattr(struct mnt_idmap *, struct dentry *, const char *, +ssize_t vfs_getxattr(const struct mnt_idmap *, struct dentry *, const char *, void *, size_t); ssize_t vfs_listxattr(struct dentry *d, char *list, size_t size); -int __vfs_setxattr(struct mnt_idmap *, struct dentry *, struct inode *, +int __vfs_setxattr(const struct mnt_idmap *, struct dentry *, struct inode *, const char *, const void *, size_t, int); -int __vfs_setxattr_noperm(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_noperm(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_setxattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_locked(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int, struct delegated_inode *); -int vfs_setxattr(struct mnt_idmap *, struct dentry *, const char *, +int vfs_setxattr(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); -int __vfs_removexattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); +int __vfs_removexattr_locked(const struct mnt_idmap *, struct dentry *, const char *, struct delegated_inode *); -int vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); +int vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); ssize_t generic_listxattr(struct dentry *dentry, char *buffer, size_t buffer_size); -int vfs_getxattr_alloc(struct mnt_idmap *idmap, +int vfs_getxattr_alloc(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, char **xattr_value, size_t size, gfp_t flags); diff --git a/include/linux/zsmalloc.h b/include/linux/zsmalloc.h index 478410c880b1..5b7298a5026b 100644 --- a/include/linux/zsmalloc.h +++ b/include/linux/zsmalloc.h @@ -40,10 +40,6 @@ unsigned int zs_lookup_class_index(struct zs_pool *pool, unsigned int size); void zs_pool_stats(struct zs_pool *pool, struct zs_pool_stats *stats); -void *zs_obj_read_begin(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *local_copy); -void zs_obj_read_end(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *handle_mem); void zs_obj_read_sg_begin(struct zs_pool *pool, unsigned long handle, struct scatterlist *sg, size_t mem_len); void zs_obj_read_sg_end(struct zs_pool *pool, unsigned long handle); diff --git a/include/linux/zswap.h b/include/linux/zswap.h index 30c193a1207e..df6cafbe95dc 100644 --- a/include/linux/zswap.h +++ b/include/linux/zswap.h @@ -27,7 +27,7 @@ struct zswap_lruvec_state { unsigned long zswap_total_pages(void); bool zswap_store(struct folio *folio); int zswap_load(struct folio *folio); -void zswap_invalidate(swp_entry_t swp); +void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries); int zswap_swapon(int type, unsigned long nr_pages); void zswap_swapoff(int type); void zswap_memcg_offline_cleanup(struct mem_cgroup *memcg); @@ -49,7 +49,11 @@ static inline int zswap_load(struct folio *folio) return -ENOENT; } -static inline void zswap_invalidate(swp_entry_t swp) {} +static inline void zswap_invalidate(int type, pgoff_t offset, + unsigned long nr_entries) +{ +} + static inline int zswap_swapon(int type, unsigned long nr_pages) { return 0; |
