From 9db11c69fcc3147d158943930ce3f2c32620054d Mon Sep 17 00:00:00 2001 From: Vova Sharaienko Date: Tue, 23 Jun 2026 00:08:20 +0000 Subject: dma-coherent: fix spacing coding style issue Fixed spacing around * coding style issue Signed-off-by: Vova Sharaienko [mszyprow: changed patch prefix] Signed-off-by: Marek Szyprowski Link: https://lore.kernel.org/r/20260623000821.2269955-1-sharaienko@google.com --- kernel/dma/coherent.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) (limited to 'kernel') diff --git a/kernel/dma/coherent.c b/kernel/dma/coherent.c index bcdc0f76d2e8..480dd1766ece 100644 --- a/kernel/dma/coherent.c +++ b/kernel/dma/coherent.c @@ -28,7 +28,7 @@ static inline struct dma_coherent_mem *dev_get_coherent_memory(struct device *de } static inline dma_addr_t dma_get_device_base(struct device *dev, - struct dma_coherent_mem * mem) + struct dma_coherent_mem *mem) { if (mem->use_dev_dma_pfn_offset) return phys_to_dma(dev, PFN_PHYS(mem->pfn_base)); -- cgit v1.2.3 From dbcc3cd58083521d77f0e43db615bca2c17b0dbf Mon Sep 17 00:00:00 2001 From: Vova Sharaienko Date: Mon, 29 Jun 2026 22:37:58 +0000 Subject: dma-coherent: use KiB in DMA allocation logs Update DMA reserved memory pool allocation log messages to display sizes in KiB instead of MiB. Using MiB caused allocations less than 1 MiB to be logged as 0 MiB due to integer truncation. KiB provides better precision for smaller memory regions specified in the Device Tree. Signed-off-by: Vova Sharaienko Signed-off-by: Marek Szyprowski Link: https://lore.kernel.org/r/20260629223759.2637162-1-sharaienko@google.com --- kernel/dma/coherent.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/coherent.c b/kernel/dma/coherent.c index 480dd1766ece..5e005e0679f8 100644 --- a/kernel/dma/coherent.c +++ b/kernel/dma/coherent.c @@ -69,8 +69,8 @@ out_free_dma_mem: kfree(dma_mem); out_unmap_membase: memunmap(mem_base); - pr_err("Reserved memory: failed to init DMA memory pool at %pa, size %zd MiB\n", - &phys_addr, size / SZ_1M); + pr_err("Reserved memory: failed to init DMA memory pool at %pa, size %zu KiB\n", + &phys_addr, size / SZ_1K); return ERR_PTR(-ENOMEM); } @@ -384,8 +384,8 @@ static int __init rmem_dma_setup(unsigned long node, struct reserved_mem *rmem) } #endif - pr_info("Reserved memory: created DMA memory pool at %pa, size %ld MiB\n", - &rmem->base, (unsigned long)rmem->size / SZ_1M); + pr_info("Reserved memory: created DMA memory pool at %pa, size %llu KiB\n", + &rmem->base, (unsigned long long)(rmem->size / SZ_1K)); return 0; } -- cgit v1.2.3 From 0188616c057d3eba8d8a4af62a59b5e7f2362571 Mon Sep 17 00:00:00 2001 From: Marek Szyprowski Date: Mon, 13 Jul 2026 10:30:22 +0200 Subject: dma-direct: Improve readability of the dma_direct_map_sg() for P2PDMA case Improve readability of the sg_dma_len assignment in the P2PDMA cases by removing duplicated code, which was a direct result of the d0d08f4bd7f6 ("dma-direct: Fix missing sg_dma_len assignment in P2PDMA bus mappings") fix. No functional change. Suggested-by: Leon Romanovsky Link: https://lore.kernel.org/all/20260604071856.GA245424@unreal/ Reviewed-by: Pranjal Shrivastava Reviewed-by: Leon Romanovsky Reviewed-by: Logan Gunthorpe Link: https://lore.kernel.org/r/20260713083022.1110993-1-m.szyprowski@samsung.com Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 4391b797d4db..d8219efe3273 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -489,9 +489,8 @@ int dma_direct_map_sg(struct device *dev, struct scatterlist *sgl, int nents, case PCI_P2PDMA_MAP_BUS_ADDR: sg->dma_address = pci_p2pdma_bus_addr_map( p2pdma_state.mem, sg_phys(sg)); - sg_dma_len(sg) = sg->length; sg_dma_mark_bus_address(sg); - continue; + break; default: ret = -EREMOTEIO; goto out_unmap; -- cgit v1.2.3 From 3cf67bbc5e5323b3af968bb289cdff33f6180535 Mon Sep 17 00:00:00 2001 From: Jagadeesh Pagadala Date: Thu, 2 Jul 2026 19:39:02 +0530 Subject: dma/swiotlb: introduce Kconfig option for compile-time default pool size MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The SWIOTLB bounce buffer pool size is hardcoded at 64 MB via IO_TLB_DEFAULT_SIZE with no compile-time knob to adjust it. On memory-constrained embedded or mobile platforms equipped with a hardware IOMMU (e.g., ARM SMMU) covering most DMA-capable devices, reserving 64 MB at boot is unnecessarily wasteful — the SWIOTLB is only exercised for devices that bypass the IOMMU or have restricted DMA address ranges. Introduce CONFIG_SWIOTLB_DEFAULT_SIZE_MB, an integer Kconfig option (range 1–64 MB, default 64) that allows platforms to set a smaller compile-time default. IO_TLB_DEFAULT_SIZE is updated to derive from this value when CONFIG_SWIOTLB is enabled, preserving the existing 64 MB default when the option is not configured. The runtime "swiotlb=" kernel parameter override remains fully supported and takes precedence over the compile-time default. Signed-off-by: Jagadeesh Pagadala Signed-off-by: Bibek Kumar Patro Reviewed-by: Michael Kelley Link: https://lore.kernel.org/r/20260702-swiotlb-v2-1-9205f3ba5408@oss.qualcomm.com Signed-off-by: Marek Szyprowski --- Documentation/core-api/swiotlb.rst | 7 +++++-- include/linux/swiotlb.h | 8 ++++++-- kernel/dma/Kconfig | 22 ++++++++++++++++++++++ 3 files changed, 33 insertions(+), 4 deletions(-) (limited to 'kernel') diff --git a/Documentation/core-api/swiotlb.rst b/Documentation/core-api/swiotlb.rst index 9e0fe027dd3b..71b4e4c27eb5 100644 --- a/Documentation/core-api/swiotlb.rst +++ b/Documentation/core-api/swiotlb.rst @@ -140,8 +140,11 @@ Data structures concepts ------------------------ Memory used for swiotlb bounce buffers is allocated from overall system memory as one or more "pools". The default pool is allocated during system boot with a -default size of 64 MiB. The default pool size may be modified with the -"swiotlb=" kernel boot line parameter. The default size may also be adjusted +default size of 64 MiB, which can be changed at compile time via +CONFIG_SWIOTLB_DEFAULT_SIZE_MB. The default pool size may also be +modified at runtime with the "swiotlb=" kernel boot line parameter, +which takes precedence over the compile-time default. The default size +may also be adjusted due to other conditions, such as running in a CoCo VM, as described above. If CONFIG_SWIOTLB_DYNAMIC is enabled, additional pools may be allocated later in the life of the system. Each pool must be a contiguous range of physical diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index 3dae0f592063..1665a9ce8f94 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -32,8 +32,12 @@ struct scatterlist; #define IO_TLB_SHIFT 11 #define IO_TLB_SIZE (1 << IO_TLB_SHIFT) -/* default to 64MB */ -#define IO_TLB_DEFAULT_SIZE (64UL<<20) +/* compile-time default; overridable via CONFIG_SWIOTLB_DEFAULT_SIZE_MB */ +#ifdef CONFIG_SWIOTLB +#define IO_TLB_DEFAULT_SIZE ((unsigned long)CONFIG_SWIOTLB_DEFAULT_SIZE_MB << 20) +#else +#define IO_TLB_DEFAULT_SIZE (64UL << 20) +#endif unsigned long swiotlb_size_or_default(void); void __init swiotlb_init_remap(bool addressing_limit, unsigned int flags, diff --git a/kernel/dma/Kconfig b/kernel/dma/Kconfig index 0a4ba21a57a7..3830a63ae032 100644 --- a/kernel/dma/Kconfig +++ b/kernel/dma/Kconfig @@ -86,6 +86,28 @@ config SWIOTLB bool select NEED_DMA_MAP_STATE +config SWIOTLB_DEFAULT_SIZE_MB + int "Default SWIOTLB bounce buffer size in MB" + depends on SWIOTLB + range 1 64 + default 64 + help + Sets the default size of the software IO TLB (SWIOTLB) bounce buffer + pool allocated at boot time. The default is 64 MB. + + On memory-constrained embedded or mobile platforms (e.g., those with + a hardware IOMMU such as ARM SMMU covering most DMA-capable devices), + a smaller value such as 4 or 8 MB may be sufficient. The SWIOTLB is + then only needed for devices that bypass the IOMMU or have restricted + DMA address ranges. + + The minimum allowed value is 1 MB. This compile-time default can be + overridden at runtime using the "swiotlb=" kernel command line + parameter. Refer to Documentation/admin-guide/kernel-parameters.txt + for details. + + If unsure, leave at the default value of 64. + config SWIOTLB_DYNAMIC bool "Dynamic allocation of DMA bounce buffers" default n -- cgit v1.2.3 From 94a04ad732c9f8b9554270fc4038a06737de5c22 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:19 +0530 Subject: dma-direct: return struct page from dma_direct_alloc_from_pool() Commit 5b138c534fda ("dma-direct: factor out a dma_direct_alloc_from_pool helper") changed dma_direct_alloc_from_pool() to return the CPU address from dma_alloc_from_pool(). That fits dma_direct_alloc(), but dma_direct_alloc_pages() also uses the helper and expects a struct page *. Fix this by making dma_direct_alloc_from_pool() return the struct page * again, and pass the CPU address back through an out-parameter for the dma_direct_alloc() caller. Fixes: 5b138c534fda ("dma-direct: factor out a dma_direct_alloc_from_pool helper") Cc: stable@vger.kernel.org Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Jason Gunthorpe Signed-off-by: Aneesh Kumar K.V (Arm) Reviewed-by: Mostafa Saleh Link: https://lore.kernel.org/r/20260717180442.110954-2-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 4391b797d4db..b4cb2c03e5d7 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -164,22 +164,21 @@ static bool dma_direct_use_pool(struct device *dev, gfp_t gfp) return !gfpflags_allow_blocking(gfp) && !is_swiotlb_for_alloc(dev); } -static void *dma_direct_alloc_from_pool(struct device *dev, size_t size, - dma_addr_t *dma_handle, gfp_t gfp) +static struct page *dma_direct_alloc_from_pool(struct device *dev, size_t size, + dma_addr_t *dma_handle, void **cpu_addr, gfp_t gfp) { struct page *page; u64 phys_limit; - void *ret; if (WARN_ON_ONCE(!IS_ENABLED(CONFIG_DMA_COHERENT_POOL))) return NULL; gfp |= dma_direct_optimal_gfp_mask(dev, &phys_limit); - page = dma_alloc_from_pool(dev, size, &ret, gfp, dma_coherent_ok); + page = dma_alloc_from_pool(dev, size, cpu_addr, gfp, dma_coherent_ok); if (!page) return NULL; *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); - return ret; + return page; } static void *dma_direct_alloc_no_mapping(struct device *dev, size_t size, @@ -247,8 +246,11 @@ void *dma_direct_alloc(struct device *dev, size_t size, * the atomic pools instead if we aren't allowed block. */ if ((remap || force_dma_unencrypted(dev)) && - dma_direct_use_pool(dev, gfp)) - return dma_direct_alloc_from_pool(dev, size, dma_handle, gfp); + dma_direct_use_pool(dev, gfp)) { + page = dma_direct_alloc_from_pool(dev, size, dma_handle, + &ret, gfp); + return page ? ret : NULL; + } /* we always manually zero the memory once we are done */ page = __dma_direct_alloc_pages(dev, size, gfp & ~__GFP_ZERO, true); @@ -357,7 +359,7 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, void *ret; if (force_dma_unencrypted(dev) && dma_direct_use_pool(dev, gfp)) - return dma_direct_alloc_from_pool(dev, size, dma_handle, gfp); + return dma_direct_alloc_from_pool(dev, size, dma_handle, &ret, gfp); page = __dma_direct_alloc_pages(dev, size, gfp, false); if (!page) -- cgit v1.2.3 From 0510cb5b23037dbaaf57b8ddd6af32cbfc173ef2 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:20 +0530 Subject: dma-pool: fix page leak in atomic_pool_expand() cleanup atomic_pool_expand() frees the allocated pages from the remove_mapping error path only when CONFIG_DMA_DIRECT_REMAP is enabled. When CONFIG_DMA_DIRECT_REMAP is disabled, failures after page allocation, such as gen_pool_add_virt(), jump to remove_mapping and return without freeing the pages. Move __free_pages(page, order) out of the CONFIG_DMA_DIRECT_REMAP block so that cleanup paths always release the allocation. Reviewed-by: Jason Gunthorpe Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-3-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/pool.c | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/pool.c b/kernel/dma/pool.c index 2b2fbb709242..b0303efbc153 100644 --- a/kernel/dma/pool.c +++ b/kernel/dma/pool.c @@ -81,6 +81,7 @@ static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, { unsigned int order; struct page *page = NULL; + bool leak_pages = false; void *addr; int ret = -ENOMEM; @@ -115,8 +116,10 @@ static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, */ ret = set_memory_decrypted((unsigned long)page_to_virt(page), 1 << order); - if (ret) + if (ret) { + leak_pages = true; goto remove_mapping; + } ret = gen_pool_add_virt(pool, (unsigned long)addr, page_to_phys(page), pool_size, NUMA_NO_NODE); if (ret) @@ -130,14 +133,15 @@ encrypt_mapping: 1 << order); if (WARN_ON_ONCE(ret)) { /* Decrypt succeeded but encrypt failed, purposely leak */ - goto out; + leak_pages = true; } remove_mapping: #ifdef CONFIG_DMA_DIRECT_REMAP dma_common_free_remap(addr, pool_size); free_page: - __free_pages(page, order); #endif + if (!leak_pages) + __free_pages(page, order); out: return ret; } -- cgit v1.2.3 From 8a9dc4a028e726ff64f0a7b2931485c9ce87e2c0 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:22 +0530 Subject: dma: free atomic pool pages by physical address dma_direct_alloc_pages() may satisfy atomic allocations from the coherent atomic pools. The pool allocation is keyed by the virtual address stored in the gen_pool, but the pages API returns only the backing struct page. On architectures with CONFIG_DMA_DIRECT_REMAP, atomic pool chunks are added to the gen_pool using their remapped virtual address. dma_direct_free_pages() reconstructs a linear-map address with page_address(page) and passes that to dma_free_from_pool(). That address does not match the gen_pool virtual range, so the pool lookup can fail and the code can fall through to freeing a pool-owned page through the normal page allocator path. Add a page-based pool free helper that looks up the owning pool chunk by physical address, translates it back to the gen_pool virtual address, and frees that address to the pool. Use it from dma_direct_free_pages() while keeping the existing virtual-address helper for coherent allocation frees. Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-5-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- include/linux/dma-map-ops.h | 1 + kernel/dma/direct.c | 4 +-- kernel/dma/pool.c | 61 +++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 64 insertions(+), 2 deletions(-) (limited to 'kernel') diff --git a/include/linux/dma-map-ops.h b/include/linux/dma-map-ops.h index bcb5b5428aea..137e015c1750 100644 --- a/include/linux/dma-map-ops.h +++ b/include/linux/dma-map-ops.h @@ -215,6 +215,7 @@ struct page *dma_alloc_from_pool(struct device *dev, size_t size, void **cpu_addr, gfp_t flags, bool (*phys_addr_ok)(struct device *, phys_addr_t, size_t)); bool dma_free_from_pool(struct device *dev, void *start, size_t size); +bool dma_free_from_pool_page(struct device *dev, struct page *page, size_t size); int dma_direct_set_offset(struct device *dev, phys_addr_t cpu_start, dma_addr_t dma_start, u64 size); diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index b4cb2c03e5d7..17f1e097499e 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -381,9 +381,9 @@ void dma_direct_free_pages(struct device *dev, size_t size, { void *vaddr = page_address(page); - /* If cpu_addr is not from an atomic pool, dma_free_from_pool() fails */ + /* If page is not from an atomic pool, dma_free_from_pool_page() fails */ if (IS_ENABLED(CONFIG_DMA_COHERENT_POOL) && - dma_free_from_pool(dev, vaddr, size)) + dma_free_from_pool_page(dev, page, size)) return; if (dma_set_encrypted(dev, vaddr, size)) diff --git a/kernel/dma/pool.c b/kernel/dma/pool.c index b0303efbc153..e981c1faaadf 100644 --- a/kernel/dma/pool.c +++ b/kernel/dma/pool.c @@ -311,3 +311,64 @@ bool dma_free_from_pool(struct device *dev, void *start, size_t size) return false; } + +struct dma_pool_phys_match { + phys_addr_t phys; + size_t size; + unsigned long addr; + bool found; +}; + +static void dma_pool_find_phys(struct gen_pool *pool, struct gen_pool_chunk *chunk, + void *data) +{ + struct dma_pool_phys_match *match = data; + phys_addr_t end = match->phys + match->size - 1; + phys_addr_t chunk_end; + + if (match->found) + return; + + chunk_end = chunk->phys_addr + (chunk->end_addr - chunk->start_addr); + if (match->phys < chunk->phys_addr || end > chunk_end) + return; + + match->addr = chunk->start_addr + (match->phys - chunk->phys_addr); + match->found = true; +} + +static bool dma_free_from_pool_phys(struct gen_pool *pool, phys_addr_t phys, + size_t size) +{ + struct dma_pool_phys_match match = { + .phys = phys, + .size = size, + }; + + gen_pool_for_each_chunk(pool, dma_pool_find_phys, &match); + if (!match.found) + return false; + + gen_pool_free(pool, match.addr, size); + return true; +} + +/* + * FIXME: We could avoid this by storing the remapped virtual address in + * struct page and using that for lookup. + */ +bool dma_free_from_pool_page(struct device *dev, struct page *page, size_t size) +{ + struct gen_pool *pool = NULL; + phys_addr_t phys = page_to_phys(page); + + if (!IS_ENABLED(CONFIG_DMA_DIRECT_REMAP)) + return dma_free_from_pool(dev, page_address(page), size); + + while ((pool = dma_guess_pool(pool, 0))) { + if (dma_free_from_pool_phys(pool, phys, size)) + return true; + } + + return false; +} -- cgit v1.2.3 From 57d29044d0f29a76c6ec0c112c8c7371d5608dc7 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:23 +0530 Subject: swiotlb: Preserve allocation virtual address for dynamic pools swiotlb_alloc_tlb() can allocate from the DMA atomic pool when a decrypted pool is needed from atomic context. With CONFIG_DMA_DIRECT_REMAP, the atomic pool is backed by remapped virtual addresses, which are not the same as the direct-map addresses returned by phys_to_virt(). swiotlb_init_io_tlb_pool() currently reconstructs the pool virtual address from the physical start address. For atomic-pool backed allocations this stores the wrong address in pool->vaddr. Later, swiotlb_free_tlb() passes that address to dma_free_from_pool(), which will fail to recognize the chunk Pass the virtual address returned by the allocation path into swiotlb_init_io_tlb_pool(), and store that address in pool->vaddr. This keeps the pool free path using the same virtual address as the allocator. Fixes: 79636caad361 ("swiotlb: if swiotlb is full, fall back to a transient memory pool") Reviewed-by: Jason Gunthorpe Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Reviewed-by: Mostafa Saleh Link: https://lore.kernel.org/r/20260717180442.110954-6-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/swiotlb.c | 31 +++++++++++++++++++------------ 1 file changed, 19 insertions(+), 12 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 1abd3e6146f4..6e8db52866bf 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -266,9 +266,9 @@ void __init swiotlb_update_mem_attributes(void) } static void swiotlb_init_io_tlb_pool(struct io_tlb_pool *mem, phys_addr_t start, - unsigned long nslabs, bool late_alloc, unsigned int nareas) + void *vaddr, unsigned long nslabs, bool late_alloc, + unsigned int nareas) { - void *vaddr = phys_to_virt(start); unsigned long bytes = nslabs << IO_TLB_SHIFT, i; mem->nslabs = nslabs; @@ -409,7 +409,7 @@ void __init swiotlb_init_remap(bool addressing_limit, unsigned int flags, return; } - swiotlb_init_io_tlb_pool(mem, __pa(tlb), nslabs, false, nareas); + swiotlb_init_io_tlb_pool(mem, __pa(tlb), tlb, nslabs, false, nareas); add_mem_pool(&io_tlb_default_mem, mem); if (flags & SWIOTLB_VERBOSE) @@ -507,7 +507,7 @@ retry: set_memory_decrypted((unsigned long)vstart, (nslabs << IO_TLB_SHIFT) >> PAGE_SHIFT); - swiotlb_init_io_tlb_pool(mem, virt_to_phys(vstart), nslabs, true, + swiotlb_init_io_tlb_pool(mem, virt_to_phys(vstart), vstart, nslabs, true, nareas); add_mem_pool(&io_tlb_default_mem, mem); @@ -605,25 +605,26 @@ error: * @bytes: Size of the buffer. * @phys_limit: Maximum allowed physical address of the buffer. * @gfp: GFP flags for the allocation. + * @vaddr: Receives the virtual address for the allocated buffer. * * Return: Allocated pages, or %NULL on allocation failure. */ static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, - u64 phys_limit, gfp_t gfp) + u64 phys_limit, gfp_t gfp, void **vaddr) { struct page *page; + *vaddr = NULL; + /* * Allocate from the atomic pools if memory is encrypted and * the allocation is atomic, because decrypting may block. */ if (!gfpflags_allow_blocking(gfp) && dev && force_dma_unencrypted(dev)) { - void *vaddr; - if (!IS_ENABLED(CONFIG_DMA_COHERENT_POOL)) return NULL; - return dma_alloc_from_pool(dev, bytes, &vaddr, gfp, + return dma_alloc_from_pool(dev, bytes, vaddr, gfp, dma_coherent_ok); } @@ -645,6 +646,8 @@ static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, return NULL; } + if (page) + *vaddr = phys_to_virt(page_to_phys(page)); return page; } @@ -685,6 +688,7 @@ static struct io_tlb_pool *swiotlb_alloc_pool(struct device *dev, { struct io_tlb_pool *pool; unsigned int slot_order; + void *tlb_vaddr; struct page *tlb; size_t pool_size; size_t tlb_size; @@ -701,7 +705,8 @@ static struct io_tlb_pool *swiotlb_alloc_pool(struct device *dev, pool->areas = (void *)pool + sizeof(*pool); tlb_size = nslabs << IO_TLB_SHIFT; - while (!(tlb = swiotlb_alloc_tlb(dev, tlb_size, phys_limit, gfp))) { + while (!(tlb = swiotlb_alloc_tlb(dev, tlb_size, phys_limit, gfp, + &tlb_vaddr))) { if (nslabs <= minslabs) goto error_tlb; nslabs = ALIGN(nslabs >> 1, IO_TLB_SEGSIZE); @@ -715,11 +720,12 @@ static struct io_tlb_pool *swiotlb_alloc_pool(struct device *dev, if (!pool->slots) goto error_slots; - swiotlb_init_io_tlb_pool(pool, page_to_phys(tlb), nslabs, true, nareas); + swiotlb_init_io_tlb_pool(pool, page_to_phys(tlb), tlb_vaddr, nslabs, + true, nareas); return pool; error_slots: - swiotlb_free_tlb(page_address(tlb), tlb_size); + swiotlb_free_tlb(tlb_vaddr, tlb_size); error_tlb: kfree(pool); error: @@ -1851,7 +1857,8 @@ static int rmem_swiotlb_device_init(struct reserved_mem *rmem, set_memory_decrypted((unsigned long)phys_to_virt(rmem->base), rmem->size >> PAGE_SHIFT); - swiotlb_init_io_tlb_pool(pool, rmem->base, nslabs, + swiotlb_init_io_tlb_pool(pool, rmem->base, phys_to_virt(rmem->base), + nslabs, false, nareas); mem->force_bounce = true; mem->for_alloc = true; -- cgit v1.2.3 From a7139dbbb1c639eff13ef021dbe0b93c8b917d1d Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:25 +0530 Subject: dma-direct: swiotlb: handle swiotlb alloc/free outside __dma_direct_alloc_pages Move swiotlb allocation out of __dma_direct_alloc_pages() and handle it in dma_direct_alloc() / dma_direct_alloc_pages(). This is needed for follow-up changes that simplify the handling of memory encryption/decryption based on the DMA attribute flags. swiotlb backing pages are already mapped decrypted by swiotlb_update_mem_attributes() and rmem_swiotlb_device_init(), so dma-direct should not call dma_set_decrypted() on allocation nor dma_set_encrypted() on free for swiotlb-backed memory. Update alloc/free paths to detect swiotlb-backed pages and skip encrypt/decrypt transitions for those paths. Keep the existing highmem rejection in dma_direct_alloc_pages() for swiotlb allocations. Only for "restricted-dma-pool", we currently set `for_alloc = true`, while rmem_swiotlb_device_init() decrypts the whole pool up front. This pool is typically used together with "shared-dma-pool", where the shared region is accessed after remap/ioremap and the returned address is suitable for decrypted memory access. So existing code paths remain valid. Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Reviewed-by: Mostafa Saleh Link: https://lore.kernel.org/r/20260717180442.110954-8-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- include/linux/swiotlb.h | 6 +++++ kernel/dma/direct.c | 71 ++++++++++++++++++++++++++++++++++++------------- kernel/dma/swiotlb.c | 6 +++++ 3 files changed, 65 insertions(+), 18 deletions(-) (limited to 'kernel') diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index 3dae0f592063..c92ff6791595 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -284,6 +284,8 @@ extern void swiotlb_print_info(void); #ifdef CONFIG_DMA_RESTRICTED_POOL struct page *swiotlb_alloc(struct device *dev, size_t size); bool swiotlb_free(struct device *dev, struct page *page, size_t size); +void swiotlb_free_from_pool(struct device *dev, + phys_addr_t tlb_addr, struct io_tlb_pool *pool); static inline bool is_swiotlb_for_alloc(struct device *dev) { @@ -299,6 +301,10 @@ static inline bool swiotlb_free(struct device *dev, struct page *page, { return false; } +static inline void swiotlb_free_from_pool(struct device *dev, + phys_addr_t tlb_addr, struct io_tlb_pool *pool) +{ +} static inline bool is_swiotlb_for_alloc(struct device *dev) { return false; diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 17f1e097499e..0cbf2b0835c4 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -96,14 +96,6 @@ static int dma_set_encrypted(struct device *dev, void *vaddr, size_t size) return ret; } -static void __dma_direct_free_pages(struct device *dev, struct page *page, - size_t size) -{ - if (swiotlb_free(dev, page, size)) - return; - dma_free_contiguous(dev, page, size); -} - static struct page *dma_direct_alloc_swiotlb(struct device *dev, size_t size) { struct page *page = swiotlb_alloc(dev, size); @@ -125,9 +117,6 @@ static struct page *__dma_direct_alloc_pages(struct device *dev, size_t size, WARN_ON_ONCE(!PAGE_ALIGNED(size)); - if (is_swiotlb_for_alloc(dev)) - return dma_direct_alloc_swiotlb(dev, size); - gfp |= dma_direct_optimal_gfp_mask(dev, &phys_limit); page = dma_alloc_contiguous(dev, size, gfp); if (page) { @@ -203,6 +192,7 @@ void *dma_direct_alloc(struct device *dev, size_t size, dma_addr_t *dma_handle, gfp_t gfp, unsigned long attrs) { bool remap = false, set_uncached = false; + bool mark_mem_decrypt = true; struct page *page; void *ret; @@ -252,11 +242,21 @@ void *dma_direct_alloc(struct device *dev, size_t size, return page ? ret : NULL; } + if (is_swiotlb_for_alloc(dev)) { + page = dma_direct_alloc_swiotlb(dev, size); + if (page) { + mark_mem_decrypt = false; + goto setup_page; + } + return NULL; + } + /* we always manually zero the memory once we are done */ page = __dma_direct_alloc_pages(dev, size, gfp & ~__GFP_ZERO, true); if (!page) return NULL; +setup_page: /* * dma_alloc_contiguous can return highmem pages depending on a * combination the cma= arguments and per-arch setup. These need to be @@ -283,7 +283,7 @@ void *dma_direct_alloc(struct device *dev, size_t size, goto out_free_pages; } else { ret = page_address(page); - if (dma_set_decrypted(dev, ret, size)) + if (mark_mem_decrypt && dma_set_decrypted(dev, ret, size)) goto out_leak_pages; } @@ -300,10 +300,11 @@ void *dma_direct_alloc(struct device *dev, size_t size, return ret; out_encrypt_pages: - if (dma_set_encrypted(dev, page_address(page), size)) + if (mark_mem_decrypt && dma_set_encrypted(dev, page_address(page), size)) return NULL; out_free_pages: - __dma_direct_free_pages(dev, page, size); + if (!swiotlb_free(dev, page, size)) + dma_free_contiguous(dev, page, size); return NULL; out_leak_pages: return NULL; @@ -312,6 +313,9 @@ out_leak_pages: void dma_direct_free(struct device *dev, size_t size, void *cpu_addr, dma_addr_t dma_addr, unsigned long attrs) { + phys_addr_t phys; + bool mark_mem_encrypted = true; + struct io_tlb_pool *swiotlb_pool; unsigned int page_order = get_order(size); if ((attrs & DMA_ATTR_NO_KERNEL_MAPPING) && @@ -340,16 +344,25 @@ void dma_direct_free(struct device *dev, size_t size, dma_free_from_pool(dev, cpu_addr, PAGE_ALIGN(size))) return; + phys = dma_to_phys(dev, dma_addr); + swiotlb_pool = swiotlb_find_pool(dev, phys); + if (swiotlb_pool) + /* Swiotlb doesn't need a page attribute update on free */ + mark_mem_encrypted = false; + if (is_vmalloc_addr(cpu_addr)) { vunmap(cpu_addr); } else { if (IS_ENABLED(CONFIG_ARCH_HAS_DMA_CLEAR_UNCACHED)) arch_dma_clear_uncached(cpu_addr, size); - if (dma_set_encrypted(dev, cpu_addr, size)) + if (mark_mem_encrypted && dma_set_encrypted(dev, cpu_addr, size)) return; } - __dma_direct_free_pages(dev, dma_direct_to_page(dev, dma_addr), size); + if (swiotlb_pool) + swiotlb_free_from_pool(dev, phys, swiotlb_pool); + else + dma_free_contiguous(dev, dma_direct_to_page(dev, dma_addr), size); } struct page *dma_direct_alloc_pages(struct device *dev, size_t size, @@ -361,6 +374,15 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, if (force_dma_unencrypted(dev) && dma_direct_use_pool(dev, gfp)) return dma_direct_alloc_from_pool(dev, size, dma_handle, &ret, gfp); + if (is_swiotlb_for_alloc(dev)) { + page = dma_direct_alloc_swiotlb(dev, size); + if (!page) + return NULL; + + ret = page_address(page); + goto setup_page; + } + page = __dma_direct_alloc_pages(dev, size, gfp, false); if (!page) return NULL; @@ -368,6 +390,7 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, ret = page_address(page); if (dma_set_decrypted(dev, ret, size)) goto out_leak_pages; +setup_page: memset(ret, 0, size); *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); return page; @@ -379,16 +402,28 @@ void dma_direct_free_pages(struct device *dev, size_t size, struct page *page, dma_addr_t dma_addr, enum dma_data_direction dir) { + phys_addr_t phys; void *vaddr = page_address(page); + struct io_tlb_pool *swiotlb_pool; + bool mark_mem_encrypted = true; /* If page is not from an atomic pool, dma_free_from_pool_page() fails */ if (IS_ENABLED(CONFIG_DMA_COHERENT_POOL) && dma_free_from_pool_page(dev, page, size)) return; - if (dma_set_encrypted(dev, vaddr, size)) + phys = page_to_phys(page); + swiotlb_pool = swiotlb_find_pool(dev, phys); + if (swiotlb_pool) + mark_mem_encrypted = false; + + if (mark_mem_encrypted && dma_set_encrypted(dev, vaddr, size)) return; - __dma_direct_free_pages(dev, page, size); + + if (swiotlb_pool) + swiotlb_free_from_pool(dev, phys, swiotlb_pool); + else + dma_free_contiguous(dev, page, size); } #if defined(CONFIG_ARCH_HAS_SYNC_DMA_FOR_DEVICE) || \ diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 6e8db52866bf..d54154c165e5 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -1815,6 +1815,12 @@ bool swiotlb_free(struct device *dev, struct page *page, size_t size) return true; } +void swiotlb_free_from_pool(struct device *dev, + phys_addr_t tlb_addr, struct io_tlb_pool *pool) +{ + swiotlb_release_slots(dev, tlb_addr, pool); +} + static int rmem_swiotlb_device_init(struct reserved_mem *rmem, struct device *dev) { -- cgit v1.2.3 From b5ef0dd28e11ea4dd993e4f06eec9970551c4087 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:28 +0530 Subject: dma-direct: use __DMA_ATTR_ALLOC_CC_SHARED in alloc/free paths Propagate force_dma_unencrypted() into __DMA_ATTR_ALLOC_CC_SHARED in the dma-direct allocation path and use the attribute to drive the related decisions. This updates dma_direct_alloc(), dma_direct_free(), and dma_direct_alloc_pages() to fold the forced unencrypted case into attrs. Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-11-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 42 +++++++++++++++++++++++++++++++++--------- kernel/dma/mapping.c | 9 +++++++++ 2 files changed, 42 insertions(+), 9 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 0cbf2b0835c4..98e47e0b332d 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -192,16 +192,22 @@ void *dma_direct_alloc(struct device *dev, size_t size, dma_addr_t *dma_handle, gfp_t gfp, unsigned long attrs) { bool remap = false, set_uncached = false; - bool mark_mem_decrypt = true; + bool mark_mem_decrypt = false; struct page *page; void *ret; + if (force_dma_unencrypted(dev)) + attrs |= __DMA_ATTR_ALLOC_CC_SHARED; + + if (attrs & __DMA_ATTR_ALLOC_CC_SHARED) + mark_mem_decrypt = true; + size = PAGE_ALIGN(size); if (attrs & DMA_ATTR_NO_WARN) gfp |= __GFP_NOWARN; - if ((attrs & DMA_ATTR_NO_KERNEL_MAPPING) && - !force_dma_unencrypted(dev) && !is_swiotlb_for_alloc(dev)) + if (((attrs & (DMA_ATTR_NO_KERNEL_MAPPING | __DMA_ATTR_ALLOC_CC_SHARED)) == + DMA_ATTR_NO_KERNEL_MAPPING) && !is_swiotlb_for_alloc(dev)) return dma_direct_alloc_no_mapping(dev, size, dma_handle, gfp); if (!dev_is_dma_coherent(dev)) { @@ -235,7 +241,7 @@ void *dma_direct_alloc(struct device *dev, size_t size, * Remapping or decrypting memory may block, allocate the memory from * the atomic pools instead if we aren't allowed block. */ - if ((remap || force_dma_unencrypted(dev)) && + if ((remap || (attrs & __DMA_ATTR_ALLOC_CC_SHARED)) && dma_direct_use_pool(dev, gfp)) { page = dma_direct_alloc_from_pool(dev, size, dma_handle, &ret, gfp); @@ -314,12 +320,22 @@ void dma_direct_free(struct device *dev, size_t size, void *cpu_addr, dma_addr_t dma_addr, unsigned long attrs) { phys_addr_t phys; - bool mark_mem_encrypted = true; + bool mark_mem_encrypted = false; struct io_tlb_pool *swiotlb_pool; unsigned int page_order = get_order(size); - if ((attrs & DMA_ATTR_NO_KERNEL_MAPPING) && - !force_dma_unencrypted(dev) && !is_swiotlb_for_alloc(dev)) { + /* + * If the allocation used decrypted/shared backing pages, restore + * the encryption state on free. + */ + if (force_dma_unencrypted(dev)) + attrs |= __DMA_ATTR_ALLOC_CC_SHARED; + + if (attrs & __DMA_ATTR_ALLOC_CC_SHARED) + mark_mem_encrypted = true; + + if (((attrs & (DMA_ATTR_NO_KERNEL_MAPPING | __DMA_ATTR_ALLOC_CC_SHARED)) == + DMA_ATTR_NO_KERNEL_MAPPING) && !is_swiotlb_for_alloc(dev)) { /* cpu_addr is a struct page cookie, not a kernel address */ dma_free_contiguous(dev, cpu_addr, size); return; @@ -368,10 +384,14 @@ void dma_direct_free(struct device *dev, size_t size, struct page *dma_direct_alloc_pages(struct device *dev, size_t size, dma_addr_t *dma_handle, enum dma_data_direction dir, gfp_t gfp) { + unsigned long attrs = 0; struct page *page; void *ret; - if (force_dma_unencrypted(dev) && dma_direct_use_pool(dev, gfp)) + if (force_dma_unencrypted(dev)) + attrs |= __DMA_ATTR_ALLOC_CC_SHARED; + + if ((attrs & __DMA_ATTR_ALLOC_CC_SHARED) && dma_direct_use_pool(dev, gfp)) return dma_direct_alloc_from_pool(dev, size, dma_handle, &ret, gfp); if (is_swiotlb_for_alloc(dev)) { @@ -405,7 +425,11 @@ void dma_direct_free_pages(struct device *dev, size_t size, phys_addr_t phys; void *vaddr = page_address(page); struct io_tlb_pool *swiotlb_pool; - bool mark_mem_encrypted = true; + /* + * if the device had requested for an unencrypted buffer, + * convert it to encrypted on free + */ + bool mark_mem_encrypted = force_dma_unencrypted(dev); /* If page is not from an atomic pool, dma_free_from_pool_page() fails */ if (IS_ENABLED(CONFIG_DMA_COHERENT_POOL) && diff --git a/kernel/dma/mapping.c b/kernel/dma/mapping.c index 4fe04669e5e6..d2f70b6ccd0f 100644 --- a/kernel/dma/mapping.c +++ b/kernel/dma/mapping.c @@ -638,6 +638,15 @@ void *dma_alloc_attrs(struct device *dev, size_t size, dma_addr_t *dma_handle, if (WARN_ON_ONCE(flag & __GFP_COMP)) return NULL; + if (attrs & (DMA_ATTR_CC_SHARED | __DMA_ATTR_ALLOC_CC_SHARED)) { + trace_dma_alloc(dev, NULL, 0, size, DMA_BIDIRECTIONAL, flag, + attrs); + return NULL; + } + + if (force_dma_unencrypted(dev)) + attrs |= __DMA_ATTR_ALLOC_CC_SHARED; + if (dma_alloc_from_dev_coherent(dev, size, dma_handle, &cpu_addr)) { trace_dma_alloc(dev, cpu_addr, *dma_handle, size, DMA_BIDIRECTIONAL, flag, attrs); -- cgit v1.2.3 From 8277a12d0d609e4b461de9f55386885fd2d4567e Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:29 +0530 Subject: dma-pool: track decrypted atomic pools and select them via attrs Teach the atomic DMA pool code to distinguish between encrypted and unencrypted pools, and make pool allocation select the matching pool based on DMA attributes. Introduce a dma_gen_pool wrapper that records whether a pool is unencrypted, initialize that state when the atomic pools are created, and use it when expanding and resizing the pools. Update dma_alloc_from_pool() to take attrs and skip pools whose encrypted state does not match __DMA_ATTR_ALLOC_CC_SHARED. Update dma_free_from_pool() accordingly. Also pass __DMA_ATTR_ALLOC_CC_SHARED from the swiotlb atomic allocation path so decrypted swiotlb allocations are taken from the correct atomic pool. Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-12-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- drivers/iommu/dma-iommu.c | 7 +- include/linux/dma-map-ops.h | 2 +- kernel/dma/direct.c | 14 +++- kernel/dma/pool.c | 182 ++++++++++++++++++++++++++++---------------- kernel/dma/swiotlb.c | 8 +- 5 files changed, 139 insertions(+), 74 deletions(-) (limited to 'kernel') diff --git a/drivers/iommu/dma-iommu.c b/drivers/iommu/dma-iommu.c index 68c686c1e81a..a7b1da5e06e6 100644 --- a/drivers/iommu/dma-iommu.c +++ b/drivers/iommu/dma-iommu.c @@ -1660,9 +1660,14 @@ void *iommu_dma_alloc(struct device *dev, size_t size, dma_addr_t *handle, { bool coherent = dev_is_dma_coherent(dev); int ioprot = dma_info_to_prot(DMA_BIDIRECTIONAL, coherent, attrs); + bool is_alloc_cc_shared = attrs & __DMA_ATTR_ALLOC_CC_SHARED; struct page *page = NULL; void *cpu_addr; + /* Not yet supported */ + if (is_alloc_cc_shared) + return NULL; + gfp |= __GFP_ZERO; if (gfpflags_allow_blocking(gfp) && @@ -1673,7 +1678,7 @@ void *iommu_dma_alloc(struct device *dev, size_t size, dma_addr_t *handle, if (IS_ENABLED(CONFIG_DMA_DIRECT_REMAP) && !gfpflags_allow_blocking(gfp) && !coherent) { page = dma_alloc_from_pool(dev, PAGE_ALIGN(size), &cpu_addr, - gfp, NULL); + gfp, attrs, NULL); if (!page) return NULL; } else { diff --git a/include/linux/dma-map-ops.h b/include/linux/dma-map-ops.h index 137e015c1750..8fae2b7deb20 100644 --- a/include/linux/dma-map-ops.h +++ b/include/linux/dma-map-ops.h @@ -212,7 +212,7 @@ void *dma_common_pages_remap(struct page **pages, size_t size, pgprot_t prot, void dma_common_free_remap(void *cpu_addr, size_t size); struct page *dma_alloc_from_pool(struct device *dev, size_t size, - void **cpu_addr, gfp_t flags, + void **cpu_addr, gfp_t flags, unsigned long attrs, bool (*phys_addr_ok)(struct device *, phys_addr_t, size_t)); bool dma_free_from_pool(struct device *dev, void *start, size_t size); bool dma_free_from_pool_page(struct device *dev, struct page *page, size_t size); diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 98e47e0b332d..7c6edd747a7a 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -154,7 +154,8 @@ static bool dma_direct_use_pool(struct device *dev, gfp_t gfp) } static struct page *dma_direct_alloc_from_pool(struct device *dev, size_t size, - dma_addr_t *dma_handle, void **cpu_addr, gfp_t gfp) + dma_addr_t *dma_handle, void **cpu_addr, gfp_t gfp, + unsigned long attrs) { struct page *page; u64 phys_limit; @@ -163,7 +164,8 @@ static struct page *dma_direct_alloc_from_pool(struct device *dev, size_t size, return NULL; gfp |= dma_direct_optimal_gfp_mask(dev, &phys_limit); - page = dma_alloc_from_pool(dev, size, cpu_addr, gfp, dma_coherent_ok); + page = dma_alloc_from_pool(dev, size, cpu_addr, gfp, attrs, + dma_coherent_ok); if (!page) return NULL; *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); @@ -240,11 +242,14 @@ void *dma_direct_alloc(struct device *dev, size_t size, /* * Remapping or decrypting memory may block, allocate the memory from * the atomic pools instead if we aren't allowed block. + * FIXME: With CONFIG_DMA_DIRECT_REMAP, the pool is also mapped as + * DMA-coherent (non-cacheable). We may want to create a separate pool + * dedicated to CC_SHARED atomic allocations. */ if ((remap || (attrs & __DMA_ATTR_ALLOC_CC_SHARED)) && dma_direct_use_pool(dev, gfp)) { page = dma_direct_alloc_from_pool(dev, size, dma_handle, - &ret, gfp); + &ret, gfp, attrs); return page ? ret : NULL; } @@ -392,7 +397,8 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, attrs |= __DMA_ATTR_ALLOC_CC_SHARED; if ((attrs & __DMA_ATTR_ALLOC_CC_SHARED) && dma_direct_use_pool(dev, gfp)) - return dma_direct_alloc_from_pool(dev, size, dma_handle, &ret, gfp); + return dma_direct_alloc_from_pool(dev, size, dma_handle, + &ret, gfp, attrs); if (is_swiotlb_for_alloc(dev)) { page = dma_direct_alloc_swiotlb(dev, size); diff --git a/kernel/dma/pool.c b/kernel/dma/pool.c index e981c1faaadf..00f422a1e896 100644 --- a/kernel/dma/pool.c +++ b/kernel/dma/pool.c @@ -12,12 +12,18 @@ #include #include #include +#include -static struct gen_pool *atomic_pool_dma __ro_after_init; +struct dma_gen_pool { + bool cc_shared; + struct gen_pool *pool; +}; + +static struct dma_gen_pool atomic_pool_dma __ro_after_init; static unsigned long pool_size_dma; -static struct gen_pool *atomic_pool_dma32 __ro_after_init; +static struct dma_gen_pool atomic_pool_dma32 __ro_after_init; static unsigned long pool_size_dma32; -static struct gen_pool *atomic_pool_kernel __ro_after_init; +static struct dma_gen_pool atomic_pool_kernel __ro_after_init; static unsigned long pool_size_kernel; /* Size can be defined by the coherent_pool command line */ @@ -76,7 +82,7 @@ static bool cma_in_zone(gfp_t gfp) return true; } -static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, +static int atomic_pool_expand(struct dma_gen_pool *dma_pool, size_t pool_size, gfp_t gfp) { unsigned int order; @@ -84,6 +90,7 @@ static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, bool leak_pages = false; void *addr; int ret = -ENOMEM; + pgprot_t prot __maybe_unused; /* Cannot allocate larger than MAX_PAGE_ORDER */ order = min(get_order(pool_size), MAX_PAGE_ORDER); @@ -102,8 +109,12 @@ static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, arch_dma_prep_coherent(page, pool_size); #ifdef CONFIG_DMA_DIRECT_REMAP - addr = dma_common_contiguous_remap(page, pool_size, - pgprot_decrypted(pgprot_dmacoherent(PAGE_KERNEL)), + if (dma_pool->cc_shared) + prot = pgprot_decrypted(pgprot_dmacoherent(PAGE_KERNEL)); + else + prot = pgprot_dmacoherent(PAGE_KERNEL); + + addr = dma_common_contiguous_remap(page, pool_size, prot, __builtin_return_address(0)); if (!addr) goto free_page; @@ -114,14 +125,17 @@ static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, * Memory in the atomic DMA pools must be unencrypted, the pools do not * shrink so no re-encryption occurs in dma_direct_free(). */ - ret = set_memory_decrypted((unsigned long)page_to_virt(page), - 1 << order); - if (ret) { - leak_pages = true; - goto remove_mapping; + if (dma_pool->cc_shared) { + ret = set_memory_decrypted((unsigned long)page_to_virt(page), + 1 << order); + if (ret) { + leak_pages = true; + goto remove_mapping; + } } - ret = gen_pool_add_virt(pool, (unsigned long)addr, page_to_phys(page), - pool_size, NUMA_NO_NODE); + + ret = gen_pool_add_virt(dma_pool->pool, (unsigned long)addr, + page_to_phys(page), pool_size, NUMA_NO_NODE); if (ret) goto encrypt_mapping; @@ -129,12 +143,10 @@ static int atomic_pool_expand(struct gen_pool *pool, size_t pool_size, return 0; encrypt_mapping: - ret = set_memory_encrypted((unsigned long)page_to_virt(page), - 1 << order); - if (WARN_ON_ONCE(ret)) { - /* Decrypt succeeded but encrypt failed, purposely leak */ + if (dma_pool->cc_shared && + set_memory_encrypted((unsigned long)page_to_virt(page), 1 << order)) leak_pages = true; - } + remove_mapping: #ifdef CONFIG_DMA_DIRECT_REMAP dma_common_free_remap(addr, pool_size); @@ -146,46 +158,52 @@ out: return ret; } -static void atomic_pool_resize(struct gen_pool *pool, gfp_t gfp) +static void atomic_pool_resize(struct dma_gen_pool *dma_pool, gfp_t gfp) { - if (pool && gen_pool_avail(pool) < atomic_pool_size) - atomic_pool_expand(pool, gen_pool_size(pool), gfp); + if (dma_pool->pool && gen_pool_avail(dma_pool->pool) < atomic_pool_size) + atomic_pool_expand(dma_pool, gen_pool_size(dma_pool->pool), gfp); } static void atomic_pool_work_fn(struct work_struct *work) { if (IS_ENABLED(CONFIG_ZONE_DMA)) - atomic_pool_resize(atomic_pool_dma, + atomic_pool_resize(&atomic_pool_dma, GFP_KERNEL | GFP_DMA); if (IS_ENABLED(CONFIG_ZONE_DMA32)) - atomic_pool_resize(atomic_pool_dma32, + atomic_pool_resize(&atomic_pool_dma32, GFP_KERNEL | GFP_DMA32); - atomic_pool_resize(atomic_pool_kernel, GFP_KERNEL); + atomic_pool_resize(&atomic_pool_kernel, GFP_KERNEL); } -static __init struct gen_pool *__dma_atomic_pool_init(size_t pool_size, - gfp_t gfp) +static __init struct dma_gen_pool *__dma_atomic_pool_init(struct dma_gen_pool *dma_pool, + size_t pool_size, gfp_t gfp) { - struct gen_pool *pool; int ret; - pool = gen_pool_create(PAGE_SHIFT, NUMA_NO_NODE); - if (!pool) + dma_pool->pool = gen_pool_create(PAGE_SHIFT, NUMA_NO_NODE); + if (!dma_pool->pool) return NULL; - gen_pool_set_algo(pool, gen_pool_first_fit_order_align, NULL); + gen_pool_set_algo(dma_pool->pool, gen_pool_first_fit_order_align, NULL); - ret = atomic_pool_expand(pool, pool_size, gfp); + /* if platform is using memory encryption atomic pools are by default shared. */ + if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) + dma_pool->cc_shared = true; + else + dma_pool->cc_shared = false; + + ret = atomic_pool_expand(dma_pool, pool_size, gfp); if (ret) { - gen_pool_destroy(pool); + gen_pool_destroy(dma_pool->pool); + dma_pool->pool = NULL; pr_err("DMA: failed to allocate %zu KiB %pGg pool for atomic allocation\n", pool_size >> 10, &gfp); return NULL; } pr_info("DMA: preallocated %zu KiB %pGg pool for atomic allocations\n", - gen_pool_size(pool) >> 10, &gfp); - return pool; + gen_pool_size(dma_pool->pool) >> 10, &gfp); + return dma_pool; } #ifdef CONFIG_ZONE_DMA32 @@ -211,21 +229,22 @@ static int __init dma_atomic_pool_init(void) /* All memory might be in the DMA zone(s) to begin with */ if (has_managed_zone(ZONE_NORMAL)) { - atomic_pool_kernel = __dma_atomic_pool_init(atomic_pool_size, - GFP_KERNEL); - if (!atomic_pool_kernel) + __dma_atomic_pool_init(&atomic_pool_kernel, atomic_pool_size, GFP_KERNEL); + if (!atomic_pool_kernel.pool) ret = -ENOMEM; } + if (has_managed_dma()) { - atomic_pool_dma = __dma_atomic_pool_init(atomic_pool_size, - GFP_KERNEL | GFP_DMA); - if (!atomic_pool_dma) + __dma_atomic_pool_init(&atomic_pool_dma, atomic_pool_size, + GFP_KERNEL | GFP_DMA); + if (!atomic_pool_dma.pool) ret = -ENOMEM; } + if (has_managed_dma32) { - atomic_pool_dma32 = __dma_atomic_pool_init(atomic_pool_size, - GFP_KERNEL | GFP_DMA32); - if (!atomic_pool_dma32) + __dma_atomic_pool_init(&atomic_pool_dma32, atomic_pool_size, + GFP_KERNEL | GFP_DMA32); + if (!atomic_pool_dma32.pool) ret = -ENOMEM; } @@ -234,19 +253,44 @@ static int __init dma_atomic_pool_init(void) } postcore_initcall(dma_atomic_pool_init); -static inline struct gen_pool *dma_guess_pool(struct gen_pool *prev, gfp_t gfp) +static inline struct dma_gen_pool *__dma_guess_pool(struct dma_gen_pool *first, + struct dma_gen_pool *second, struct dma_gen_pool *third) { - if (prev == NULL) { + if (first->pool) + return first; + if (second && second->pool) + return second; + if (third && third->pool) + return third; + return NULL; +} + +static inline struct dma_gen_pool *dma_guess_pool(struct dma_gen_pool *prev, + gfp_t gfp) +{ + if (!prev) { if (gfp & GFP_DMA) - return atomic_pool_dma ?: atomic_pool_dma32 ?: atomic_pool_kernel; + return __dma_guess_pool(&atomic_pool_dma, + &atomic_pool_dma32, + &atomic_pool_kernel); + if (gfp & GFP_DMA32) - return atomic_pool_dma32 ?: atomic_pool_dma ?: atomic_pool_kernel; - return atomic_pool_kernel ?: atomic_pool_dma32 ?: atomic_pool_dma; + return __dma_guess_pool(&atomic_pool_dma32, + &atomic_pool_dma, + &atomic_pool_kernel); + + return __dma_guess_pool(&atomic_pool_kernel, + &atomic_pool_dma32, + &atomic_pool_dma); } - if (prev == atomic_pool_kernel) - return atomic_pool_dma32 ? atomic_pool_dma32 : atomic_pool_dma; - if (prev == atomic_pool_dma32) - return atomic_pool_dma; + + if (prev == &atomic_pool_kernel) + return __dma_guess_pool(&atomic_pool_dma32, + &atomic_pool_dma, NULL); + + if (prev == &atomic_pool_dma32) + return __dma_guess_pool(&atomic_pool_dma, NULL, NULL); + return NULL; } @@ -276,16 +320,20 @@ static struct page *__dma_alloc_from_pool(struct device *dev, size_t size, } struct page *dma_alloc_from_pool(struct device *dev, size_t size, - void **cpu_addr, gfp_t gfp, + void **cpu_addr, gfp_t gfp, unsigned long attrs, bool (*phys_addr_ok)(struct device *, phys_addr_t, size_t)) { - struct gen_pool *pool = NULL; + struct dma_gen_pool *dma_pool = NULL; struct page *page; bool pool_found = false; - while ((pool = dma_guess_pool(pool, gfp))) { + while ((dma_pool = dma_guess_pool(dma_pool, gfp))) { + + if (dma_pool->cc_shared != !!(attrs & __DMA_ATTR_ALLOC_CC_SHARED)) + continue; + pool_found = true; - page = __dma_alloc_from_pool(dev, size, pool, cpu_addr, + page = __dma_alloc_from_pool(dev, size, dma_pool->pool, cpu_addr, phys_addr_ok); if (page) return page; @@ -300,12 +348,14 @@ struct page *dma_alloc_from_pool(struct device *dev, size_t size, bool dma_free_from_pool(struct device *dev, void *start, size_t size) { - struct gen_pool *pool = NULL; + struct dma_gen_pool *dma_pool = NULL; + + while ((dma_pool = dma_guess_pool(dma_pool, 0))) { - while ((pool = dma_guess_pool(pool, 0))) { - if (!gen_pool_has_addr(pool, (unsigned long)start, size)) + if (!gen_pool_has_addr(dma_pool->pool, (unsigned long)start, size)) continue; - gen_pool_free(pool, (unsigned long)start, size); + + gen_pool_free(dma_pool->pool, (unsigned long)start, size); return true; } @@ -337,7 +387,7 @@ static void dma_pool_find_phys(struct gen_pool *pool, struct gen_pool_chunk *chu match->found = true; } -static bool dma_free_from_pool_phys(struct gen_pool *pool, phys_addr_t phys, +static bool dma_free_from_pool_phys(struct dma_gen_pool *dma_pool, phys_addr_t phys, size_t size) { struct dma_pool_phys_match match = { @@ -345,11 +395,11 @@ static bool dma_free_from_pool_phys(struct gen_pool *pool, phys_addr_t phys, .size = size, }; - gen_pool_for_each_chunk(pool, dma_pool_find_phys, &match); + gen_pool_for_each_chunk(dma_pool->pool, dma_pool_find_phys, &match); if (!match.found) return false; - gen_pool_free(pool, match.addr, size); + gen_pool_free(dma_pool->pool, match.addr, size); return true; } @@ -359,14 +409,14 @@ static bool dma_free_from_pool_phys(struct gen_pool *pool, phys_addr_t phys, */ bool dma_free_from_pool_page(struct device *dev, struct page *page, size_t size) { - struct gen_pool *pool = NULL; + struct dma_gen_pool *dma_pool = NULL; phys_addr_t phys = page_to_phys(page); if (!IS_ENABLED(CONFIG_DMA_DIRECT_REMAP)) return dma_free_from_pool(dev, page_address(page), size); - while ((pool = dma_guess_pool(pool, 0))) { - if (dma_free_from_pool_phys(pool, phys, size)) + while ((dma_pool = dma_guess_pool(dma_pool, 0))) { + if (dma_free_from_pool_phys(dma_pool, phys, size)) return true; } diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index d54154c165e5..908de28aceb2 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -613,9 +613,9 @@ static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, u64 phys_limit, gfp_t gfp, void **vaddr) { struct page *page; + unsigned long attrs = 0; *vaddr = NULL; - /* * Allocate from the atomic pools if memory is encrypted and * the allocation is atomic, because decrypting may block. @@ -624,8 +624,12 @@ static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, if (!IS_ENABLED(CONFIG_DMA_COHERENT_POOL)) return NULL; + /* swiotlb considered decrypted by default */ + if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) + attrs = __DMA_ATTR_ALLOC_CC_SHARED; + return dma_alloc_from_pool(dev, bytes, vaddr, gfp, - dma_coherent_ok); + attrs, dma_coherent_ok); } gfp &= ~GFP_ZONEMASK; -- cgit v1.2.3 From 7fcad5e3715adb31ec3a5a522c8fe87caa2569af Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:30 +0530 Subject: dma: swiotlb: pass mapping attributes by reference Change swiotlb_tbl_map_single() to take the DMA mapping attributes by reference and update the direct callers accordingly. This is a preparatory change for a follow-up patch which updates the attributes based on the selected swiotlb pool. Keeping the signature change separate makes the follow-up patch easier to review. No functional change in this patch. Reviewed-by: Jason Gunthorpe Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-13-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- drivers/iommu/dma-iommu.c | 2 +- drivers/xen/swiotlb-xen.c | 2 +- include/linux/swiotlb.h | 2 +- kernel/dma/swiotlb.c | 6 +++--- 4 files changed, 6 insertions(+), 6 deletions(-) (limited to 'kernel') diff --git a/drivers/iommu/dma-iommu.c b/drivers/iommu/dma-iommu.c index a7b1da5e06e6..fe387829ee92 100644 --- a/drivers/iommu/dma-iommu.c +++ b/drivers/iommu/dma-iommu.c @@ -1180,7 +1180,7 @@ static phys_addr_t iommu_dma_map_swiotlb(struct device *dev, phys_addr_t phys, trace_swiotlb_bounced(dev, phys, size); phys = swiotlb_tbl_map_single(dev, phys, size, iova_mask(iovad), dir, - attrs); + &attrs); /* * Untrusted devices should not see padding areas with random leftover diff --git a/drivers/xen/swiotlb-xen.c b/drivers/xen/swiotlb-xen.c index 2cbf2b588f5b..8c4abe65cd49 100644 --- a/drivers/xen/swiotlb-xen.c +++ b/drivers/xen/swiotlb-xen.c @@ -243,7 +243,7 @@ static dma_addr_t xen_swiotlb_map_phys(struct device *dev, phys_addr_t phys, */ trace_swiotlb_bounced(dev, dev_addr, size); - map = swiotlb_tbl_map_single(dev, phys, size, 0, dir, attrs); + map = swiotlb_tbl_map_single(dev, phys, size, 0, dir, &attrs); if (map == (phys_addr_t)DMA_MAPPING_ERROR) return DMA_MAPPING_ERROR; diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index c92ff6791595..ea4c0a292dea 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -238,7 +238,7 @@ static inline phys_addr_t default_swiotlb_limit(void) phys_addr_t swiotlb_tbl_map_single(struct device *hwdev, phys_addr_t phys, size_t mapping_size, unsigned int alloc_aligned_mask, - enum dma_data_direction dir, unsigned long attrs); + enum dma_data_direction dir, unsigned long *attrs); dma_addr_t swiotlb_map(struct device *dev, phys_addr_t phys, size_t size, enum dma_data_direction dir, unsigned long attrs); diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 908de28aceb2..046ae92c4832 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -1396,7 +1396,7 @@ static unsigned long mem_used(struct io_tlb_mem *mem) */ phys_addr_t swiotlb_tbl_map_single(struct device *dev, phys_addr_t orig_addr, size_t mapping_size, unsigned int alloc_align_mask, - enum dma_data_direction dir, unsigned long attrs) + enum dma_data_direction dir, unsigned long *attrs) { struct io_tlb_mem *mem = dev->dma_io_tlb_mem; unsigned int offset; @@ -1430,7 +1430,7 @@ phys_addr_t swiotlb_tbl_map_single(struct device *dev, phys_addr_t orig_addr, size = ALIGN(mapping_size + offset, alloc_align_mask + 1); index = swiotlb_find_slots(dev, orig_addr, size, alloc_align_mask, &pool); if (index == -1) { - if (!(attrs & DMA_ATTR_NO_WARN)) + if (!(*attrs & DMA_ATTR_NO_WARN)) dev_warn_ratelimited(dev, "swiotlb buffer is full (sz: %zd bytes), total %lu (slots), used %lu (slots)\n", size, mem->nslabs, mem_used(mem)); @@ -1609,7 +1609,7 @@ dma_addr_t swiotlb_map(struct device *dev, phys_addr_t paddr, size_t size, trace_swiotlb_bounced(dev, phys_to_dma(dev, paddr), size); - swiotlb_addr = swiotlb_tbl_map_single(dev, paddr, size, 0, dir, attrs); + swiotlb_addr = swiotlb_tbl_map_single(dev, paddr, size, 0, dir, &attrs); if (swiotlb_addr == (phys_addr_t)DMA_MAPPING_ERROR) return DMA_MAPPING_ERROR; -- cgit v1.2.3 From d5fd03cdd6c0837bd0d6ae7210a1c2c8649a0160 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:31 +0530 Subject: dma: swiotlb: track pool encryption state and honor DMA_ATTR_CC_SHARED Teach swiotlb to distinguish between encrypted and decrypted bounce buffer pools, and make allocation and mapping paths select a pool whose state matches the requested DMA attributes. Add a cc_shared flag to io_tlb_mem, initialize it for the default and restricted pools, and propagate __DMA_ATTR_ALLOC_CC_SHARED into swiotlb pool allocation. Reject swiotlb alloc/map requests when the selected pool does not match the required encrypted/decrypted state. Also return DMA addresses with the matching phys_to_dma_{encrypted, unencrypted} helper so the DMA address encoding stays consistent with the chosen pool. Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-14-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- include/linux/dma-direct.h | 10 +++ include/linux/swiotlb.h | 10 ++- kernel/dma/direct.c | 13 +++- kernel/dma/swiotlb.c | 177 ++++++++++++++++++++++++++++++++++----------- 4 files changed, 162 insertions(+), 48 deletions(-) (limited to 'kernel') diff --git a/include/linux/dma-direct.h b/include/linux/dma-direct.h index c249912456f9..94fad4e7c11e 100644 --- a/include/linux/dma-direct.h +++ b/include/linux/dma-direct.h @@ -77,6 +77,10 @@ static inline dma_addr_t dma_range_map_max(const struct bus_dma_region *map) #ifndef phys_to_dma_unencrypted #define phys_to_dma_unencrypted phys_to_dma #endif + +#ifndef phys_to_dma_encrypted +#define phys_to_dma_encrypted phys_to_dma +#endif #else static inline dma_addr_t __phys_to_dma(struct device *dev, phys_addr_t paddr) { @@ -90,6 +94,12 @@ static inline dma_addr_t phys_to_dma_unencrypted(struct device *dev, { return dma_addr_unencrypted(__phys_to_dma(dev, paddr)); } + +static inline dma_addr_t phys_to_dma_encrypted(struct device *dev, + phys_addr_t paddr) +{ + return dma_addr_encrypted(__phys_to_dma(dev, paddr)); +} /* * If memory encryption is supported, phys_to_dma will set the memory encryption * bit in the DMA address, and dma_to_phys will clear it. diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index ea4c0a292dea..ee42f7588847 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -66,6 +66,7 @@ extern void __init swiotlb_update_mem_attributes(void); * @node: Member of the IO TLB memory pool list. * @rcu: RCU head for swiotlb_dyn_free(). * @transient: %true if transient memory pool. + * @cc_shared: %true if the pool memory is shared for confidential computing. */ struct io_tlb_pool { phys_addr_t start; @@ -81,6 +82,7 @@ struct io_tlb_pool { struct list_head node; struct rcu_head rcu; bool transient; + bool cc_shared; #endif }; @@ -92,6 +94,7 @@ struct io_tlb_pool { * @debugfs: The dentry to debugfs. * @force_bounce: %true if swiotlb bouncing is forced * @for_alloc: %true if the pool is used for memory allocation + * @cc_shared: %true if the pool memory is shared for confidential computing. * @can_grow: %true if more pools can be allocated dynamically. * @phys_limit: Maximum allowed physical address. * @lock: Lock to synchronize changes to the list. @@ -111,6 +114,7 @@ struct io_tlb_mem { struct dentry *debugfs; bool force_bounce; bool for_alloc; + bool cc_shared; #ifdef CONFIG_SWIOTLB_DYNAMIC bool can_grow; u64 phys_limit; @@ -282,7 +286,8 @@ static inline void swiotlb_sync_single_for_cpu(struct device *dev, extern void swiotlb_print_info(void); #ifdef CONFIG_DMA_RESTRICTED_POOL -struct page *swiotlb_alloc(struct device *dev, size_t size); +struct page *swiotlb_alloc(struct device *dev, size_t size, + unsigned long attrs); bool swiotlb_free(struct device *dev, struct page *page, size_t size); void swiotlb_free_from_pool(struct device *dev, phys_addr_t tlb_addr, struct io_tlb_pool *pool); @@ -292,7 +297,8 @@ static inline bool is_swiotlb_for_alloc(struct device *dev) return dev->dma_io_tlb_mem->for_alloc; } #else -static inline struct page *swiotlb_alloc(struct device *dev, size_t size) +static inline struct page *swiotlb_alloc(struct device *dev, size_t size, + unsigned long attrs) { return NULL; } diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 7c6edd747a7a..c6561ed05cb0 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -96,9 +96,10 @@ static int dma_set_encrypted(struct device *dev, void *vaddr, size_t size) return ret; } -static struct page *dma_direct_alloc_swiotlb(struct device *dev, size_t size) +static struct page *dma_direct_alloc_swiotlb(struct device *dev, size_t size, + unsigned long attrs) { - struct page *page = swiotlb_alloc(dev, size); + struct page *page = swiotlb_alloc(dev, size, attrs); if (page && !dma_coherent_ok(dev, page_to_phys(page), size)) { swiotlb_free(dev, page, size); @@ -254,8 +255,12 @@ void *dma_direct_alloc(struct device *dev, size_t size, } if (is_swiotlb_for_alloc(dev)) { - page = dma_direct_alloc_swiotlb(dev, size); + page = dma_direct_alloc_swiotlb(dev, size, attrs); if (page) { + /* + * swiotlb allocations comes from pool already marked + * decrypted + */ mark_mem_decrypt = false; goto setup_page; } @@ -401,7 +406,7 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, &ret, gfp, attrs); if (is_swiotlb_for_alloc(dev)) { - page = dma_direct_alloc_swiotlb(dev, size); + page = dma_direct_alloc_swiotlb(dev, size, attrs); if (!page) return NULL; diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 046ae92c4832..335e27bc1e1f 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -259,10 +259,21 @@ void __init swiotlb_update_mem_attributes(void) struct io_tlb_pool *mem = &io_tlb_default_mem.defpool; unsigned long bytes; + /* + * if platform support memory encryption, swiotlb buffers are + * shared by default. + */ + if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) + io_tlb_default_mem.cc_shared = true; + else + io_tlb_default_mem.cc_shared = false; + if (!mem->nslabs || mem->late_alloc) return; bytes = PAGE_ALIGN(mem->nslabs << IO_TLB_SHIFT); - set_memory_decrypted((unsigned long)mem->vaddr, bytes >> PAGE_SHIFT); + + if (io_tlb_default_mem.cc_shared) + set_memory_decrypted((unsigned long)mem->vaddr, bytes >> PAGE_SHIFT); } static void swiotlb_init_io_tlb_pool(struct io_tlb_pool *mem, phys_addr_t start, @@ -505,8 +516,10 @@ retry: if (!mem->slots) goto error_slots; - set_memory_decrypted((unsigned long)vstart, - (nslabs << IO_TLB_SHIFT) >> PAGE_SHIFT); + if (io_tlb_default_mem.cc_shared) + set_memory_decrypted((unsigned long)vstart, + (nslabs << IO_TLB_SHIFT) >> PAGE_SHIFT); + swiotlb_init_io_tlb_pool(mem, virt_to_phys(vstart), vstart, nslabs, true, nareas); add_mem_pool(&io_tlb_default_mem, mem); @@ -539,7 +552,9 @@ void __init swiotlb_exit(void) tbl_size = PAGE_ALIGN(mem->end - mem->start); slots_size = PAGE_ALIGN(array_size(sizeof(*mem->slots), mem->nslabs)); - set_memory_encrypted(tbl_vaddr, tbl_size >> PAGE_SHIFT); + if (io_tlb_default_mem.cc_shared) + set_memory_encrypted(tbl_vaddr, tbl_size >> PAGE_SHIFT); + if (mem->late_alloc) { area_order = get_order(array_size(sizeof(*mem->areas), mem->nareas)); @@ -563,6 +578,7 @@ void __init swiotlb_exit(void) * @gfp: GFP flags for the allocation. * @bytes: Size of the buffer. * @phys_limit: Maximum allowed physical address of the buffer. + * @attrs: DMA attributes for the allocation. * * Allocate pages from the buddy allocator. If successful, make the allocated * pages decrypted that they can be used for DMA. @@ -570,9 +586,11 @@ void __init swiotlb_exit(void) * Return: Decrypted pages, %NULL on allocation failure, or ERR_PTR(-EAGAIN) * if the allocated physical address was above @phys_limit. */ -static struct page *alloc_dma_pages(gfp_t gfp, size_t bytes, u64 phys_limit) +static struct page *alloc_dma_pages(gfp_t gfp, size_t bytes, + u64 phys_limit, unsigned long attrs) { unsigned int order = get_order(bytes); + bool cc_shared = attrs & __DMA_ATTR_ALLOC_CC_SHARED; struct page *page; phys_addr_t paddr; void *vaddr; @@ -588,13 +606,13 @@ static struct page *alloc_dma_pages(gfp_t gfp, size_t bytes, u64 phys_limit) } vaddr = phys_to_virt(paddr); - if (set_memory_decrypted((unsigned long)vaddr, PFN_UP(bytes))) + if (cc_shared && set_memory_decrypted((unsigned long)vaddr, PFN_UP(bytes))) goto error; return page; error: /* Intentional leak if pages cannot be encrypted again. */ - if (!set_memory_encrypted((unsigned long)vaddr, PFN_UP(bytes))) + if (cc_shared && !set_memory_encrypted((unsigned long)vaddr, PFN_UP(bytes))) __free_pages(page, order); return NULL; } @@ -602,6 +620,7 @@ error: /** * swiotlb_alloc_tlb() - allocate a dynamic IO TLB buffer * @dev: Device for which a memory pool is allocated. + * @mem: SWIOTLB allocator for the pool. * @bytes: Size of the buffer. * @phys_limit: Maximum allowed physical address of the buffer. * @gfp: GFP flags for the allocation. @@ -609,25 +628,23 @@ error: * * Return: Allocated pages, or %NULL on allocation failure. */ -static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, +static struct page *swiotlb_alloc_tlb(struct device *dev, + struct io_tlb_mem *mem, size_t bytes, u64 phys_limit, gfp_t gfp, void **vaddr) { struct page *page; - unsigned long attrs = 0; + unsigned long attrs = mem->cc_shared ? __DMA_ATTR_ALLOC_CC_SHARED : 0; *vaddr = NULL; /* * Allocate from the atomic pools if memory is encrypted and * the allocation is atomic, because decrypting may block. */ - if (!gfpflags_allow_blocking(gfp) && dev && force_dma_unencrypted(dev)) { + if (!gfpflags_allow_blocking(gfp) && dev && mem->cc_shared) { + if (!IS_ENABLED(CONFIG_DMA_COHERENT_POOL)) return NULL; - /* swiotlb considered decrypted by default */ - if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) - attrs = __DMA_ATTR_ALLOC_CC_SHARED; - return dma_alloc_from_pool(dev, bytes, vaddr, gfp, attrs, dma_coherent_ok); } @@ -638,7 +655,7 @@ static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, else if (phys_limit <= DMA_BIT_MASK(32)) gfp |= __GFP_DMA32; - while (IS_ERR(page = alloc_dma_pages(gfp, bytes, phys_limit))) { + while (IS_ERR(page = alloc_dma_pages(gfp, bytes, phys_limit, attrs))) { if (IS_ENABLED(CONFIG_ZONE_DMA32) && phys_limit < DMA_BIT_MASK(64) && !(gfp & (__GFP_DMA32 | __GFP_DMA))) @@ -659,21 +676,25 @@ static struct page *swiotlb_alloc_tlb(struct device *dev, size_t bytes, * swiotlb_free_tlb() - free a dynamically allocated IO TLB buffer * @vaddr: Virtual address of the buffer. * @bytes: Size of the buffer. + * @cc_shared: true if @vaddr was allocated decrypted and must be + * re-encrypted before being freed */ -static void swiotlb_free_tlb(void *vaddr, size_t bytes) +static void swiotlb_free_tlb(void *vaddr, size_t bytes, bool cc_shared) { if (IS_ENABLED(CONFIG_DMA_COHERENT_POOL) && dma_free_from_pool(NULL, vaddr, bytes)) return; /* Intentional leak if pages cannot be encrypted again. */ - if (!set_memory_encrypted((unsigned long)vaddr, PFN_UP(bytes))) + if (!cc_shared || + !set_memory_encrypted((unsigned long)vaddr, PFN_UP(bytes))) __free_pages(virt_to_page(vaddr), get_order(bytes)); } /** * swiotlb_alloc_pool() - allocate a new IO TLB memory pool * @dev: Device for which a memory pool is allocated. + * @mem: SWIOTLB allocator for the pool. * @minslabs: Minimum number of slabs. * @nslabs: Desired (maximum) number of slabs. * @nareas: Number of areas. @@ -687,8 +708,9 @@ static void swiotlb_free_tlb(void *vaddr, size_t bytes) * Return: New memory pool, or %NULL on allocation failure. */ static struct io_tlb_pool *swiotlb_alloc_pool(struct device *dev, - unsigned long minslabs, unsigned long nslabs, - unsigned int nareas, u64 phys_limit, gfp_t gfp) + struct io_tlb_mem *mem, unsigned long minslabs, + unsigned long nslabs, unsigned int nareas, u64 phys_limit, + gfp_t gfp) { struct io_tlb_pool *pool; unsigned int slot_order; @@ -707,10 +729,11 @@ static struct io_tlb_pool *swiotlb_alloc_pool(struct device *dev, if (!pool) goto error; pool->areas = (void *)pool + sizeof(*pool); + pool->cc_shared = mem->cc_shared; tlb_size = nslabs << IO_TLB_SHIFT; - while (!(tlb = swiotlb_alloc_tlb(dev, tlb_size, phys_limit, gfp, - &tlb_vaddr))) { + while (!(tlb = swiotlb_alloc_tlb(dev, mem, tlb_size, + phys_limit, gfp, &tlb_vaddr))) { if (nslabs <= minslabs) goto error_tlb; nslabs = ALIGN(nslabs >> 1, IO_TLB_SEGSIZE); @@ -729,7 +752,7 @@ static struct io_tlb_pool *swiotlb_alloc_pool(struct device *dev, return pool; error_slots: - swiotlb_free_tlb(tlb_vaddr, tlb_size); + swiotlb_free_tlb(tlb_vaddr, tlb_size, mem->cc_shared); error_tlb: kfree(pool); error: @@ -746,7 +769,7 @@ static void swiotlb_dyn_alloc(struct work_struct *work) container_of(work, struct io_tlb_mem, dyn_alloc); struct io_tlb_pool *pool; - pool = swiotlb_alloc_pool(NULL, IO_TLB_MIN_SLABS, default_nslabs, + pool = swiotlb_alloc_pool(NULL, mem, IO_TLB_MIN_SLABS, default_nslabs, default_nareas, mem->phys_limit, GFP_KERNEL); if (!pool) { pr_warn_ratelimited("Failed to allocate new pool"); @@ -767,7 +790,7 @@ static void swiotlb_dyn_free(struct rcu_head *rcu) size_t tlb_size = pool->end - pool->start; free_pages((unsigned long)pool->slots, get_order(slots_size)); - swiotlb_free_tlb(pool->vaddr, tlb_size); + swiotlb_free_tlb(pool->vaddr, tlb_size, pool->cc_shared); kfree(pool); } @@ -1031,6 +1054,7 @@ static void dec_transient_used(struct io_tlb_mem *mem, unsigned int nslots) * @pool: Memory pool to be searched. * @area_index: Index of the IO TLB memory area to be searched. * @orig_addr: Original (non-bounced) IO buffer address. + * @tbl_dma_addr: DMA address of the bounce buffer. * @alloc_size: Total requested size of the bounce buffer, * including initial alignment padding. * @alloc_align_mask: Required alignment of the allocated buffer. @@ -1042,13 +1066,11 @@ static void dec_transient_used(struct io_tlb_mem *mem, unsigned int nslots) * Return: Index of the first allocated slot, or -1 on error. */ static int swiotlb_search_pool_area(struct device *dev, struct io_tlb_pool *pool, - int area_index, phys_addr_t orig_addr, size_t alloc_size, - unsigned int alloc_align_mask) + int area_index, phys_addr_t orig_addr, dma_addr_t tbl_dma_addr, + size_t alloc_size, unsigned int alloc_align_mask) { struct io_tlb_area *area = pool->areas + area_index; unsigned long boundary_mask = dma_get_seg_boundary(dev); - dma_addr_t tbl_dma_addr = - phys_to_dma_unencrypted(dev, pool->start) & boundary_mask; unsigned long max_slots = get_max_slots(boundary_mask); unsigned int iotlb_align_mask = dma_get_min_align_mask(dev); unsigned int nslots = nr_slots(alloc_size), stride; @@ -1061,6 +1083,8 @@ static int swiotlb_search_pool_area(struct device *dev, struct io_tlb_pool *pool BUG_ON(!nslots); BUG_ON(area_index >= pool->nareas); + tbl_dma_addr &= boundary_mask; + /* * Historically, swiotlb allocations >= PAGE_SIZE were guaranteed to be * page-aligned in the absence of any other alignment requirements. @@ -1172,6 +1196,7 @@ static int swiotlb_search_area(struct device *dev, int start_cpu, { struct io_tlb_mem *mem = dev->dma_io_tlb_mem; struct io_tlb_pool *pool; + dma_addr_t tbl_dma_addr; int area_index; int index = -1; @@ -1180,9 +1205,15 @@ static int swiotlb_search_area(struct device *dev, int start_cpu, if (cpu_offset >= pool->nareas) continue; area_index = (start_cpu + cpu_offset) & (pool->nareas - 1); + + if (mem->cc_shared) + tbl_dma_addr = phys_to_dma_unencrypted(dev, pool->start); + else + tbl_dma_addr = phys_to_dma_encrypted(dev, pool->start); + index = swiotlb_search_pool_area(dev, pool, area_index, - orig_addr, alloc_size, - alloc_align_mask); + orig_addr, tbl_dma_addr, + alloc_size, alloc_align_mask); if (index >= 0) { *retpool = pool; break; @@ -1212,6 +1243,7 @@ static int swiotlb_find_slots(struct device *dev, phys_addr_t orig_addr, { struct io_tlb_mem *mem = dev->dma_io_tlb_mem; struct io_tlb_pool *pool; + dma_addr_t tbl_dma_addr; unsigned long nslabs; unsigned long flags; u64 phys_limit; @@ -1236,12 +1268,17 @@ static int swiotlb_find_slots(struct device *dev, phys_addr_t orig_addr, nslabs = nr_slots(alloc_size); phys_limit = min_not_zero(*dev->dma_mask, dev->bus_dma_limit); - pool = swiotlb_alloc_pool(dev, nslabs, nslabs, 1, phys_limit, + pool = swiotlb_alloc_pool(dev, mem, nslabs, nslabs, 1, phys_limit, GFP_NOWAIT); if (!pool) return -1; - index = swiotlb_search_pool_area(dev, pool, 0, orig_addr, + if (mem->cc_shared) + tbl_dma_addr = phys_to_dma_unencrypted(dev, pool->start); + else + tbl_dma_addr = phys_to_dma_encrypted(dev, pool->start); + + index = swiotlb_search_pool_area(dev, pool, 0, orig_addr, tbl_dma_addr, alloc_size, alloc_align_mask); if (index < 0) { swiotlb_dyn_free(&pool->rcu); @@ -1286,15 +1323,23 @@ static int swiotlb_find_slots(struct device *dev, phys_addr_t orig_addr, size_t alloc_size, unsigned int alloc_align_mask, struct io_tlb_pool **retpool) { + struct io_tlb_mem *mem = dev->dma_io_tlb_mem; struct io_tlb_pool *pool; + dma_addr_t tbl_dma_addr; int start, i; int index; - *retpool = pool = &dev->dma_io_tlb_mem->defpool; + *retpool = pool = &mem->defpool; + if (mem->cc_shared) + tbl_dma_addr = phys_to_dma_unencrypted(dev, pool->start); + else + tbl_dma_addr = phys_to_dma_encrypted(dev, pool->start); + i = start = raw_smp_processor_id() & (pool->nareas - 1); do { index = swiotlb_search_pool_area(dev, pool, i, orig_addr, - alloc_size, alloc_align_mask); + tbl_dma_addr, alloc_size, + alloc_align_mask); if (index >= 0) return index; if (++i >= pool->nareas) @@ -1377,9 +1422,19 @@ static unsigned long mem_used(struct io_tlb_mem *mem) * any pre- or post-padding for alignment * @alloc_align_mask: Required start and end alignment of the allocated buffer * @dir: DMA direction - * @attrs: Optional DMA attributes for the map operation + * @attrs: Optional DMA attributes for the map operation, updated + * to match the selected SWIOTLB pool * * Find and allocate a suitable sequence of IO TLB slots for the request. + * The device's SWIOTLB pool must match the device's current DMA encryption + * requirements. If the device requires decrypted DMA, bouncing is done through + * an unencrypted pool and the mapping is marked shared. If the device can DMA + * to encrypted memory, bouncing is done through an encrypted pool even when the + * original DMA address was unencrypted. Enabling encrypted DMA for a device is + * therefore expected to update its default io_tlb_mem to an encrypted pool, so + * later bounce mappings for both encrypted and decrypted original memory use + * that encrypted pool. + * * The allocated space starts at an alignment specified by alloc_align_mask, * and the size of the allocated space is rounded up so that the total amount * of allocated space is a multiple of (alloc_align_mask + 1). If @@ -1416,6 +1471,30 @@ phys_addr_t swiotlb_tbl_map_single(struct device *dev, phys_addr_t orig_addr, if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) pr_warn_once("Memory encryption is active and system is using DMA bounce buffers\n"); + if (cc_platform_has(CC_ATTR_GUEST_MEM_ENCRYPT)) { + + /* swiotlb pool is incorrect for this device */ + if (unlikely(mem->cc_shared != force_dma_unencrypted(dev))) + return (phys_addr_t)DMA_MAPPING_ERROR; + + } else if (cc_platform_has(CC_ATTR_HOST_MEM_ENCRYPT)) { + /* + * On hosts with memory encryption, SWIOTLB-backed memory is + * unencrypted. DMA addresses returned for bounce buffers must + * therefore be marked unencrypted, even for devices that can + * address encrypted memory. This also preserves swiotlb=force + * behavior for those devices. + */ + if (unlikely(!mem->cc_shared)) + return (phys_addr_t)DMA_MAPPING_ERROR; + } + + /* Force attrs to match the kind of memory in the pool */ + if (mem->cc_shared) + *attrs |= DMA_ATTR_CC_SHARED; + else + *attrs &= ~DMA_ATTR_CC_SHARED; + /* * The default swiotlb memory pool is allocated with PAGE_SIZE * alignment. If a mapping is requested with larger alignment, @@ -1613,8 +1692,11 @@ dma_addr_t swiotlb_map(struct device *dev, phys_addr_t paddr, size_t size, if (swiotlb_addr == (phys_addr_t)DMA_MAPPING_ERROR) return DMA_MAPPING_ERROR; - /* Ensure that the address returned is DMA'ble */ - dma_addr = phys_to_dma_unencrypted(dev, swiotlb_addr); + if (attrs & DMA_ATTR_CC_SHARED) + dma_addr = phys_to_dma_unencrypted(dev, swiotlb_addr); + else + dma_addr = phys_to_dma_encrypted(dev, swiotlb_addr); + if (unlikely(!dma_capable(dev, dma_addr, size, true))) { __swiotlb_tbl_unmap_single(dev, swiotlb_addr, size, dir, attrs | DMA_ATTR_SKIP_CPU_SYNC, @@ -1778,7 +1860,7 @@ static inline void swiotlb_create_debugfs_files(struct io_tlb_mem *mem, #ifdef CONFIG_DMA_RESTRICTED_POOL -struct page *swiotlb_alloc(struct device *dev, size_t size) +struct page *swiotlb_alloc(struct device *dev, size_t size, unsigned long attrs) { struct io_tlb_mem *mem = dev->dma_io_tlb_mem; struct io_tlb_pool *pool; @@ -1789,6 +1871,9 @@ struct page *swiotlb_alloc(struct device *dev, size_t size) if (!mem) return NULL; + if (mem->cc_shared != !!(attrs & __DMA_ATTR_ALLOC_CC_SHARED)) + return NULL; + align = (1 << (get_order(size) + PAGE_SHIFT)) - 1; index = swiotlb_find_slots(dev, 0, size, align, &pool); if (index == -1) @@ -1864,12 +1949,20 @@ static int rmem_swiotlb_device_init(struct reserved_mem *rmem, kfree(mem); return -ENOMEM; } + /* + * if platform supports memory encryption, + * restricted mem pool is shared by default + */ + if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) { + mem->cc_shared = true; + set_memory_decrypted((unsigned long)phys_to_virt(rmem->base), + rmem->size >> PAGE_SHIFT); + } else { + mem->cc_shared = false; + } - set_memory_decrypted((unsigned long)phys_to_virt(rmem->base), - rmem->size >> PAGE_SHIFT); swiotlb_init_io_tlb_pool(pool, rmem->base, phys_to_virt(rmem->base), - nslabs, - false, nareas); + nslabs, false, nareas); mem->force_bounce = true; mem->for_alloc = true; #ifdef CONFIG_SWIOTLB_DYNAMIC -- cgit v1.2.3 From b45e3d35e2df444878d62ce241a4e53a83ebd36c Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:32 +0530 Subject: dma-mapping: make dma_pgprot() honor __DMA_ATTR_ALLOC_CC_SHARED Fold encrypted/decrypted pgprot selection into dma_pgprot() so callers do not need to adjust the page protection separately. Update dma_pgprot() to apply pgprot_decrypted() when DMA_ATTR_CC_SHARED or __DMA_ATTR_ALLOC_CC_SHARED is set and pgprot_encrypted() otherwise Convert the dma-direct mmap paths to pass DMA_ATTR_CC_SHARED instead of open-coding force_dma_unencrypted() handling around dma_pgprot(). Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-15-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 8 +++----- kernel/dma/mapping.c | 16 ++++++++++++---- 2 files changed, 15 insertions(+), 9 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index c6561ed05cb0..b58bd1610aff 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -286,9 +286,6 @@ setup_page: if (remap) { pgprot_t prot = dma_pgprot(dev, PAGE_KERNEL, attrs); - if (force_dma_unencrypted(dev)) - prot = pgprot_decrypted(prot); - /* remove any dirty cache lines on the kernel alias */ arch_dma_prep_coherent(page, size); @@ -608,9 +605,10 @@ int dma_direct_mmap(struct device *dev, struct vm_area_struct *vma, unsigned long pfn = PHYS_PFN(dma_to_phys(dev, dma_addr)); int ret = -ENXIO; - vma->vm_page_prot = dma_pgprot(dev, vma->vm_page_prot, attrs); if (force_dma_unencrypted(dev)) - vma->vm_page_prot = pgprot_decrypted(vma->vm_page_prot); + attrs |= DMA_ATTR_CC_SHARED; + + vma->vm_page_prot = dma_pgprot(dev, vma->vm_page_prot, attrs); if (dma_mmap_from_dev_coherent(dev, vma, cpu_addr, size, &ret)) return ret; diff --git a/kernel/dma/mapping.c b/kernel/dma/mapping.c index d2f70b6ccd0f..a628820fd10e 100644 --- a/kernel/dma/mapping.c +++ b/kernel/dma/mapping.c @@ -537,13 +537,21 @@ EXPORT_SYMBOL(dma_get_sgtable_attrs); */ pgprot_t dma_pgprot(struct device *dev, pgprot_t prot, unsigned long attrs) { + pgprot_t dma_prot; + if (dev_is_dma_coherent(dev)) - return prot; + dma_prot = prot; #ifdef CONFIG_ARCH_HAS_DMA_WRITE_COMBINE - if (attrs & DMA_ATTR_WRITE_COMBINE) - return pgprot_writecombine(prot); + else if (attrs & DMA_ATTR_WRITE_COMBINE) + dma_prot = pgprot_writecombine(prot); #endif - return pgprot_dmacoherent(prot); + else + dma_prot = pgprot_dmacoherent(prot); + + if (attrs & (DMA_ATTR_CC_SHARED | __DMA_ATTR_ALLOC_CC_SHARED)) + return pgprot_decrypted(dma_prot); + else + return pgprot_encrypted(dma_prot); } #endif /* CONFIG_MMU */ -- cgit v1.2.3 From 20beb6884b83ce750df93c4e38db6fa1b0d50c29 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:33 +0530 Subject: dma-direct: pass attrs to dma_capable() for DMA_ATTR_CC_SHARED checks Teach dma_capable() about DMA_ATTR_CC_SHARED so the capability check can reject encrypted DMA addresses for devices that require unencrypted/shared DMA. Also propagate DMA_ATTR_CC_SHARED in swiotlb_map() when the selected SWIOTLB pool is decrypted so the capability check sees the correct DMA address attribute. Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-16-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- arch/x86/kernel/amd_gart_64.c | 30 ++++++++++++++++-------------- drivers/xen/swiotlb-xen.c | 6 +++--- include/linux/dma-direct.h | 10 +++++++++- kernel/dma/direct.h | 6 +++--- kernel/dma/swiotlb.c | 2 +- 5 files changed, 32 insertions(+), 22 deletions(-) (limited to 'kernel') diff --git a/arch/x86/kernel/amd_gart_64.c b/arch/x86/kernel/amd_gart_64.c index e8000a56732e..b5f1f031d45b 100644 --- a/arch/x86/kernel/amd_gart_64.c +++ b/arch/x86/kernel/amd_gart_64.c @@ -180,22 +180,23 @@ static void iommu_full(struct device *dev, size_t size, int dir) } static inline int -need_iommu(struct device *dev, unsigned long addr, size_t size) +need_iommu(struct device *dev, unsigned long addr, size_t size, unsigned long attrs) { - return force_iommu || !dma_capable(dev, addr, size, true); + return force_iommu || !dma_capable(dev, addr, size, true, attrs); } static inline int -nonforced_iommu(struct device *dev, unsigned long addr, size_t size) +nonforced_iommu(struct device *dev, unsigned long addr, size_t size, + unsigned long attrs) { - return !dma_capable(dev, addr, size, true); + return !dma_capable(dev, addr, size, true, attrs); } /* Map a single continuous physical area into the IOMMU. * Caller needs to check if the iommu is needed and flush. */ static dma_addr_t dma_map_area(struct device *dev, dma_addr_t phys_mem, - size_t size, int dir, unsigned long align_mask) + size_t size, int dir, unsigned long align_mask, unsigned long attrs) { unsigned long npages = iommu_num_pages(phys_mem, size, PAGE_SIZE); unsigned long iommu_page; @@ -206,7 +207,7 @@ static dma_addr_t dma_map_area(struct device *dev, dma_addr_t phys_mem, iommu_page = alloc_iommu(dev, npages, align_mask); if (iommu_page == -1) { - if (!nonforced_iommu(dev, phys_mem, size)) + if (!nonforced_iommu(dev, phys_mem, size, attrs)) return phys_mem; if (panic_on_overflow) panic("dma_map_area overflow %lu bytes\n", size); @@ -231,10 +232,10 @@ static dma_addr_t gart_map_phys(struct device *dev, phys_addr_t paddr, if (unlikely(attrs & DMA_ATTR_MMIO)) return DMA_MAPPING_ERROR; - if (!need_iommu(dev, paddr, size)) + if (!need_iommu(dev, paddr, size, attrs)) return paddr; - bus = dma_map_area(dev, paddr, size, dir, 0); + bus = dma_map_area(dev, paddr, size, dir, 0, attrs); flush_gart(); return bus; @@ -289,7 +290,7 @@ static void gart_unmap_sg(struct device *dev, struct scatterlist *sg, int nents, /* Fallback for dma_map_sg in case of overflow */ static int dma_map_sg_nonforce(struct device *dev, struct scatterlist *sg, - int nents, int dir) + int nents, int dir, unsigned long attrs) { struct scatterlist *s; int i; @@ -301,8 +302,8 @@ static int dma_map_sg_nonforce(struct device *dev, struct scatterlist *sg, for_each_sg(sg, s, nents, i) { unsigned long addr = sg_phys(s); - if (nonforced_iommu(dev, addr, s->length)) { - addr = dma_map_area(dev, addr, s->length, dir, 0); + if (nonforced_iommu(dev, addr, s->length, attrs)) { + addr = dma_map_area(dev, addr, s->length, dir, 0, attrs); if (addr == DMA_MAPPING_ERROR) { if (i > 0) gart_unmap_sg(dev, sg, i, dir, 0); @@ -401,7 +402,7 @@ static int gart_map_sg(struct device *dev, struct scatterlist *sg, int nents, s->dma_address = addr; BUG_ON(s->length == 0); - nextneed = need_iommu(dev, addr, s->length); + nextneed = need_iommu(dev, addr, s->length, attrs); /* Handle the previous not yet processed entries */ if (i > start) { @@ -449,7 +450,7 @@ error: /* When it was forced or merged try again in a dumb way */ if (force_iommu || iommu_merge) { - out = dma_map_sg_nonforce(dev, sg, nents, dir); + out = dma_map_sg_nonforce(dev, sg, nents, dir, attrs); if (out > 0) return out; } @@ -473,7 +474,8 @@ gart_alloc_coherent(struct device *dev, size_t size, dma_addr_t *dma_addr, return vaddr; *dma_addr = dma_map_area(dev, virt_to_phys(vaddr), size, - DMA_BIDIRECTIONAL, (1UL << get_order(size)) - 1); + DMA_BIDIRECTIONAL, + (1UL << get_order(size)) - 1, attrs); flush_gart(); if (unlikely(*dma_addr == DMA_MAPPING_ERROR)) goto out_free; diff --git a/drivers/xen/swiotlb-xen.c b/drivers/xen/swiotlb-xen.c index 8c4abe65cd49..e2538824ef52 100644 --- a/drivers/xen/swiotlb-xen.c +++ b/drivers/xen/swiotlb-xen.c @@ -212,7 +212,7 @@ static dma_addr_t xen_swiotlb_map_phys(struct device *dev, phys_addr_t phys, BUG_ON(dir == DMA_NONE); if (attrs & DMA_ATTR_MMIO) { - if (unlikely(!dma_capable(dev, phys, size, false))) { + if (unlikely(!dma_capable(dev, phys, size, false, attrs))) { dev_err_once( dev, "DMA addr %pa+%zu overflow (mask %llx, bus limit %llx).\n", @@ -231,7 +231,7 @@ static dma_addr_t xen_swiotlb_map_phys(struct device *dev, phys_addr_t phys, * we can safely return the device addr and not worry about bounce * buffering it. */ - if (dma_capable(dev, dev_addr, size, true) && + if (dma_capable(dev, dev_addr, size, true, attrs) && !dma_kmalloc_needs_bounce(dev, size, dir) && !range_straddles_page_boundary(phys, size) && !xen_arch_need_swiotlb(dev, phys, dev_addr) && @@ -253,7 +253,7 @@ static dma_addr_t xen_swiotlb_map_phys(struct device *dev, phys_addr_t phys, /* * Ensure that the address returned is DMA'ble */ - if (unlikely(!dma_capable(dev, dev_addr, size, true))) { + if (unlikely(!dma_capable(dev, dev_addr, size, true, attrs))) { __swiotlb_tbl_unmap_single(dev, map, size, dir, attrs | DMA_ATTR_SKIP_CPU_SYNC, swiotlb_find_pool(dev, map)); diff --git a/include/linux/dma-direct.h b/include/linux/dma-direct.h index 94fad4e7c11e..daa31a1adf7b 100644 --- a/include/linux/dma-direct.h +++ b/include/linux/dma-direct.h @@ -135,12 +135,20 @@ static inline bool force_dma_unencrypted(struct device *dev) #endif /* CONFIG_ARCH_HAS_FORCE_DMA_UNENCRYPTED */ static inline bool dma_capable(struct device *dev, dma_addr_t addr, size_t size, - bool is_ram) + bool is_ram, unsigned long attrs) { dma_addr_t end = addr + size - 1; if (addr == DMA_MAPPING_ERROR) return false; + /* + * The DMA address was derived from encrypted RAM, but this device + * requires unencrypted DMA addresses. Treat it as not DMA-capable + * so the caller can fall back to a suitable SWIOTLB pool. + */ + if (!(attrs & DMA_ATTR_CC_SHARED) && force_dma_unencrypted(dev)) + return false; + if (is_ram && !IS_ENABLED(CONFIG_ARCH_DMA_ADDR_T_64BIT) && min(addr, end) < phys_to_dma(dev, PFN_PHYS(min_low_pfn))) return false; diff --git a/kernel/dma/direct.h b/kernel/dma/direct.h index 7140c208c123..e05dc7649366 100644 --- a/kernel/dma/direct.h +++ b/kernel/dma/direct.h @@ -101,15 +101,15 @@ static inline dma_addr_t dma_direct_map_phys(struct device *dev, if (attrs & DMA_ATTR_MMIO) { dma_addr = phys; - if (unlikely(!dma_capable(dev, dma_addr, size, false))) + if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) goto err_overflow; } else if (attrs & DMA_ATTR_CC_SHARED) { dma_addr = phys_to_dma_unencrypted(dev, phys); - if (unlikely(!dma_capable(dev, dma_addr, size, false))) + if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) goto err_overflow; } else { dma_addr = phys_to_dma(dev, phys); - if (unlikely(!dma_capable(dev, dma_addr, size, true)) || + if (unlikely(!dma_capable(dev, dma_addr, size, true, attrs)) || dma_kmalloc_needs_bounce(dev, size, dir)) { if (is_swiotlb_active(dev) && !(attrs & DMA_ATTR_REQUIRE_COHERENT)) diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 335e27bc1e1f..b5960c2e98d8 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -1697,7 +1697,7 @@ dma_addr_t swiotlb_map(struct device *dev, phys_addr_t paddr, size_t size, else dma_addr = phys_to_dma_encrypted(dev, swiotlb_addr); - if (unlikely(!dma_capable(dev, dma_addr, size, true))) { + if (unlikely(!dma_capable(dev, dma_addr, size, true, attrs))) { __swiotlb_tbl_unmap_single(dev, swiotlb_addr, size, dir, attrs | DMA_ATTR_SKIP_CPU_SYNC, swiotlb_find_pool(dev, swiotlb_addr)); -- cgit v1.2.3 From b4f38ddd7943797ac171643a8f451fde74a42ace Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:34 +0530 Subject: dma-direct: Move dma_direct_map_phys() to dma/direct.c dma_direct_map_phys() is too large to benefit from being inlined. Move its implementation to direct.c and leave the declaration in direct.h. No functional change in this patch Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-17-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 53 ++++++++++++++++++++++++++++++++++++++++++++++++++ kernel/dma/direct.h | 56 +++-------------------------------------------------- 2 files changed, 56 insertions(+), 53 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index b58bd1610aff..c2acdadc3b16 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -621,6 +621,59 @@ int dma_direct_mmap(struct device *dev, struct vm_area_struct *vma, user_count << PAGE_SHIFT, vma->vm_page_prot); } +dma_addr_t dma_direct_map_phys(struct device *dev, phys_addr_t phys, + size_t size, enum dma_data_direction dir, + unsigned long attrs, bool flush) +{ + dma_addr_t dma_addr; + + if (is_swiotlb_force_bounce(dev)) { + if (!(attrs & DMA_ATTR_CC_SHARED)) { + if (attrs & (DMA_ATTR_MMIO | DMA_ATTR_REQUIRE_COHERENT)) + return DMA_MAPPING_ERROR; + + return swiotlb_map(dev, phys, size, dir, attrs); + } + } else if (attrs & DMA_ATTR_CC_SHARED) { + return DMA_MAPPING_ERROR; + } + + if (attrs & DMA_ATTR_MMIO) { + dma_addr = phys; + if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) + goto err_overflow; + } else if (attrs & DMA_ATTR_CC_SHARED) { + dma_addr = phys_to_dma_unencrypted(dev, phys); + if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) + goto err_overflow; + } else { + dma_addr = phys_to_dma(dev, phys); + if (unlikely(!dma_capable(dev, dma_addr, size, true, attrs)) || + dma_kmalloc_needs_bounce(dev, size, dir)) { + if (is_swiotlb_active(dev) && + !(attrs & DMA_ATTR_REQUIRE_COHERENT)) + return swiotlb_map(dev, phys, size, dir, attrs); + + goto err_overflow; + } + } + + if (!dev_is_dma_coherent(dev) && + !(attrs & (DMA_ATTR_SKIP_CPU_SYNC | DMA_ATTR_MMIO))) { + arch_sync_dma_for_device(phys, size, dir); + if (flush) + arch_sync_dma_flush(); + } + return dma_addr; + +err_overflow: + dev_WARN_ONCE( + dev, 1, + "DMA addr %pad+%zu overflow (mask %llx, bus limit %llx).\n", + &dma_addr, size, *dev->dma_mask, dev->bus_dma_limit); + return DMA_MAPPING_ERROR; +} + int dma_direct_supported(struct device *dev, u64 mask) { u64 min_mask = ((u64)max_pfn << PAGE_SHIFT) - 1; diff --git a/kernel/dma/direct.h b/kernel/dma/direct.h index e05dc7649366..a7adadb1b2a5 100644 --- a/kernel/dma/direct.h +++ b/kernel/dma/direct.h @@ -17,6 +17,9 @@ bool dma_direct_can_mmap(struct device *dev); int dma_direct_mmap(struct device *dev, struct vm_area_struct *vma, void *cpu_addr, dma_addr_t dma_addr, size_t size, unsigned long attrs); +dma_addr_t dma_direct_map_phys(struct device *dev, phys_addr_t phys, + size_t size, enum dma_data_direction dir, + unsigned long attrs, bool flush); bool dma_direct_need_sync(struct device *dev, dma_addr_t dma_addr); int dma_direct_map_sg(struct device *dev, struct scatterlist *sgl, int nents, enum dma_data_direction dir, unsigned long attrs); @@ -82,59 +85,6 @@ static inline void dma_direct_sync_single_for_cpu(struct device *dev, swiotlb_sync_single_for_cpu(dev, paddr, size, dir); } -static inline dma_addr_t dma_direct_map_phys(struct device *dev, - phys_addr_t phys, size_t size, enum dma_data_direction dir, - unsigned long attrs, bool flush) -{ - dma_addr_t dma_addr; - - if (is_swiotlb_force_bounce(dev)) { - if (!(attrs & DMA_ATTR_CC_SHARED)) { - if (attrs & (DMA_ATTR_MMIO | DMA_ATTR_REQUIRE_COHERENT)) - return DMA_MAPPING_ERROR; - - return swiotlb_map(dev, phys, size, dir, attrs); - } - } else if (attrs & DMA_ATTR_CC_SHARED) { - return DMA_MAPPING_ERROR; - } - - if (attrs & DMA_ATTR_MMIO) { - dma_addr = phys; - if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) - goto err_overflow; - } else if (attrs & DMA_ATTR_CC_SHARED) { - dma_addr = phys_to_dma_unencrypted(dev, phys); - if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) - goto err_overflow; - } else { - dma_addr = phys_to_dma(dev, phys); - if (unlikely(!dma_capable(dev, dma_addr, size, true, attrs)) || - dma_kmalloc_needs_bounce(dev, size, dir)) { - if (is_swiotlb_active(dev) && - !(attrs & DMA_ATTR_REQUIRE_COHERENT)) - return swiotlb_map(dev, phys, size, dir, attrs); - - goto err_overflow; - } - } - - if (!dev_is_dma_coherent(dev) && - !(attrs & (DMA_ATTR_SKIP_CPU_SYNC | DMA_ATTR_MMIO))) { - arch_sync_dma_for_device(phys, size, dir); - if (flush) - arch_sync_dma_flush(); - } - return dma_addr; - -err_overflow: - dev_WARN_ONCE( - dev, 1, - "DMA addr %pad+%zu overflow (mask %llx, bus limit %llx).\n", - &dma_addr, size, *dev->dma_mask, dev->bus_dma_limit); - return DMA_MAPPING_ERROR; -} - static inline void dma_direct_unmap_phys(struct device *dev, dma_addr_t addr, size_t size, enum dma_data_direction dir, unsigned long attrs, bool flush) -- cgit v1.2.3 From 30c5e45ee2c536de4a2c4357c7757e30feeef7ca Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:35 +0530 Subject: dma-direct: make dma_direct_map_phys() honor DMA_ATTR_CC_SHARED Teach dma_direct_map_phys() to select the DMA address encoding based on DMA_ATTR_CC_SHARED. Use phys_to_dma_unencrypted() for decrypted mappings and phys_to_dma_encrypted() otherwise. If a device requires unencrypted DMA but the source physical address is still encrypted, force the mapping through swiotlb so the DMA address and backing memory attributes remain consistent. Update the arm64, x86, s390 and powerpc secure-guest setup to not use swiotlb force option Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-18-aneesh.kumar@kernel.org [mszyprow: rebased onto latest changes in arch/arm64/mm/init.c] Signed-off-by: Marek Szyprowski --- arch/arm64/mm/init.c | 4 +-- arch/powerpc/platforms/pseries/svm.c | 2 +- arch/s390/mm/init.c | 2 +- arch/x86/kernel/pci-dma.c | 4 +-- kernel/dma/direct.c | 52 +++++++++++++++++++++--------------- 5 files changed, 34 insertions(+), 30 deletions(-) (limited to 'kernel') diff --git a/arch/arm64/mm/init.c b/arch/arm64/mm/init.c index 8ab8349d6126..e308a7cabd12 100644 --- a/arch/arm64/mm/init.c +++ b/arch/arm64/mm/init.c @@ -339,9 +339,7 @@ void __init arch_mm_preinit(void) { unsigned int flags = SWIOTLB_VERBOSE; - if (is_realm_world() || is_protected_kvm_guest()) { - flags |= SWIOTLB_FORCE; - } else if (max_pfn <= PFN_DOWN(arm64_dma_phys_limit)) { + if (max_pfn <= PFN_DOWN(arm64_dma_phys_limit)) { /* * If no bouncing needed for ZONE_DMA, reduce the swiotlb * buffer for kmalloc() bouncing to 1MB per 1GB of RAM. diff --git a/arch/powerpc/platforms/pseries/svm.c b/arch/powerpc/platforms/pseries/svm.c index 384c9dc1899a..7a403dbd35ee 100644 --- a/arch/powerpc/platforms/pseries/svm.c +++ b/arch/powerpc/platforms/pseries/svm.c @@ -29,7 +29,7 @@ static int __init init_svm(void) * need to use the SWIOTLB buffer for DMA even if dma_capable() says * otherwise. */ - ppc_swiotlb_flags |= SWIOTLB_ANY | SWIOTLB_FORCE; + ppc_swiotlb_flags |= SWIOTLB_ANY; /* Share the SWIOTLB buffer with the host. */ swiotlb_update_mem_attributes(); diff --git a/arch/s390/mm/init.c b/arch/s390/mm/init.c index 6b1c5a4fa9ce..8d1de5a2e554 100644 --- a/arch/s390/mm/init.c +++ b/arch/s390/mm/init.c @@ -166,7 +166,7 @@ static void __init pv_init(void) virtio_set_mem_acc_cb(virtio_require_restricted_mem_acc); /* make sure bounce buffers are shared */ - swiotlb_init(true, SWIOTLB_FORCE | SWIOTLB_VERBOSE); + swiotlb_init(true, SWIOTLB_VERBOSE); swiotlb_update_mem_attributes(); } diff --git a/arch/x86/kernel/pci-dma.c b/arch/x86/kernel/pci-dma.c index 6267363e0189..75cf8f6ae8cd 100644 --- a/arch/x86/kernel/pci-dma.c +++ b/arch/x86/kernel/pci-dma.c @@ -59,10 +59,8 @@ static void __init pci_swiotlb_detect(void) * bounce buffers as the hypervisor can't access arbitrary VM memory * that is not explicitly shared with it. */ - if (cc_platform_has(CC_ATTR_GUEST_MEM_ENCRYPT)) { + if (cc_platform_has(CC_ATTR_GUEST_MEM_ENCRYPT)) x86_swiotlb_enable = true; - x86_swiotlb_flags |= SWIOTLB_FORCE; - } } #else static inline void __init pci_swiotlb_detect(void) diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index c2acdadc3b16..c24f26a590b7 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -14,6 +14,8 @@ #include #include #include +#include + #include "direct.h" /* @@ -627,37 +629,41 @@ dma_addr_t dma_direct_map_phys(struct device *dev, phys_addr_t phys, { dma_addr_t dma_addr; + if (attrs & DMA_ATTR_MMIO) { + /* + * For host memory encryption treat MMIO memory as shared + */ + if (cc_platform_has(CC_ATTR_HOST_MEM_ENCRYPT)) + attrs |= DMA_ATTR_CC_SHARED; + } + if (is_swiotlb_force_bounce(dev)) { - if (!(attrs & DMA_ATTR_CC_SHARED)) { - if (attrs & (DMA_ATTR_MMIO | DMA_ATTR_REQUIRE_COHERENT)) - return DMA_MAPPING_ERROR; + if (attrs & (DMA_ATTR_MMIO | DMA_ATTR_REQUIRE_COHERENT)) + return DMA_MAPPING_ERROR; - return swiotlb_map(dev, phys, size, dir, attrs); - } - } else if (attrs & DMA_ATTR_CC_SHARED) { - return DMA_MAPPING_ERROR; + return swiotlb_map(dev, phys, size, dir, attrs); } - if (attrs & DMA_ATTR_MMIO) { - dma_addr = phys; - if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) - goto err_overflow; - } else if (attrs & DMA_ATTR_CC_SHARED) { + if (attrs & DMA_ATTR_CC_SHARED) dma_addr = phys_to_dma_unencrypted(dev, phys); + else + dma_addr = phys_to_dma_encrypted(dev, phys); + + if (attrs & DMA_ATTR_MMIO) { if (unlikely(!dma_capable(dev, dma_addr, size, false, attrs))) goto err_overflow; - } else { - dma_addr = phys_to_dma(dev, phys); - if (unlikely(!dma_capable(dev, dma_addr, size, true, attrs)) || - dma_kmalloc_needs_bounce(dev, size, dir)) { - if (is_swiotlb_active(dev) && - !(attrs & DMA_ATTR_REQUIRE_COHERENT)) - return swiotlb_map(dev, phys, size, dir, attrs); + goto dma_mapped; + } - goto err_overflow; - } + if (unlikely(!dma_capable(dev, dma_addr, size, true, attrs)) || + dma_kmalloc_needs_bounce(dev, size, dir)) { + if (is_swiotlb_active(dev) && + !(attrs & DMA_ATTR_REQUIRE_COHERENT)) + return swiotlb_map(dev, phys, size, dir, attrs); + goto err_overflow; } +dma_mapped: if (!dev_is_dma_coherent(dev) && !(attrs & (DMA_ATTR_SKIP_CPU_SYNC | DMA_ATTR_MMIO))) { arch_sync_dma_for_device(phys, size, dir); @@ -749,8 +755,10 @@ size_t dma_direct_max_mapping_size(struct device *dev) { /* If SWIOTLB is active, use its maximum mapping size */ if (is_swiotlb_active(dev) && - (dma_addressing_limited(dev) || is_swiotlb_force_bounce(dev))) + (dma_addressing_limited(dev) || is_swiotlb_force_bounce(dev) || + force_dma_unencrypted(dev))) return swiotlb_max_mapping_size(dev); + return SIZE_MAX; } -- cgit v1.2.3 From f253a83ff94c5c5107d1e23ffadd8fc1fe0854c9 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:36 +0530 Subject: dma-direct: set decrypted flag for remapped DMA allocations Devices that are DMA non-coherent and require a remap were skipping dma_set_decrypted(), leaving DMA buffers encrypted even when the device requires unencrypted access. Move the call after the if (remap) branch so that both the direct and remapped allocation paths correctly mark the allocation as decrypted (or fail cleanly) before use. Fix dma_direct_alloc() and dma_direct_free() to apply set_memory_*() to the linear-map alias of the backing pages instead of the remapped CPU address. Also disallow highmem pages for __DMA_ATTR_ALLOC_CC_SHARED, because highmem buffers do not provide a usable linear-map address. Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-19-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 56 +++++++++++++++++++++++++++++++++++++++++------------ 1 file changed, 44 insertions(+), 12 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index c24f26a590b7..7f9f65121137 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -198,14 +198,23 @@ void *dma_direct_alloc(struct device *dev, size_t size, { bool remap = false, set_uncached = false; bool mark_mem_decrypt = false; + bool allow_highmem = true; struct page *page; void *ret; if (force_dma_unencrypted(dev)) attrs |= __DMA_ATTR_ALLOC_CC_SHARED; - if (attrs & __DMA_ATTR_ALLOC_CC_SHARED) + if (attrs & __DMA_ATTR_ALLOC_CC_SHARED) { + /* + * Unencrypted/shared DMA requires a linear-mapped buffer + * address to look up the PFN and set architecture-required PFN + * attributes. This is not possible with HighMem. Avoid HighMem + * allocation. + */ + allow_highmem = false; mark_mem_decrypt = true; + } size = PAGE_ALIGN(size); if (attrs & DMA_ATTR_NO_WARN) @@ -270,7 +279,7 @@ void *dma_direct_alloc(struct device *dev, size_t size, } /* we always manually zero the memory once we are done */ - page = __dma_direct_alloc_pages(dev, size, gfp & ~__GFP_ZERO, true); + page = __dma_direct_alloc_pages(dev, size, gfp & ~__GFP_ZERO, allow_highmem); if (!page) return NULL; @@ -285,6 +294,14 @@ setup_page: set_uncached = false; } + if (mark_mem_decrypt) { + void *lm_addr; + + lm_addr = page_address(page); + if (set_memory_decrypted((unsigned long)lm_addr, PFN_UP(size))) + goto out_leak_pages; + } + if (remap) { pgprot_t prot = dma_pgprot(dev, PAGE_KERNEL, attrs); @@ -295,29 +312,36 @@ setup_page: ret = dma_common_contiguous_remap(page, size, prot, __builtin_return_address(0)); if (!ret) - goto out_free_pages; + goto out_encrypt_pages; } else { ret = page_address(page); - if (mark_mem_decrypt && dma_set_decrypted(dev, ret, size)) - goto out_leak_pages; } memset(ret, 0, size); if (set_uncached) { + void *uncached_cpu_addr; + arch_dma_prep_coherent(page, size); - ret = arch_dma_set_uncached(ret, size); - if (IS_ERR(ret)) - goto out_encrypt_pages; + uncached_cpu_addr = arch_dma_set_uncached(ret, size); + if (IS_ERR(uncached_cpu_addr)) + goto out_free_remap_pages; + ret = uncached_cpu_addr; } *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); return ret; + +out_free_remap_pages: + if (remap) + dma_common_free_remap(ret, size); + out_encrypt_pages: - if (mark_mem_decrypt && dma_set_encrypted(dev, page_address(page), size)) - return NULL; -out_free_pages: + if (mark_mem_decrypt && + dma_set_encrypted(dev, page_address(page), size)) + goto out_leak_pages; + if (!swiotlb_free(dev, page, size)) dma_free_contiguous(dev, page, size); return NULL; @@ -380,8 +404,16 @@ void dma_direct_free(struct device *dev, size_t size, } else { if (IS_ENABLED(CONFIG_ARCH_HAS_DMA_CLEAR_UNCACHED)) arch_dma_clear_uncached(cpu_addr, size); - if (mark_mem_encrypted && dma_set_encrypted(dev, cpu_addr, size)) + } + + if (mark_mem_encrypted) { + void *lm_addr; + + lm_addr = phys_to_virt(phys); + if (set_memory_encrypted((unsigned long)lm_addr, PFN_UP(size))) { + pr_warn_ratelimited("leaking DMA memory that can't be re-encrypted\n"); return; + } } if (swiotlb_pool) -- cgit v1.2.3 From fd01136e25c6256dfc6ed41e7030c4cdfa5597de Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:37 +0530 Subject: dma-direct: select DMA address encoding from __DMA_ATTR_ALLOC_CC_SHARED Make the dma-direct helpers derive the DMA address encoding from __DMA_ATTR_ALLOC_CC_SHARED instead of implicitly relying on force_dma_unencrypted() inside phys_to_dma_direct() Pass an explicit unencrypted/decrypted state into phys_to_dma_direct(), make the alloc paths return DMA addresses that match the requested buffer encryption state. Also only call dma_set_decrypted() when __DMA_ATTR_ALLOC_CC_SHARED is actually set. Reviewed-by: Jason Gunthorpe Tested-by: Jiri Pirko Tested-by: Michael Kelley Tested-by: Mostafa Saleh Signed-off-by: Aneesh Kumar K.V (Arm) Reviewed-by: Mostafa Saleh Link: https://lore.kernel.org/r/20260717180442.110954-20-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 43 ++++++++++++++++++++++++++----------------- 1 file changed, 26 insertions(+), 17 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 7f9f65121137..83b9b747f4a5 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -26,11 +26,11 @@ u64 zone_dma_limit __ro_after_init = DMA_BIT_MASK(24); static inline dma_addr_t phys_to_dma_direct(struct device *dev, - phys_addr_t phys) + phys_addr_t phys, bool unencrypted) { - if (force_dma_unencrypted(dev)) + if (unencrypted) return phys_to_dma_unencrypted(dev, phys); - return phys_to_dma(dev, phys); + return phys_to_dma_encrypted(dev, phys); } static inline struct page *dma_direct_to_page(struct device *dev, @@ -41,8 +41,9 @@ static inline struct page *dma_direct_to_page(struct device *dev, u64 dma_direct_get_required_mask(struct device *dev) { + bool require_decrypted = force_dma_unencrypted(dev); phys_addr_t phys = ((phys_addr_t)max_pfn << PAGE_SHIFT) - 1; - u64 max_dma = phys_to_dma_direct(dev, phys); + u64 max_dma = phys_to_dma_direct(dev, phys, require_decrypted); return (1ULL << (fls64(max_dma) - 1)) * 2 - 1; } @@ -71,7 +72,8 @@ static gfp_t dma_direct_optimal_gfp_mask(struct device *dev, u64 *phys_limit) bool dma_coherent_ok(struct device *dev, phys_addr_t phys, size_t size) { - dma_addr_t dma_addr = phys_to_dma_direct(dev, phys); + bool require_decrypted = force_dma_unencrypted(dev); + dma_addr_t dma_addr = phys_to_dma_direct(dev, phys, require_decrypted); if (dma_addr == DMA_MAPPING_ERROR) return false; @@ -81,17 +83,18 @@ bool dma_coherent_ok(struct device *dev, phys_addr_t phys, size_t size) static int dma_set_decrypted(struct device *dev, void *vaddr, size_t size) { - if (!force_dma_unencrypted(dev)) - return 0; - return set_memory_decrypted((unsigned long)vaddr, PFN_UP(size)); + int ret; + + ret = set_memory_decrypted((unsigned long)vaddr, PFN_UP(size)); + if (ret) + pr_warn_ratelimited("leaking DMA memory that can't be decrypted\n"); + return ret; } static int dma_set_encrypted(struct device *dev, void *vaddr, size_t size) { int ret; - if (!force_dma_unencrypted(dev)) - return 0; ret = set_memory_encrypted((unsigned long)vaddr, PFN_UP(size)); if (ret) pr_warn_ratelimited("leaking DMA memory that can't be re-encrypted\n"); @@ -171,7 +174,8 @@ static struct page *dma_direct_alloc_from_pool(struct device *dev, size_t size, dma_coherent_ok); if (!page) return NULL; - *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); + *dma_handle = phys_to_dma_direct(dev, page_to_phys(page), + attrs & __DMA_ATTR_ALLOC_CC_SHARED); return page; } @@ -187,9 +191,11 @@ static void *dma_direct_alloc_no_mapping(struct device *dev, size_t size, /* remove any dirty cache lines on the kernel alias */ if (!PageHighMem(page)) arch_dma_prep_coherent(page, size); - - /* return the page pointer as the opaque cookie */ - *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); + /* + * return the page pointer as the opaque cookie. + * Never used for unencrypted allocation + */ + *dma_handle = phys_to_dma_encrypted(dev, page_to_phys(page)); return page; } @@ -329,7 +335,8 @@ setup_page: ret = uncached_cpu_addr; } - *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); + *dma_handle = phys_to_dma_direct(dev, page_to_phys(page), + attrs & __DMA_ATTR_ALLOC_CC_SHARED); return ret; @@ -450,11 +457,13 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, return NULL; ret = page_address(page); - if (dma_set_decrypted(dev, ret, size)) + if ((attrs & __DMA_ATTR_ALLOC_CC_SHARED) && + dma_set_decrypted(dev, ret, size)) goto out_leak_pages; setup_page: memset(ret, 0, size); - *dma_handle = phys_to_dma_direct(dev, page_to_phys(page)); + *dma_handle = phys_to_dma_direct(dev, page_to_phys(page), + attrs & __DMA_ATTR_ALLOC_CC_SHARED); return page; out_leak_pages: return NULL; -- cgit v1.2.3 From 63e62b63e853946cb9a75aead89b0a6ec435fe89 Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:38 +0530 Subject: dma-direct: rename ret to cpu_addr in alloc helpers ret in dma_direct_alloc() and dma_direct_alloc_pages() holds the returned CPU mapping, not a generic return value. Rename it to cpu_addr and update the remaining uses to match. This makes the allocation paths easier to follow and keeps the local naming consistent with what the variable actually represents. Reviewed-by: Jason Gunthorpe Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-21-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/direct.c | 40 ++++++++++++++++++++-------------------- 1 file changed, 20 insertions(+), 20 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/direct.c b/kernel/dma/direct.c index 83b9b747f4a5..0a7943d31e20 100644 --- a/kernel/dma/direct.c +++ b/kernel/dma/direct.c @@ -206,7 +206,7 @@ void *dma_direct_alloc(struct device *dev, size_t size, bool mark_mem_decrypt = false; bool allow_highmem = true; struct page *page; - void *ret; + void *cpu_addr; if (force_dma_unencrypted(dev)) attrs |= __DMA_ATTR_ALLOC_CC_SHARED; @@ -266,9 +266,10 @@ void *dma_direct_alloc(struct device *dev, size_t size, */ if ((remap || (attrs & __DMA_ATTR_ALLOC_CC_SHARED)) && dma_direct_use_pool(dev, gfp)) { - page = dma_direct_alloc_from_pool(dev, size, dma_handle, - &ret, gfp, attrs); - return page ? ret : NULL; + page = dma_direct_alloc_from_pool(dev, size, + dma_handle, &cpu_addr, + gfp, attrs); + return page ? cpu_addr : NULL; } if (is_swiotlb_for_alloc(dev)) { @@ -315,34 +316,33 @@ setup_page: arch_dma_prep_coherent(page, size); /* create a coherent mapping */ - ret = dma_common_contiguous_remap(page, size, prot, - __builtin_return_address(0)); - if (!ret) + cpu_addr = dma_common_contiguous_remap(page, size, prot, + __builtin_return_address(0)); + if (!cpu_addr) goto out_encrypt_pages; } else { - ret = page_address(page); + cpu_addr = page_address(page); } - memset(ret, 0, size); + memset(cpu_addr, 0, size); if (set_uncached) { void *uncached_cpu_addr; arch_dma_prep_coherent(page, size); - uncached_cpu_addr = arch_dma_set_uncached(ret, size); + uncached_cpu_addr = arch_dma_set_uncached(cpu_addr, size); if (IS_ERR(uncached_cpu_addr)) goto out_free_remap_pages; - ret = uncached_cpu_addr; + cpu_addr = uncached_cpu_addr; } *dma_handle = phys_to_dma_direct(dev, page_to_phys(page), attrs & __DMA_ATTR_ALLOC_CC_SHARED); - return ret; - + return cpu_addr; out_free_remap_pages: if (remap) - dma_common_free_remap(ret, size); + dma_common_free_remap(cpu_addr, size); out_encrypt_pages: if (mark_mem_decrypt && @@ -434,21 +434,21 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, { unsigned long attrs = 0; struct page *page; - void *ret; + void *cpu_addr; if (force_dma_unencrypted(dev)) attrs |= __DMA_ATTR_ALLOC_CC_SHARED; if ((attrs & __DMA_ATTR_ALLOC_CC_SHARED) && dma_direct_use_pool(dev, gfp)) return dma_direct_alloc_from_pool(dev, size, dma_handle, - &ret, gfp, attrs); + &cpu_addr, gfp, attrs); if (is_swiotlb_for_alloc(dev)) { page = dma_direct_alloc_swiotlb(dev, size, attrs); if (!page) return NULL; - ret = page_address(page); + cpu_addr = page_address(page); goto setup_page; } @@ -456,12 +456,12 @@ struct page *dma_direct_alloc_pages(struct device *dev, size_t size, if (!page) return NULL; - ret = page_address(page); + cpu_addr = page_address(page); if ((attrs & __DMA_ATTR_ALLOC_CC_SHARED) && - dma_set_decrypted(dev, ret, size)) + dma_set_decrypted(dev, cpu_addr, size)) goto out_leak_pages; setup_page: - memset(ret, 0, size); + memset(cpu_addr, 0, size); *dma_handle = phys_to_dma_direct(dev, page_to_phys(page), attrs & __DMA_ATTR_ALLOC_CC_SHARED); return page; -- cgit v1.2.3 From e13d4d9a4915d9dbd0961594442704a4f26160dc Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:39 +0530 Subject: dma: swiotlb: free dynamic pools from process context swiotlb_dyn_free() is used after removing a dynamic swiotlb pool from RCU-protected lists. It can call swiotlb_free_tlb(), which may need to restore the encryption state of an unencrypted pool with set_memory_encrypted() before freeing the pages. RCU callbacks run in atomic context, but set_memory_encrypted() is not guaranteed to be atomic-safe on all architectures. For example, page attribute updates may allocate page tables or take sleeping locks. Use queue_rcu_work() for dynamic pool freeing instead. This keeps the RCU grace period before freeing a published pool, while running the actual pool teardown from workqueue context. Use the same helper for the transient-pool error path, since that path may also be reached from atomic DMA mapping context. Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-22-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- include/linux/swiotlb.h | 4 ++-- kernel/dma/swiotlb.c | 19 +++++++++++-------- 2 files changed, 13 insertions(+), 10 deletions(-) (limited to 'kernel') diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index ee42f7588847..c3bf7ed6f7a6 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -64,7 +64,7 @@ extern void __init swiotlb_update_mem_attributes(void); * @areas: Array of memory area descriptors. * @slots: Array of slot descriptors. * @node: Member of the IO TLB memory pool list. - * @rcu: RCU head for swiotlb_dyn_free(). + * @dyn_free: RCU work item used to free the pool from process context. * @transient: %true if transient memory pool. * @cc_shared: %true if the pool memory is shared for confidential computing. */ @@ -80,7 +80,7 @@ struct io_tlb_pool { struct io_tlb_slot *slots; #ifdef CONFIG_SWIOTLB_DYNAMIC struct list_head node; - struct rcu_head rcu; + struct rcu_work dyn_free; bool transient; bool cc_shared; #endif diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index b5960c2e98d8..4d0f2c04d891 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -779,13 +779,10 @@ static void swiotlb_dyn_alloc(struct work_struct *work) add_mem_pool(mem, pool); } -/** - * swiotlb_dyn_free() - RCU callback to free a memory pool - * @rcu: RCU head in the corresponding struct io_tlb_pool. - */ -static void swiotlb_dyn_free(struct rcu_head *rcu) +static void swiotlb_dyn_free_work(struct work_struct *work) { - struct io_tlb_pool *pool = container_of(rcu, struct io_tlb_pool, rcu); + struct io_tlb_pool *pool = + container_of(to_rcu_work(work), struct io_tlb_pool, dyn_free); size_t slots_size = array_size(sizeof(*pool->slots), pool->nslabs); size_t tlb_size = pool->end - pool->start; @@ -794,6 +791,12 @@ static void swiotlb_dyn_free(struct rcu_head *rcu) kfree(pool); } +static void swiotlb_schedule_dyn_free(struct io_tlb_pool *pool) +{ + INIT_RCU_WORK(&pool->dyn_free, swiotlb_dyn_free_work); + queue_rcu_work(system_wq, &pool->dyn_free); +} + /** * __swiotlb_find_pool() - find the IO TLB pool for a physical address * @dev: Device which has mapped the DMA buffer. @@ -840,7 +843,7 @@ static void swiotlb_del_pool(struct device *dev, struct io_tlb_pool *pool) list_del_rcu(&pool->node); spin_unlock_irqrestore(&dev->dma_io_tlb_lock, flags); - call_rcu(&pool->rcu, swiotlb_dyn_free); + swiotlb_schedule_dyn_free(pool); } #endif /* CONFIG_SWIOTLB_DYNAMIC */ @@ -1281,7 +1284,7 @@ static int swiotlb_find_slots(struct device *dev, phys_addr_t orig_addr, index = swiotlb_search_pool_area(dev, pool, 0, orig_addr, tbl_dma_addr, alloc_size, alloc_align_mask); if (index < 0) { - swiotlb_dyn_free(&pool->rcu); + swiotlb_schedule_dyn_free(pool); return -1; } -- cgit v1.2.3 From ad8d2326d2640020527fb632cd8e18651115dfcc Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:40 +0530 Subject: dma: swiotlb: handle set_memory_decrypted() failures Check the return value when converting swiotlb pools between encrypted and decrypted mappings. If the default pool cannot be decrypted after early initialization, mark the pool fully used so it cannot satisfy future bounce allocations. For late initialization, return the `set_memory_decrypted()` failure. For restricted DMA pools, fail device initialization if the reserved pool cannot be decrypted. This prevents swiotlb from using pools whose encryption attributes do not match their metadata, and avoids returning pages with uncertain encryption state back to the allocator. Reviewed-by: Jason Gunthorpe Tested-by: Michael Kelley Tested-by: Mostafa Saleh Reviewed-by: Petr Tesarik Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-23-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- kernel/dma/swiotlb.c | 80 ++++++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 65 insertions(+), 15 deletions(-) (limited to 'kernel') diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 4d0f2c04d891..8b7e47504304 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -248,6 +248,23 @@ static inline unsigned long nr_slots(u64 val) return DIV_ROUND_UP(val, IO_TLB_SIZE); } +static void swiotlb_mark_pool_used(struct io_tlb_pool *pool) +{ + unsigned long i; + + for (i = 0; i < pool->nareas; i++) { + pool->areas[i].index = 0; + pool->areas[i].used = pool->area_nslabs; + } + + for (i = 0; i < pool->nslabs; i++) { + pool->slots[i].list = 0; + pool->slots[i].orig_addr = INVALID_PHYS_ADDR; + pool->slots[i].alloc_size = 0; + pool->slots[i].pad_slots = 0; + } +} + /* * Early SWIOTLB allocation may be too early to allow an architecture to * perform the desired operations. This function allows the architecture to @@ -272,8 +289,16 @@ void __init swiotlb_update_mem_attributes(void) return; bytes = PAGE_ALIGN(mem->nslabs << IO_TLB_SHIFT); - if (io_tlb_default_mem.cc_shared) - set_memory_decrypted((unsigned long)mem->vaddr, bytes >> PAGE_SHIFT); + if (io_tlb_default_mem.cc_shared) { + int ret; + + ret = set_memory_decrypted((unsigned long)mem->vaddr, + bytes >> PAGE_SHIFT); + if (ret) { + pr_warn("Failed to decrypt default memory pool, disabling it\n"); + swiotlb_mark_pool_used(mem); + } + } } static void swiotlb_init_io_tlb_pool(struct io_tlb_pool *mem, phys_addr_t start, @@ -442,9 +467,10 @@ int swiotlb_init_late(size_t size, gfp_t gfp_mask, { struct io_tlb_pool *mem = &io_tlb_default_mem.defpool; unsigned long nslabs = ALIGN(size >> IO_TLB_SHIFT, IO_TLB_SEGSIZE); + unsigned int order, area_order, slot_order; + bool leak_pages = false; unsigned int nareas; unsigned char *vstart = NULL; - unsigned int order, area_order; bool retried = false; int rc = 0; @@ -504,6 +530,7 @@ retry: (PAGE_SIZE << order) >> 20); } + rc = -ENOMEM; nareas = limit_nareas(default_nareas, nslabs); area_order = get_order(array_size(sizeof(*mem->areas), nareas)); mem->areas = (struct io_tlb_area *) @@ -511,14 +538,20 @@ retry: if (!mem->areas) goto error_area; + slot_order = get_order(array_size(sizeof(*mem->slots), nslabs)); mem->slots = (void *)__get_free_pages(GFP_KERNEL | __GFP_ZERO, - get_order(array_size(sizeof(*mem->slots), nslabs))); + slot_order); if (!mem->slots) goto error_slots; - if (io_tlb_default_mem.cc_shared) - set_memory_decrypted((unsigned long)vstart, - (nslabs << IO_TLB_SHIFT) >> PAGE_SHIFT); + if (io_tlb_default_mem.cc_shared) { + rc = set_memory_decrypted((unsigned long)vstart, + (nslabs << IO_TLB_SHIFT) >> PAGE_SHIFT); + if (rc) { + leak_pages = true; + goto error_decrypt; + } + } swiotlb_init_io_tlb_pool(mem, virt_to_phys(vstart), vstart, nslabs, true, nareas); @@ -527,16 +560,20 @@ retry: swiotlb_print_info(); return 0; +error_decrypt: + free_pages((unsigned long)mem->slots, slot_order); error_slots: free_pages((unsigned long)mem->areas, area_order); error_area: - free_pages((unsigned long)vstart, order); - return -ENOMEM; + if (!leak_pages) + free_pages((unsigned long)vstart, order); + return rc; } void __init swiotlb_exit(void) { struct io_tlb_pool *mem = &io_tlb_default_mem.defpool; + bool leak_pages = false; unsigned long tbl_vaddr; size_t tbl_size, slots_size; unsigned int area_order; @@ -552,19 +589,23 @@ void __init swiotlb_exit(void) tbl_size = PAGE_ALIGN(mem->end - mem->start); slots_size = PAGE_ALIGN(array_size(sizeof(*mem->slots), mem->nslabs)); - if (io_tlb_default_mem.cc_shared) - set_memory_encrypted(tbl_vaddr, tbl_size >> PAGE_SHIFT); + if (io_tlb_default_mem.cc_shared) { + if (set_memory_encrypted(tbl_vaddr, tbl_size >> PAGE_SHIFT)) + leak_pages = true; + } if (mem->late_alloc) { area_order = get_order(array_size(sizeof(*mem->areas), mem->nareas)); free_pages((unsigned long)mem->areas, area_order); - free_pages(tbl_vaddr, get_order(tbl_size)); + if (!leak_pages) + free_pages(tbl_vaddr, get_order(tbl_size)); free_pages((unsigned long)mem->slots, get_order(slots_size)); } else { memblock_free(mem->areas, array_size(sizeof(*mem->areas), mem->nareas)); - memblock_phys_free(mem->start, tbl_size); + if (!leak_pages) + memblock_phys_free(mem->start, tbl_size); memblock_free(mem->slots, slots_size); } @@ -1957,9 +1998,18 @@ static int rmem_swiotlb_device_init(struct reserved_mem *rmem, * restricted mem pool is shared by default */ if (cc_platform_has(CC_ATTR_MEM_ENCRYPT)) { + int ret; + mem->cc_shared = true; - set_memory_decrypted((unsigned long)phys_to_virt(rmem->base), - rmem->size >> PAGE_SHIFT); + ret = set_memory_decrypted((unsigned long)phys_to_virt(rmem->base), + rmem->size >> PAGE_SHIFT); + if (ret) { + dev_err(dev, "Failed to decrypt restricted DMA pool\n"); + kfree(pool->areas); + kfree(pool->slots); + kfree(mem); + return ret; + } } else { mem->cc_shared = false; } -- cgit v1.2.3 From 04a19b35dc05354445f0096befad63af885bb77e Mon Sep 17 00:00:00 2001 From: "Aneesh Kumar K.V (Arm)" Date: Fri, 17 Jul 2026 23:34:41 +0530 Subject: swiotlb: remove unused SWIOTLB_FORCE flag SWIOTLB_FORCE has no remaining in-tree users. Forced bouncing is now controlled through the swiotlb=force command line option via swiotlb_force_bounce. Remove the unused flag and simplify the force_bounce initialization. Reviewed-by: Jason Gunthorpe Signed-off-by: Aneesh Kumar K.V (Arm) Link: https://lore.kernel.org/r/20260717180442.110954-24-aneesh.kumar@kernel.org Signed-off-by: Marek Szyprowski --- include/linux/swiotlb.h | 3 +-- kernel/dma/swiotlb.c | 3 +-- 2 files changed, 2 insertions(+), 4 deletions(-) (limited to 'kernel') diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index c3bf7ed6f7a6..9caca923c380 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -15,8 +15,7 @@ struct page; struct scatterlist; #define SWIOTLB_VERBOSE (1 << 0) /* verbose initialization */ -#define SWIOTLB_FORCE (1 << 1) /* force bounce buffering */ -#define SWIOTLB_ANY (1 << 2) /* allow any memory for the buffer */ +#define SWIOTLB_ANY (1 << 1) /* allow any memory for the buffer */ /* * Maximum allowable number of contiguous slabs to map, diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 8b7e47504304..897aba538c5b 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -400,8 +400,7 @@ void __init swiotlb_init_remap(bool addressing_limit, unsigned int flags, if (swiotlb_force_disable) return; - io_tlb_default_mem.force_bounce = - swiotlb_force_bounce || (flags & SWIOTLB_FORCE); + io_tlb_default_mem.force_bounce = swiotlb_force_bounce; #ifdef CONFIG_SWIOTLB_DYNAMIC if (!remap) -- cgit v1.2.3 From 121f9fd1c3083312055126205f3e40b8ee999fe6 Mon Sep 17 00:00:00 2001 From: chenhuguanshen Date: Wed, 12 Aug 2026 15:04:59 +0800 Subject: dma/swiotlb: decouple high watermark tracking from CONFIG_DEBUG_FS Under heavy concurrent DMA traffic on CoCo VMs, inc_used_and_hiwater() performs an atomic_long_add_return() plus a CAS loop on the global used_hiwater, and dec_used() performs an atomic_long_sub() on total_used. All CPUs contend on the same cacheline, causing measurable throughput degradation at scale. Historically these counters were only compiled in under CONFIG_DEBUG_FS, which means production kernels with debugfs paid the atomic overhead unconditionally. Make the tracking boot-time opt-in instead so that it is disabled by default with near-zero overhead via static_call, and can be enabled via "swiotlb=track_hiwater" parameter on demand for debugging. Note that when CONFIG_DEBUG_FS is enabled but hiwater tracking is disabled, the "io_tlb_used" metric reports an approximate value rather than an instantaneously exact one. Suggested-by: Fan Du Signed-off-by: Jun Miao Co-developed-by: Fan Du Signed-off-by: Fan Du Tested-by: chenhuguanshen Signed-off-by: chenhuguanshen Reviewed-by: Michael Kelley Tested-by: Michael Kelley Link: https://lore.kernel.org/r/20260812070459.637077-1-frankchen158@126.com Signed-off-by: Marek Szyprowski --- Documentation/admin-guide/kernel-parameters.txt | 4 +- include/linux/swiotlb.h | 8 +- kernel/dma/swiotlb.c | 153 +++++++++++++++--------- 3 files changed, 101 insertions(+), 64 deletions(-) (limited to 'kernel') diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index b5493a7f8f22..eb3e8db2c178 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -7477,7 +7477,7 @@ Kernel parameters Execution Facility on pSeries. swiotlb= [ARM,PPC,MIPS,X86,S390,EARLY] - Format: { [,] | force | noforce } + Format: { [,] | force | noforce | track_hiwater} -- Number of I/O TLB slabs -- Second integer after comma. Number of swiotlb areas with their own lock. Will be rounded up @@ -7485,6 +7485,8 @@ Kernel parameters force -- force using of bounce buffers even if they wouldn't be automatically used by the kernel noforce -- Never use bounce buffers (for debugging) + track_hiwater -- Track high watermark of swiotlb buffers. + Only available when CONFIG_DEBUG_FS is set. switches= [HW,M68k,EARLY] diff --git a/include/linux/swiotlb.h b/include/linux/swiotlb.h index 1665a9ce8f94..318bb1beb07f 100644 --- a/include/linux/swiotlb.h +++ b/include/linux/swiotlb.h @@ -102,10 +102,10 @@ struct io_tlb_pool { * @pools: List of IO TLB memory pool descriptors (if dynamic). * @dyn_alloc: Dynamic IO TLB pool allocation work. * @total_used: The total number of slots in the pool that are currently used - * across all areas. Used only for calculating used_hiwater in - * debugfs. - * @used_hiwater: The high water mark for total_used. Used only for reporting - * in debugfs. + * across all areas. Used only for calculating used_hiwater via boot + * parameter swiotlb=track_hiwater and exposed via debugfs. + * @used_hiwater: The high water mark for total_used. Can be enabled at boot + * time via swiotlb=track_hiwater and exposed via debugfs. * @transient_nslabs: The total number of slots in all transient pools that * are currently used across all areas. */ diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index 1abd3e6146f4..e0bc7ca4a7ad 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -180,6 +180,74 @@ static unsigned int limit_nareas(unsigned int nareas, unsigned long nslots) return nareas; } +#ifdef CONFIG_DEBUG_FS +/* + * Track the total used slots with a global atomic value in order to have + * correct information to determine the high water mark. + */ +static void inc_used_and_hiwater_real(struct io_tlb_mem *mem, + unsigned int nslots) +{ + unsigned long old_hiwater, new_used; + + new_used = atomic_long_add_return(nslots, &mem->total_used); + old_hiwater = atomic_long_read(&mem->used_hiwater); + do { + if (new_used <= old_hiwater) + break; + } while (!atomic_long_try_cmpxchg(&mem->used_hiwater, + &old_hiwater, new_used)); +} + +static void dec_used_real(struct io_tlb_mem *mem, unsigned int nslots) +{ + atomic_long_sub(nslots, &mem->total_used); +} + +static void inc_used_and_hiwater_nop(struct io_tlb_mem *mem, + unsigned int nslots) +{ +} +static void dec_used_nop(struct io_tlb_mem *mem, unsigned int nslots) +{ +} + +DEFINE_STATIC_CALL(swiotlb_inc_used, inc_used_and_hiwater_nop); +DEFINE_STATIC_CALL(swiotlb_dec_used, dec_used_nop); + +static __always_inline void inc_used_and_hiwater(struct io_tlb_mem *mem, + unsigned int nslots) +{ + static_call(swiotlb_inc_used)(mem, nslots); +} + +static __always_inline void dec_used(struct io_tlb_mem *mem, + unsigned int nslots) +{ + static_call(swiotlb_dec_used)(mem, nslots); +} + +static bool track_hiwater_enabled __read_mostly; + +#else + +static __always_inline void inc_used_and_hiwater(struct io_tlb_mem *mem, + unsigned int nslots) +{ +} + +static __always_inline void dec_used(struct io_tlb_mem *mem, + unsigned int nslots) +{ +} +#endif + +/* + * The tracking of used slots high watermark can be enabled + * by appending "track_hiwater" to the swiotlb= boot parameter. + * When disabled the tracking functions are no-ops with near-zero + * overhead via static_call. + */ static int __init setup_io_tlb_npages(char *str) { @@ -194,10 +262,24 @@ setup_io_tlb_npages(char *str) swiotlb_adjust_nareas(simple_strtoul(str, &str, 0)); if (*str == ',') ++str; - if (!strcmp(str, "force")) + if (!strncmp(str, "force", 5)) { swiotlb_force_bounce = true; - else if (!strcmp(str, "noforce")) + str += 5; + } else if (!strncmp(str, "noforce", 7)) { swiotlb_force_disable = true; + str += 7; + } + +#ifdef CONFIG_DEBUG_FS + if (*str == ',') + ++str; + if (!strncmp(str, "track_hiwater", 13)) { + track_hiwater_enabled = true; + static_call_update(swiotlb_inc_used, + inc_used_and_hiwater_real); + static_call_update(swiotlb_dec_used, dec_used_real); + } +#endif return 0; } @@ -959,40 +1041,6 @@ static unsigned int wrap_area_index(struct io_tlb_pool *mem, unsigned int index) return index; } -/* - * Track the total used slots with a global atomic value in order to have - * correct information to determine the high water mark. The mem_used() - * function gives imprecise results because there's no locking across - * multiple areas. - */ -#ifdef CONFIG_DEBUG_FS -static void inc_used_and_hiwater(struct io_tlb_mem *mem, unsigned int nslots) -{ - unsigned long old_hiwater, new_used; - - new_used = atomic_long_add_return(nslots, &mem->total_used); - old_hiwater = atomic_long_read(&mem->used_hiwater); - do { - if (new_used <= old_hiwater) - break; - } while (!atomic_long_try_cmpxchg(&mem->used_hiwater, - &old_hiwater, new_used)); -} - -static void dec_used(struct io_tlb_mem *mem, unsigned int nslots) -{ - atomic_long_sub(nslots, &mem->total_used); -} - -#else /* !CONFIG_DEBUG_FS */ -static void inc_used_and_hiwater(struct io_tlb_mem *mem, unsigned int nslots) -{ -} -static void dec_used(struct io_tlb_mem *mem, unsigned int nslots) -{ -} -#endif /* CONFIG_DEBUG_FS */ - #ifdef CONFIG_SWIOTLB_DYNAMIC #ifdef CONFIG_DEBUG_FS static void inc_transient_used(struct io_tlb_mem *mem, unsigned int nslots) @@ -1295,24 +1343,6 @@ static int swiotlb_find_slots(struct device *dev, phys_addr_t orig_addr, #endif /* CONFIG_SWIOTLB_DYNAMIC */ -#ifdef CONFIG_DEBUG_FS - -/** - * mem_used() - get number of used slots in an allocator - * @mem: Software IO TLB allocator. - * - * The result is accurate in this version of the function, because an atomic - * counter is available if CONFIG_DEBUG_FS is set. - * - * Return: Number of used slots. - */ -static unsigned long mem_used(struct io_tlb_mem *mem) -{ - return atomic_long_read(&mem->total_used); -} - -#else /* !CONFIG_DEBUG_FS */ - /** * mem_pool_used() - get number of used slots in a memory pool * @pool: Software IO TLB memory pool. @@ -1335,13 +1365,20 @@ static unsigned long mem_pool_used(struct io_tlb_pool *pool) * mem_used() - get number of used slots in an allocator * @mem: Software IO TLB allocator. * - * The result is not accurate, because there is no locking of individual - * areas. + * When trace_hiwater and CONFIG_DEBUG_FS is enabled, the result is accurate + * because the total number of used slots is tracked in mem->total_used. + * Otherwise, the result is an approximation, because there is no locking of + * individual areas. * - * Return: Approximate number of used slots. + * Return: Number of used slots. */ static unsigned long mem_used(struct io_tlb_mem *mem) { +#ifdef CONFIG_DEBUG_FS + if (track_hiwater_enabled) + return atomic_long_read(&mem->total_used); +#endif + #ifdef CONFIG_SWIOTLB_DYNAMIC struct io_tlb_pool *pool; unsigned long used = 0; @@ -1357,8 +1394,6 @@ static unsigned long mem_used(struct io_tlb_mem *mem) #endif } -#endif /* CONFIG_DEBUG_FS */ - /** * swiotlb_tbl_map_single() - bounce buffer map a single contiguous physical area * @dev: Device which maps the buffer. -- cgit v1.2.3