diff options
| author | Dave Airlie <airlied@redhat.com> | 2026-09-08 16:51:58 +1000 |
|---|---|---|
| committer | Dave Airlie <airlied@redhat.com> | 2026-09-08 21:05:35 +1000 |
| commit | 483ece619b64a6596bdaf69479ea0ecbbd197a00 (patch) | |
| tree | 44af1491a66bbce99c99396405552afb77f4512f /include/linux | |
| parent | 94376186f25bfb86842c9b7551d223415cc6e4f0 (diff) | |
| parent | 99c95ce1b07081d7944d637ba7d72d835c0d520a (diff) | |
| download | linux-next-483ece619b64a6596bdaf69479ea0ecbbd197a00.tar.gz linux-next-483ece619b64a6596bdaf69479ea0ecbbd197a00.zip | |
Merge tag 'drm-misc-next-2026-09-03' of https://gitlab.freedesktop.org/drm/misc/kernel into drm-next
drm-misc-next for v7.4:
UAPI Changes:
colorop:
- provide DRM_COLOR_OP_FIXED_MATRIX
Cross-subsystem Changes:
cgroups:
- fix typos
dma-buf:
- fix typos
sound:
- adapt to changes in omapdrm
Core Changes:
atomic:
- convert most of DRM from state reset callbacks to atomic_create_state
- remove drm_simple_encoder_init(); update drivers
buddy:
- improve dirty-page tracking
clients:
- log: Improve vmap handling
display:
- export HDMI SCDC status data via debugfs
edid:
- parse AMD VSDB entries
- parse ALLM/VRR capabilities
log:
- add drm_warn_ratelimited()
sched:
- add missing locking
Driver Changes:
amd:
- display: Use AMD VSDB for FreeSync
- display: Implement YUV-to-RGB with fixed-matrix colorop
amdxdma:
- various fixes
ast:
- support 256-byte EDID data
bridge:
- clean up redundant error reporting
- ti-sn65dsi83: Simplify error condition logic
hyperv_drm:
- remove support for pre-Win10 hosts
komeda:
- fix usage of GLB_CORE_ID
nouveau:
- improve runtime PM on R570 GSP firmware
- various fixes throughout the driver
- dispnv50: Support 2.147 GHz pixel clock in GB20x
omap:
- report HDMI hotplug events to ASoC HDMI codec
panel:
- panel-edp: Support MNE007QS3-F, TM140VDXP15, and KD116N36-30NB-A001
- samsung-s6d16d0: Use mipi_dsi_*_multi() functions
- support Ilitek ILI7836A OLED plus DT bindings
- support Novatek NT36532 plus DT bindings
- convert several drivers to managed cleanup
- fix Kconfig selections
panthor:
- provide gpu_cache_flush tracepoint
- improve dma_fence signalling latency
- improve locking
qaic:
- reject BOs that exceed maximum page count
- add missing include statements
verisilicon:
- fix hardware cursor offsets
vkms:
- implement YUV-to-RGB with fixed-matrix colorop
Signed-off-by: Dave Airlie <airlied@redhat.com>
From: Thomas Zimmermann <tzimmermann@suse.de>
Link: https://patch.msgid.link/20260903130548.GA91506@2a02-2455-9062-2500-3419-2212-e55c-8a45.dyn6.pyur.net
Diffstat (limited to 'include/linux')
| -rw-r--r-- | include/linux/gpu_buddy.h | 101 |
1 files changed, 78 insertions, 23 deletions
diff --git a/include/linux/gpu_buddy.h b/include/linux/gpu_buddy.h index 2c36124bb696..ddc4eca84175 100644 --- a/include/linux/gpu_buddy.h +++ b/include/linux/gpu_buddy.h @@ -43,8 +43,8 @@ /** * GPU_BUDDY_CLEAR_ALLOCATION - Prefer pre-cleared (zeroed) memory * - * Attempt to allocate from the clear tree first. If insufficient clear - * memory is available, falls back to dirty memory. Useful when the + * Attempt to allocate outside dirty-tracked ranges first. If insufficient + * clear memory is available, falls back to dirty memory. Useful when the * caller needs zeroed memory and wants to avoid GPU clear operations. */ #define GPU_BUDDY_CLEAR_ALLOCATION BIT(3) @@ -53,8 +53,8 @@ * GPU_BUDDY_CLEARED - Mark returned blocks as cleared * * Used with gpu_buddy_free_list() to indicate that the memory being - * freed has been cleared (zeroed). The blocks will be placed in the - * clear tree for future GPU_BUDDY_CLEAR_ALLOCATION requests. + * freed has been cleared (zeroed). The blocks will be removed from the + * dirty tracker for future GPU_BUDDY_CLEAR_ALLOCATION requests. */ #define GPU_BUDDY_CLEARED BIT(4) @@ -67,15 +67,17 @@ */ #define GPU_BUDDY_TRIM_DISABLE BIT(5) -enum gpu_buddy_free_tree { - GPU_BUDDY_CLEAR_TREE = 0, - GPU_BUDDY_DIRTY_TREE, - GPU_BUDDY_MAX_FREE_TREES, +/* + * Clear/dirty state of a free block. Ordered so a numerically larger value + * is "more clear" (DIRTY < MIXED < CLEAR) which lets subtree_block_state be + * maintained as a simple max-augment over the per-order free tree. + */ +enum gpu_block_state { + GPU_BLOCK_DIRTY = 0, + GPU_BLOCK_MIXED = 1, + GPU_BLOCK_CLEAR = 2, }; -#define for_each_free_tree(tree) \ - for ((tree) = 0; (tree) < GPU_BUDDY_MAX_FREE_TREES; (tree)++) - /** * struct gpu_buddy_block - Block within a buddy allocator * @@ -103,6 +105,13 @@ struct gpu_buddy_block { #define GPU_BUDDY_ALLOCATED (1 << 10) #define GPU_BUDDY_FREE (2 << 10) #define GPU_BUDDY_SPLIT (3 << 10) +/* + * GPU_BUDDY_HEADER_CLEAR has two roles: + * - FREE state: set when the block's full range is cleared (dirty + * tracker confirmed no overlap). + * - ALLOCATED state: set when the block was served from cleared memory, + * informing the caller that no GPU clear pass is needed. + */ #define GPU_BUDDY_HEADER_CLEAR GENMASK_ULL(9, 9) /* Free to be used, if needed in the future */ #define GPU_BUDDY_HEADER_UNUSED GENMASK_ULL(8, 6) @@ -128,14 +137,45 @@ struct gpu_buddy_block { struct list_head link; }; /* private: */ - struct list_head tmp_link; + enum gpu_block_state subtree_block_state; unsigned int subtree_max_alignment; + struct list_head tmp_link; + bool has_clear; }; /* Order-zero must be at least SZ_4K */ #define GPU_BUDDY_MAX_ORDER (63 - 12) /** + * struct gpu_dirty_extent - a contiguous dirty address range + * + * Tracks a single contiguous address range whose memory content is known + * to be dirty. Extents are non-overlapping and stored in an augmented + * red-black tree sorted by @start. The augmented value @subtree_max_size + * allows O(log N) search for an extent of at least a given size. + */ +struct gpu_dirty_extent { +/* private: */ + struct rb_node rb; + u64 start; + u64 end; + u64 subtree_max_size; +}; + +/** + * struct gpu_dirty_tracker - tracks dirty address intervals + * + * Maintains a set of non-overlapping dirty extents as an augmented + * red-black tree. + */ +struct gpu_dirty_tracker { +/* private: */ + struct rb_root root; + /* Total bytes of dirty memory currently tracked. */ + u64 total_dirty; +}; + +/** * struct gpu_buddy - GPU binary buddy allocator * * The buddy allocator provides efficient power-of-two memory allocation @@ -152,20 +192,21 @@ struct gpu_buddy_block { * @chunk_size: Minimum allocation granularity in bytes. Must be at least SZ_4K. * @size: Total size of the address space managed by this allocator in bytes. * @avail: Total free space currently available for allocation in bytes. - * @clear_avail: Free space available in the clear tree (zeroed memory) in bytes. - * This is a subset of @avail. * @lock_dep_map: Annotates gpu_buddy API with a driver provided lock. */ struct gpu_buddy { /* private: */ + /* Tracker of dirty address ranges (decoupled from free_tree). */ + struct gpu_dirty_tracker dirty; /* - * Array of red-black trees for free block management. - * Indexed as free_trees[clear/dirty][order] where: - * - Index 0 (GPU_BUDDY_CLEAR_TREE): blocks with zeroed content - * - Index 1 (GPU_BUDDY_DIRTY_TREE): blocks with unknown content - * Each tree holds free blocks of the corresponding order. + * One RB-tree per order containing all free blocks (clear and + * dirty alike). The augment field subtree_block_state (a max over + * the subtree of each block's state) lets clear allocations + * find the right-most fully-clear or mixed block in O(log N). + * Dirty free blocks coexist here but are also indexed by the + * @dirty tracker for fast dirty allocation lookups. */ - struct rb_root **free_trees; + struct rb_root *free_tree; /* * Array of root blocks representing the top-level blocks of the * binary tree(s). Multiple roots exist when the total size is not @@ -194,7 +235,6 @@ struct gpu_buddy { u64 chunk_size; u64 size; u64 avail; - u64 clear_avail; #ifdef CONFIG_LOCKDEP struct lockdep_map *lock_dep_map; #endif @@ -226,17 +266,32 @@ struct gpu_buddy { * * Ensure driver lock is held. */ -static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm) +static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm) { if (mm->lock_dep_map) lockdep_assert(lock_is_held_type(mm->lock_dep_map, 0)); } #else -static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm) +static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm) { } #endif +/** + * gpu_buddy_clear_avail - free space that is clear (zeroed), in bytes + * @mm: gpu buddy allocator + * + * A subset of @mm->avail. Derived on demand as @mm->avail minus the bytes + * the dirty tracker records as dirty, so it is always consistent with the + * tracker without a cached field to keep in sync. Zero for a fresh pool, + * which is fully dirty. + */ +static inline u64 gpu_buddy_clear_avail(const struct gpu_buddy *mm) +{ + gpu_buddy_driver_lock_held(mm); + return mm->avail - mm->dirty.total_dirty; +} + static inline u64 gpu_buddy_block_offset(const struct gpu_buddy_block *block) { |
