summaryrefslogtreecommitdiff
path: root/include/linux
diff options
context:
space:
mode:
authorDave Airlie <airlied@redhat.com>2026-09-08 16:51:58 +1000
committerDave Airlie <airlied@redhat.com>2026-09-08 21:05:35 +1000
commit483ece619b64a6596bdaf69479ea0ecbbd197a00 (patch)
tree44af1491a66bbce99c99396405552afb77f4512f /include/linux
parent94376186f25bfb86842c9b7551d223415cc6e4f0 (diff)
parent99c95ce1b07081d7944d637ba7d72d835c0d520a (diff)
downloadlinux-next-483ece619b64a6596bdaf69479ea0ecbbd197a00.tar.gz
linux-next-483ece619b64a6596bdaf69479ea0ecbbd197a00.zip
Merge tag 'drm-misc-next-2026-09-03' of https://gitlab.freedesktop.org/drm/misc/kernel into drm-next
drm-misc-next for v7.4: UAPI Changes: colorop: - provide DRM_COLOR_OP_FIXED_MATRIX Cross-subsystem Changes: cgroups: - fix typos dma-buf: - fix typos sound: - adapt to changes in omapdrm Core Changes: atomic: - convert most of DRM from state reset callbacks to atomic_create_state - remove drm_simple_encoder_init(); update drivers buddy: - improve dirty-page tracking clients: - log: Improve vmap handling display: - export HDMI SCDC status data via debugfs edid: - parse AMD VSDB entries - parse ALLM/VRR capabilities log: - add drm_warn_ratelimited() sched: - add missing locking Driver Changes: amd: - display: Use AMD VSDB for FreeSync - display: Implement YUV-to-RGB with fixed-matrix colorop amdxdma: - various fixes ast: - support 256-byte EDID data bridge: - clean up redundant error reporting - ti-sn65dsi83: Simplify error condition logic hyperv_drm: - remove support for pre-Win10 hosts komeda: - fix usage of GLB_CORE_ID nouveau: - improve runtime PM on R570 GSP firmware - various fixes throughout the driver - dispnv50: Support 2.147 GHz pixel clock in GB20x omap: - report HDMI hotplug events to ASoC HDMI codec panel: - panel-edp: Support MNE007QS3-F, TM140VDXP15, and KD116N36-30NB-A001 - samsung-s6d16d0: Use mipi_dsi_*_multi() functions - support Ilitek ILI7836A OLED plus DT bindings - support Novatek NT36532 plus DT bindings - convert several drivers to managed cleanup - fix Kconfig selections panthor: - provide gpu_cache_flush tracepoint - improve dma_fence signalling latency - improve locking qaic: - reject BOs that exceed maximum page count - add missing include statements verisilicon: - fix hardware cursor offsets vkms: - implement YUV-to-RGB with fixed-matrix colorop Signed-off-by: Dave Airlie <airlied@redhat.com> From: Thomas Zimmermann <tzimmermann@suse.de> Link: https://patch.msgid.link/20260903130548.GA91506@2a02-2455-9062-2500-3419-2212-e55c-8a45.dyn6.pyur.net
Diffstat (limited to 'include/linux')
-rw-r--r--include/linux/gpu_buddy.h101
1 files changed, 78 insertions, 23 deletions
diff --git a/include/linux/gpu_buddy.h b/include/linux/gpu_buddy.h
index 2c36124bb696..ddc4eca84175 100644
--- a/include/linux/gpu_buddy.h
+++ b/include/linux/gpu_buddy.h
@@ -43,8 +43,8 @@
/**
* GPU_BUDDY_CLEAR_ALLOCATION - Prefer pre-cleared (zeroed) memory
*
- * Attempt to allocate from the clear tree first. If insufficient clear
- * memory is available, falls back to dirty memory. Useful when the
+ * Attempt to allocate outside dirty-tracked ranges first. If insufficient
+ * clear memory is available, falls back to dirty memory. Useful when the
* caller needs zeroed memory and wants to avoid GPU clear operations.
*/
#define GPU_BUDDY_CLEAR_ALLOCATION BIT(3)
@@ -53,8 +53,8 @@
* GPU_BUDDY_CLEARED - Mark returned blocks as cleared
*
* Used with gpu_buddy_free_list() to indicate that the memory being
- * freed has been cleared (zeroed). The blocks will be placed in the
- * clear tree for future GPU_BUDDY_CLEAR_ALLOCATION requests.
+ * freed has been cleared (zeroed). The blocks will be removed from the
+ * dirty tracker for future GPU_BUDDY_CLEAR_ALLOCATION requests.
*/
#define GPU_BUDDY_CLEARED BIT(4)
@@ -67,15 +67,17 @@
*/
#define GPU_BUDDY_TRIM_DISABLE BIT(5)
-enum gpu_buddy_free_tree {
- GPU_BUDDY_CLEAR_TREE = 0,
- GPU_BUDDY_DIRTY_TREE,
- GPU_BUDDY_MAX_FREE_TREES,
+/*
+ * Clear/dirty state of a free block. Ordered so a numerically larger value
+ * is "more clear" (DIRTY < MIXED < CLEAR) which lets subtree_block_state be
+ * maintained as a simple max-augment over the per-order free tree.
+ */
+enum gpu_block_state {
+ GPU_BLOCK_DIRTY = 0,
+ GPU_BLOCK_MIXED = 1,
+ GPU_BLOCK_CLEAR = 2,
};
-#define for_each_free_tree(tree) \
- for ((tree) = 0; (tree) < GPU_BUDDY_MAX_FREE_TREES; (tree)++)
-
/**
* struct gpu_buddy_block - Block within a buddy allocator
*
@@ -103,6 +105,13 @@ struct gpu_buddy_block {
#define GPU_BUDDY_ALLOCATED (1 << 10)
#define GPU_BUDDY_FREE (2 << 10)
#define GPU_BUDDY_SPLIT (3 << 10)
+/*
+ * GPU_BUDDY_HEADER_CLEAR has two roles:
+ * - FREE state: set when the block's full range is cleared (dirty
+ * tracker confirmed no overlap).
+ * - ALLOCATED state: set when the block was served from cleared memory,
+ * informing the caller that no GPU clear pass is needed.
+ */
#define GPU_BUDDY_HEADER_CLEAR GENMASK_ULL(9, 9)
/* Free to be used, if needed in the future */
#define GPU_BUDDY_HEADER_UNUSED GENMASK_ULL(8, 6)
@@ -128,14 +137,45 @@ struct gpu_buddy_block {
struct list_head link;
};
/* private: */
- struct list_head tmp_link;
+ enum gpu_block_state subtree_block_state;
unsigned int subtree_max_alignment;
+ struct list_head tmp_link;
+ bool has_clear;
};
/* Order-zero must be at least SZ_4K */
#define GPU_BUDDY_MAX_ORDER (63 - 12)
/**
+ * struct gpu_dirty_extent - a contiguous dirty address range
+ *
+ * Tracks a single contiguous address range whose memory content is known
+ * to be dirty. Extents are non-overlapping and stored in an augmented
+ * red-black tree sorted by @start. The augmented value @subtree_max_size
+ * allows O(log N) search for an extent of at least a given size.
+ */
+struct gpu_dirty_extent {
+/* private: */
+ struct rb_node rb;
+ u64 start;
+ u64 end;
+ u64 subtree_max_size;
+};
+
+/**
+ * struct gpu_dirty_tracker - tracks dirty address intervals
+ *
+ * Maintains a set of non-overlapping dirty extents as an augmented
+ * red-black tree.
+ */
+struct gpu_dirty_tracker {
+/* private: */
+ struct rb_root root;
+ /* Total bytes of dirty memory currently tracked. */
+ u64 total_dirty;
+};
+
+/**
* struct gpu_buddy - GPU binary buddy allocator
*
* The buddy allocator provides efficient power-of-two memory allocation
@@ -152,20 +192,21 @@ struct gpu_buddy_block {
* @chunk_size: Minimum allocation granularity in bytes. Must be at least SZ_4K.
* @size: Total size of the address space managed by this allocator in bytes.
* @avail: Total free space currently available for allocation in bytes.
- * @clear_avail: Free space available in the clear tree (zeroed memory) in bytes.
- * This is a subset of @avail.
* @lock_dep_map: Annotates gpu_buddy API with a driver provided lock.
*/
struct gpu_buddy {
/* private: */
+ /* Tracker of dirty address ranges (decoupled from free_tree). */
+ struct gpu_dirty_tracker dirty;
/*
- * Array of red-black trees for free block management.
- * Indexed as free_trees[clear/dirty][order] where:
- * - Index 0 (GPU_BUDDY_CLEAR_TREE): blocks with zeroed content
- * - Index 1 (GPU_BUDDY_DIRTY_TREE): blocks with unknown content
- * Each tree holds free blocks of the corresponding order.
+ * One RB-tree per order containing all free blocks (clear and
+ * dirty alike). The augment field subtree_block_state (a max over
+ * the subtree of each block's state) lets clear allocations
+ * find the right-most fully-clear or mixed block in O(log N).
+ * Dirty free blocks coexist here but are also indexed by the
+ * @dirty tracker for fast dirty allocation lookups.
*/
- struct rb_root **free_trees;
+ struct rb_root *free_tree;
/*
* Array of root blocks representing the top-level blocks of the
* binary tree(s). Multiple roots exist when the total size is not
@@ -194,7 +235,6 @@ struct gpu_buddy {
u64 chunk_size;
u64 size;
u64 avail;
- u64 clear_avail;
#ifdef CONFIG_LOCKDEP
struct lockdep_map *lock_dep_map;
#endif
@@ -226,17 +266,32 @@ struct gpu_buddy {
*
* Ensure driver lock is held.
*/
-static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm)
+static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm)
{
if (mm->lock_dep_map)
lockdep_assert(lock_is_held_type(mm->lock_dep_map, 0));
}
#else
-static inline void gpu_buddy_driver_lock_held(struct gpu_buddy *mm)
+static inline void gpu_buddy_driver_lock_held(const struct gpu_buddy *mm)
{
}
#endif
+/**
+ * gpu_buddy_clear_avail - free space that is clear (zeroed), in bytes
+ * @mm: gpu buddy allocator
+ *
+ * A subset of @mm->avail. Derived on demand as @mm->avail minus the bytes
+ * the dirty tracker records as dirty, so it is always consistent with the
+ * tracker without a cached field to keep in sync. Zero for a fresh pool,
+ * which is fully dirty.
+ */
+static inline u64 gpu_buddy_clear_avail(const struct gpu_buddy *mm)
+{
+ gpu_buddy_driver_lock_held(mm);
+ return mm->avail - mm->dirty.total_dirty;
+}
+
static inline u64
gpu_buddy_block_offset(const struct gpu_buddy_block *block)
{