From 9a8e740ee9f69ec857ffa0c76cf1a01d1cb7360f Mon Sep 17 00:00:00 2001 From: Shrikanth Hegde Date: Mon, 28 Sep 2026 11:07:25 +0530 Subject: virt: Introduce steal governor driver Introduce a new driver in virt named steal_governor. This driver will compute the steal time and drive the policy decisions regarding the preferred CPU state. Note that this driver is strictly intended for actual guests. Hence block it on Xen dom0. More details can be found in Documentation/driver-api/steal-governor.rst. A new kconfig called STEAL_GOVERNOR is introduced in subsequent patches, which enables this driver. This driver will select CONFIG_PREFERRED_CPU. This makes configs driven by user preference/configuration. When the driver is disabled, preferred CPUs remain the same as active CPUs. The file layout of the driver is kept simple for now. The code is in drivers/virt/steal_governor.c, and the configs are part of drivers/virt/Kconfig. The main structure of the steal governor contains: - work, delay: Deferred periodic work function variables. - steal, time: Used to calculate deltas during periodic work. - interval_ms, high_threshold, low_threshold: Tuning knobs for the steal governor. While there, add MAINTAINERS entry for this new driver. Suggested-by: Yury Norov Suggested-by: K Prateek Nayak Signed-off-by: Shrikanth Hegde Signed-off-by: Peter Zijlstra (Intel) Link: https://patch.msgid.link/20260928053728.797539-11-sshegde@linux.ibm.com --- drivers/virt/steal_governor.c | 78 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 78 insertions(+) create mode 100644 drivers/virt/steal_governor.c (limited to 'drivers/virt') diff --git a/drivers/virt/steal_governor.c b/drivers/virt/steal_governor.c new file mode 100644 index 000000000000..2320cbfa4b47 --- /dev/null +++ b/drivers/virt/steal_governor.c @@ -0,0 +1,78 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Steal time governor driver periodically computes steal time. + * Based on the thresholds it either reduce/increase the preferred + * CPUs which can be used by the workload to avoid vCPU preemption + * to an extent possible in paravirtualized environment. + * + * Available with CONFIG_STEAL_GOVERNOR + * + * Copyright (C) 2026 IBM + * Author: Shrikanth Hegde + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#ifdef CONFIG_XEN +#include +#endif + +#if !IS_ENABLED(CONFIG_PREFERRED_CPU) +#error "Steal Governor requires CONFIG_PREFERRED_CPU" +#endif + +struct steal_governor { + ktime_t time; + u64 steal; + unsigned long delay; + unsigned int interval_ms; + unsigned int high_threshold; + unsigned int low_threshold; + struct delayed_work work; +}; + +static struct steal_governor sg_ctx; + +static void restore_preferred_to_active(void) +{ + int cpu; + + guard(cpus_read_lock)(); + for_each_cpu(cpu, cpu_active_mask) + set_cpu_preferred(cpu, true); +} + +static int __init steal_governor_init(void) +{ +#ifdef CONFIG_XEN + if (xen_initial_domain()) { + pr_err("Cannot load in Xen Dom0 (Host OS). Driver is for guests only.\n"); + return -ENODEV; + } +#endif + + pr_info("enabled\n"); + return 0; +} + +static void __exit steal_governor_exit(void) +{ + restore_preferred_to_active(); + pr_info("disabled\n"); +} + +module_init(steal_governor_init); +module_exit(steal_governor_exit); + +MODULE_LICENSE("GPL"); +MODULE_AUTHOR("IBM Corporation"); +MODULE_DESCRIPTION("Virtualization Steal Time Governor"); -- cgit v1.2.3 From 4b9302d494fffee7318bd9cdd2021198b6a7badd Mon Sep 17 00:00:00 2001 From: Shrikanth Hegde Date: Mon, 28 Sep 2026 11:07:26 +0530 Subject: virt/steal_governor: Add control knobs for handling steal values These are the knobs to control the steal_governor. interval_ms: How often steal governor checks for steal time. (Default: 1000 i.e 1 second) This controls how fast steal governor driver reacts to changes to the contention of physical CPUs. Can be set between 100 to 100000. i.e. 100ms to 100seconds. 100ms is kept as minimum to ensure few meaningful steal values accumulate even with HZ=100. low_threshold: lower threshold value in percentage * 100. (Default: 200, i.e 2% steal is considered as low threshold) This determines what values should be considered as nil/no steal values. When steal governor see steal ratio is below or equal to this value, it will increase the preferred CPUs by 1 core. Having value as zero might cause oscillations high_threshold: higher threshold value in percentage * 100 (Default: 500, i.e 5% steal is considered as high threshold) This determines what values should be considered as high steal values. When steal governor sees steal ratio is higher than this value, it will reduce the preferred CPUs by 1 core. module_param_cb methods are used to do the validation checks. This helps to ensure one configures sane values. Since low and high are dependent, that check is done at module init. Notes: - Parameters values can't be changed at runtime. One has to unload the module and change it. Hence recommended to build it as module. - Default values may not work well for all configurations. Tune it according to the system under test. Documentation is available at: Documentation/driver-api/steal-governor.rst Suggested-by: Yury Norov Signed-off-by: Shrikanth Hegde Signed-off-by: Peter Zijlstra (Intel) Link: https://patch.msgid.link/20260928053728.797539-12-sshegde@linux.ibm.com --- drivers/virt/steal_governor.c | 73 +++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 71 insertions(+), 2 deletions(-) (limited to 'drivers/virt') diff --git a/drivers/virt/steal_governor.c b/drivers/virt/steal_governor.c index 2320cbfa4b47..27f53ea16498 100644 --- a/drivers/virt/steal_governor.c +++ b/drivers/virt/steal_governor.c @@ -40,7 +40,11 @@ struct steal_governor { struct delayed_work work; }; -static struct steal_governor sg_ctx; +static struct steal_governor sg_ctx = { + .interval_ms = 1000, /* 1 second */ + .high_threshold = 500, /* 5% */ + .low_threshold = 200, /* 2% */ +}; static void restore_preferred_to_active(void) { @@ -51,6 +55,62 @@ static void restore_preferred_to_active(void) set_cpu_preferred(cpu, true); } +static int param_set_interval_ms(const char *val, const struct kernel_param *kp) +{ + unsigned int interval; + int ret; + + ret = kstrtouint(val, 0, &interval); + if (ret) + return ret; + + if (interval < 100 || interval > 100000) { + pr_err("interval_ms must be between 100 and 100000\n"); + return -EINVAL; + } + + return param_set_uint(val, kp); +} + +static const struct kernel_param_ops interval_ms_ops = { + .set = param_set_interval_ms, + .get = param_get_uint, +}; + +module_param_cb(interval_ms, &interval_ms_ops, &sg_ctx.interval_ms, 0444); +MODULE_PARM_DESC(interval_ms, + "Sampling frequency in milliseconds. default: 1000"); + +static int param_set_high_threshold(const char *val, const struct kernel_param *kp) +{ + unsigned int threshold; + int ret; + + ret = kstrtouint(val, 0, &threshold); + if (ret) + return ret; + + if (threshold >= 100 * 100) { + pr_err("high_threshold (%u) can't be more than 99.99%%\n", threshold); + return -EINVAL; + } + + return param_set_uint(val, kp); +} + +static const struct kernel_param_ops high_threshold_ops = { + .set = param_set_high_threshold, + .get = param_get_uint, +}; + +module_param_cb(high_threshold, &high_threshold_ops, &sg_ctx.high_threshold, 0444); +MODULE_PARM_DESC(high_threshold, + "High steal threshold. default: 500 i.e 5%. Must be > low_threshold"); + +module_param_named(low_threshold, sg_ctx.low_threshold, uint, 0444); +MODULE_PARM_DESC(low_threshold, + "Low steal threshold. default: 200 i.e 2%. Must be < high_threshold"); + static int __init steal_governor_init(void) { #ifdef CONFIG_XEN @@ -60,7 +120,16 @@ static int __init steal_governor_init(void) } #endif - pr_info("enabled\n"); + if (sg_ctx.low_threshold >= sg_ctx.high_threshold) { + pr_err("low_threshold (%u) must be less than high_threshold (%u)\n", + sg_ctx.low_threshold, sg_ctx.high_threshold); + return -EINVAL; + } + + sg_ctx.delay = msecs_to_jiffies(sg_ctx.interval_ms); + pr_info("enabled. interval: %ums, high_threshold: %u, low_threshold: %u\n", + sg_ctx.interval_ms, sg_ctx.high_threshold, sg_ctx.low_threshold); + return 0; } -- cgit v1.2.3 From 27d47ebce4d6490ce093031a48cc2d81bf9997a3 Mon Sep 17 00:00:00 2001 From: Shrikanth Hegde Date: Mon, 28 Sep 2026 11:07:27 +0530 Subject: virt/steal_governor: Implement steal_governor policy loop Schedule work at regular intervals to implement the steal_governor policy loop, which monitors steal time and takes action on the state of preferred CPUs. The interval is determined by the interval_ms parameter. schedule_delayed_work() is used since interval_ms is on the order of milliseconds and the work does not need to happen instantly. Periodic policy loop essentially does: - Gets the total/delta steal values and cpus to use steal_ratio. - Calculate the steal_ratio as below. steal_ratio = (delta_steal * 100*100)/(delta_ns * num_cpus()) It is calculated this way to consider the fractional values of steal time. I.e 10 means 0.1% steal time. A few tricks such as divide by 10,000 are used to avoid possible overflow. - If steal ratio is higher than high threshold, call the method to reduce the preferred CPUs. - If steal ratio is lower or equal to low threshold, call the method to increase the preferred CPUs. - If the steal ratio falls in between, no action is taken. - Ensures design constraints always met. 1. At least one core/CPU must be there in preferred mask. 2. preferred CPUs is subset of active CPUs. If not met, then restore preferred CPUs to active and stop requeue of the work. Driver is effectively non-functional after that. Note that design checks are always performed. This helps avoid placing driver-specific design constraints inside the core CPU hotplug mechanism. User may offline specific set of CPUs that could leave the preferred mask as empty. With the design check performed always, driver gracefully shuts down upon detecting that edge case. In order to help the above loop, a few helper functions have been added. 1. get_system_steal_time() - steal governor takes global view of steal time instead of individual vCPU. Collect the steal values across the vCPUs of interest. - Sum up steal time values across possible CPUs. This helps to keep it a monotonically increasing number and avoids spikes due to CPU hotplug. 2. decrease_preferred_cpus() - Called when there is high steal time. It needs to decide which CPUs to mark as non-preferred. - Get first housekeeping CPU and its core mask. Mark it as protected core. This helps to keep at least one core as preferred. (kernel ensures at least one housekeeping CPU stays active.) - Find the last CPU outside of this protected core mask. i.e target CPU - Based on that target CPU, get its sibling and mark them as non-preferred. 3. increase_preferred_cpus() - Called when there is low steal time. It needs to decide which CPUs to mark as preferred and set that state. - Get the first active non-preferred CPUs. This likely is the last set of CPUs being marked as non-preferred. - get the siblings of that CPU and mark them as preferred. 4. get_system_cpus() - informs how many CPUs needs to be considered for steal_ratio calculations. - Since only active CPUs effectively contribute to steal time delta, returns number of active CPUs. This also helps to avoid dilution of thresholds in sparsely populated systems. Notes: 1. Using core instead of individual CPUs performs better as SMT is quite common and some hypervisor such as powerVM does core scheduling. 2. This doesn't do any NUMA splicing to keep the code simpler and minimal overhead. Current code expects CPUs spread uniformly across NUMA nodes. Signed-off-by: Shrikanth Hegde Signed-off-by: Peter Zijlstra (Intel) Link: https://patch.msgid.link/20260928053728.797539-13-sshegde@linux.ibm.com --- drivers/virt/steal_governor.c | 149 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 149 insertions(+) (limited to 'drivers/virt') diff --git a/drivers/virt/steal_governor.c b/drivers/virt/steal_governor.c index 27f53ea16498..6e31f9923dea 100644 --- a/drivers/virt/steal_governor.c +++ b/drivers/virt/steal_governor.c @@ -13,13 +13,18 @@ #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt +#include #include #include #include #include +#include #include #include +#include #include +#include +#include #include #include #ifdef CONFIG_XEN @@ -111,6 +116,145 @@ module_param_named(low_threshold, sg_ctx.low_threshold, uint, 0444); MODULE_PARM_DESC(low_threshold, "Low steal threshold. default: 200 i.e 2%. Must be < high_threshold"); +/* Return collective steal time across system. */ +static u64 get_system_steal_time(void) +{ + return kcpustat_field_total(CPUTIME_STEAL, cpu_possible_mask); +} + +/* Return number of CPUs to consider for steal ratio. */ +static unsigned int get_system_cpus(void) +{ + return num_active_cpus(); +} + +/* + * Called when the steal governor detects high physical CPU contention. + * It finds the last active core in the preferred mask and mark those + * CPUs as non-preferred. + * + * Must ensure: + * - at least one core is always kept as preferred + * - preferred is always subset of active. + */ +static void decrease_preferred_cpus(void) +{ + const struct cpumask *first_hk_core; + int target_cpu = nr_cpu_ids; + int cpu; + + guard(cpus_read_lock)(); + cpu = cpumask_first_and(housekeeping_cpumask(HK_TYPE_KERNEL_NOISE), + cpu_preferred_mask); + if (cpu >= nr_cpu_ids) + return; + + /* Always leave first housekeeping core as preferred. */ + first_hk_core = topology_sibling_cpumask(cpu); + cpu = cpumask_last(cpu_preferred_mask); + if (cpu >= nr_cpu_ids) + return; + + /* Find the last CPU which doesn't belong to that first hk_core. */ + if (!cpumask_test_cpu(cpu, first_hk_core)) { + target_cpu = cpu; + } else { + for_each_cpu_andnot(cpu, cpu_preferred_mask, first_hk_core) + target_cpu = cpu; + } + + /* Only the first housekeeping core remains */ + if (target_cpu >= nr_cpu_ids) + return; + + for_each_cpu_and(cpu, topology_sibling_cpumask(target_cpu), + cpu_preferred_mask) + set_cpu_preferred(cpu, false); +} + +/* + * Called when the steal governor detects no/low physical CPU contention. + * It finds the first active core outside of preferred mask and mark + * those CPUs as preferred. + * + * Must ensure preferred is subset of active. + */ +static void increase_preferred_cpus(void) +{ + int first_cpu, cpu; + + guard(cpus_read_lock)(); + first_cpu = cpumask_first_andnot(cpu_active_mask, cpu_preferred_mask); + + /* All CPUs are preferred. Nothing to increase further */ + if (first_cpu >= nr_cpu_ids) + return; + + for_each_cpu_and(cpu, topology_sibling_cpumask(first_cpu), + cpu_active_mask) + set_cpu_preferred(cpu, true); +} + +static bool preferred_cpus_valid(void) +{ + if (cpumask_empty(cpu_preferred_mask)) { + pr_err("empty preferred mask. stopping\n"); + return false; + } + + if (!cpumask_subset(cpu_preferred_mask, cpu_active_mask)) { + pr_err("preferred: %*pbl is not subset of active: %*pbl, stopping\n", + cpumask_pr_args(cpu_preferred_mask), + cpumask_pr_args(cpu_active_mask)); + return false; + } + + return true; +} + +static void steal_governor_loop(struct work_struct *work) +{ + u64 curr_steal, delta_steal, delta_ns, steal_ratio; + ktime_t now; + + now = ktime_get(); + delta_ns = ktime_to_ns(ktime_sub(now, sg_ctx.time)); + + if (unlikely(delta_ns < NSEC_PER_MSEC)) { + pr_err_ratelimited("work scheduled too soon delta_ns: %llu\n", delta_ns); + goto requeue_work; + } + + curr_steal = get_system_steal_time(); + delta_steal = curr_steal > sg_ctx.steal ? curr_steal - sg_ctx.steal : 0; + sg_ctx.steal = curr_steal; + sg_ctx.time = now; + + /* + * steal_ratio = (delta_steal * 100*100)/(delta_ns * num_cpus()) + * To avoid possible overflow, divide the denominator early. + * Note minimum interval is 100ms. + */ + delta_ns = max_t(u64, div_u64(delta_ns * get_system_cpus(), 10000), 1); + steal_ratio = div64_u64(delta_steal, delta_ns); + + if (steal_ratio > sg_ctx.high_threshold) + decrease_preferred_cpus(); + else if (steal_ratio <= sg_ctx.low_threshold) + increase_preferred_cpus(); + /* + * else: steal ratio is within bounds. Still do design checks so that + * module restores to active if CPU hotplug breaks those assumptions. + */ + if (!preferred_cpus_valid()) { + restore_preferred_to_active(); + return; + } + +requeue_work: + schedule_delayed_work(&sg_ctx.work, sg_ctx.delay); +} + static int __init steal_governor_init(void) { #ifdef CONFIG_XEN @@ -127,6 +271,10 @@ static int __init steal_governor_init(void) } sg_ctx.delay = msecs_to_jiffies(sg_ctx.interval_ms); + INIT_DELAYED_WORK(&sg_ctx.work, steal_governor_loop); + sg_ctx.steal = get_system_steal_time(); + sg_ctx.time = ktime_get(); + schedule_delayed_work(&sg_ctx.work, sg_ctx.delay); pr_info("enabled. interval: %ums, high_threshold: %u, low_threshold: %u\n", sg_ctx.interval_ms, sg_ctx.high_threshold, sg_ctx.low_threshold); @@ -135,6 +283,7 @@ static int __init steal_governor_init(void) static void __exit steal_governor_exit(void) { + disable_delayed_work_sync(&sg_ctx.work); restore_preferred_to_active(); pr_info("disabled\n"); } -- cgit v1.2.3 From 1fb28c664a19df8d45a6afa04d28d102b04ea680 Mon Sep 17 00:00:00 2001 From: Shrikanth Hegde Date: Mon, 28 Sep 2026 11:07:28 +0530 Subject: virt/steal_governor: Enable the driver Provide a config option to enable the steal_governor driver. Since the feature targets paravirtualized environments and requires SMP, enforce those dependencies. The driver selects CONFIG_PREFERRED_CPU for the core scheduler mechanisms to work. It is recommended to build the driver as a module (m) instead of built-in (y) due to the following reasons: - Module parameters are read-only after initialization. Building as a module allows updating these parameters by simply reloading the module. Default module parameters cannot work in all configurations. - The driver can be completely disabled by unloading the module. - This feature works best when all VMs operate in a cooperative manner. Requiring an explicit module load ensures intentional deployment across all VMs by the system administrator. Suggested-by: Yury Norov Signed-off-by: Shrikanth Hegde Signed-off-by: Peter Zijlstra (Intel) Link: https://patch.msgid.link/20260928053728.797539-14-sshegde@linux.ibm.com --- drivers/virt/Kconfig | 17 +++++++++++++++++ drivers/virt/Makefile | 1 + 2 files changed, 18 insertions(+) (limited to 'drivers/virt') diff --git a/drivers/virt/Kconfig b/drivers/virt/Kconfig index 52eb7e4ba71f..eeb84e578ddf 100644 --- a/drivers/virt/Kconfig +++ b/drivers/virt/Kconfig @@ -41,6 +41,23 @@ config FSL_HV_MANAGER 4) A kernel interface for receiving callbacks when a managed partition shuts down. +config STEAL_GOVERNOR + tristate "Dynamic vCPU management based on steal time" + depends on PARAVIRT && SMP + select PREFERRED_CPU + default m + help + This driver helps to reduce the steal time in paravirtualized + environments, thereby reducing vCPU preemption costs. + + By default preferred CPUs will be same as active CPUs. Depending + on the steal time when steal_governor driver is enabled, + preferred CPUs could become subset of active CPUs. + More details are at: Documentation/driver-api/steal-governor.rst + + It is recommended to build it as module and load the module + to enable it. + source "drivers/virt/vboxguest/Kconfig" source "drivers/virt/nitro_enclaves/Kconfig" diff --git a/drivers/virt/Makefile b/drivers/virt/Makefile index f29901bd7820..05fb075ef5b8 100644 --- a/drivers/virt/Makefile +++ b/drivers/virt/Makefile @@ -5,6 +5,7 @@ obj-$(CONFIG_FSL_HV_MANAGER) += fsl_hypervisor.o obj-$(CONFIG_VMGENID) += vmgenid.o +obj-$(CONFIG_STEAL_GOVERNOR) += steal_governor.o obj-y += vboxguest/ obj-$(CONFIG_NITRO_ENCLAVES) += nitro_enclaves/ -- cgit v1.2.3 From d48b2d347105436365280a3697a265aa4d37e96c Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 28 Sep 2026 08:36:00 -0700 Subject: virt: tdx-guest: Remove unused and confusing function argument The tsm_report_ops->report_new() function takes a void* argument for implementations to use. But, TDX does not use the argument. It relies entirely on the 'struct tsm_report'. Despite that, the TDX code passes 'data' around needlessly from tdx_report_new()=>tdx_report_new_locked() where it is completely unused and ignored. This is not just a cleanup or bike-shedding rename. The variable is a real liability because there are 'quote_data' variables and even a tdx_quote_buf->data[] that can get literally referred to as "data". Remove the unused argument. Signed-off-by: Dave Hansen --- drivers/virt/coco/tdx-guest/tdx-guest.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) (limited to 'drivers/virt') diff --git a/drivers/virt/coco/tdx-guest/tdx-guest.c b/drivers/virt/coco/tdx-guest/tdx-guest.c index d0303e31e816..a21bd0376b74 100644 --- a/drivers/virt/coco/tdx-guest/tdx-guest.c +++ b/drivers/virt/coco/tdx-guest/tdx-guest.c @@ -265,7 +265,7 @@ static int wait_for_quote_completion(struct tdx_quote_buf *quote_buf, u32 timeou return (i == timeout) ? -ETIMEDOUT : 0; } -static int tdx_report_new_locked(struct tsm_report *report, void *data) +static int tdx_report_new_locked(struct tsm_report *report) { u8 *buf; struct tdx_quote_buf *quote_buf = quote_data; @@ -333,10 +333,10 @@ static int tdx_report_new_locked(struct tsm_report *report, void *data) return ret; } -static int tdx_report_new(struct tsm_report *report, void *data) +static int tdx_report_new(struct tsm_report *report, void *unused) { scoped_cond_guard(mutex_intr, return -EINTR, "e_lock) - return tdx_report_new_locked(report, data); + return tdx_report_new_locked(report); } static bool tdx_report_attr_visible(int n) -- cgit v1.2.3