summaryrefslogtreecommitdiff
path: root/arch
diff options
context:
space:
mode:
authorAmirmohammad Eftekhar <amirmohammad.eftekhar@cispa.de>2026-09-30 10:36:15 -0700
committerSean Christopherson <seanjc@google.com>2026-09-30 14:23:45 -0700
commitb701a8139fd6d51462a5dcac79c92c3ff6524f6c (patch)
tree5fff84eb193925fef8ccec0a330c0d77d317056f /arch
parent3bac3b77c5e6b4403701cab9df43030daf984a2e (diff)
downloadlinux-next-b701a8139fd6d51462a5dcac79c92c3ff6524f6c.tar.gz
linux-next-b701a8139fd6d51462a5dcac79c92c3ff6524f6c.zip
KVM: x86: Saturate L2's TSC frequency if it exceeds hardware supports
When computing the TSC multiplier for L2, use the minimum/maximum frequency supported by hardware if the resulting multiplier can't be programmed into hardware, i.e. saturate L2's frequency on both sides. This can happen if userspace is running L1 at a low/high frequency relative to L0, and L1 is doing the same for L2 (relative to L1), because although each of the L1 and L2 inputs are constrained and validated, the end result can exceed SVM's and VMX's architectural limits. Don't bother trying to log an error or do something more sophisticated, because in practice only a misbehaving L1 (and/or host userspace) will run afoul of the flaw. On SVM, which has the smallest "range" by far (8 bits for the integer multiplier, versus 16 bits on VMX), KVM can still run L2 with a frequency 255x that of L0. Even assuming a pessimistic L0 TSC frequency of 1GHz (modern CPUs run TSC at 2GHz+), L2 would need to be running at a whopping ~255GHz for the limitation to be a problem. Not to mention that if L2 is running at a legitimate frequency, then running that VM on existing hardware is already doomed irrespective of the nested angle. Open code the mul_u64_u64_shr() if the compiler natively supports 128-bit integers, mostly to avoid having to do the multiply twice for what should be a very rare scenario, but also so that there's a slightly more scrutable sequence that can be read to understand what all is going on. Signed-off-by: Amirmohammad Eftekhar <amirmohammad.eftekhar@cispa.de> [sean: optimize for INT128, rewrite comment+changelog to be less AI] Link: https://patch.msgid.link/20260930173635.3362655-2-seanjc@google.com Signed-off-by: Sean Christopherson <seanjc@google.com>
Diffstat (limited to 'arch')
-rw-r--r--arch/x86/kvm/x86.c55
1 files changed, 51 insertions, 4 deletions
diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c
index 79468ddfe473..b7eb410c423c 100644
--- a/arch/x86/kvm/x86.c
+++ b/arch/x86/kvm/x86.c
@@ -1164,11 +1164,58 @@ EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_calc_nested_tsc_offset);
u64 kvm_calc_nested_tsc_multiplier(u64 l1_multiplier, u64 l2_multiplier)
{
- if (l2_multiplier != kvm_caps.default_tsc_scaling_ratio)
- return mul_u64_u64_shr(l1_multiplier, l2_multiplier,
- kvm_caps.tsc_scaling_ratio_frac_bits);
+ u8 frac_bits = kvm_caps.tsc_scaling_ratio_frac_bits;
+ u64 nested_multiplier;
- return l1_multiplier;
+ if (l2_multiplier == kvm_caps.default_tsc_scaling_ratio)
+ return l1_multiplier;
+
+ /*
+ * The shift is fixed on both AMD and Intel, and operates on a 64-bit
+ * value. I.e. a shift greater than 63 is completely nonsensical.
+ */
+ if (WARN_ON_ONCE(frac_bits > 63))
+ return l1_multiplier;
+
+ /*
+ * If the resulting multiplier can't be programmed into hardware, run
+ * L2 at the minimum/maximum frequency supported by hardware, i.e.
+ * saturate L2's frequency on both sides. Because L2's frequency needs
+ * to be distilled down to a single multiplier to get from:
+ *
+ * L2 = (((L0 * L1_mult) >> frac) * L2_mult) >> frac)
+ *
+ * to:
+ *
+ * L2 = (L0 * mult) >> frac
+ *
+ * very small/large L1 and L2 multipliers can underflow/overflow the
+ * minimum/maximum multiplier supported by hardware when combined into
+ * a single value.
+ *
+ * Manually check for the case where the result would overflow a u64,
+ * i.e. if the multiplier would be silently truncated before the "too
+ * large" check. Avoid doing the multiply twice in the common case
+ * where the compiler natively supports 128-bit values.
+ */
+#ifdef CONFIG_ARCH_SUPPORTS_INT128
+ unsigned __int128 m = (unsigned __int128)l1_multiplier * l2_multiplier;
+
+ if (m >> (64 + frac_bits))
+ return kvm_caps.max_tsc_scaling_ratio;
+
+ nested_multiplier = m >> frac_bits;
+#else
+ if (mul_u64_u64_shr(l1_multiplier, l2_multiplier, 64 + frac_bits))
+ return kvm_caps.max_tsc_scaling_ratio;
+
+ nested_multiplier = mul_u64_u64_shr(l1_multiplier, l2_multiplier, frac_bits);
+#endif
+ if (nested_multiplier > kvm_caps.max_tsc_scaling_ratio)
+ return kvm_caps.max_tsc_scaling_ratio;
+
+ /* The minimum multiplier is '1' on both AMD and Intel. */
+ return nested_multiplier ?: 1;
}
EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_calc_nested_tsc_multiplier);