diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-29 19:14:26 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-29 19:14:26 +0100 |
| commit | 0fbfaeecfd6f5925bfcd7cfcbd62860a9e868619 (patch) | |
| tree | 1a7726ceef53338e457951374e9d1d559080856a /arch/arm64/kvm | |
| parent | f9aac7cea1ffe9f4a942a773a9be0d8c065f99f4 (diff) | |
| parent | 995b96380ba5d98131acc162608da3538e1395d8 (diff) | |
| download | linux-next-0fbfaeecfd6f5925bfcd7cfcbd62860a9e868619.tar.gz linux-next-0fbfaeecfd6f5925bfcd7cfcbd62860a9e868619.zip | |
Merge asoc/for-7.4 into asoc-next
Diffstat (limited to 'arch/arm64/kvm')
| -rw-r--r-- | arch/arm64/kvm/arm.c | 10 | ||||
| -rw-r--r-- | arch/arm64/kvm/emulate-nested.c | 2 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/include/nvhe/mem_protect.h | 2 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/include/nvhe/mm.h | 1 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/nvhe/mem_protect.c | 20 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/nvhe/mm.c | 79 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/nvhe/pkvm.c | 29 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/nvhe/setup.c | 16 | ||||
| -rw-r--r-- | arch/arm64/kvm/hypercalls.c | 3 | ||||
| -rw-r--r-- | arch/arm64/kvm/mmu.c | 15 | ||||
| -rw-r--r-- | arch/arm64/kvm/nested.c | 111 | ||||
| -rw-r--r-- | arch/arm64/kvm/sys_regs.c | 1 | ||||
| -rw-r--r-- | arch/arm64/kvm/vgic/vgic-its.c | 46 |
13 files changed, 231 insertions, 104 deletions
diff --git a/arch/arm64/kvm/arm.c b/arch/arm64/kvm/arm.c index 8b080804bc90..0576c2022ef5 100644 --- a/arch/arm64/kvm/arm.c +++ b/arch/arm64/kvm/arm.c @@ -236,8 +236,6 @@ int kvm_arch_init_vm(struct kvm *kvm, unsigned long type) mutex_unlock(&kvm->lock); #endif - kvm_init_nested(kvm); - ret = kvm_share_hyp(kvm, kvm + 1); if (ret) return ret; @@ -252,6 +250,10 @@ int kvm_arch_init_vm(struct kvm *kvm, unsigned long type) if (ret) goto err_free_cpumask; + ret = kvm_init_nested(kvm); + if (ret) + goto err_uninit_mmu; + if (is_protected_kvm_enabled()) { /* * If any failures occur after this is successful, make sure to @@ -280,6 +282,7 @@ int kvm_arch_init_vm(struct kvm *kvm, unsigned long type) err_uninit_mmu: kvm_uninit_stage2_mmu(kvm); + kvm_destroy_nested(kvm); err_free_cpumask: free_cpumask_var(kvm->arch.supported_cpus); err_unshare_kvm: @@ -337,6 +340,7 @@ void kvm_arch_destroy_vm(struct kvm *kvm) kvm_unshare_hyp(kvm, kvm + 1); + kvm_destroy_nested(kvm); kvm_arm_teardown_hypercalls(kvm); } @@ -1129,7 +1133,7 @@ static int kvm_vcpu_suspend(struct kvm_vcpu *vcpu) static int check_vcpu_requests(struct kvm_vcpu *vcpu) { if (kvm_request_pending(vcpu)) { - if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) + if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) return -EIO; if (kvm_check_request(KVM_REQ_SLEEP, vcpu)) diff --git a/arch/arm64/kvm/emulate-nested.c b/arch/arm64/kvm/emulate-nested.c index 625604019fb3..3806ff0920fe 100644 --- a/arch/arm64/kvm/emulate-nested.c +++ b/arch/arm64/kvm/emulate-nested.c @@ -1445,7 +1445,7 @@ static const struct encoding_to_trap_config encoding_to_fgt[] __initconst = { SR_FGT(OP_AT_S1E1A, HFGITR, ATS1E1A, 1), SR_FGT(OP_COSP_RCTX, HFGITR, COSPRCTX, 1), SR_FGT(OP_GCSPUSHX, HFGITR, nGCSEPP, 0), - SR_FGT(OP_GCSPOPX, HFGITR, nGCSEPP, 0), + SR_FGT(OP_GCSPOPCX, HFGITR, nGCSEPP, 0), SR_FGT(OP_GCSPUSHM, HFGITR, nGCSPUSHM_EL1, 0), SR_FGT(OP_BRB_IALL, HFGITR, nBRBIALL, 0), SR_FGT(OP_BRB_INJ, HFGITR, nBRBINJ, 0), diff --git a/arch/arm64/kvm/hyp/include/nvhe/mem_protect.h b/arch/arm64/kvm/hyp/include/nvhe/mem_protect.h index 29935c7da1de..ec85a9547120 100644 --- a/arch/arm64/kvm/hyp/include/nvhe/mem_protect.h +++ b/arch/arm64/kvm/hyp/include/nvhe/mem_protect.h @@ -52,8 +52,10 @@ int __pkvm_host_test_clear_young_guest(u64 gfn, u64 nr_pages, bool mkold, struct int __pkvm_host_mkyoung_guest(u64 gfn, struct pkvm_hyp_vcpu *vcpu); bool addr_is_memory(phys_addr_t phys); +bool addr_is_hyp_text(phys_addr_t phys); int host_stage2_idmap_locked(phys_addr_t addr, u64 size, enum kvm_pgtable_prot prot); int host_stage2_set_owner_locked(phys_addr_t addr, u64 size, u8 owner_id); +bool host_stage2_pte_is_hyp_owned(kvm_pte_t pte); int kvm_host_prepare_stage2(void *pgt_pool_base); int kvm_guest_prepare_stage2(struct pkvm_hyp_vm *vm, void *pgd); void kvm_guest_destroy_stage2(struct pkvm_hyp_vm *vm); diff --git a/arch/arm64/kvm/hyp/include/nvhe/mm.h b/arch/arm64/kvm/hyp/include/nvhe/mm.h index 6e83ce35c2f2..31cae95ddb71 100644 --- a/arch/arm64/kvm/hyp/include/nvhe/mm.h +++ b/arch/arm64/kvm/hyp/include/nvhe/mm.h @@ -29,6 +29,7 @@ int __pkvm_create_private_mapping(phys_addr_t phys, size_t size, enum kvm_pgtable_prot prot, unsigned long *haddr); int pkvm_create_stack(phys_addr_t phys, unsigned long *haddr); +int pkvm_check_host_ownership(void); int pkvm_alloc_private_va_range(size_t size, unsigned long *haddr); #endif /* __KVM_HYP_MM_H */ diff --git a/arch/arm64/kvm/hyp/nvhe/mem_protect.c b/arch/arm64/kvm/hyp/nvhe/mem_protect.c index 39aa8911f62c..a6a47c1e058b 100644 --- a/arch/arm64/kvm/hyp/nvhe/mem_protect.c +++ b/arch/arm64/kvm/hyp/nvhe/mem_protect.c @@ -450,6 +450,14 @@ bool addr_is_memory(phys_addr_t phys) return !!find_mem_range(phys, &range); } +bool addr_is_hyp_text(phys_addr_t phys) +{ + phys_addr_t start = ALIGN_DOWN(__hyp_pa(__hyp_text_start), PAGE_SIZE); + phys_addr_t end = PAGE_ALIGN(__hyp_pa(__hyp_text_end)); + + return phys >= start && phys < end; +} + static bool is_in_mem_range(u64 addr, struct kvm_mem_range *range) { return range->start <= addr && addr < range->end; @@ -633,6 +641,18 @@ int host_stage2_set_owner_locked(phys_addr_t addr, u64 size, u8 owner_id) return ret; } +bool host_stage2_pte_is_hyp_owned(kvm_pte_t pte) +{ + if (kvm_pte_valid(pte)) + return false; + + if (FIELD_GET(KVM_INVALID_PTE_TYPE_MASK, pte) != + KVM_HOST_INVALID_PTE_TYPE_DONATION) + return false; + + return FIELD_GET(KVM_HOST_DONATION_PTE_OWNER_MASK, pte) == PKVM_ID_HYP; +} + #define KVM_HOST_PTE_OWNER_GUEST_HANDLE_MASK GENMASK(15, 0) /* We need 40 bits for the GFN to cover a 52-bit IPA with 4k pages and LPA2 */ #define KVM_HOST_PTE_OWNER_GUEST_GFN_MASK GENMASK(55, 16) diff --git a/arch/arm64/kvm/hyp/nvhe/mm.c b/arch/arm64/kvm/hyp/nvhe/mm.c index 3b0bee496bff..29ab5ee9d57f 100644 --- a/arch/arm64/kvm/hyp/nvhe/mm.c +++ b/arch/arm64/kvm/hyp/nvhe/mm.c @@ -25,6 +25,7 @@ struct memblock_region hyp_memory[HYP_MEMBLOCK_REGIONS]; unsigned int hyp_memblock_nr; static u64 __io_map_base; +static u64 __io_map_next; struct hyp_fixmap_slot { u64 addr; @@ -50,7 +51,7 @@ static int __pkvm_alloc_private_va_range(unsigned long start, size_t size) hyp_assert_lock_held(&pkvm_pgd_lock); - if (!start || start < __io_map_base) + if (!start || start < __io_map_next) return -EINVAL; /* The allocated size is always a multiple of PAGE_SIZE */ @@ -60,7 +61,7 @@ static int __pkvm_alloc_private_va_range(unsigned long start, size_t size) if (cur > __hyp_vmemmap) return -ENOMEM; - __io_map_base = cur; + __io_map_next = cur; return 0; } @@ -70,7 +71,7 @@ static int __pkvm_alloc_private_va_range(unsigned long start, size_t size) * @size: The size of the VA range to reserve. * @haddr: The hypervisor virtual start address of the allocation. * - * The private virtual address (VA) range is allocated above __io_map_base + * The private virtual address (VA) range is allocated above __io_map_next * and aligned based on the order of @size. * * Return: 0 on success or negative error code on failure. @@ -81,7 +82,7 @@ int pkvm_alloc_private_va_range(size_t size, unsigned long *haddr) int ret; hyp_spin_lock(&pkvm_pgd_lock); - addr = __io_map_base; + addr = __io_map_next; ret = __pkvm_alloc_private_va_range(addr, size); hyp_spin_unlock(&pkvm_pgd_lock); @@ -341,7 +342,7 @@ static int create_fixblock(void) return -EINVAL; hyp_spin_lock(&pkvm_pgd_lock); - addr = ALIGN(__io_map_base, PMD_SIZE); + addr = ALIGN(__io_map_next, PMD_SIZE); ret = __pkvm_alloc_private_va_range(addr, PMD_SIZE); if (ret) goto unlock; @@ -426,6 +427,7 @@ int hyp_create_idmap(u32 hyp_va_bits) */ __io_map_base = start & BIT(hyp_va_bits - 2); __io_map_base ^= BIT(hyp_va_bits - 2); + __io_map_next = __io_map_base; __hyp_vmemmap = __io_map_base | BIT(hyp_va_bits - 3); return __pkvm_create_mappings(start, end - start, start, PAGE_HYP_EXEC); @@ -433,19 +435,19 @@ int hyp_create_idmap(u32 hyp_va_bits) int pkvm_create_stack(phys_addr_t phys, unsigned long *haddr) { - unsigned long addr, prev_base; + unsigned long addr, prev_next; size_t size; int ret; hyp_spin_lock(&pkvm_pgd_lock); - prev_base = __io_map_base; + prev_next = __io_map_next; /* * Efficient stack verification using the NVHE_STACK_SHIFT bit implies * an alignment of our allocation on the order of the size. */ size = NVHE_STACK_SIZE * 2; - addr = ALIGN(__io_map_base, size); + addr = ALIGN(__io_map_next, size); ret = __pkvm_alloc_private_va_range(addr, size); if (!ret) { @@ -461,7 +463,7 @@ int pkvm_create_stack(phys_addr_t phys, unsigned long *haddr) ret = kvm_pgtable_hyp_map(&pkvm_pgtable, addr + NVHE_STACK_SIZE, NVHE_STACK_SIZE, phys, PAGE_HYP); if (ret) - __io_map_base = prev_base; + __io_map_next = prev_next; } hyp_spin_unlock(&pkvm_pgd_lock); @@ -470,6 +472,65 @@ int pkvm_create_stack(phys_addr_t phys, unsigned long *haddr) return ret; } +static int check_page_ownership(phys_addr_t phys) +{ + kvm_pte_t pte; + bool host_ok; + int ret; + + if (addr_is_memory(phys)) { + struct hyp_page *page = hyp_phys_to_page(phys); + + if (get_hyp_state(page) != PKVM_PAGE_OWNED || + get_host_state(page) != PKVM_NOPAGE) + return -EPERM; + } + + ret = kvm_pgtable_get_leaf(&host_mmu.pgt, phys, &pte, NULL); + if (ret) + return ret; + + /* Hyp text may stay host-readable, see fix_host_ownership_walker(). */ + if (kvm_pte_valid(pte) && addr_is_hyp_text(phys)) + host_ok = !(kvm_pgtable_stage2_pte_prot(pte) & KVM_PGTABLE_PROT_W); + else + host_ok = host_stage2_pte_is_hyp_owned(pte); + + return host_ok ? 0 : -EPERM; +} + +static int check_host_ownership_walker(const struct kvm_pgtable_visit_ctx *ctx, + enum kvm_pgtable_walk_flags visit) +{ + phys_addr_t phys, end; + int ret; + + if (!kvm_pte_valid(ctx->old)) + return 0; + + phys = kvm_pte_to_phys(ctx->old); + end = phys + kvm_granule_size(ctx->level); + for (; phys < end; phys += PAGE_SIZE) { + ret = check_page_ownership(phys); + if (ret) + return ret; + } + + return 0; +} + +int pkvm_check_host_ownership(void) +{ + struct kvm_pgtable_walker walker = { + .cb = check_host_ownership_walker, + .flags = KVM_PGTABLE_WALK_LEAF, + }; + + /* The private range and the vmemmap share one quarter of the VA space. */ + return kvm_pgtable_walk(&pkvm_pgtable, __io_map_base, + BIT(pkvm_pgtable.ia_bits - 2), &walker); +} + static void *admit_host_page(void *arg) { struct kvm_hyp_memcache *host_mc = arg; diff --git a/arch/arm64/kvm/hyp/nvhe/pkvm.c b/arch/arm64/kvm/hyp/nvhe/pkvm.c index 459bd9eb7e4b..bb3e0dc0676e 100644 --- a/arch/arm64/kvm/hyp/nvhe/pkvm.c +++ b/arch/arm64/kvm/hyp/nvhe/pkvm.c @@ -360,7 +360,7 @@ static void pkvm_init_features_from_host(struct pkvm_hyp_vm *hyp_vm, const struc if (test_bit(KVM_ARCH_FLAG_WRITABLE_IMP_ID_REGS, &host_arch_flags)) hyp_vm->kvm.arch.midr_el1 = host_kvm->arch.midr_el1; - return; + goto out; } if (kvm_pkvm_ext_allowed(kvm, KVM_CAP_ARM_MTE)) @@ -379,13 +379,14 @@ static void pkvm_init_features_from_host(struct pkvm_hyp_vm *hyp_vm, const struc if (kvm_pkvm_ext_allowed(kvm, KVM_CAP_ARM_PTRAUTH_GENERIC)) set_bit(KVM_ARM_VCPU_PTRAUTH_GENERIC, allowed_features); - if (kvm_pkvm_ext_allowed(kvm, KVM_CAP_ARM_SVE)) { + if (kvm_pkvm_ext_allowed(kvm, KVM_CAP_ARM_SVE)) set_bit(KVM_ARM_VCPU_SVE, allowed_features); - kvm->arch.flags |= host_arch_flags & BIT(KVM_ARCH_FLAG_GUEST_HAS_SVE); - } bitmap_and(kvm->arch.vcpu_features, host_kvm->arch.vcpu_features, allowed_features, KVM_VCPU_MAX_FEATURES); +out: + __assign_bit(KVM_ARCH_FLAG_GUEST_HAS_SVE, &kvm->arch.flags, + kvm_vcpu_has_feature(kvm, KVM_ARM_VCPU_SVE)); } static void unpin_host_vcpu(struct kvm_vcpu *host_vcpu) @@ -398,10 +399,10 @@ static void unpin_host_sve_state(struct pkvm_hyp_vcpu *hyp_vcpu) { void *sve_state; - if (!vcpu_has_feature(&hyp_vcpu->vcpu, KVM_ARM_VCPU_SVE)) + sve_state = hyp_vcpu->vcpu.arch.sve_state; + if (!sve_state) return; - sve_state = hyp_vcpu->vcpu.arch.sve_state; hyp_unpin_shared_mem(sve_state, sve_state + vcpu_sve_state_size(&hyp_vcpu->vcpu)); } @@ -450,7 +451,7 @@ static int pkvm_vcpu_init_sve(struct pkvm_hyp_vcpu *hyp_vcpu, struct kvm_vcpu *h unsigned int sve_max_vl; size_t sve_state_size; void *sve_state; - int ret = 0; + int ret; if (!vcpu_has_feature(vcpu, KVM_ARM_VCPU_SVE)) { vcpu_clear_flag(vcpu, VCPU_SVE_FINALIZED); @@ -459,25 +460,21 @@ static int pkvm_vcpu_init_sve(struct pkvm_hyp_vcpu *hyp_vcpu, struct kvm_vcpu *h /* Limit guest vector length to the maximum supported by the host. */ sve_max_vl = min(READ_ONCE(host_vcpu->arch.sve_max_vl), kvm_host_sve_max_vl); - sve_state_size = sve_state_size_from_vl(sve_max_vl); sve_state = kern_hyp_va(READ_ONCE(host_vcpu->arch.sve_state)); - if (!sve_state || !sve_state_size) { - ret = -EINVAL; - goto err; - } + if (!sve_vl_valid(sve_max_vl) || !sve_state) + return -EINVAL; + + sve_state_size = sve_state_size_from_vl(sve_max_vl); ret = hyp_pin_shared_mem(sve_state, sve_state + sve_state_size); if (ret) - goto err; + return ret; vcpu->arch.sve_state = sve_state; vcpu->arch.sve_max_vl = sve_max_vl; return 0; -err: - clear_bit(KVM_ARM_VCPU_SVE, vcpu->kvm->arch.vcpu_features); - return ret; } static int vm_copy_id_regs(struct pkvm_hyp_vcpu *hyp_vcpu) diff --git a/arch/arm64/kvm/hyp/nvhe/setup.c b/arch/arm64/kvm/hyp/nvhe/setup.c index 75b00c323310..45ac5f2ba4f7 100644 --- a/arch/arm64/kvm/hyp/nvhe/setup.c +++ b/arch/arm64/kvm/hyp/nvhe/setup.c @@ -217,7 +217,7 @@ static int fix_host_ownership_walker(const struct kvm_pgtable_visit_ctx *ctx, case PKVM_PAGE_OWNED: set_hyp_state(page, PKVM_PAGE_OWNED); /* hyp text is RO in the host stage-2 to be inspected on panic. */ - if (prot == PAGE_HYP_EXEC) { + if (addr_is_hyp_text(phys)) { set_host_state(page, PKVM_NOPAGE); return host_stage2_idmap_locked(phys, PAGE_SIZE, KVM_PGTABLE_PROT_R); } else { @@ -269,6 +269,16 @@ static int fix_host_ownership(void) return ret; } + /* The stacks sit in the private VA range, not the linear map. */ + for (i = 0; i < hyp_nr_cpus; i++) { + struct kvm_nvhe_init_params *params = per_cpu_ptr(&kvm_init_params, i); + u64 start = params->stack_hyp_va - NVHE_STACK_SIZE; + + ret = kvm_pgtable_walk(&pkvm_pgtable, start, NVHE_STACK_SIZE, &walker); + if (ret) + return ret; + } + return 0; } @@ -324,6 +334,10 @@ void __noreturn __pkvm_init_finalise(void) if (ret) goto out; + ret = pkvm_check_host_ownership(); + if (ret) + goto out; + ret = hyp_ffa_init(ffa_proxy_pages); if (ret) goto out; diff --git a/arch/arm64/kvm/hypercalls.c b/arch/arm64/kvm/hypercalls.c index b11b8821c9fb..dfa25bb6f25d 100644 --- a/arch/arm64/kvm/hypercalls.c +++ b/arch/arm64/kvm/hypercalls.c @@ -185,7 +185,8 @@ static int kvm_smccc_set_filter(struct kvm *kvm, struct kvm_smccc_filter __user start = filter.base; end = start + filter.nr_functions - 1; - if (end < start || filter.action >= NR_SMCCC_FILTER_ACTIONS) + if (!filter.nr_functions || end < start || + filter.action >= NR_SMCCC_FILTER_ACTIONS) return -EINVAL; mutex_lock(&kvm->arch.config_lock); diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c index 9ba86450fe4a..2d44cd6a5aed 100644 --- a/arch/arm64/kvm/mmu.c +++ b/arch/arm64/kvm/mmu.c @@ -59,27 +59,36 @@ static phys_addr_t stage2_range_addr_end(phys_addr_t addr, phys_addr_t end) * long will also starve other vCPUs. We have to also make sure that the page * tables are not freed while we released the lock. */ -static int stage2_apply_range(struct kvm_s2_mmu *mmu, phys_addr_t addr, +static int stage2_apply_range(struct kvm_s2_mmu *mmu, phys_addr_t start, phys_addr_t end, int (*fn)(struct kvm_pgtable *, u64, u64), bool resched) { struct kvm *kvm = kvm_s2_mmu_to_kvm(mmu); + bool lock_dropped = false; + phys_addr_t addr = start; int ret; u64 next; do { struct kvm_pgtable *pgt = mmu->pgt; + /* + * We may be raced on PGT teardown when we release the + * kvm->mmu_lock. That's fine as the PGT is legitimately no + * longer present. + */ if (!pgt) - return -EINVAL; + return lock_dropped ? 0 : -EINVAL; next = stage2_range_addr_end(addr, end); ret = fn(pgt, addr, next - addr); if (ret) break; - if (resched && next != end) + if (resched && next != end) { cond_resched_rwlock_write(&kvm->mmu_lock); + lock_dropped = true; + } } while (addr = next, addr != end); return ret; diff --git a/arch/arm64/kvm/nested.c b/arch/arm64/kvm/nested.c index 3c4fc566eafc..b191365d97cc 100644 --- a/arch/arm64/kvm/nested.c +++ b/arch/arm64/kvm/nested.c @@ -45,11 +45,24 @@ struct vncr_tlb { */ #define S2_MMU_PER_VCPU 2 -void kvm_init_nested(struct kvm *kvm) +int kvm_init_nested(struct kvm *kvm) { - kvm->arch.nested_mmus = NULL; + kvm->arch.nested_mmus = kvmalloc_objs(struct kvm_s2_mmu *, + KVM_MAX_VCPUS * S2_MMU_PER_VCPU, + GFP_KERNEL_ACCOUNT); kvm->arch.nested_mmus_size = 0; atomic_set(&kvm->arch.vncr_tlb_count, 0); + + return kvm->arch.nested_mmus ? 0 : -ENOMEM; +} + +void kvm_destroy_nested(struct kvm *kvm) +{ + for (int i = 0; i < kvm->arch.nested_mmus_size; i+= S2_MMU_PER_VCPU) + kvfree(kvm->arch.nested_mmus[i]); + + kvm->arch.nested_mmus_size = 0; + kvfree(kvm->arch.nested_mmus); } static int init_nested_s2_mmu(struct kvm *kvm, struct kvm_s2_mmu *mmu) @@ -70,8 +83,9 @@ static int init_nested_s2_mmu(struct kvm *kvm, struct kvm_s2_mmu *mmu) int kvm_vcpu_init_nested(struct kvm_vcpu *vcpu) { struct kvm *kvm = vcpu->kvm; - struct kvm_s2_mmu *tmp; - int num_mmus, ret = 0; + int num_mmus; + + lockdep_assert_held(&kvm->arch.config_lock); if (test_bit(KVM_ARM_VCPU_HAS_EL2_E2H0, kvm->arch.vcpu_features) && !cpus_have_final_cap(ARM64_HAS_HCR_NV1)) @@ -84,51 +98,40 @@ int kvm_vcpu_init_nested(struct kvm_vcpu *vcpu) if (!vcpu->arch.ctxt.vncr_array) return -ENOMEM; - /* - * Let's treat memory allocation failures as benign: If we fail to - * allocate anything, return an error and keep the allocated array - * alive. Userspace may try to recover by initializing the vcpu - * again, and there is no reason to affect the whole VM for this. - */ num_mmus = atomic_read(&kvm->online_vcpus) * S2_MMU_PER_VCPU; if (num_mmus > kvm->arch.nested_mmus_size) { - tmp = kvzalloc_objs(*tmp, num_mmus, GFP_KERNEL_ACCOUNT); - if (!tmp) - return -ENOMEM; + struct kvm_s2_mmu *tmp; + int i, ret = 0; - write_lock(&kvm->mmu_lock); - - if (kvm->arch.nested_mmus_size) { - memcpy(tmp, kvm->arch.nested_mmus, - size_mul(sizeof(*tmp), kvm->arch.nested_mmus_size)); + tmp = kvzalloc_objs(*tmp, S2_MMU_PER_VCPU, GFP_KERNEL_ACCOUNT); + if (!tmp) + ret = -ENOMEM; - for (int i = 0; i < kvm->arch.nested_mmus_size; i++) - tmp[i].pgt->mmu = &tmp[i]; + for (i = 0; !ret && i < S2_MMU_PER_VCPU; i++) { + ret = init_nested_s2_mmu(kvm, &tmp[i]); + if (ret) + break; } - swap(kvm->arch.nested_mmus, tmp); - - write_unlock(&kvm->mmu_lock); - - kvfree(tmp); - } + if (ret) { + while (--i >= 0) + kvm_free_stage2_pgd(&tmp[i]); - for (int i = kvm->arch.nested_mmus_size; !ret && i < num_mmus; i++) - ret = init_nested_s2_mmu(kvm, &kvm->arch.nested_mmus[i]); + kvfree(tmp); + free_page((unsigned long)vcpu->arch.ctxt.vncr_array); + vcpu->arch.ctxt.vncr_array = NULL; + return ret; + } - if (ret) { - for (int i = kvm->arch.nested_mmus_size; i < num_mmus; i++) - kvm_free_stage2_pgd(&kvm->arch.nested_mmus[i]); + guard(write_lock)(&kvm->mmu_lock); - free_page((unsigned long)vcpu->arch.ctxt.vncr_array); - vcpu->arch.ctxt.vncr_array = NULL; + for (i = 0; i < S2_MMU_PER_VCPU; i++) + kvm->arch.nested_mmus[i + kvm->arch.nested_mmus_size] = &tmp[i]; - return ret; + kvm->arch.nested_mmus_size += S2_MMU_PER_VCPU; } - kvm->arch.nested_mmus_size = num_mmus; - return 0; } @@ -742,7 +745,7 @@ void kvm_s2_mmu_iterate_by_vmid(struct kvm *kvm, u16 vmid, write_lock(&kvm->mmu_lock); for (int i = 0; i < kvm->arch.nested_mmus_size; i++) { - struct kvm_s2_mmu *mmu = &kvm->arch.nested_mmus[i]; + struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; if (!kvm_s2_mmu_valid(mmu)) continue; @@ -784,7 +787,7 @@ struct kvm_s2_mmu *lookup_s2_mmu(struct kvm_vcpu *vcpu) * if S2 translation is disabled. */ for (int i = 0; i < kvm->arch.nested_mmus_size; i++) { - struct kvm_s2_mmu *mmu = &kvm->arch.nested_mmus[i]; + struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; if (!kvm_s2_mmu_valid(mmu)) continue; @@ -823,7 +826,7 @@ static struct kvm_s2_mmu *get_s2_mmu_nested(struct kvm_vcpu *vcpu) for (i = kvm->arch.nested_mmus_next; i < (kvm->arch.nested_mmus_size + kvm->arch.nested_mmus_next); i++) { - s2_mmu = &kvm->arch.nested_mmus[i % kvm->arch.nested_mmus_size]; + s2_mmu = kvm->arch.nested_mmus[i % kvm->arch.nested_mmus_size]; if (atomic_read(&s2_mmu->refcnt) == 0) break; @@ -1260,6 +1263,17 @@ void kvm_handle_s1e2_tlbi(struct kvm_vcpu *vcpu, u32 inst, u64 val) invalidate_vncr_va(vcpu->kvm, &scope); } +static void kvm_invalidate_vncr_ipa_all(struct kvm *kvm) +{ + struct kvm_pgtable *pgt = kvm->arch.mmu.pgt; + + lockdep_assert_held_write(&kvm->mmu_lock); + + /* if the mmu lock was dropped, pgt teardown may have raced. */ + if (pgt) + kvm_invalidate_vncr_ipa(kvm, 0, BIT(pgt->ia_bits)); +} + void kvm_nested_s2_wp(struct kvm *kvm) { int i; @@ -1270,13 +1284,13 @@ void kvm_nested_s2_wp(struct kvm *kvm) return; for (i = 0; i < kvm->arch.nested_mmus_size; i++) { - struct kvm_s2_mmu *mmu = &kvm->arch.nested_mmus[i]; + struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; if (kvm_s2_mmu_valid(mmu)) kvm_stage2_wp_range(mmu, 0, kvm_phys_size(mmu)); } - kvm_invalidate_vncr_ipa(kvm, 0, BIT(kvm->arch.mmu.pgt->ia_bits)); + kvm_invalidate_vncr_ipa_all(kvm); } void kvm_nested_s2_unmap(struct kvm *kvm, bool may_block) @@ -1289,13 +1303,13 @@ void kvm_nested_s2_unmap(struct kvm *kvm, bool may_block) return; for (i = 0; i < kvm->arch.nested_mmus_size; i++) { - struct kvm_s2_mmu *mmu = &kvm->arch.nested_mmus[i]; + struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; if (kvm_s2_mmu_valid(mmu)) kvm_stage2_unmap_range(mmu, 0, kvm_phys_size(mmu), may_block); } - kvm_invalidate_vncr_ipa(kvm, 0, BIT(kvm->arch.mmu.pgt->ia_bits)); + kvm_invalidate_vncr_ipa_all(kvm); } void kvm_nested_s2_flush(struct kvm *kvm) @@ -1308,7 +1322,7 @@ void kvm_nested_s2_flush(struct kvm *kvm) return; for (i = 0; i < kvm->arch.nested_mmus_size; i++) { - struct kvm_s2_mmu *mmu = &kvm->arch.nested_mmus[i]; + struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; if (kvm_s2_mmu_valid(mmu)) kvm_stage2_flush_range(mmu, 0, kvm_phys_size(mmu)); @@ -1317,17 +1331,12 @@ void kvm_nested_s2_flush(struct kvm *kvm) void kvm_arch_flush_shadow_all(struct kvm *kvm) { - int i; - - for (i = 0; i < kvm->arch.nested_mmus_size; i++) { - struct kvm_s2_mmu *mmu = &kvm->arch.nested_mmus[i]; + for (int i = 0; i < kvm->arch.nested_mmus_size; i++) { + struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; if (!WARN_ON(atomic_read(&mmu->refcnt))) kvm_free_stage2_pgd(mmu); } - kvfree(kvm->arch.nested_mmus); - kvm->arch.nested_mmus = NULL; - kvm->arch.nested_mmus_size = 0; kvm_uninit_stage2_mmu(kvm); } diff --git a/arch/arm64/kvm/sys_regs.c b/arch/arm64/kvm/sys_regs.c index 44aae52c473d..a2f4e769a428 100644 --- a/arch/arm64/kvm/sys_regs.c +++ b/arch/arm64/kvm/sys_regs.c @@ -926,6 +926,7 @@ static u64 *demux_wb_reg(struct kvm_vcpu *vcpu, const struct sys_reg_desc *rd) struct kvm_guest_debug_arch *dbg = &vcpu->arch.vcpu_debug_state; switch (rd->Op2) { + case 0b001: case 0b100: return &dbg->dbg_bvr[rd->CRm]; case 0b101: diff --git a/arch/arm64/kvm/vgic/vgic-its.c b/arch/arm64/kvm/vgic/vgic-its.c index 9e782a4fea7e..0904ae850c35 100644 --- a/arch/arm64/kvm/vgic/vgic-its.c +++ b/arch/arm64/kvm/vgic/vgic-its.c @@ -1658,7 +1658,7 @@ static void vgic_mmio_write_its_baser(struct kvm *kvm, unsigned long val) { const struct vgic_its_abi *abi = vgic_its_get_abi(its); - u64 entry_size, table_type; + u64 old, entry_size, table_type; u64 reg, *regptr, clearbits = 0; /* When GITS_CTLR.Enable is 1, we ignore write accesses. */ @@ -1681,7 +1681,9 @@ static void vgic_mmio_write_its_baser(struct kvm *kvm, return; } - reg = update_64bit_reg(*regptr, addr & 7, len, val); + old = *regptr; + + reg = update_64bit_reg(old, addr & 7, len, val); reg &= ~GITS_BASER_RO_MASK; reg &= ~clearbits; @@ -1691,7 +1693,8 @@ static void vgic_mmio_write_its_baser(struct kvm *kvm, *regptr = reg; - if (!(reg & GITS_BASER_VALID)) { + /* The ITS driver rewrites an unchanged GITS_BASER<n> on resume. */ + if (reg != old) { /* Take the its_lock to prevent a race with a save/restore */ mutex_lock(&its->its_lock); switch (table_type) { @@ -1702,6 +1705,8 @@ static void vgic_mmio_write_its_baser(struct kvm *kvm, vgic_its_free_collection_list(kvm, its); break; } + /* A concurrent injection may have cached a translation. */ + vgic_its_invalidate_cache(its); mutex_unlock(&its->its_lock); } } @@ -2019,18 +2024,22 @@ out: return ret; } -static u32 compute_next_devid_offset(struct list_head *h, +static u32 compute_next_devid_offset(struct vgic_its *its, u64 baser, struct its_device *dev) { - struct its_device *next; - u32 next_offset; + struct its_device *next = dev; - if (list_is_last(&dev->dev_list, h)) - return 0; - next = list_next_entry(dev, dev_list); - next_offset = next->device_id - dev->device_id; + /* + * Point at the next device vgic_its_save_device_tables() saves. It + * sorts device_list first, so the subtraction cannot underflow. + */ + list_for_each_entry_continue(next, &its->device_list, dev_list) { + if (vgic_its_check_id(its, baser, next->device_id, NULL)) + return min_t(u32, next->device_id - dev->device_id, + VITS_DTE_MAX_DEVID_OFFSET); + } - return min_t(u32, next_offset, VITS_DTE_MAX_DEVID_OFFSET); + return 0; } static u32 compute_next_eventid_offset(struct list_head *h, struct its_ite *ite) @@ -2271,17 +2280,18 @@ static int vgic_its_restore_itt(struct vgic_its *its, struct its_device *dev) * vgic_its_save_dte - Save a device table entry at a given GPA * * @its: ITS handle + * @baser: GITS_BASER<dev> the caller is saving against * @dev: ITS device * @ptr: GPA */ -static int vgic_its_save_dte(struct vgic_its *its, struct its_device *dev, - gpa_t ptr) +static int vgic_its_save_dte(struct vgic_its *its, u64 baser, + struct its_device *dev, gpa_t ptr) { u64 val, itt_addr_field; u32 next_offset; itt_addr_field = dev->itt_addr >> 8; - next_offset = compute_next_devid_offset(&its->device_list, dev); + next_offset = compute_next_devid_offset(its, baser, dev); val = (1ULL << KVM_ITS_DTE_VALID_SHIFT | ((u64)next_offset << KVM_ITS_DTE_NEXT_SHIFT) | (itt_addr_field << KVM_ITS_DTE_ITTADDR_SHIFT) | @@ -2380,15 +2390,16 @@ static int vgic_its_save_device_tables(struct vgic_its *its) int ret; gpa_t eaddr; + /* Don't fail a save that userspace must be able to issue. */ if (!vgic_its_check_id(its, baser, dev->device_id, &eaddr)) - return -EINVAL; + continue; ret = vgic_its_save_itt(its, dev); if (ret) return ret; - ret = vgic_its_save_dte(its, dev, eaddr); + ret = vgic_its_save_dte(its, baser, dev, eaddr); if (ret) return ret; } @@ -2541,9 +2552,6 @@ static int vgic_its_save_collection_table(struct vgic_its *its) max_size = GITS_BASER_NR_PAGES(baser) * SZ_64K; list_for_each_entry(collection, &its->collection_list, coll_list) { - if (!vgic_its_check_id(its, baser, collection->collection_id, NULL)) - return -EINVAL; - ret = vgic_its_save_cte(its, collection, gpa); if (ret) return ret; |
