From: David Woodhouse <[email protected]> Reads of L1's MSRPM go through kvm_vcpu_read_guest() on two paths: the vmcb02 MSRPM merge on nested VMRUN (~10 reads per entry), and the per-intercept check on every trapped L2 MSR access. With unmanaged guest memory each read is a memremap/memunmap cycle.
Cache the two pages of L1's MSRPM in a pair of gfn_to_pfn_caches, keyed on msrpm_base_pa (re-keying automatically when L1 switches to a different L2 with a different bitmap). Reads preserve the existing failure semantics: an unreadable bitmap means NESTED_EXIT_DONE (reflect to L1) on the intercept path and a failed VMRUN on the merge path. An 8-byte merge read is naturally aligned and a one-byte intercept read cannot cross a page, so each access hits exactly one cached page; a WARN in the accessor enforces that invariant. Signed-off-by: David Woodhouse <[email protected]> Assisted-by: Claude:claude-mythos-5 --- arch/x86/kvm/svm/nested.c | 46 +++++++++++++++++++++++++++++++++------ arch/x86/kvm/svm/svm.h | 7 ++++++ 2 files changed, 46 insertions(+), 7 deletions(-) diff --git a/arch/x86/kvm/svm/nested.c b/arch/x86/kvm/svm/nested.c index a8ee0a03d9f7..07609e4802f9 100644 --- a/arch/x86/kvm/svm/nested.c +++ b/arch/x86/kvm/svm/nested.c @@ -292,6 +292,34 @@ int __init nested_svm_init_msrpm_merge_offsets(void) * is optimized in that it only merges the parts where KVM MSR permission bitmap * may contain zero bits. */ +/* + * Read from L1's MSR permissions map through the per-page caches. + * @offset is a byte offset into the (two-page) MSRPM; the read must not + * cross a page boundary. msrpm_base_pa was validated page-aligned and + * in-bounds by nested_svm_check_controls(). + */ +static int nested_svm_read_l1_msrpm(struct vcpu_svm *svm, unsigned long offset, + void *val, unsigned int len) +{ + gpa_t base = svm->nested.ctl.msrpm_base_pa; + struct gfn_to_pfn_cache *gpc; + int idx; + + if (WARN_ON_ONCE(offset >= MSRPM_SIZE || + offset_in_page(offset) + len > PAGE_SIZE)) + return -EINVAL; + + gpc = &svm->nested.msrpm12_cache[offset >> PAGE_SHIFT]; + + idx = kvm_gpc_lock_page(gpc, base + (offset & PAGE_MASK)); + if (idx < 0) + return idx; + + memcpy(val, gpc->khva + offset_in_page(offset), len); + kvm_gpc_unlock(gpc, idx); + return 0; +} + static bool nested_svm_merge_msrpm(struct kvm_vcpu *vcpu) { struct vcpu_svm *svm = to_svm(vcpu); @@ -324,11 +352,9 @@ static bool nested_svm_merge_msrpm(struct kvm_vcpu *vcpu) for (i = 0; i < nested_svm_nr_msrpm_merge_offsets; i++) { const int p = nested_svm_msrpm_merge_offsets[i]; nsvm_msrpm_merge_t l1_val; - gpa_t gpa; - gpa = svm->nested.ctl.msrpm_base_pa + (p * sizeof(l1_val)); - - if (kvm_vcpu_read_guest(vcpu, gpa, &l1_val, sizeof(l1_val))) + if (nested_svm_read_l1_msrpm(svm, p * sizeof(l1_val), + &l1_val, sizeof(l1_val))) return false; msrpm02[p] = msrpm01[p] | l1_val; @@ -1488,6 +1514,7 @@ static void nested_svm_triple_fault(struct kvm_vcpu *vcpu) int svm_allocate_nested(struct vcpu_svm *svm) { struct page *vmcb02_page; + int i; if (svm->nested.initialized) return 0; @@ -1503,6 +1530,8 @@ int svm_allocate_nested(struct vcpu_svm *svm) goto err_free_vmcb02; kvm_gpc_init(&svm->nested.vmcb12_cache, svm->vcpu.kvm); + for (i = 0; i < ARRAY_SIZE(svm->nested.msrpm12_cache); i++) + kvm_gpc_init(&svm->nested.msrpm12_cache[i], svm->vcpu.kvm); svm->nested.initialized = true; return 0; @@ -1514,6 +1543,8 @@ int svm_allocate_nested(struct vcpu_svm *svm) void svm_free_nested(struct vcpu_svm *svm) { + int i; + if (!svm->nested.initialized) return; @@ -1524,6 +1555,8 @@ void svm_free_nested(struct vcpu_svm *svm) svm->nested.msrpm = NULL; kvm_gpc_deactivate(&svm->nested.vmcb12_cache); + for (i = 0; i < ARRAY_SIZE(svm->nested.msrpm12_cache); i++) + kvm_gpc_deactivate(&svm->nested.msrpm12_cache[i]); __free_page(virt_to_page(svm->nested.vmcb02.ptr)); svm->nested.vmcb02.ptr = NULL; @@ -1575,7 +1608,6 @@ void svm_leave_nested(struct kvm_vcpu *vcpu) static int nested_svm_exit_handled_msr(struct vcpu_svm *svm) { - gpa_t base = svm->nested.ctl.msrpm_base_pa; int write, bit_nr; u8 value, mask; u32 msr; @@ -1590,8 +1622,8 @@ static int nested_svm_exit_handled_msr(struct vcpu_svm *svm) if (bit_nr < 0) return NESTED_EXIT_DONE; - if (kvm_vcpu_read_guest(&svm->vcpu, base + bit_nr / BITS_PER_BYTE, - &value, sizeof(value))) + if (nested_svm_read_l1_msrpm(svm, bit_nr / BITS_PER_BYTE, + &value, sizeof(value))) return NESTED_EXIT_DONE; mask = BIT(write) << (bit_nr & (BITS_PER_BYTE - 1)); diff --git a/arch/x86/kvm/svm/svm.h b/arch/x86/kvm/svm/svm.h index cd0755919062..ac2f9731bf5d 100644 --- a/arch/x86/kvm/svm/svm.h +++ b/arch/x86/kvm/svm/svm.h @@ -221,6 +221,13 @@ struct svm_nested_state { /* Cached mapping of the vmcb12 page, keyed on vmcb12_gpa */ struct gfn_to_pfn_cache vmcb12_cache; + /* + * Cached mappings of the two pages of L1's MSR permissions map, + * keyed on ctl.msrpm_base_pa, for the vmcb02 merge and the + * per-intercept checks. + */ + struct gfn_to_pfn_cache msrpm12_cache[MSRPM_SIZE / PAGE_SIZE]; + /* * The MSR permissions map used for vmcb02, which is the merge result * of vmcb01 and vmcb12 -- 2.55.0

