On Wed, 23 Sept 2026 at 16:17, Lorenzo Stoakes (ARM) <[email protected]> wrote:
>
> When stage 2 page tables fault the net result may either be that a page is
> mapped, an error occurred or the fault should be retried (-EAGAIN).
>
> When a fault succeeds it may be upgraded to a PMD size via
> transparent_hugepage_adjust().
>
> In order to support KVM pre-faulting the outcome of the fault and the
> mapping size must be recorded.
>
> Track the mapping size via new kvm_s2_fault_result struct, which is
> threaded through gmem_abort(), user_mem_abort() and kvm_s2_fault_map().
>
> PKVM and SEA aren't relevant to synthetic pre-faulting so neither
> kvm_inject_sea() nor pkvm_mem_abort() are altered.
>
> Actual hardware faulting doesn't require this information, so
> kvm_handle_guest_abort() simply passes NULL kvm_s2_fault_result to
> gmem_abort() and user_mem_abort().
>
> A non-NULL result also tells the abort handlers that the caller is
> pre-faulting rather than a vcpu, in which case -EAGAIN is propagated to the
> caller to allow the pre-fault to be retried.
>
> Likewise, only prepare a KVM_EXIT_MEMORY_FAULT in gmem_abort() when a real
> fault is being handled, as this updates vcpu state.
>
> No functional change intended.
>
> Suggested-by: Vincent Donnefort <[email protected]>
> Reviewed-by: Oliver Upton <[email protected]>
> Signed-off-by: Lorenzo Stoakes (ARM) <[email protected]>

Reviewed-by: Fuad Tabba <[email protected]>

Cheers,
/fuad

> ---
>  arch/arm64/kvm/mmu.c | 43 +++++++++++++++++++++++++++++++------------
>  1 file changed, 31 insertions(+), 12 deletions(-)
>
> diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
> index 9d799f7c1e62..ee2a4f0faa14 100644
> --- a/arch/arm64/kvm/mmu.c
> +++ b/arch/arm64/kvm/mmu.c
> @@ -1659,6 +1659,10 @@ struct kvm_s2_fault_desc {
>         struct kvm_s2_mmu       *mmu;
>  };
>
> +struct kvm_s2_fault_result {
> +       unsigned long mapping_size;
> +};
> +
>  static bool kvm_s2_fault_is_perm(const struct kvm_s2_fault_desc *s2fd)
>  {
>         return esr_fsc_is_permission_fault(s2fd->esr);
> @@ -1684,7 +1688,8 @@ static u64 kvm_s2_perm_fault_granule(const struct 
> kvm_s2_fault_desc *s2fd)
>         return BIT(ARM64_HW_PGTABLE_LEVEL_SHIFT(level));
>  }
>
> -static int gmem_abort(const struct kvm_s2_fault_desc *s2fd)
> +static int gmem_abort(const struct kvm_s2_fault_desc *s2fd,
> +                     struct kvm_s2_fault_result *result)
>  {
>         bool write_fault, exec_fault;
>         bool perm_fault = kvm_s2_fault_is_perm(s2fd);
> @@ -1728,8 +1733,10 @@ static int gmem_abort(const struct kvm_s2_fault_desc 
> *s2fd)
>
>         ret = kvm_gmem_get_pfn(kvm, s2fd->memslot, gfn, &pfn, &page, NULL);
>         if (ret) {
> -               kvm_prepare_memory_fault_exit(s2fd->vcpu, s2fd->fault_ipa, 
> PAGE_SIZE,
> -                                             write_fault, exec_fault, false);
> +               /* If result is non-NULL this is a synthetic fault. */
> +               if (!result)
> +                       kvm_prepare_memory_fault_exit(s2fd->vcpu, 
> s2fd->fault_ipa, PAGE_SIZE,
> +                                                     write_fault, 
> exec_fault, false);
>                 kfree(mapping);
>                 return ret;
>         }
> @@ -1784,7 +1791,13 @@ static int gmem_abort(const struct kvm_s2_fault_desc 
> *s2fd)
>         if ((prot & KVM_PGTABLE_PROT_W) && !ret)
>                 mark_page_dirty_in_slot(kvm, s2fd->memslot, gfn);
>
> -       return ret != -EAGAIN ? ret : 0;
> +       if (ret == -EAGAIN)
> +               return result ? ret : 0;
> +
> +       if (result && !ret)
> +               result->mapping_size = PAGE_SIZE;
> +
> +       return ret;
>  }
>
>  struct kvm_s2_fault_vma_info {
> @@ -2108,7 +2121,8 @@ static int kvm_s2_fault_compute_prot(const struct 
> kvm_s2_fault_desc *s2fd,
>  static int kvm_s2_fault_map(const struct kvm_s2_fault_desc *s2fd,
>                             const struct kvm_s2_fault_vma_info *s2vi,
>                             enum kvm_pgtable_prot prot,
> -                           void *memcache)
> +                           void *memcache,
> +                           struct kvm_s2_fault_result *result)
>  {
>         enum kvm_pgtable_walk_flags flags = KVM_PGTABLE_WALK_SHARED;
>         struct kvm_guest_s2_mapping *mapping = NULL;
> @@ -2209,12 +2223,17 @@ static int kvm_s2_fault_map(const struct 
> kvm_s2_fault_desc *s2fd,
>                 mark_page_dirty_in_slot(kvm, s2fd->memslot,
>                                         gpa_to_gfn(canonical_ipa));
>
> -       if (ret != -EAGAIN)
> -               return ret;
> -       return 0;
> +       if (ret == -EAGAIN)
> +               return result ? ret : 0;
> +
> +       if (result && !ret)
> +               result->mapping_size = mapping_size;
> +
> +       return ret;
>  }
>
> -static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd)
> +static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd,
> +                         struct kvm_s2_fault_result *result)
>  {
>         bool perm_fault = kvm_s2_fault_is_perm(s2fd);
>         struct kvm_s2_fault_vma_info s2vi = {};
> @@ -2253,7 +2272,7 @@ static int user_mem_abort(const struct 
> kvm_s2_fault_desc *s2fd)
>                 return ret;
>         }
>
> -       return kvm_s2_fault_map(s2fd, &s2vi, prot, memcache);
> +       return kvm_s2_fault_map(s2fd, &s2vi, prot, memcache, result);
>  }
>
>  /* Resolve the access fault by making the page young again. */
> @@ -2532,9 +2551,9 @@ int kvm_handle_guest_abort(struct kvm_vcpu *vcpu)
>                                 !kvm_s2_fault_is_exec(&s2fd));
>
>                 if (kvm_slot_has_gmem(memslot))
> -                       ret = gmem_abort(&s2fd);
> +                       ret = gmem_abort(&s2fd, NULL);
>                 else
> -                       ret = user_mem_abort(&s2fd);
> +                       ret = user_mem_abort(&s2fd, NULL);
>         }
>
>         if (ret == 0)
>
> --
> 2.55.0
>

Reply via email to