[PATCH v2] RISC-V: KVM: Add minimal guest_memfd support
"Bingyu.Xian" <[email protected]> Thu, 30 Jul 2026 10:45:10 +0800
| Newsgroups | org.infradead.lists.kvm-riscv,org.infradead.lists.linux-riscv,org.kernel.vger.kvm |
|---|---|
| Message-ID | <[email protected]> |
Add minimal guest_memfd support for RISC-V KVM, following the same pattern as x86 and arm64. 1. Select KVM_GUEST_MEMFD in Kconfig 2. In kvm_riscv_mmu_map(), check kvm_slot_has_gmem() early, before taking mmap_read_lock(). For gmem memslots, resolve the PFN via kvm_gmem_get_pfn(), then acquire kvm->mmu_lock and jump to G-stage mapping, bypassing vma_lookup()/__kvm_faultin_pfn(). The gmem PFN is resolved before mmap_read_lock() so the mmu_lock taken for the mapping is never held while waiting for mmap_lock, which would deadlock against mmu_notifier invalidation (mmu_lock -> mmap_lock). 3. Skip kvm_release_faultin_page() for gmem pages; their lifetime is managed by the gmem inode and invalidation notifier. 4. Skip hva lookup and validity check in gstage_page_fault() for gmem, which does not have a userspace VMA backing it. Changes since RFC v1: - Add vcpu_exit.c hva lookup bypass for gmem slots - Call kvm_prepare_memory_fault_exit() on gmem fault failure - Respect KVM_MEM_READONLY for gmem slot writability - Skip THP adjustment for gmem (only PAGE_SIZE supported) - Skip dirty tracking for gmem (handled by gmem notifier) - Replace goto with is_gmem variable for cleaner flow Assisted-by: YuanSheng: deepseek-v4-pro Co-developed-by: Quan Zhou <[email protected]> Signed-off-by: Quan Zhou <[email protected]> Signed-off-by: Bingyu.Xian <[email protected]> --- arch/riscv/kvm/Kconfig | 1 + arch/riscv/kvm/mmu.c | 134 +++++++++++++++++++++++-------------- arch/riscv/kvm/vcpu_exit.c | 17 ++++- 3 files changed, 99 insertions(+), 53 deletions(-) diff --git a/arch/riscv/kvm/Kconfig b/arch/riscv/kvm/Kconfig index ec2cee0a39e0..77121100579f 100644 --- a/arch/riscv/kvm/Kconfig +++ b/arch/riscv/kvm/Kconfig @@ -29,6 +29,7 @@ config KVM select KVM_GENERIC_DIRTYLOG_READ_PROTECT select KVM_GENERIC_HARDWARE_ENABLING select KVM_MMIO + select KVM_GUEST_MEMFD select VIRT_XFER_TO_GUEST_WORK select SCHED_INFO select GUEST_PERF_EVENTS if PERF_EVENTS diff --git a/arch/riscv/kvm/mmu.c b/arch/riscv/kvm/mmu.c index d2a06a54be17..5d37d8a93a1d 100644 --- a/arch/riscv/kvm/mmu.c +++ b/arch/riscv/kvm/mmu.c @@ -551,6 +551,8 @@ int kvm_riscv_mmu_map(struct kvm_vcpu *vcpu, struct kvm_memory_slot *memslot, unsigned long vma_pagesize, mmu_seq; struct kvm_gstage gstage; struct page *page; + int max_order; + bool is_gmem; kvm_riscv_gstage_init(&gstage, kvm); @@ -568,60 +570,79 @@ int kvm_riscv_mmu_map(struct kvm_vcpu *vcpu, struct kvm_memory_slot *memslot, return ret; } - mmap_read_lock(current->mm); - - vma = vma_lookup(current->mm, hva); - if (unlikely(!vma)) { - kvm_err("Failed to find VMA for hva 0x%lx\n", hva); - mmap_read_unlock(current->mm); - return -EFAULT; - } + is_gmem = kvm_slot_has_gmem(memslot); + if (is_gmem) { + /* + * For guest_memfd-backed slots, bypass the userspace VMA lookup + * and get the PFN directly from the guest_memfd inode. + */ + ret = kvm_gmem_get_pfn(kvm, memslot, gfn, &hfn, &page, &max_order); + if (ret) { + kvm_prepare_memory_fault_exit(vcpu, gpa, PAGE_SIZE, + is_write, false, false); + return ret; + } - is_hugetlb = is_vm_hugetlb_page(vma); - if (is_hugetlb) - vma_pageshift = huge_page_shift(hstate_vma(vma)); - else - vma_pageshift = PAGE_SHIFT; - vma_pagesize = 1ULL << vma_pageshift; - if (logging || (vma->vm_flags & VM_PFNMAP)) vma_pagesize = PAGE_SIZE; - else if (is_hugetlb) - vma_pagesize = hugetlb_mapping_size(memslot, hva, vma_pagesize); + writable = !(memslot->flags & KVM_MEM_READONLY); + /* gmem faultin does not go through mmu notifier invalidation */ + mmu_seq = kvm->mmu_invalidate_seq; + } else { + mmap_read_lock(current->mm); - /* - * For hugetlb mappings, vma_pagesize might have been reduced from the - * VMA size to a smaller safe mapping size. - */ - if (vma_pagesize == PMD_SIZE || vma_pagesize == PUD_SIZE) - gfn = ALIGN_DOWN(gpa, vma_pagesize) >> PAGE_SHIFT; + vma = vma_lookup(current->mm, hva); + if (unlikely(!vma)) { + kvm_err("Failed to find VMA for hva 0x%lx\n", hva); + mmap_read_unlock(current->mm); + return -EFAULT; + } - /* - * Read mmu_invalidate_seq so that KVM can detect if the results of - * vma_lookup() or __kvm_faultin_pfn() become stale prior to acquiring - * kvm->mmu_lock. - * - * Rely on mmap_read_unlock() for an implicit smp_rmb(), which pairs - * with the smp_wmb() in kvm_mmu_invalidate_end(). - */ - mmu_seq = kvm->mmu_invalidate_seq; - mmap_read_unlock(current->mm); + is_hugetlb = is_vm_hugetlb_page(vma); + if (is_hugetlb) + vma_pageshift = huge_page_shift(hstate_vma(vma)); + else + vma_pageshift = PAGE_SHIFT; + vma_pagesize = 1ULL << vma_pageshift; + if (logging || (vma->vm_flags & VM_PFNMAP)) + vma_pagesize = PAGE_SIZE; + else if (is_hugetlb) + vma_pagesize = hugetlb_mapping_size(memslot, hva, vma_pagesize); - if (vma_pagesize != PUD_SIZE && - vma_pagesize != PMD_SIZE && - vma_pagesize != PAGE_SIZE) { - kvm_err("Invalid VMA page size 0x%lx\n", vma_pagesize); - return -EFAULT; - } + /* + * For hugetlb mappings, vma_pagesize might have been reduced from the + * VMA size to a smaller safe mapping size. + */ + if (vma_pagesize == PMD_SIZE || vma_pagesize == PUD_SIZE) + gfn = ALIGN_DOWN(gpa, vma_pagesize) >> PAGE_SHIFT; - hfn = __kvm_faultin_pfn(memslot, gfn, is_write ? FOLL_WRITE : 0, - &writable, &page); - if (hfn == KVM_PFN_ERR_HWPOISON) { - send_sig_mceerr(BUS_MCEERR_AR, (void __user *)hva, - vma_pageshift, current); - return 0; + /* + * Read mmu_invalidate_seq so that KVM can detect if the results of + * vma_lookup() or __kvm_faultin_pfn() become stale prior to acquiring + * kvm->mmu_lock. + * + * Rely on mmap_read_unlock() for an implicit smp_rmb(), which pairs + * with the smp_wmb() in kvm_mmu_invalidate_end(). + */ + mmu_seq = kvm->mmu_invalidate_seq; + mmap_read_unlock(current->mm); + + if (vma_pagesize != PUD_SIZE && + vma_pagesize != PMD_SIZE && + vma_pagesize != PAGE_SIZE) { + kvm_err("Invalid VMA page size 0x%lx\n", vma_pagesize); + return -EFAULT; + } + + hfn = __kvm_faultin_pfn(memslot, gfn, is_write ? FOLL_WRITE : 0, + &writable, &page); + if (hfn == KVM_PFN_ERR_HWPOISON) { + send_sig_mceerr(BUS_MCEERR_AR, (void __user *)hva, + vma_pageshift, current); + return 0; + } + if (is_error_noslot_pfn(hfn)) + return -EFAULT; } - if (is_error_noslot_pfn(hfn)) - return -EFAULT; /* * If logging is active then we allow writable pages only @@ -639,12 +660,20 @@ int kvm_riscv_mmu_map(struct kvm_vcpu *vcpu, struct kvm_memory_slot *memslot, * Check if we are backed by a THP and thus use block mapping if * possible. Hugetlb mappings already selected their target size above, * so do not promote them through the THP helper. + * + * Skip THP adjustment for gmem, which only supports PAGE_SIZE mappings + * and does not have a valid hva for follow_pte(). */ - if (!logging && !is_hugetlb && vma_pagesize == PAGE_SIZE) + if (!is_gmem && !logging && !is_hugetlb && vma_pagesize == PAGE_SIZE) vma_pagesize = transparent_hugepage_adjust(kvm, memslot, hva, &hfn, &gpa); if (writable) { - mark_page_dirty_in_slot(kvm, memslot, gfn); + /* + * Skip dirty page tracking for gmem; page dirtying is handled + * via the gmem invalidation notifier and page cache. + */ + if (!is_gmem) + mark_page_dirty_in_slot(kvm, memslot, gfn); ret = kvm_riscv_gstage_map_page(&gstage, pcache, gpa, hfn << PAGE_SHIFT, vma_pagesize, false, true, out_map); } else { @@ -658,7 +687,12 @@ int kvm_riscv_mmu_map(struct kvm_vcpu *vcpu, struct kvm_memory_slot *memslot, kvm_err("Failed to map in G-stage\n"); out_unlock: - kvm_release_faultin_page(kvm, page, ret, writable); + /* + * Skip kvm_release_faultin_page() for gmem pages; their lifetime is + * managed by the gmem inode and invalidation notifier. + */ + if (!is_gmem) + kvm_release_faultin_page(kvm, page, ret, writable); write_unlock(&kvm->mmu_lock); return ret; } diff --git a/arch/riscv/kvm/vcpu_exit.c b/arch/riscv/kvm/vcpu_exit.c index 28cf9b27bb07..5c381a4f9977 100644 --- a/arch/riscv/kvm/vcpu_exit.c +++ b/arch/riscv/kvm/vcpu_exit.c @@ -26,10 +26,21 @@ static int gstage_page_fault(struct kvm_vcpu *vcpu, struct kvm_run *run, fault_addr = ((gpa_t)trap->htval << 2) | (trap->stval & 0x3); gfn = fault_addr >> PAGE_SHIFT; memslot = gfn_to_memslot(vcpu->kvm, gfn); - hva = gfn_to_hva_memslot_prot(memslot, gfn, &writable); - if (kvm_is_error_hva(hva) || - (trap->scause == EXC_STORE_GUEST_PAGE_FAULT && !writable)) { + /* + * For guest_memfd-backed slots, bypass the hva lookup and validity check + * since gmem does not have a userspace VMA backing it. + */ + if (kvm_slot_has_gmem(memslot)) { + hva = 0; + writable = !(memslot->flags & KVM_MEM_READONLY); + } else { + hva = gfn_to_hva_memslot_prot(memslot, gfn, &writable); + } + + if (!kvm_slot_has_gmem(memslot) && + (kvm_is_error_hva(hva) || + (trap->scause == EXC_STORE_GUEST_PAGE_FAULT && !writable))) { switch (trap->scause) { case EXC_LOAD_GUEST_PAGE_FAULT: return kvm_riscv_vcpu_mmio_load(vcpu, run, -- 2.54.0 -- kvm-riscv mailing list [email protected] http://lists.infradead.org/mailman/listinfo/kvm-riscv