Re: [PATCH v7 08/11] KVM: TDX: Get/put PAMT pages when (un)mapping private memory

Nikolay Borisov <[email protected]> Wed, 22 Jul 2026 16:46:58 +0300
Newsgroups dev.linux.lists.linux-coco,org.kernel.vger.kvm,org.kernel.vger.linux-doc,org.kernel.vger.linux-kernel
Message-ID <[email protected]>

On 7/18/26 04:44, Rick Edgecombe wrote:
> From: "Kirill A. Shutemov" <[email protected]>
> 
> Add Dynamic PAMT support to KVM's S-EPT MMU by "getting" a PAMT page when
> adding guest memory (PAGE.ADD or PAGE.AUG), and "putting" the page when
> removing guest memory (PAGE.REMOVE).
> 
> To access the per-vCPU PAMT caches without plumbing @vcpu throughout the
> TDP MMU, begrudgingly use kvm_get_running_vcpu() to get the vCPU, and bug
> the VM if KVM attempts to set an S-EPT leaf without an active vCPU.  KVM
> only supports creating _new_ mappings in page (pre)fault paths, all of
> which require an active vCPU.
> 
> The PAMT memory holds metadata for TDX protected memory. With Dynamic
> PAMT, PAMT_4K is allocated on demand. The kernel supplies the TDX module
> with a few pages that cover 2MB of host physical memory.
> 
> Releases are balanced via tdx_pamt_put(): every control-page free goes
> through tdx_free_control_page(), and guest data pages are put directly on
> the successful tdh_mem_page_remove() path and in the
> tdx_mem_page_add/aug() error path.
> 
> Signed-off-by: Kirill A. Shutemov <[email protected]>
> Co-developed-by: Sean Christopherson <[email protected]>
> Signed-off-by: Sean Christopherson <[email protected]>
> [rick: enhance log, reviewing, rebase, with help from AI tooling]
> Co-developed-by: Rick Edgecombe <[email protected]>
> Signed-off-by: Rick Edgecombe <[email protected]>
> Reviewed-by: Binbin Wu <[email protected]>
> Reviewed-by: Tony Lindgren <[email protected]>
> ---
> v7:
>   - Don't do to_tdx() before NULL check for readability (Binbin, Sean)
>   - Fixup tags (Sean)
>   - Export tdx_supports_dynamic_pamt() since it's used in KVM here and no
>     longer an static inline.
> v6:
>   - Don't have topup op take a min param (Yan, Sean)
>   - Make log match style of the rest of the series
>   - Adjustments from dropping error helper patches
> ---
>   arch/x86/include/asm/kvm-x86-ops.h |  1 +
>   arch/x86/include/asm/kvm_host.h    |  2 +
>   arch/x86/kvm/mmu/mmu.c             |  4 ++
>   arch/x86/kvm/vmx/tdx.c             | 63 ++++++++++++++++++++++++++----
>   arch/x86/kvm/vmx/tdx.h             |  2 +
>   arch/x86/virt/vmx/tdx/tdx.c        |  1 +
>   6 files changed, 65 insertions(+), 8 deletions(-)
> 
> diff --git a/arch/x86/include/asm/kvm-x86-ops.h b/arch/x86/include/asm/kvm-x86-ops.h
> index 83dc5086138b3..588563dfe88d5 100644
> --- a/arch/x86/include/asm/kvm-x86-ops.h
> +++ b/arch/x86/include/asm/kvm-x86-ops.h
> @@ -98,6 +98,7 @@ KVM_X86_OP_OPTIONAL_RET0(tdp_has_smep)
>   KVM_X86_OP(load_mmu_pgd)
>   KVM_X86_OP_OPTIONAL_RET0(set_external_spte)
>   KVM_X86_OP_OPTIONAL(free_external_spt)
> +KVM_X86_OP_OPTIONAL_RET0(topup_external_cache)
>   KVM_X86_OP(has_wbinvd_exit)
>   KVM_X86_OP(get_l2_tsc_offset)
>   KVM_X86_OP(get_l2_tsc_multiplier)
> diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_host.h
> index 5f6c1ce9673b7..1c706e2d773b0 100644
> --- a/arch/x86/include/asm/kvm_host.h
> +++ b/arch/x86/include/asm/kvm_host.h
> @@ -1922,6 +1922,8 @@ struct kvm_x86_ops {
>   	/* Update external page tables for page table about to be freed. */
>   	void (*free_external_spt)(struct kvm *kvm, struct kvm_mmu_page *sp);
>   
> +	int (*topup_external_cache)(struct kvm_vcpu *vcpu, int min_nr_spts);
> +
>   
>   	bool (*has_wbinvd_exit)(void);
>   
> diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c
> index 234d0a95abf53..6dab99654f170 100644
> --- a/arch/x86/kvm/mmu/mmu.c
> +++ b/arch/x86/kvm/mmu/mmu.c
> @@ -614,6 +614,10 @@ static int mmu_topup_memory_caches(struct kvm_vcpu *vcpu, bool maybe_indirect)
>   					       PT64_ROOT_MAX_LEVEL);
>   		if (r)
>   			return r;
> +
> +		r = kvm_x86_call(topup_external_cache)(vcpu, PT64_ROOT_MAX_LEVEL);
> +		if (r)
> +			return r;
>   	}
>   	r = kvm_mmu_topup_memory_cache(&vcpu->arch.mmu_shadow_page_cache,
>   				       PT64_ROOT_MAX_LEVEL);
> diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c
> index f6c01eab8113b..c3b1d1f056cea 100644
> --- a/arch/x86/kvm/vmx/tdx.c
> +++ b/arch/x86/kvm/vmx/tdx.c
> @@ -681,6 +681,8 @@ int tdx_vcpu_create(struct kvm_vcpu *vcpu)
>   	if (!irqchip_split(vcpu->kvm))
>   		return -EINVAL;
>   
> +	tdx_init_pamt_cache(&tdx->pamt_cache);
> +
>   	fpstate_set_confidential(&vcpu->arch.guest_fpu);
>   	vcpu->arch.apic->guest_apic_protected = true;
>   	INIT_LIST_HEAD(&tdx->vt.pi_wakeup_list);
> @@ -866,6 +868,8 @@ void tdx_vcpu_free(struct kvm_vcpu *vcpu)
>   	struct vcpu_tdx *tdx = to_tdx(vcpu);
>   	int i;
>   
> +	tdx_free_pamt_cache(&tdx->pamt_cache);
> +
>   	if (vcpu->cpu != -1) {
>   		KVM_BUG_ON(tdx->state == VCPU_TD_STATE_INITIALIZED, vcpu->kvm);
>   		tdx_flush_vp_on_cpu(vcpu);
> @@ -1621,6 +1625,16 @@ void tdx_load_mmu_pgd(struct kvm_vcpu *vcpu, hpa_t root_hpa, int pgd_level)
>   	td_vmcs_write64(to_tdx(vcpu), SHARED_EPT_POINTER, root_hpa);
>   }
>   
> +static int tdx_topup_external_pamt_cache(struct kvm_vcpu *vcpu, int min_nr_spts)
> +{
> +	/*
> +	 * Don't cover the root SPT, but cover a possible 4KB private
> +	 * page in addition to the SPTs. So -1 to exclude the root
> +	 * SPT, and +1 for the guest page cancel out.
> +	 */

That comment here doesn't seem to correspond to the code in anyway. 
There are no +-1 adjustments.
> +	return tdx_topup_pamt_cache(&to_tdx(vcpu)->pamt_cache, min_nr_spts);
> +}

<snip>