Kernel KVM virtualization development
 help / color / mirror / Atom feed
From: Paolo Bonzini <pbonzini@redhat.com>
To: linux-kernel@vger.kernel.org, kvm@vger.kernel.org
Cc: nsaenz@amazon.com, vkuznets@redhat.com, snambakam@linux.microsoft.com
Subject: [PATCH v2 21/28] KVM: x86/mmu: Take memory protection attributes into account during faults
Date: Fri, 18 Sep 2026 04:15:36 -0400	[thread overview]
Message-ID: <20260918081543.139871-22-pbonzini@redhat.com> (raw)
In-Reply-To: <20260918081543.139871-1-pbonzini@redhat.com>

From: Nicolas Saenz Julienne <nsaenz@amazon.com>

Take memory protection attributes when faulting guest memory.  Prohibited
memory accesses will cause a user-space -EFAULT exit just like private
memory accesses.  Userspace will either enable the access or bump it to
the guest as some kind of exception (e.g. a VTL return).

Since the struct kvm_page_fault already has the access type in PFERR_*
format, the check is done via the kvm_page_format permissions table.
This means that it supports naturally all page table format variants,
and it can even handle mode-based memory protection when the host
uses MBEC/GMET.  The only thing that needs some care is to build the
restricted ACC_* mask with the root page's own access mask as a base
(and not ACC_ALL).  Otherwise, supervisor mode execution would be
handled incorrectly on AMD processors with GMET.

To avoid spamming the trace buffer too much, the new trace event only
kicks in if memory protection attributes are present for the faulted gfn.

Signed-off-by: Nicolas Saenz Julienne <nsaenz@amazon.com>
Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
---
 arch/x86/kvm/mmu/mmu.c          | 49 +++++++++++++++++++++++++++++++++
 arch/x86/kvm/mmu/mmu_internal.h | 19 +++++++++++++
 arch/x86/kvm/mmu/mmutrace.h     | 36 ++++++++++++++++++++++++
 arch/x86/kvm/mmu/paging_tmpl.h  |  2 +-
 arch/x86/kvm/mmu/spte.h         | 11 ++------
 5 files changed, 107 insertions(+), 10 deletions(-)

diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c
index 797b18201d98..3fd4cc3c57a5 100644
--- a/arch/x86/kvm/mmu/mmu.c
+++ b/arch/x86/kvm/mmu/mmu.c
@@ -4642,6 +4642,50 @@ static int kvm_mmu_faultin_pfn_gmem(struct kvm_vcpu *vcpu,
 	return RET_PF_CONTINUE;
 }
 
+static inline unsigned kvm_get_gfn_protections(struct kvm_vcpu *vcpu, gfn_t gfn)
+{
+	struct kvm *kvm = vcpu->kvm;
+	unsigned int access = vcpu->arch.mmu->root_role.access;
+	unsigned long attrs = kvm_get_memory_attributes(kvm, gfn);
+	if (!attrs)
+		return access;
+
+	WARN_ON_ONCE(!kvm_mem_attributes_valid(kvm, attrs));
+
+	if (!kvm_mem_attributes_may_read(attrs))
+		access &= ~ACC_READ_MASK;
+	if (!kvm_mem_attributes_may_write(attrs))
+		access &= ~ACC_WRITE_MASK;
+	if (!kvm_mem_attributes_may_exec(attrs)) {
+		access &= ~ACC_EXEC_MASK;
+		if (shadow_xu_mask)
+			access &= ~ACC_USER_EXEC_MASK;
+	}
+
+	return access;
+}
+
+static int kvm_faultin_memory_protections(struct kvm_vcpu *vcpu,
+					  struct kvm_page_fault *fault)
+{
+	unsigned access;
+
+	/* Memory attributes don't apply to MMIO regions */
+	if (unlikely(!fault->slot))
+		return RET_PF_CONTINUE;
+
+	access = kvm_get_gfn_protections(vcpu, fault->gfn);
+	if (access == ACC_ALL)
+		return RET_PF_CONTINUE;
+
+	trace_kvm_faultin_memory_protections(vcpu, fault, access);
+	if (__permission_fault(vcpu->arch.mmu, access, fault))
+		return -EFAULT;
+
+	fault->host_access &= access;
+	return RET_PF_CONTINUE;
+}
+
 static int __kvm_mmu_faultin_pfn(struct kvm_vcpu *vcpu,
 				 struct kvm_page_fault *fault)
 {
@@ -4722,6 +4766,11 @@ static int kvm_mmu_faultin_pfn(struct kvm_vcpu *vcpu,
 	if (unlikely(!slot))
 		return kvm_handle_noslot_fault(vcpu, fault, access);
 
+	if (kvm_faultin_memory_protections(vcpu, fault)) {
+		kvm_mmu_prepare_memory_fault_exit(vcpu, fault);
+		return -EFAULT;
+	}
+
 	/*
 	 * Retry the page fault if the gfn hit a memslot that is being deleted
 	 * or moved.  This ensures any existing SPTEs for the old memslot will
diff --git a/arch/x86/kvm/mmu/mmu_internal.h b/arch/x86/kvm/mmu/mmu_internal.h
index 00215b9f309f..0bad60c5aab6 100644
--- a/arch/x86/kvm/mmu/mmu_internal.h
+++ b/arch/x86/kvm/mmu/mmu_internal.h
@@ -290,6 +290,25 @@ struct kvm_page_fault {
 	bool write_fault_to_shadow_pgtable;
 };
 
+/*
+ * Returns true if the access indicated by @fault is forbidden by the existing
+ * SPTE protections.
+ */
+static inline bool __permission_fault(struct kvm_mmu *mmu, unsigned access,
+				      struct kvm_page_fault *fault)
+{
+	unsigned pfec;
+
+	/*
+	 * RSVD is handled elsewhere, and is used for SMAP in the context
+	 * of accessing fmt.permissions[].  SPTEs never use PK or SS, as
+	 * they are not supported for shadow paging and irrelevant for TDP.
+	 */
+	pfec = fault->error_code & (
+		PFERR_WRITE_MASK | PFERR_USER_MASK | PFERR_FETCH_MASK);
+	return (mmu->fmt.permissions[pfec >> 1] >> access) & 1;
+}
+
 /*
  * Return values of handle_mmio_page_fault(), mmu.page_fault(), fast_page_fault(),
  * and of course kvm_mmu_do_page_fault().
diff --git a/arch/x86/kvm/mmu/mmutrace.h b/arch/x86/kvm/mmu/mmutrace.h
index 8354d9f39777..744f111644cb 100644
--- a/arch/x86/kvm/mmu/mmutrace.h
+++ b/arch/x86/kvm/mmu/mmutrace.h
@@ -8,6 +8,13 @@
 #undef TRACE_SYSTEM
 #define TRACE_SYSTEM kvmmmu
 
+#ifdef CREATE_TRACE_POINTS
+#define tracing_kvm_rip_read(vcpu) ({					\
+	typeof(vcpu) __vcpu = vcpu;					\
+	__vcpu->arch.guest_state_protected ? 0 : kvm_rip_read(__vcpu);	\
+	})
+#endif
+
 #define KVM_MMU_PAGE_FIELDS		\
 	__field(__u8, mmu_valid_gen)	\
 	__field(__u64, gfn)		\
@@ -447,6 +454,35 @@ TRACE_EVENT(
 		  __entry->gfn, __entry->spte, __entry->level, __entry->errno)
 );
 
+TRACE_EVENT(kvm_faultin_memory_protections,
+	TP_PROTO(struct kvm_vcpu *vcpu, struct kvm_page_fault *fault,
+		 unsigned access),
+	TP_ARGS(vcpu, fault, access),
+
+	TP_STRUCT__entry(
+		__field(unsigned int, vcpu_id)
+		__field(unsigned long, guest_rip)
+		__field(u64, fault_address)
+		__field(bool, write)
+		__field(bool, exec)
+		__field(unsigned, access)
+	),
+
+	TP_fast_assign(
+		__entry->vcpu_id = vcpu->vcpu_id;
+		__entry->guest_rip = tracing_kvm_rip_read(vcpu);
+		__entry->fault_address = fault->gfn;
+		__entry->write = fault->write;
+		__entry->exec = fault->exec;
+		__entry->access = access;
+	),
+
+	TP_printk("vcpu %d rip 0x%lx gfn 0x%016llx access %s protections 0x%x",
+		  __entry->vcpu_id, __entry->guest_rip, __entry->fault_address,
+		  __entry->exec ? "X" : (__entry->write ? "W" : "R"),
+		  __entry->access)
+);
+
 #endif /* _TRACE_KVMMMU_H */
 
 #undef TRACE_INCLUDE_PATH
diff --git a/arch/x86/kvm/mmu/paging_tmpl.h b/arch/x86/kvm/mmu/paging_tmpl.h
index e6ec14165f40..b031123fcb36 100644
--- a/arch/x86/kvm/mmu/paging_tmpl.h
+++ b/arch/x86/kvm/mmu/paging_tmpl.h
@@ -992,7 +992,7 @@ static int FNAME(sync_spte)(struct kvm_vcpu *vcpu, struct kvm_mmu_page *sp, int
 
 	sptep = &sp->spt[i];
 	spte = *sptep;
-	host_access = ACC_ALL;
+	host_access = kvm_get_gfn_protections(vcpu, gfn);
 	if (!(spte & shadow_host_writable_mask))
 		host_access &= ~ACC_WRITE_MASK;
 	slot = kvm_vcpu_gfn_to_memslot(vcpu, gfn);
diff --git a/arch/x86/kvm/mmu/spte.h b/arch/x86/kvm/mmu/spte.h
index 589f3954633e..14f322944671 100644
--- a/arch/x86/kvm/mmu/spte.h
+++ b/arch/x86/kvm/mmu/spte.h
@@ -491,7 +491,7 @@ static inline bool is_mmu_writable_spte(u64 spte)
 static inline bool spte_permission_fault(struct kvm_mmu *mmu, u64 spte,
 					 struct kvm_page_fault *fault)
 {
-	unsigned pfec, pte_access;
+	unsigned pte_access;
 
 	if (!is_shadow_present_pte(spte))
 		return true;
@@ -511,14 +511,7 @@ static inline bool spte_permission_fault(struct kvm_mmu *mmu, u64 spte,
 		pte_access |= spte & shadow_xu_mask ? ACC_USER_EXEC_MASK : 0;
 	}
 
-	/*
-	 * RSVD is handled elsewhere, and is used for SMAP in the context
-	 * of accessing fmt.permissions[].  SPTEs never use PK or SS, as
-	 * they are not supported for shadow paging and irrelevant for TDP.
-	 */
-	pfec = fault->error_code & (
-		PFERR_WRITE_MASK | PFERR_USER_MASK | PFERR_FETCH_MASK);
-	return (mmu->fmt.permissions[pfec >> 1] >> pte_access) & 1;
+	return __permission_fault(mmu, pte_access, fault);
 }
 
 /*
-- 
2.52.0



  parent reply	other threads:[~2026-09-18  8:16 UTC|newest]

Thread overview: 44+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-18  8:15 [PATCH v2 00/28] KVM: x86: Introduce memory protection attributes Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 01/28] KVM: selftests: Take into account mixed memory fault flags Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 02/28] KVM: Define and communicate KVM_EXIT_MEMORY_FAULT RWX flags to userspace Paolo Bonzini
2026-09-18  8:27   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 03/28] KVM: selftests: Test address translation for Hyper-V direct L2 hypercalls Paolo Bonzini
2026-09-18  8:30   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 04/28] KVM: apply nGPA->GPA translation to KVM_HC_CLOCK_PAIRING Paolo Bonzini
2026-09-18  8:34   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 05/28] KVM: x86: Introduce memory fault on invalid hypercalls reads/writes Paolo Bonzini
2026-09-18  8:33   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 06/28] KVM: selftests: test hypercall memory fault exits Paolo Bonzini
2026-09-18  8:24   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 07/28] KVM: x86/mmu: intersect writability from __kvm_faultin_pfn with fault->map_writable Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 08/28] KVM: x86/mmu: Extend map_writable to a full ACC_* mask Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 09/28] KVM: x86/mmu: Init memslot hugepage information for non-private_mem VMs too Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 10/28] KVM: pass kvm == NULL case to kvm_arch_has_private_mem Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 11/28] KVM: adjust for presence of more than one attribute Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 12/28] KVM: Introduce NR/NW/NX memory attributes Paolo Bonzini
2026-09-18  8:36   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 13/28] KVM: Include memory protections in result of gfn->hva conversion Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 14/28] KVM: Introduce kvm_fetch_guest_page() and use it for x86 Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 15/28] KVM: Take memory protections into account for memory read/write/fetch Paolo Bonzini
2026-09-18  8:34   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 16/28] KVM: Encapsulate memattrs array into anonymous struct Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 17/28] KVM: Introduce kvm_check_gen()/kvm_memslots_check_gen() Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 18/28] KVM: Introduce a generation number for memory attributes Paolo Bonzini
2026-09-18  8:39   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 19/28] KVM: Take memory protections into account for accesses with cached gfn->hva Paolo Bonzini
2026-09-18  8:38   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 20/28] KVM: pfncache: Fail to refresh if it contains memory protections Paolo Bonzini
2026-09-18  8:15 ` Paolo Bonzini [this message]
2026-09-18  8:46   ` [PATCH v2 21/28] KVM: x86/mmu: Take memory protection attributes into account during faults sashiko-bot
2026-09-18  8:15 ` [PATCH v2 22/28] KVM: x86/mmu: Issue memory fault exit if walk failed due to memory attribute Paolo Bonzini
2026-09-18  8:37   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 23/28] KVM: x86/mmu: Do not update accessed/dirty if guest PTE is read-only Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 24/28] KVM: x86/mmu: Do not prefetch sptes on gfns backed by memory attributes Paolo Bonzini
2026-09-18  8:36   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 25/28] KVM: x86/mmu: Obsolete all roots if memattr contains gPTEs Paolo Bonzini
2026-09-18  8:43   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 26/28] KVM: x86: selftests: Introduce memory protection attributes test Paolo Bonzini
2026-09-18  8:39   ` sashiko-bot
2026-09-18  8:15 ` [PATCH v2 27/28] KVM: x86: selftests: Introduce memory attributes PTE test Paolo Bonzini
2026-09-18  8:15 ` [PATCH v2 28/28] KVM: x86: selftests: Introduce memory attributes side-channel tests Paolo Bonzini
2026-09-18  8:43   ` sashiko-bot

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260918081543.139871-22-pbonzini@redhat.com \
    --to=pbonzini@redhat.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=nsaenz@amazon.com \
    --cc=snambakam@linux.microsoft.com \
    --cc=vkuznets@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox