From: Andrew Cooper <andrew.cooper3@citrix.com>
To: Xen-devel <xen-devel@lists.xenproject.org>
Cc: "Andrew Cooper" <andrew.cooper3@citrix.com>,
"Jan Beulich" <JBeulich@suse.com>,
"Roger Pau Monné" <roger.pau@citrix.com>
Subject: [PATCH] RFC x86/msr: Use WRMSRNS $imm when available
Date: Fri, 8 Aug 2025 23:20:13 +0100 [thread overview]
Message-ID: <20250808222013.1071291-1-andrew.cooper3@citrix.com> (raw)
Signed-off-by: Andrew Cooper <andrew.cooper3@citrix.com>
---
CC: Jan Beulich <JBeulich@suse.com>
CC: Roger Pau Monné <roger.pau@citrix.com>
This is on top of the FRED series for the wrmsrns() cleanup, but otherwise
unrelated.
The code generation isn't entirely ideal
Function old new delta
init_fred 255 274 +19
vmx_set_reg 248 256 +8
enter_state_helper.cold 1014 1018 +4
__start_xen 8893 8897 +4
but made worse by the the prior codegen for wrmsrns(MSR_STAR, ...) being mad:
mov $0xc0000081,%ecx
mov $0xe023e008,%edx
movabs $0xe023e00800000000,%rax
cs wrmsr
The two sources of code expansion come from the compiler not being able to
construct %eax and %edx separately, and not being able propagate constants.
Loading 0 is possibly common enough to warrant another specialisation where we
can use "a" (0), "d" (0) and forgo the MOV+SHR.
I'm probably overthinking things. The addition will be in the noise in
practice, and Intel are sure the advantage of MSR_IMM will not be.
---
xen/arch/x86/include/asm/alternative.h | 7 ++++
xen/arch/x86/include/asm/msr.h | 39 ++++++++++++++++++++-
xen/include/public/arch-x86/cpufeatureset.h | 1 +
3 files changed, 46 insertions(+), 1 deletion(-)
diff --git a/xen/arch/x86/include/asm/alternative.h b/xen/arch/x86/include/asm/alternative.h
index 0482bbf7cbf1..fe87b15ec72c 100644
--- a/xen/arch/x86/include/asm/alternative.h
+++ b/xen/arch/x86/include/asm/alternative.h
@@ -151,6 +151,13 @@ extern void alternative_instructions(void);
ALTERNATIVE(oldinstr, newinstr, feature) \
:: input )
+#define alternative_input_2(oldinstr, newinstr1, feature1, \
+ newinstr2, feature2, input...) \
+ asm_inline volatile ( \
+ ALTERNATIVE_2(oldinstr, newinstr1, feature1, \
+ newinstr2, feature2) \
+ :: input )
+
/* Like alternative_input, but with a single output argument */
#define alternative_io(oldinstr, newinstr, feature, output, input...) \
asm_inline volatile ( \
diff --git a/xen/arch/x86/include/asm/msr.h b/xen/arch/x86/include/asm/msr.h
index 01f510315ffe..434fcac854e1 100644
--- a/xen/arch/x86/include/asm/msr.h
+++ b/xen/arch/x86/include/asm/msr.h
@@ -38,9 +38,46 @@ static inline void wrmsrl(unsigned int msr, uint64_t val)
wrmsr(msr, lo, hi);
}
+/*
+ * Non-serialising WRMSR with a compile-time constant index, when available.
+ * Falls back to plain WRMSRNS, or to a serialising WRMSR.
+ */
+static always_inline void __wrmsrns_imm(uint32_t msr, uint64_t val)
+{
+ /*
+ * For best performance, WRMSRNS %r64, $msr is recommended. For
+ * compatibility, we need to fall back to plain WRMSRNS, or to WRMSR.
+ *
+ * The combined ABI is awkward, because WRMSRNS $imm takes a single r64,
+ * whereas WRMSR{,NS} takes a split edx:eax pair.
+ *
+ * Always use WRMSRNS %rax, $imm, because it has the most in common with
+ * the legacy forms. When MSR_IMM isn't available, emit setup logic for
+ * %ecx and %edx too.
+ */
+ alternative_input_2(
+ "mov $%c[msr], %%ecx\n\t"
+ "mov %%rax, %%rdx\n\t"
+ "shr $32, %%rdx\n\t"
+ ".byte 0x2e; wrmsr",
+
+ /* WRMSRNS %rax, $msr */
+ ".byte 0xc4,0xe7,0x7a,0xf6,0xc0; .long %c[msr]", X86_FEATURE_MSR_IMM,
+
+ "mov $%c[msr], %%ecx\n\t"
+ "mov %%rax, %%rdx\n\t"
+ "shr $32, %%rdx\n\t"
+ ".byte 0x0f,0x01,0xc6", X86_FEATURE_WRMSRNS,
+
+ [msr] "i" (msr), "a" (val) : "rcx", "rdx");
+}
+
/* Non-serialising WRMSR, when available. Falls back to a serialising WRMSR. */
-static inline void wrmsrns(uint32_t msr, uint64_t val)
+static always_inline void wrmsrns(uint32_t msr, uint64_t val)
{
+ if ( __builtin_constant_p(msr) )
+ return __wrmsrns_imm(msr, val);
+
/*
* WRMSR is 2 bytes. WRMSRNS is 3 bytes. Pad WRMSR with a redundant CS
* prefix to avoid a trailing NOP.
diff --git a/xen/include/public/arch-x86/cpufeatureset.h b/xen/include/public/arch-x86/cpufeatureset.h
index 9cd778586f10..af69cf3822eb 100644
--- a/xen/include/public/arch-x86/cpufeatureset.h
+++ b/xen/include/public/arch-x86/cpufeatureset.h
@@ -352,6 +352,7 @@ XEN_CPUFEATURE(MCDT_NO, 13*32+ 5) /*A MCDT_NO */
XEN_CPUFEATURE(UC_LOCK_DIS, 13*32+ 6) /* UC-lock disable */
/* Intel-defined CPU features, CPUID level 0x00000007:1.ecx, word 14 */
+XEN_CPUFEATURE(MSR_IMM, 14*32+ 5) /* {RD,WR}MSR $imm32 */
/* Intel-defined CPU features, CPUID level 0x00000007:1.edx, word 15 */
XEN_CPUFEATURE(AVX_VNNI_INT8, 15*32+ 4) /*A AVX-VNNI-INT8 Instructions */
--
2.39.5
next reply other threads:[~2025-08-08 22:20 UTC|newest]
Thread overview: 6+ messages / expand[flat|nested] mbox.gz Atom feed top
2025-08-08 22:20 Andrew Cooper [this message]
2025-08-11 8:16 ` [PATCH] RFC x86/msr: Use WRMSRNS $imm when available Jan Beulich
2025-08-11 9:50 ` Andrew Cooper
2025-08-11 10:06 ` Jan Beulich
2025-08-11 10:16 ` Andrew Cooper
2025-08-11 10:31 ` Jan Beulich
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20250808222013.1071291-1-andrew.cooper3@citrix.com \
--to=andrew.cooper3@citrix.com \
--cc=JBeulich@suse.com \
--cc=roger.pau@citrix.com \
--cc=xen-devel@lists.xenproject.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.