[PATCH 4/6] arm64: io: replace ARM erratum 832075 alternative with callback
From: Ada Couprie Diaz <hidden>
Date: 2026-09-28 13:31:31
Subsystem:
arm64 port (aarch64 architecture), the rest · Maintainers:
Catalin Marinas, Will Deacon, Linus Torvalds
`__raw_read{b,w,l,q}()` represent about 10k call sites
that need to be patched.
Commit 5afaa1fc1b32 ("arm64: add Cortex-A57 erratum 832075 workaround")
implements its workaround with alternative instructions,
adding 10k extra instructions growing the size of the image.
Implement and use `__io_arm_a57_patch_ladr()` as a callback alternative
instead, saving close to 40kB of image size with a defconfig.
Unsigned offset loads and ordered loads encode the size of the load and
the operand registers identically, but ordered loads cannot offset
the base address.
As the alternatives in device loads do not have offsets, we can safely
bitmask the loads and convert them to ordered load as per the erratum fix.
The callback needs to be added to the KVM NVHE namespace as `readl()`
is used in the vgic-v2 driver.
Cc: Andre Przywara <andre.przywara@arm.com>
Signed-off-by: Ada Couprie Diaz <redacted>
---
arch/arm64/include/asm/io.h | 27 +++++++++++++++------------
arch/arm64/kernel/image-vars.h | 1 +
arch/arm64/kernel/io.c | 33 +++++++++++++++++++++++++++++++++
3 files changed, 49 insertions(+), 12 deletions(-)
diff --git a/arch/arm64/include/asm/io.h b/arch/arm64/include/asm/io.h
index 0f1ca651c5c79..1791081d97dcb 100644
--- a/arch/arm64/include/asm/io.h
+++ b/arch/arm64/include/asm/io.h@@ -22,6 +22,9 @@ /* IO-specific callbacks for alternative patching. */ void __io_nvidia_olympus_patch_dmb(struct alt_instr *alt, __le32 *origptr, __le32 *updptr, int nr_inst); +void __io_arm_a57_patch_ladr(struct alt_instr *alt, __le32 *origptr, + __le32 *updptr, int nr_inst); + /* * Generic IO read/write. These perform native-endian accesses.
@@ -61,9 +64,9 @@ static __always_inline u8 __raw_readb(const volatile void __iomem *addr) asm volatile(ALTERNATIVE_CB("nop", ARM64_WORKAROUND_NVIDIA_OLYMPUS_1027, __io_nvidia_olympus_patch_dmb) - ALTERNATIVE("ldrb %w0, [%1]", - "ldarb %w0, [%1]", - ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE) + ALTERNATIVE_CB("ldrb %w0, [%1]", + ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE, + __io_arm_a57_patch_ladr) : "=r" (val) : "r" (addr)); return val; }
@@ -76,9 +79,9 @@ static __always_inline u16 __raw_readw(const volatile void __iomem *addr) asm volatile(ALTERNATIVE_CB("nop", ARM64_WORKAROUND_NVIDIA_OLYMPUS_1027, __io_nvidia_olympus_patch_dmb) - ALTERNATIVE("ldrh %w0, [%1]", - "ldarh %w0, [%1]", - ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE) + ALTERNATIVE_CB("ldrh %w0, [%1]", + ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE, + __io_arm_a57_patch_ladr) : "=r" (val) : "r" (addr)); return val; }
@@ -90,9 +93,9 @@ static __always_inline u32 __raw_readl(const volatile void __iomem *addr) asm volatile(ALTERNATIVE_CB("nop", ARM64_WORKAROUND_NVIDIA_OLYMPUS_1027, __io_nvidia_olympus_patch_dmb) - ALTERNATIVE("ldr %w0, [%1]", - "ldar %w0, [%1]", - ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE) + ALTERNATIVE_CB("ldr %w0, [%1]", + ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE, + __io_arm_a57_patch_ladr) : "=r" (val) : "r" (addr)); return val; }
@@ -104,9 +107,9 @@ static __always_inline u64 __raw_readq(const volatile void __iomem *addr) asm volatile(ALTERNATIVE_CB("nop", ARM64_WORKAROUND_NVIDIA_OLYMPUS_1027, __io_nvidia_olympus_patch_dmb) - ALTERNATIVE("ldr %0, [%1]", - "ldar %0, [%1]", - ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE) + ALTERNATIVE_CB("ldr %0, [%1]", + ARM64_WORKAROUND_DEVICE_LOAD_ACQUIRE, + __io_arm_a57_patch_ladr) : "=r" (val) : "r" (addr)); return val; }
diff --git a/arch/arm64/kernel/image-vars.h b/arch/arm64/kernel/image-vars.h
index a22519c5223b4..615b77ef403fe 100644
--- a/arch/arm64/kernel/image-vars.h
+++ b/arch/arm64/kernel/image-vars.h@@ -95,6 +95,7 @@ KVM_NVHE_ALIAS(alt_cb_patch_nops); KVM_NVHE_ALIAS(kvm_compute_ich_hcr_trap_bits); KVM_NVHE_ALIAS(kvm_patch_ich_vtr_el2); KVM_NVHE_ALIAS(__io_nvidia_olympus_patch_dmb); +KVM_NVHE_ALIAS(__io_arm_a57_patch_ladr); /* Global kernel state accessed by nVHE hyp code. */ KVM_NVHE_ALIAS(kvm_vgic_global_state);
diff --git a/arch/arm64/kernel/io.c b/arch/arm64/kernel/io.c
index 8ad3cd1773d20..f56bdf0a70822 100644
--- a/arch/arm64/kernel/io.c
+++ b/arch/arm64/kernel/io.c@@ -66,3 +66,36 @@ noinstr void __io_nvidia_olympus_patch_dmb(struct alt_instr *alt, __le32 *origpt updptr[0] = cpu_to_le32(aarch64_insn_gen_dmb(AARCH64_INSN_MB_OSH)); } EXPORT_SYMBOL(__io_nvidia_olympus_patch_dmb); + +/* + * Patch unsigned immediate loads to ordered loads for Arm erratum 832075. + * The immediate offset of the load MUST BE 0 for this to make any sense, + * as ordered loads do not encode any offset. + * + * This can patch 8, 16, 32 and 64 bits loads as they share the same encoding, + * with the two highest bits encoding size. + * See Arm ARM DDI 0487 C4.1 "Load/store register (unsigned immediate)" and + * "Load/store ordered" for the complete encodings. + */ +noinstr void __io_arm_a57_patch_ladr(struct alt_instr *alt, __le32 *origptr, + __le32 *updptr, int nr_inst) +{ + u32 orinst, altinst; + + BUG_ON(nr_inst != 1); + + orinst = le32_to_cpu(origptr[0]); + BUG_ON(!aarch64_insn_is_load_imm(orinst)); + BUG_ON((orinst & GENMASK(21, 10)) != 0); + + /* + * Preserve the size (31, 30) and registers (9,0) of the immediate load, + * as they are encoded identically for ordered loads. + */ + altinst = orinst & ~GENMASK(29, 10); + /* The value defined in insn.h includes the RES1 bits and o0. */ + altinst |= aarch64_insn_get_load_acq_value(); + + updptr[0] = cpu_to_le32(altinst); +} +EXPORT_SYMBOL(__io_arm_a57_patch_ladr);
--
2.43.0