Thread (20 messages) 20 messages, 7 authors, 3d ago

[RFC 11/12] arm64: tlbid: Pass domain to TLBI instructions

flat view

From: Kristina Martšenko <hidden>
Date: 2026-10-01 11:08:21
Also in: kvmarm, linux-acpi, linux-efi
Subsystem: arm64 port (aarch64 architecture), kernel virtual machine for arm64 (kvm/arm64), the rest · Maintainers: Catalin Marinas, Will Deacon, Marc Zyngier, Oliver Upton, Linus Torvalds

Update the TLB invalidation interfaces in tlbflush.h to pass a TLBI
domain to TLBI instructions (when available). This may improve
performance by sending the TLBI to only the subset of CPUs that needs
it.

Only use TLBI domains for user mappings, as the kernel is mapped on all
CPUs. Use the mm domain bitmap to choose the smallest domain that
contains all the CPUs that an mm has been installed on.

Most invalidations need to use the TLBIP instruction in order to use a
TLBI domain. The exception is TLBI ASID* (flush_tlb_mm()) which can pass
a domain to a regular TLBI instruction.

To avoid runtime checks on the TLBI domain, use TLBIP instructions for all
invalidations (except local/non-shareable) when supported. When no
domain is needed, callers pass 0 which is the fallback domain
broadcasting to the entire Inner Shareable domain. Local invalidations
continue to use TLBI instructions since the TLBID architecture does not
have non-shareable versions of TLBIP instructions.

The mm domain bitmap may be updated concurrently by another CPU while it
is being read. Use a DSB to ensure that invalidations don't miss any CPU
that has relevant mappings.

Assisted-by: LLM
Signed-off-by: Kristina Martšenko <redacted>
---
 arch/arm64/include/asm/mmu.h      |  2 +
 arch/arm64/include/asm/tlbflush.h | 90 ++++++++++++++++++++++---------
 arch/arm64/kvm/hyp/nvhe/tlb.c     |  2 +-
 arch/arm64/kvm/hyp/vhe/tlb.c      |  2 +-
 arch/arm64/mm/tlbid.c             | 36 +++++++++++++
 5 files changed, 104 insertions(+), 28 deletions(-)
diff --git a/arch/arm64/include/asm/mmu.h b/arch/arm64/include/asm/mmu.h
index fc4e759f9f64..ae43a91ddabc 100644
--- a/arch/arm64/include/asm/mmu.h
+++ b/arch/arm64/include/asm/mmu.h
@@ -111,10 +111,12 @@ extern bool page_alloc_available;
 int init_tlbid_context(struct mm_struct *mm);
 void destroy_tlbid_context(struct mm_struct *mm);
 void update_tlbid_domains(struct mm_struct *mm, unsigned int cpu);
+u16 get_tlbid_domain(struct mm_struct *mm);
 #else
 static inline int init_tlbid_context(struct mm_struct *mm) { return 0; }
 static inline void destroy_tlbid_context(struct mm_struct *mm) {}
 static inline void update_tlbid_domains(struct mm_struct *mm, unsigned int cpu) {}
+static inline u16 get_tlbid_domain(struct mm_struct *mm) { return 0; }
 #endif
 
 #endif	/* !__ASSEMBLER__ */
diff --git a/arch/arm64/include/asm/tlbflush.h b/arch/arm64/include/asm/tlbflush.h
index dac35568570f..864d7459d5be 100644
--- a/arch/arm64/include/asm/tlbflush.h
+++ b/arch/arm64/include/asm/tlbflush.h
@@ -61,7 +61,9 @@
 	}									\
 } while (0)
 
+#define TLBI_ASID_MASK		GENMASK_ULL(63, 48)
 #define TLBI_TLBID_MASK		GENMASK_ULL(15, 0)
+#define TLBIP_ADDR_MASK		GENMASK_U128(107, 64)
 
 /* This macro creates a properly formatted VA operand for the TLBI */
 #define __TLBI_VADDR(addr, asid)				\
@@ -144,6 +146,7 @@ static inline void sme_dvmsync_batch(void)
  * in asm/stage2_pgtable.h.
  */
 #define TLBI_TTL_MASK		GENMASK_ULL(47, 44)
+#define TLBIP_TTL64		BIT(32)
 
 #define TLBI_TTL_UNKNOWN	INT_MAX
 
@@ -219,22 +222,38 @@ static __always_inline void ipas2e1is(bool tlbip, u128 arg)
 }
 
 static __always_inline void __tlbi_level_asid(tlbi_op op, u64 addr, u32 level,
-					      u16 asid)
+					      u16 asid, bool local, int domain)
 {
-	u64 arg = __TLBI_VADDR(addr, asid);
+	bool tlbip = system_supports_tlbid() && !local;
+	u128 arg = 0;
+
+	if (tlbip) {
+		arg |= (((u128)addr >> 12) << 64) & TLBIP_ADDR_MASK;
+		arg |= FIELD_PREP(TLBI_ASID_MASK, asid);
+		arg |= domain;
+	} else {
+		arg |= __TLBI_VADDR(addr, asid);
+	}
 
 	if (alternative_has_cap_unlikely(ARM64_HAS_ARMv8_4_TTL) && level <= 3) {
 		u64 ttl = level | (get_trans_granule() << 2);
+		arg |= FIELD_PREP(TLBI_TTL_MASK, ttl);
 
-		FIELD_MODIFY(TLBI_TTL_MASK, &arg, ttl);
+		if (tlbip)
+			arg |= TLBIP_TTL64;
 	}
 
-	op(false, arg);
+	op(tlbip, arg);
 }
 
 static inline void __tlbi_level(tlbi_op op, u64 addr, u32 level)
 {
-	__tlbi_level_asid(op, addr, level, 0);
+	__tlbi_level_asid(op, addr, level, 0, false, 0);
+}
+
+static inline void __tlbi_level_local(tlbi_op op, u64 addr, u32 level)
+{
+	__tlbi_level_asid(op, addr, level, 0, true, -1);
 }
 
 /*
@@ -418,12 +437,13 @@ static inline void flush_tlb_all(void)
 
 static inline void flush_tlb_mm(struct mm_struct *mm)
 {
-	unsigned long asid;
+	unsigned long arg;
 
 	dsb(ishst);
-	asid = __TLBI_VADDR(0, ASID(mm));
-	__tlbi(aside1is, asid);
-	__tlbi_user(aside1is, asid);
+	arg = __TLBI_VADDR(0, ASID(mm));
+	arg |= get_tlbid_domain(mm);
+	__tlbi(aside1is, arg);
+	__tlbi_user(aside1is, arg);
 	__tlbi_sync_s1ish(mm);
 	mmu_notifier_arch_invalidate_secondary_tlbs(mm, 0, -1UL);
 }
@@ -534,24 +554,38 @@ static __always_inline void ripas2e1is(bool tlbip, u128 arg)
 
 static __always_inline void __tlbi_range(tlbi_op op, u64 addr,
 					 u16 asid, int scale, int num,
-					 u32 level, bool lpa2)
+					 u32 level, bool lpa2, bool local,
+					 int domain)
 {
-	u64 arg = 0;
+	bool tlbip = system_supports_tlbid() && !local;
+	u128 arg = 0;
+
+	if (tlbip) {
+		arg |= (((u128)addr >> 12) << 64) & TLBIP_ADDR_MASK;
+		arg |= domain;
+	} else {
+		arg |= FIELD_PREP(TLBIR_BADDR_MASK, addr >> (lpa2 ? 16 : PAGE_SHIFT));
+	}
+
+	if (level <= 3) {
+		arg |= FIELD_PREP(TLBIR_TTL_MASK, level);
+		if (tlbip)
+			arg |= TLBIP_TTL64;
+	}
 
-	arg |= FIELD_PREP(TLBIR_BADDR_MASK, addr >> (lpa2 ? 16 : PAGE_SHIFT));
-	arg |= FIELD_PREP(TLBIR_TTL_MASK, level > 3 ? 0 : level);
 	arg |= FIELD_PREP(TLBIR_NUM_MASK, num);
 	arg |= FIELD_PREP(TLBIR_SCALE_MASK, scale);
 	arg |= FIELD_PREP(TLBIR_TG_MASK, get_trans_granule());
 	arg |= FIELD_PREP(TLBIR_ASID_MASK, asid);
 
-	op(false, arg);
+	op(tlbip, arg);
 }
 
 static __always_inline void __flush_tlb_range_op(tlbi_op lop, tlbi_op rop,
 						 u64 start, size_t pages,
 						 u64 stride, u16 asid,
-						 u32 level, bool lpa2)
+						 u32 level, bool lpa2,
+						 bool local, int domain)
 {
 	u64 addr = start, end = start + pages * PAGE_SIZE;
 	int scale = 3;
@@ -569,23 +603,24 @@ static __always_inline void __flush_tlb_range_op(tlbi_op lop, tlbi_op rop,
 
 		num = __TLBI_RANGE_NUM(pages, scale);
 		if (num >= 0) {
-			__tlbi_range(rop, addr, asid, scale, num, level, lpa2);
+			__tlbi_range(rop, addr, asid, scale, num, level, lpa2,
+				     local, domain);
 			addr += __TLBI_RANGE_PAGES(num, scale) << PAGE_SHIFT;
 		}
 
 		scale--;
 		continue;
 invalidate_one:
-		__tlbi_level_asid(lop, addr, level, asid);
+		__tlbi_level_asid(lop, addr, level, asid, local, domain);
 		addr += stride;
 	}
 }
 
-#define __flush_s1_tlb_range_op(op, start, pages, stride, asid, tlb_level) \
-	__flush_tlb_range_op(op, r##op, start, pages, stride, asid, tlb_level, lpa2_is_enabled())
+#define __flush_s1_tlb_range_op(op, start, pages, stride, asid, tlb_level, local, domain) \
+	__flush_tlb_range_op(op, r##op, start, pages, stride, asid, tlb_level, lpa2_is_enabled(), local, domain)
 
 #define __flush_s2_tlb_range_op(op, start, pages, stride, tlb_level) \
-	__flush_tlb_range_op(op, r##op, start, pages, stride, 0, tlb_level, kvm_lpa2_is_enabled())
+	__flush_tlb_range_op(op, r##op, start, pages, stride, 0, tlb_level, kvm_lpa2_is_enabled(), false, 0)
 
 static inline bool __flush_tlb_range_limit_excess(unsigned long pages,
 						  unsigned long stride)
@@ -626,6 +661,7 @@ static __always_inline void __do_flush_tlb_range(struct vm_area_struct *vma,
 {
 	struct mm_struct *mm = vma->vm_mm;
 	unsigned long asid, pages;
+	int domain;
 
 	pages = (end - start) >> PAGE_SHIFT;
 
@@ -634,21 +670,23 @@ static __always_inline void __do_flush_tlb_range(struct vm_area_struct *vma,
 		return;
 	}
 
-	if (!(flags & TLBF_NOBROADCAST))
+	if (!(flags & TLBF_NOBROADCAST)) {
 		dsb(ishst);
-	else
+		domain = get_tlbid_domain(mm);
+	} else {
 		dsb(nshst);
+	}
 
 	asid = ASID(mm);
 
 	switch (flags & (TLBF_NOWALKCACHE | TLBF_NOBROADCAST)) {
 	case TLBF_NONE:
 		__flush_s1_tlb_range_op(vae1is, start, pages, stride,
-					asid, tlb_level);
+					asid, tlb_level, false, domain);
 		break;
 	case TLBF_NOWALKCACHE:
 		__flush_s1_tlb_range_op(vale1is, start, pages, stride,
-					asid, tlb_level);
+					asid, tlb_level, false, domain);
 		break;
 	case TLBF_NOBROADCAST:
 		/* Combination unused */
@@ -656,7 +694,7 @@ static __always_inline void __do_flush_tlb_range(struct vm_area_struct *vma,
 		break;
 	case TLBF_NOWALKCACHE | TLBF_NOBROADCAST:
 		__flush_s1_tlb_range_op(vale1, start, pages, stride,
-					asid, tlb_level);
+					asid, tlb_level, true, -1);
 		break;
 	}
 
@@ -725,7 +763,7 @@ static inline void flush_tlb_kernel_range(unsigned long start, unsigned long end
 
 	dsb(ishst);
 	__flush_s1_tlb_range_op(vaale1is, start, pages, stride, 0,
-				TLBI_TTL_UNKNOWN);
+				TLBI_TTL_UNKNOWN, false, 0);
 	__tlbi_sync_s1ish_kernel();
 	isb();
 }
diff --git a/arch/arm64/kvm/hyp/nvhe/tlb.c b/arch/arm64/kvm/hyp/nvhe/tlb.c
index fdb90483340c..f7cac7f1f4fe 100644
--- a/arch/arm64/kvm/hyp/nvhe/tlb.c
+++ b/arch/arm64/kvm/hyp/nvhe/tlb.c
@@ -187,7 +187,7 @@ void __kvm_tlb_flush_vmid_ipa_nsh(struct kvm_s2_mmu *mmu,
 	 * Instead, we invalidate Stage-2 for this IPA, and the
 	 * whole of Stage-1. Weep...
 	 */
-	__tlbi_level(ipas2e1, ipa, level);
+	__tlbi_level_local(ipas2e1, ipa, level);
 
 	/*
 	 * We have to ensure completion of the invalidation at Stage-2,
diff --git a/arch/arm64/kvm/hyp/vhe/tlb.c b/arch/arm64/kvm/hyp/vhe/tlb.c
index e2e6a2df1533..ac1c5e623cf3 100644
--- a/arch/arm64/kvm/hyp/vhe/tlb.c
+++ b/arch/arm64/kvm/hyp/vhe/tlb.c
@@ -135,7 +135,7 @@ void __kvm_tlb_flush_vmid_ipa_nsh(struct kvm_s2_mmu *mmu,
 	 * Instead, we invalidate Stage-2 for this IPA, and the
 	 * whole of Stage-1. Weep...
 	 */
-	__tlbi_level(ipas2e1, ipa, level);
+	__tlbi_level_local(ipas2e1, ipa, level);
 
 	/*
 	 * We have to ensure completion of the invalidation at Stage-2,
diff --git a/arch/arm64/mm/tlbid.c b/arch/arm64/mm/tlbid.c
index b44fab2dca20..c748e710e2ee 100644
--- a/arch/arm64/mm/tlbid.c
+++ b/arch/arm64/mm/tlbid.c
@@ -10,6 +10,7 @@
 #include <linux/atomic.h>
 #include <linux/bitops.h>
 #include <linux/bitmap.h>
+#include <linux/bug.h>
 #include <linux/efi.h>
 #include <linux/jump_label.h>
 #include <linux/mm_types.h>
@@ -24,6 +25,33 @@ static u16 *domain_ids;
 static unsigned long **cpu_domains;
 static int domain_count;
 
+u16 get_tlbid_domain(struct mm_struct *mm)
+{
+	unsigned long mask = BITMAP_LAST_WORD_MASK(domain_count);
+	int idx = (domain_count - 1) / 64;
+	unsigned long domain_idx;
+	unsigned long domains;
+
+	if (!system_supports_tlbid())
+		return 0;
+
+	/* Open-coded find_last_bit() that uses atomic reads */
+	do {
+		domains = atomic64_read(&mm->context.tlbid_domains[idx]);
+		domains &= mask;
+		if (domains) {
+			domain_idx = idx * 64 + __fls(domains);
+			break;
+		}
+		mask = ~0UL;
+	} while (idx--);
+
+	if (WARN_ON(idx < 0))
+		return 0;
+
+	return domain_ids[domain_idx];
+}
+
 void update_tlbid_domains(struct mm_struct *mm, unsigned int cpu)
 {
 	if (!system_supports_tlbid())
@@ -31,6 +59,14 @@ void update_tlbid_domains(struct mm_struct *mm, unsigned int cpu)
 
 	for (int i = 0; i < BITS_TO_U64(domain_count); i++)
 		atomic64_and(cpu_domains[cpu][i], &mm->context.tlbid_domains[i]);
+
+	/*
+	 * Pairs with the DSB before reading tlbid_domains in the TLBI paths.
+	 * Either a concurrent PTE update is visible to this CPU before it uses
+	 * the mm, or the PTE updater observes the reduced domain bitmap and
+	 * includes this CPU in its TLBI.
+	 */
+	dsb(ishst);
 }
 
 int init_tlbid_context(struct mm_struct *mm)
-- 
2.43.0

Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help