Thread (15 messages) flat view 15 messages, 2 authors, 3d ago
WARM3d

Revision v7 of 7 in this series.

Revisions (7)
  1. rfc [diff vs current]
  2. v2 [diff vs current]
  3. v3 [diff vs current]
  4. v4 [diff vs current]
  5. v5 [diff vs current]
  6. v6 [diff vs current]
  7. v7 current

[RFC PATCH v7 03/10] trace: add stackmap statistics interface

From: Li Pengfei <hidden>
Date: 2026-09-12 08:39:05
Also in: linux-doc, linux-kselftest, lkml
Subsystem: the rest, tracing · Maintainers: Linus Torvalds, Steven Rostedt, Masami Hiramatsu

From: Pengfei Li <redacted>

Export stackmap counters through stack_map_stat: entries, table size,
successes, drops and success rate.

entries is the element-pool allocation cursor: the number of element
records claimed since the last reset. It is not a strict unique-stack
count. Concurrent insertion races can claim duplicate records, and a
sample can include a record claimed before it is published in the hash
table.

successes counts map operations that returned a stack id, while drops
counts capacity and probe-limit failures. The rate is
successes / (successes + drops), so it excludes bypasses that never
call the map, including deep stacks, reset windows and ring-buffer
reservation failures. A fresh or reset map reports 0%.

Take the complete sample under reader_sem. Reset clears next_elt and
the per-CPU atomic_long_t counters under the write side; read those
counters with atomic_long_read() under the read side to avoid combining
values from different generations.

The file is auxiliary. Failure to create it does not disable
stackmap because stack_map remains available to resolve and reset ids.

Signed-off-by: Pengfei Li <redacted>
---
 kernel/trace/trace.c          |  14 +++-
 kernel/trace/trace_stackmap.c | 137 ++++++++++++++++++++++++++++++++++
 kernel/trace/trace_stackmap.h |   1 +
 3 files changed, 151 insertions(+), 1 deletion(-)
diff --git a/kernel/trace/trace.c b/kernel/trace/trace.c
index 06ae8ed11475..17df5f85da7a 100644
--- a/kernel/trace/trace.c
+++ b/kernel/trace/trace.c
@@ -2199,7 +2199,7 @@ void __ftrace_trace_stack(struct trace_array *tr,
 	 *   - get_id() fails      -> discard the reserved slot, then try
 	 *                            full-stack fallback
 	 * A failed stack-id reservation therefore never consumes a map slot
-	 * or updates the map counters.
+	 * or updates stack_map_stat.
 	 */
 	if (tr->trace_flags & TRACE_ITER(STACKMAP)) {
 		struct ftrace_stackmap *smap;
@@ -9411,6 +9411,18 @@ static __init void tracer_init_tracefs_work_func(struct work_struct *work)
 				 */
 				smp_store_release(&global_trace.stackmap, smap);
 				WRITE_ONCE(stackmap_init_state, STACKMAP_INIT_DONE);
+				/*
+				 * stat is an auxiliary observability
+				 * surface. If it fails to be created we keep
+				 * dedup enabled -- the kernel side still
+				 * works and stack_map alone is enough to
+				 * resolve and reset; trace_create_file()
+				 * already pr_warn()s on failure.
+				 */
+				trace_create_file("stack_map_stat",
+						  TRACE_MODE_READ, NULL,
+						  smap,
+						  &ftrace_stackmap_stat_fops);
 			}
 		} else {
 			pr_warn("ftrace stackmap init failed, dedup disabled\n");
diff --git a/kernel/trace/trace_stackmap.c b/kernel/trace/trace_stackmap.c
index b2e2115a15f9..2382c7459712 100644
--- a/kernel/trace/trace_stackmap.c
+++ b/kernel/trace/trace_stackmap.c
@@ -56,6 +56,8 @@
 #include <linux/random.h>
 #include <linux/rcupdate.h>
 #include <linux/log2.h>
+#include <linux/math64.h>
+#include <linux/overflow.h>
 #include <asm/local.h>
 
 #include "trace.h"
@@ -716,3 +718,138 @@ const struct file_operations ftrace_stackmap_fops = {
 	.llseek		= seq_lseek,
 	.release	= stackmap_release,
 };
+
+/* --- Stats --- */
+
+static u64 stackmap_u64_add_sat(u64 left, u64 right)
+{
+	u64 sum;
+
+	return check_add_overflow(left, right, &sum) ? U64_MAX : sum;
+}
+
+static int stackmap_scaled_cmp(u64 left, u32 left_scale,
+			       u64 right, u32 right_scale)
+{
+	u64 left_hi = mul_u64_u64_shr(left, left_scale, 64);
+	u64 right_hi = mul_u64_u64_shr(right, right_scale, 64);
+	u64 left_lo = left * left_scale;
+	u64 right_lo = right * right_scale;
+
+	if (left_hi != right_hi)
+		return left_hi < right_hi ? -1 : 1;
+	if (left_lo != right_lo)
+		return left_lo < right_lo ? -1 : 1;
+	return 0;
+}
+
+static u64 stackmap_success_rate(u64 successes, u64 drops)
+{
+	u32 low = 0, high = 100;
+
+	if (!successes)
+		return 0;
+	if (!drops)
+		return 100;
+
+	/*
+	 * Find the largest percentage p satisfying
+	 *
+	 *   p * drops <= (100 - p) * successes
+	 *
+	 * which is equivalent to p <= 100 * successes / (successes + drops),
+	 * without forming the potentially 65-bit denominator. Compare the
+	 * products as 128-bit values split into high and low halves.
+	 */
+	while (low < high) {
+		u32 mid = (low + high + 1) / 2;
+
+		if (stackmap_scaled_cmp(drops, mid, successes, 100 - mid) <= 0)
+			low = mid;
+		else
+			high = mid - 1;
+	}
+
+	return low;
+}
+
+static int stackmap_stat_show(struct seq_file *m, void *v)
+{
+	struct ftrace_stackmap *smap = m->private;
+	u64 successes = 0, drops = 0;
+	u64 cpu_successes, cpu_drops;
+	u32 entries;
+	int cpu;
+
+	if (!smap) {
+		seq_puts(m, "stackmap not initialized\n");
+		return 0;
+	}
+
+	/*
+	 * Sample every counter under the read side of reader_sem. Reset
+	 * clears next_elt and the per-CPU counters under the write side,
+	 * so without this an unserialized read could straddle a reset and
+	 * report a mix of the two generations -- a non-zero entry count
+	 * next to counters that have already been zeroed, for instance.
+	 */
+	down_read(&smap->reader_sem);
+
+	entries = atomic_read(&smap->next_elt);
+	for_each_possible_cpu(cpu) {
+		cpu_successes = atomic_long_read(
+			per_cpu_ptr(smap->successes, cpu));
+		cpu_drops = atomic_long_read(per_cpu_ptr(smap->drops, cpu));
+		successes = stackmap_u64_add_sat(successes, cpu_successes);
+		drops = stackmap_u64_add_sat(drops, cpu_drops);
+	}
+
+	seq_printf(m, "entries:      %u / %u\n", entries, smap->max_elts);
+	seq_printf(m, "table_size:   %u\n", smap->map_size);
+	seq_printf(m, "successes:    %llu\n", successes);
+	seq_printf(m, "drops:        %llu\n", drops);
+	seq_printf(m, "success_rate: %llu%%\n",
+		   stackmap_success_rate(successes, drops));
+
+	up_read(&smap->reader_sem);
+	return 0;
+}
+
+static int stackmap_stat_open(struct inode *inode, struct file *file)
+{
+	struct ftrace_stackmap *smap = inode->i_private;
+	int ret;
+
+	if (!smap)
+		return -ENODEV;
+
+	/* Same open-time tracing policy as the other stackmap files. */
+	ret = tracing_check_open_get_tr(smap->tr);
+	if (ret)
+		return ret;
+
+	ret = single_open(file, stackmap_stat_show, smap);
+	if (ret) {
+		trace_array_put(smap->tr);
+		return ret;
+	}
+	return 0;
+}
+
+static int stackmap_stat_release(struct inode *inode, struct file *file)
+{
+	struct seq_file *m = file->private_data;
+	struct ftrace_stackmap *smap = m->private;
+	int ret;
+
+	ret = single_release(inode, file);
+	trace_array_put(smap->tr);
+	return ret;
+}
+
+const struct file_operations ftrace_stackmap_stat_fops = {
+	.open		= stackmap_stat_open,
+	.read		= seq_read,
+	.llseek		= seq_lseek,
+	.release	= stackmap_stat_release,
+};
diff --git a/kernel/trace/trace_stackmap.h b/kernel/trace/trace_stackmap.h
index 979d6fd76460..7615e346dfa6 100644
--- a/kernel/trace/trace_stackmap.h
+++ b/kernel/trace/trace_stackmap.h
@@ -19,6 +19,7 @@ int ftrace_stackmap_get_id(struct ftrace_stackmap *smap,
 			   unsigned long *ips, unsigned int nr_entries);
 
 extern const struct file_operations ftrace_stackmap_fops;
+extern const struct file_operations ftrace_stackmap_stat_fops;
 
 #else
 
-- 
2.34.1
Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help