aboutsummaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
authorKumar Kartikeya Dwivedi <memxor@gmail.com>2026-10-01 18:40:06 +0200
committerKumar Kartikeya Dwivedi <memxor@gmail.com>2026-10-01 18:40:06 +0200
commit7b12c538697f2d5014dfd378c73be193d700c7a9 (patch)
treec1c2bdbb4b6b7174a5bbe9e4c2c1f1c0204adcb5 /kernel
parent33a154a96e71a34a1bcca9f40da343dbbf7b38b4 (diff)
parentaeb709c767aeca996e6011133e4f7bdb2a4af652 (diff)
Merge branch 'bpf-fix-objects-stuck-in-free_by_rcu_ttrace'
Alexei Starovoitov says: ==================== bpf: Fix objects stuck in free_by_rcu_ttrace From: Alexei Starovoitov <ast@kernel.org> The objects that bpf_mem_alloc frees while RCU tasks trace GP is in flight stay in free_by_rcu_ttrace list until the same bpf_mem_cache frees or allocates in bulk again. Patch 1 - refactoring. No functional change. Patch 2 - the fix. Patch 3 - selftest. Signed-off-by: Alexei Starovoitov <ast@kernel.org> ==================== Link: https://patch.msgid.link/20260930095920.601738-1-alexei.starovoitov@gmail.com Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
Diffstat (limited to 'kernel')
-rw-r--r--kernel/bpf/memalloc.c57
1 files changed, 47 insertions, 10 deletions
diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c
index 8a8f088e83e6..15684d0fc883 100644
--- a/kernel/bpf/memalloc.c
+++ b/kernel/bpf/memalloc.c
@@ -118,6 +118,11 @@ struct bpf_mem_cache {
struct llist_head free_by_rcu_ttrace;
struct llist_head waiting_for_gp_ttrace;
struct rcu_head rcu_ttrace;
+ /*
+ * 0 - idle
+ * 1 - __free_rcu() is queued
+ * 2 - __free_rcu() is queued and free_by_rcu_ttrace got more objects since
+ */
atomic_t call_rcu_ttrace_in_progress;
raw_spinlock_t lock;
};
@@ -276,6 +281,8 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per
return cnt;
}
+static void __do_call_rcu_ttrace(struct bpf_mem_cache *c);
+
static void __free_rcu(struct rcu_head *head)
{
struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace);
@@ -285,7 +292,19 @@ static void __free_rcu(struct rcu_head *head)
llnode = llist_del_all(&c->waiting_for_gp_ttrace);
free_all(c, llnode, !!c->percpu_size);
- atomic_set(&c->call_rcu_ttrace_in_progress, 0);
+
+ /*
+ * do_call_rcu_ttrace() that ran while GP was in flight left its objects
+ * in free_by_rcu_ttrace. This cache may never free or alloc in bulk
+ * again, so start the next GP from here.
+ * 'c' can be freed as soon as call_rcu_ttrace_in_progress is zero.
+ */
+ if (atomic_cmpxchg(&c->call_rcu_ttrace_in_progress, 1, 0) == 1)
+ return;
+
+ /* Pairs with synchronize_rcu() in free_mem_alloc() */
+ guard(rcu)();
+ __do_call_rcu_ttrace(c);
}
static void enque_to_free(struct bpf_mem_cache *c, void *obj)
@@ -298,18 +317,15 @@ static void enque_to_free(struct bpf_mem_cache *c, void *obj)
llist_add(llnode, &c->free_by_rcu_ttrace);
}
-static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
+static void __do_call_rcu_ttrace(struct bpf_mem_cache *c)
{
struct llist_node *llnode, *t;
- if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) {
- if (unlikely(READ_ONCE(c->draining))) {
- scoped_guard(raw_spinlock_irqsave, &c->lock)
- llnode = llist_del_all(&c->free_by_rcu_ttrace);
- free_all(c, llnode, !!c->percpu_size);
- }
- return;
- }
+ /*
+ * Must be done before llist_del_all(). Objects that it misses were
+ * added by do_call_rcu_ttrace() that will set 2 after this store.
+ */
+ atomic_set(&c->call_rcu_ttrace_in_progress, 1);
WARN_ON_ONCE(!llist_empty(&c->waiting_for_gp_ttrace));
llist_for_each_safe(llnode, t, llist_del_all(&c->free_by_rcu_ttrace))
@@ -328,6 +344,22 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
call_rcu_tasks_trace(&c->rcu_ttrace, __free_rcu);
}
+static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
+{
+ struct llist_node *llnode;
+
+ if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 2)) {
+ if (unlikely(READ_ONCE(c->draining))) {
+ scoped_guard(raw_spinlock_irqsave, &c->lock)
+ llnode = llist_del_all(&c->free_by_rcu_ttrace);
+ free_all(c, llnode, !!c->percpu_size);
+ }
+ return;
+ }
+
+ __do_call_rcu_ttrace(c);
+}
+
static void free_bulk(struct bpf_mem_cache *c)
{
struct bpf_mem_cache *tgt = c->tgt;
@@ -700,7 +732,12 @@ static void free_mem_alloc(struct bpf_mem_alloc *ma)
* to wait for the pending __free_by_rcu(), and __free_rcu(). RCU Tasks
* Trace grace period implies RCU grace period, so all __free_rcu don't
* need extra call_rcu() (and thus extra rcu_barrier() here).
+ *
+ * __free_rcu() queues itself again unless it sees 'draining'. After
+ * synchronize_rcu() it either did that already or will not do it, so
+ * rcu_barrier_tasks_trace() cannot miss it.
*/
+ synchronize_rcu();
rcu_barrier(); /* wait for __free_by_rcu */
rcu_barrier_tasks_trace(); /* wait for __free_rcu */
free_mem_alloc_no_barrier(ma);