aboutsummaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:59:32 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:59:32 -0700
commit8f150ccedfbd610aa25509ff42365d70fb20478f (patch)
tree1b22e147a9f8464940fcfa0483c96b8842c953c5 /kernel
parentd2dbe503fd806082acb0ca79a9d6641822988c2c (diff)
parentde020dc8049bfb2b22e3b6d99c031feb2e22d112 (diff)
Merge tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf
Pull bpf fixes from Alexei Starovoitov: - Fix overflow of backward jump offset in constant blinding (Alexei Starovoitov) - Fix packet range of packet pointers sharing an id when var_off tightens umax of one pointer and not the other (Alexei Starovoitov) - Fix objects stuck in free_by_rcu_ttrace list of bpf memalloc (Alexei Starovoitov) - Fix use-after-free of progs detached from busy trampolines: wait for an RCU tasks grace period before freeing trampoline progs, and patch detached progs out of trampoline images that are still in use (Florent Revest) - Hold map BTF for the memory allocator destructor record to fix UAF in deferred bpf_mem_alloc destruction (Kumar Kartikeya Dwivedi) - Fix missing migration protection in resizable hashtab lookup_and_delete batch operation (Ă–mer Mete Kaya) * tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf: bpf: Fix missing migration protection in __rhtab_map_lookup_and_delete_batch() selftests/bpf: Add a test for objects stuck in free_by_rcu_ttrace bpf: Fix objects stuck in free_by_rcu_ttrace bpf: Factor out __do_call_rcu_ttrace() selftests/bpf: Test packet range of pointers sharing an id bpf: Fix packet range of pointers sharing an id selftests/bpf: Detach a trampoline prog while a task sleeps before it bpf: Skip detached progs in trampoline images that are still in use bpf: Wait for an RCU tasks grace period before freeing trampoline progs bpf: Hold map BTF for the memory allocator destructor record bpf: Fix overflow of jump offset in constant blinding
Diffstat (limited to 'kernel')
-rw-r--r--kernel/bpf/core.c10
-rw-r--r--kernel/bpf/hashtab.c19
-rw-r--r--kernel/bpf/memalloc.c57
-rw-r--r--kernel/bpf/syscall.c19
-rw-r--r--kernel/bpf/trampoline.c85
-rw-r--r--kernel/bpf/verifier.c10
6 files changed, 168 insertions, 32 deletions
diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c
index 2e3bf8113ae9..7f11555a5070 100644
--- a/kernel/bpf/core.c
+++ b/kernel/bpf/core.c
@@ -1362,7 +1362,7 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from,
{
struct bpf_insn *to = to_buff;
u32 imm_rnd = get_random_u32();
- s16 off;
+ int off;
BUILD_BUG_ON(BPF_REG_PARAMS + 2 != MAX_BPF_JIT_REG);
BUILD_BUG_ON(BPF_REG_AX + 1 != MAX_BPF_JIT_REG);
@@ -1438,6 +1438,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from,
off = from->off;
if (off < 0)
off -= 2;
+ if (off < S16_MIN)
+ return -ERANGE;
*to++ = BPF_ALU64_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm);
*to++ = BPF_ALU64_IMM(BPF_XOR, BPF_REG_AX, imm_rnd);
*to++ = BPF_JMP_REG(from->code, from->dst_reg, BPF_REG_AX, off);
@@ -1458,6 +1460,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from,
off = from->off;
if (off < 0)
off -= 2;
+ if (off < S16_MIN)
+ return -ERANGE;
*to++ = BPF_ALU32_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm);
*to++ = BPF_ALU32_IMM(BPF_XOR, BPF_REG_AX, imm_rnd);
*to++ = BPF_JMP32_REG(from->code, from->dst_reg, BPF_REG_AX,
@@ -1606,7 +1610,9 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp
if (!rewritten)
continue;
- if (env)
+ if (rewritten < 0)
+ tmp = ERR_PTR(rewritten);
+ else if (env)
tmp = bpf_patch_insn_data(env, i, insn_buff, rewritten);
else
tmp = bpf_patch_insn_single(clone, i, insn_buff, rewritten);
diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c
index f9464e566f10..13a2356c84cf 100644
--- a/kernel/bpf/hashtab.c
+++ b/kernel/bpf/hashtab.c
@@ -128,6 +128,7 @@ struct htab_elem {
struct htab_btf_record {
struct btf_record *record;
+ struct btf *btf;
u32 key_size;
};
@@ -497,8 +498,13 @@ static void htab_dtor_ctx_free(void *ctx)
{
struct htab_btf_record *hrec = ctx;
+ /*
+ * The duplicated record still points into the map BTF, so free it
+ * before dropping the reference that keeps that BTF alive.
+ */
btf_record_free(hrec->record);
- kfree(ctx);
+ btf_put(hrec->btf);
+ kfree(hrec);
}
static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma,
@@ -521,6 +527,15 @@ static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma,
kfree(hrec);
return err;
}
+ /*
+ * btf_record_dup() only acquires kernel and module BTF. Fields whose
+ * types live in the map BTF keep pointing into it: kptrs to local
+ * types refer to map->btf, and graph roots carry a value record owned
+ * by its struct meta table. The context can outlive the map when the
+ * allocator defers its teardown, so hold a reference of our own.
+ */
+ hrec->btf = map->btf;
+ btf_get(hrec->btf);
bpf_mem_alloc_set_dtor(ma, dtor, htab_dtor_ctx_free, hrec);
return 0;
}
@@ -3359,8 +3374,10 @@ static int __rhtab_map_lookup_and_delete_batch(struct bpf_map *map,
}
if (do_delete) {
+ migrate_disable();
for (i = 0; i < total; i++)
rhtab_delete_elem(rhtab, del_elems[i], NULL, 0);
+ migrate_enable();
}
rcu_read_unlock();
diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c
index 8a8f088e83e6..15684d0fc883 100644
--- a/kernel/bpf/memalloc.c
+++ b/kernel/bpf/memalloc.c
@@ -118,6 +118,11 @@ struct bpf_mem_cache {
struct llist_head free_by_rcu_ttrace;
struct llist_head waiting_for_gp_ttrace;
struct rcu_head rcu_ttrace;
+ /*
+ * 0 - idle
+ * 1 - __free_rcu() is queued
+ * 2 - __free_rcu() is queued and free_by_rcu_ttrace got more objects since
+ */
atomic_t call_rcu_ttrace_in_progress;
raw_spinlock_t lock;
};
@@ -276,6 +281,8 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per
return cnt;
}
+static void __do_call_rcu_ttrace(struct bpf_mem_cache *c);
+
static void __free_rcu(struct rcu_head *head)
{
struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace);
@@ -285,7 +292,19 @@ static void __free_rcu(struct rcu_head *head)
llnode = llist_del_all(&c->waiting_for_gp_ttrace);
free_all(c, llnode, !!c->percpu_size);
- atomic_set(&c->call_rcu_ttrace_in_progress, 0);
+
+ /*
+ * do_call_rcu_ttrace() that ran while GP was in flight left its objects
+ * in free_by_rcu_ttrace. This cache may never free or alloc in bulk
+ * again, so start the next GP from here.
+ * 'c' can be freed as soon as call_rcu_ttrace_in_progress is zero.
+ */
+ if (atomic_cmpxchg(&c->call_rcu_ttrace_in_progress, 1, 0) == 1)
+ return;
+
+ /* Pairs with synchronize_rcu() in free_mem_alloc() */
+ guard(rcu)();
+ __do_call_rcu_ttrace(c);
}
static void enque_to_free(struct bpf_mem_cache *c, void *obj)
@@ -298,18 +317,15 @@ static void enque_to_free(struct bpf_mem_cache *c, void *obj)
llist_add(llnode, &c->free_by_rcu_ttrace);
}
-static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
+static void __do_call_rcu_ttrace(struct bpf_mem_cache *c)
{
struct llist_node *llnode, *t;
- if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) {
- if (unlikely(READ_ONCE(c->draining))) {
- scoped_guard(raw_spinlock_irqsave, &c->lock)
- llnode = llist_del_all(&c->free_by_rcu_ttrace);
- free_all(c, llnode, !!c->percpu_size);
- }
- return;
- }
+ /*
+ * Must be done before llist_del_all(). Objects that it misses were
+ * added by do_call_rcu_ttrace() that will set 2 after this store.
+ */
+ atomic_set(&c->call_rcu_ttrace_in_progress, 1);
WARN_ON_ONCE(!llist_empty(&c->waiting_for_gp_ttrace));
llist_for_each_safe(llnode, t, llist_del_all(&c->free_by_rcu_ttrace))
@@ -328,6 +344,22 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
call_rcu_tasks_trace(&c->rcu_ttrace, __free_rcu);
}
+static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
+{
+ struct llist_node *llnode;
+
+ if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 2)) {
+ if (unlikely(READ_ONCE(c->draining))) {
+ scoped_guard(raw_spinlock_irqsave, &c->lock)
+ llnode = llist_del_all(&c->free_by_rcu_ttrace);
+ free_all(c, llnode, !!c->percpu_size);
+ }
+ return;
+ }
+
+ __do_call_rcu_ttrace(c);
+}
+
static void free_bulk(struct bpf_mem_cache *c)
{
struct bpf_mem_cache *tgt = c->tgt;
@@ -700,7 +732,12 @@ static void free_mem_alloc(struct bpf_mem_alloc *ma)
* to wait for the pending __free_by_rcu(), and __free_rcu(). RCU Tasks
* Trace grace period implies RCU grace period, so all __free_rcu don't
* need extra call_rcu() (and thus extra rcu_barrier() here).
+ *
+ * __free_rcu() queues itself again unless it sees 'draining'. After
+ * synchronize_rcu() it either did that already or will not do it, so
+ * rcu_barrier_tasks_trace() cannot miss it.
*/
+ synchronize_rcu();
rcu_barrier(); /* wait for __free_by_rcu */
rcu_barrier_tasks_trace(); /* wait for __free_rcu */
free_mem_alloc_no_barrier(ma);
diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c
index 244a939b9d2d..96217b99399d 100644
--- a/kernel/bpf/syscall.c
+++ b/kernel/bpf/syscall.c
@@ -2448,6 +2448,21 @@ static void __bpf_prog_put_rcu(struct rcu_head *rcu)
bpf_prog_free(aux->prog);
}
+/*
+ * Progs called from a trampoline can also be reached by a task that was
+ * preempted in the trampoline before the prog's enter helper took its RCU
+ * read lock, wait for those first.
+ */
+static void __bpf_prog_put_rcu_tasks(struct rcu_head *rcu)
+{
+ struct bpf_prog *prog = container_of(rcu, struct bpf_prog_aux, rcu)->prog;
+
+ if (prog->sleepable)
+ call_rcu_tasks_trace(rcu, __bpf_prog_put_rcu);
+ else
+ call_rcu(rcu, __bpf_prog_put_rcu);
+}
+
static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred)
{
bpf_prog_kallsyms_del_all(prog);
@@ -2461,7 +2476,9 @@ static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred)
btf_put(prog->aux->attach_btf);
if (deferred) {
- if (prog->sleepable)
+ if (IS_ENABLED(CONFIG_TASKS_RCU) && prog->aux->tramp_linked)
+ call_rcu_tasks(&prog->aux->rcu, __bpf_prog_put_rcu_tasks);
+ else if (prog->sleepable)
call_rcu_tasks_trace(&prog->aux->rcu, __bpf_prog_put_rcu);
else
call_rcu(&prog->aux->rcu, __bpf_prog_put_rcu);
diff --git a/kernel/bpf/trampoline.c b/kernel/bpf/trampoline.c
index 90b70ea0d370..bf4cb3dd444d 100644
--- a/kernel/bpf/trampoline.c
+++ b/kernel/bpf/trampoline.c
@@ -401,6 +401,7 @@ static struct bpf_trampoline *bpf_trampoline_lookup(u64 key, unsigned long ip)
head = &trampoline_ip_table[hash_64(tr->ip, TRAMPOLINE_HASH_BITS)];
hlist_add_head(&tr->hlist_ip, head);
refcount_set(&tr->refcnt, 1);
+ INIT_LIST_HEAD(&tr->images);
for (i = 0; i < BPF_TRAMP_MAX; i++)
INIT_HLIST_HEAD(&tr->progs_hlist[i]);
out:
@@ -565,15 +566,22 @@ static void bpf_tramp_image_free(struct bpf_tramp_image *im)
arch_free_bpf_trampoline(im->image, im->size);
bpf_jit_uncharge_modmem(im->size);
percpu_ref_exit(&im->pcref);
+ kfree(im->skips);
kfree_rcu(im, rcu);
}
static void __bpf_tramp_image_put_deferred(struct work_struct *work)
{
struct bpf_tramp_image *im;
+ struct bpf_trampoline *tr;
im = container_of(work, struct bpf_tramp_image, work);
+ tr = im->tr;
+ trampoline_lock(tr);
+ list_del(&im->list);
+ trampoline_unlock(tr);
bpf_tramp_image_free(im);
+ bpf_trampoline_put(tr);
}
/* callback, fexit step 3 or fentry step 2 */
@@ -601,7 +609,7 @@ static void __bpf_tramp_image_put_rcu_tasks(struct rcu_head *rcu)
struct bpf_tramp_image *im;
im = container_of(rcu, struct bpf_tramp_image, rcu);
- if (im->ip_after_call)
+ if (im->call_orig)
/* the case of fmod_ret/fexit trampoline and CONFIG_PREEMPTION=y */
percpu_ref_kill(&im->pcref);
else
@@ -621,9 +629,9 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im)
*
* The trampoline is unreachable before bpf_tramp_image_put().
*
- * First, patch the trampoline to avoid calling into fexit progs.
- * The progs will be freed even if the original function is still
- * executing or sleeping.
+ * Progs are patched out of the image when they are detached, see
+ * bpf_trampoline_skip_prog(), so they can be freed even if a task is
+ * still in the image.
* In case of CONFIG_PREEMPT=y use call_rcu_tasks() to wait on
* first few asm instructions to execute and call into
* __bpf_tramp_enter->percpu_ref_get.
@@ -637,11 +645,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im)
* percpu_ref_kill will be waiting for. Hence the first
* call_rcu_tasks() is not necessary.
*/
- if (im->ip_after_call) {
- int err = bpf_arch_text_poke(im->ip_after_call, BPF_MOD_NOP,
- BPF_MOD_JUMP, NULL,
- im->ip_epilogue);
- WARN_ON(err);
+ if (im->call_orig) {
if (IS_ENABLED(CONFIG_TASKS_RCU))
call_rcu_tasks(&im->rcu, __bpf_tramp_image_put_rcu_tasks);
else
@@ -658,7 +662,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im)
call_rcu_tasks_trace(&im->rcu, __bpf_tramp_image_put_rcu_tasks);
}
-static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size)
+static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size, int nr_progs)
{
struct bpf_tramp_image *im;
struct bpf_ksym *ksym;
@@ -669,6 +673,10 @@ static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size)
if (!im)
goto out;
+ im->skips = kzalloc_objs(*im->skips, nr_progs);
+ if (!im->skips)
+ goto out_free_im;
+
err = bpf_jit_charge_modmem(size);
if (err)
goto out_free_im;
@@ -695,6 +703,7 @@ out_free_image:
out_uncharge:
bpf_jit_uncharge_modmem(size);
out_free_im:
+ kfree(im->skips);
kfree(im);
out:
return ERR_PTR(err);
@@ -771,11 +780,12 @@ again:
goto out;
}
- im = bpf_tramp_image_alloc(tr->key, size);
+ im = bpf_tramp_image_alloc(tr->key, size, total);
if (IS_ERR(im)) {
err = PTR_ERR(im);
goto out;
}
+ im->call_orig = tr->flags & BPF_TRAMP_F_CALL_ORIG;
err = arch_prepare_bpf_trampoline(im, im->image, im->image + size,
&tr->func.model, tr->flags, tnodes,
@@ -806,8 +816,14 @@ again:
#endif
out_free:
- if (err)
+ if (err) {
bpf_tramp_image_free(im);
+ } else {
+ /* track the image until it is freed, for bpf_trampoline_skip_prog() */
+ refcount_inc(&tr->refcnt);
+ im->tr = tr;
+ list_add(&im->list, &tr->images);
+ }
out:
/* If any error happens, restore previous flags */
if (err)
@@ -907,6 +923,7 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr,
}
hlist_add_head(&node->tramp_hlist, prog_list);
+ node->link->prog->aux->tramp_linked = true;
if (kind == BPF_TRAMP_FSESSION) {
tr->progs_cnt[BPF_TRAMP_FENTRY]++;
fexit = fsession_exit(node);
@@ -920,6 +937,41 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr,
return 0;
}
+/*
+ * Patch the nop in front of a prog call to a jump over it. A task can be
+ * preempted anywhere in the image, so archs that need several instructions for
+ * a jump of any range patch a single near branch here instead.
+ */
+int __weak arch_bpf_trampoline_skip(void *nop, void *target)
+{
+ return bpf_arch_text_poke(nop, BPF_MOD_NOP, BPF_MOD_JUMP, NULL, target);
+}
+
+/*
+ * prog was detached and can be freed, but tasks may still be running in images
+ * that call it, sleeping in an earlier prog for example. They can be in any
+ * image that is not freed yet, not only in cur_image, so patch all of them to
+ * jump over prog.
+ */
+static void bpf_trampoline_skip_prog(struct bpf_trampoline *tr, struct bpf_prog *prog)
+{
+ struct bpf_tramp_image *im;
+ int i, err;
+
+ list_for_each_entry(im, &tr->images, list) {
+ for (i = 0; i < im->nr_skips; i++) {
+ struct bpf_tramp_skip *skip = &im->skips[i];
+
+ if (skip->prog != prog)
+ continue;
+ err = arch_bpf_trampoline_skip(skip->nop, skip->target);
+ WARN_ON_ONCE(err);
+ /* not a nop anymore, and prog's address can be reused */
+ skip->prog = NULL;
+ }
+ }
+}
+
static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr,
struct bpf_tramp_node *node)
{
@@ -937,6 +989,7 @@ static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr,
}
hlist_del_init(&node->tramp_hlist);
tr->progs_cnt[kind]--;
+ bpf_trampoline_skip_prog(tr, node->link->prog);
}
static int __bpf_trampoline_link_prog(struct bpf_tramp_node *node,
@@ -1245,11 +1298,9 @@ void bpf_trampoline_put(struct bpf_trampoline *tr)
if (WARN_ON_ONCE(!hlist_empty(&tr->progs_hlist[i])))
goto out;
- /* This code will be executed even when the last bpf_tramp_image
- * is alive. All progs are detached from the trampoline and the
- * trampoline image is patched with jmp into epilogue to skip
- * fexit progs. The fentry-only trampoline will be freed via
- * multiple rcu callbacks.
+ /*
+ * All progs are detached and the last image has been freed, images
+ * hold a reference on the trampoline until then.
*/
hlist_del(&tr->hlist_key);
hlist_del(&tr->hlist_ip);
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 41b49c56e123..5f874979b8d7 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -14826,7 +14826,15 @@ static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn
"Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range.");
return -EINVAL;
}
- reg_bounds_sync(dst_reg);
+ /*
+ * A packet pointer that keeps its id or range is checked against a
+ * range set from the checked pointer's umax, so var_off must not tighten
+ * its umax. r32 must still match var_off for reg_bounds_sanity_check().
+ */
+ if (reg_is_pkt_pointer(dst_reg) && (known || dst_reg->range > 0))
+ __update_reg32_bounds(dst_reg);
+ else
+ reg_bounds_sync(dst_reg);
bounds_ret = sanitize_check_bounds(env, insn, dst_reg);
if (bounds_ret == -EACCES)
return bounds_ret;