diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-02 12:59:32 -0700 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-02 12:59:32 -0700 |
| commit | 8f150ccedfbd610aa25509ff42365d70fb20478f (patch) | |
| tree | 1b22e147a9f8464940fcfa0483c96b8842c953c5 /kernel | |
| parent | d2dbe503fd806082acb0ca79a9d6641822988c2c (diff) | |
| parent | de020dc8049bfb2b22e3b6d99c031feb2e22d112 (diff) | |
Merge tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf
Pull bpf fixes from Alexei Starovoitov:
- Fix overflow of backward jump offset in constant blinding
(Alexei Starovoitov)
- Fix packet range of packet pointers sharing an id when var_off
tightens umax of one pointer and not the other (Alexei Starovoitov)
- Fix objects stuck in free_by_rcu_ttrace list of bpf memalloc
(Alexei Starovoitov)
- Fix use-after-free of progs detached from busy trampolines: wait for
an RCU tasks grace period before freeing trampoline progs, and patch
detached progs out of trampoline images that are still in use
(Florent Revest)
- Hold map BTF for the memory allocator destructor record to fix UAF in
deferred bpf_mem_alloc destruction (Kumar Kartikeya Dwivedi)
- Fix missing migration protection in resizable hashtab
lookup_and_delete batch operation (Ă–mer Mete Kaya)
* tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf:
bpf: Fix missing migration protection in __rhtab_map_lookup_and_delete_batch()
selftests/bpf: Add a test for objects stuck in free_by_rcu_ttrace
bpf: Fix objects stuck in free_by_rcu_ttrace
bpf: Factor out __do_call_rcu_ttrace()
selftests/bpf: Test packet range of pointers sharing an id
bpf: Fix packet range of pointers sharing an id
selftests/bpf: Detach a trampoline prog while a task sleeps before it
bpf: Skip detached progs in trampoline images that are still in use
bpf: Wait for an RCU tasks grace period before freeing trampoline progs
bpf: Hold map BTF for the memory allocator destructor record
bpf: Fix overflow of jump offset in constant blinding
Diffstat (limited to 'kernel')
| -rw-r--r-- | kernel/bpf/core.c | 10 | ||||
| -rw-r--r-- | kernel/bpf/hashtab.c | 19 | ||||
| -rw-r--r-- | kernel/bpf/memalloc.c | 57 | ||||
| -rw-r--r-- | kernel/bpf/syscall.c | 19 | ||||
| -rw-r--r-- | kernel/bpf/trampoline.c | 85 | ||||
| -rw-r--r-- | kernel/bpf/verifier.c | 10 |
6 files changed, 168 insertions, 32 deletions
diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c index 2e3bf8113ae9..7f11555a5070 100644 --- a/kernel/bpf/core.c +++ b/kernel/bpf/core.c @@ -1362,7 +1362,7 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from, { struct bpf_insn *to = to_buff; u32 imm_rnd = get_random_u32(); - s16 off; + int off; BUILD_BUG_ON(BPF_REG_PARAMS + 2 != MAX_BPF_JIT_REG); BUILD_BUG_ON(BPF_REG_AX + 1 != MAX_BPF_JIT_REG); @@ -1438,6 +1438,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from, off = from->off; if (off < 0) off -= 2; + if (off < S16_MIN) + return -ERANGE; *to++ = BPF_ALU64_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm); *to++ = BPF_ALU64_IMM(BPF_XOR, BPF_REG_AX, imm_rnd); *to++ = BPF_JMP_REG(from->code, from->dst_reg, BPF_REG_AX, off); @@ -1458,6 +1460,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from, off = from->off; if (off < 0) off -= 2; + if (off < S16_MIN) + return -ERANGE; *to++ = BPF_ALU32_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm); *to++ = BPF_ALU32_IMM(BPF_XOR, BPF_REG_AX, imm_rnd); *to++ = BPF_JMP32_REG(from->code, from->dst_reg, BPF_REG_AX, @@ -1606,7 +1610,9 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp if (!rewritten) continue; - if (env) + if (rewritten < 0) + tmp = ERR_PTR(rewritten); + else if (env) tmp = bpf_patch_insn_data(env, i, insn_buff, rewritten); else tmp = bpf_patch_insn_single(clone, i, insn_buff, rewritten); diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c index f9464e566f10..13a2356c84cf 100644 --- a/kernel/bpf/hashtab.c +++ b/kernel/bpf/hashtab.c @@ -128,6 +128,7 @@ struct htab_elem { struct htab_btf_record { struct btf_record *record; + struct btf *btf; u32 key_size; }; @@ -497,8 +498,13 @@ static void htab_dtor_ctx_free(void *ctx) { struct htab_btf_record *hrec = ctx; + /* + * The duplicated record still points into the map BTF, so free it + * before dropping the reference that keeps that BTF alive. + */ btf_record_free(hrec->record); - kfree(ctx); + btf_put(hrec->btf); + kfree(hrec); } static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma, @@ -521,6 +527,15 @@ static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma, kfree(hrec); return err; } + /* + * btf_record_dup() only acquires kernel and module BTF. Fields whose + * types live in the map BTF keep pointing into it: kptrs to local + * types refer to map->btf, and graph roots carry a value record owned + * by its struct meta table. The context can outlive the map when the + * allocator defers its teardown, so hold a reference of our own. + */ + hrec->btf = map->btf; + btf_get(hrec->btf); bpf_mem_alloc_set_dtor(ma, dtor, htab_dtor_ctx_free, hrec); return 0; } @@ -3359,8 +3374,10 @@ static int __rhtab_map_lookup_and_delete_batch(struct bpf_map *map, } if (do_delete) { + migrate_disable(); for (i = 0; i < total; i++) rhtab_delete_elem(rhtab, del_elems[i], NULL, 0); + migrate_enable(); } rcu_read_unlock(); diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c index 8a8f088e83e6..15684d0fc883 100644 --- a/kernel/bpf/memalloc.c +++ b/kernel/bpf/memalloc.c @@ -118,6 +118,11 @@ struct bpf_mem_cache { struct llist_head free_by_rcu_ttrace; struct llist_head waiting_for_gp_ttrace; struct rcu_head rcu_ttrace; + /* + * 0 - idle + * 1 - __free_rcu() is queued + * 2 - __free_rcu() is queued and free_by_rcu_ttrace got more objects since + */ atomic_t call_rcu_ttrace_in_progress; raw_spinlock_t lock; }; @@ -276,6 +281,8 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per return cnt; } +static void __do_call_rcu_ttrace(struct bpf_mem_cache *c); + static void __free_rcu(struct rcu_head *head) { struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace); @@ -285,7 +292,19 @@ static void __free_rcu(struct rcu_head *head) llnode = llist_del_all(&c->waiting_for_gp_ttrace); free_all(c, llnode, !!c->percpu_size); - atomic_set(&c->call_rcu_ttrace_in_progress, 0); + + /* + * do_call_rcu_ttrace() that ran while GP was in flight left its objects + * in free_by_rcu_ttrace. This cache may never free or alloc in bulk + * again, so start the next GP from here. + * 'c' can be freed as soon as call_rcu_ttrace_in_progress is zero. + */ + if (atomic_cmpxchg(&c->call_rcu_ttrace_in_progress, 1, 0) == 1) + return; + + /* Pairs with synchronize_rcu() in free_mem_alloc() */ + guard(rcu)(); + __do_call_rcu_ttrace(c); } static void enque_to_free(struct bpf_mem_cache *c, void *obj) @@ -298,18 +317,15 @@ static void enque_to_free(struct bpf_mem_cache *c, void *obj) llist_add(llnode, &c->free_by_rcu_ttrace); } -static void do_call_rcu_ttrace(struct bpf_mem_cache *c) +static void __do_call_rcu_ttrace(struct bpf_mem_cache *c) { struct llist_node *llnode, *t; - if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) { - if (unlikely(READ_ONCE(c->draining))) { - scoped_guard(raw_spinlock_irqsave, &c->lock) - llnode = llist_del_all(&c->free_by_rcu_ttrace); - free_all(c, llnode, !!c->percpu_size); - } - return; - } + /* + * Must be done before llist_del_all(). Objects that it misses were + * added by do_call_rcu_ttrace() that will set 2 after this store. + */ + atomic_set(&c->call_rcu_ttrace_in_progress, 1); WARN_ON_ONCE(!llist_empty(&c->waiting_for_gp_ttrace)); llist_for_each_safe(llnode, t, llist_del_all(&c->free_by_rcu_ttrace)) @@ -328,6 +344,22 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c) call_rcu_tasks_trace(&c->rcu_ttrace, __free_rcu); } +static void do_call_rcu_ttrace(struct bpf_mem_cache *c) +{ + struct llist_node *llnode; + + if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 2)) { + if (unlikely(READ_ONCE(c->draining))) { + scoped_guard(raw_spinlock_irqsave, &c->lock) + llnode = llist_del_all(&c->free_by_rcu_ttrace); + free_all(c, llnode, !!c->percpu_size); + } + return; + } + + __do_call_rcu_ttrace(c); +} + static void free_bulk(struct bpf_mem_cache *c) { struct bpf_mem_cache *tgt = c->tgt; @@ -700,7 +732,12 @@ static void free_mem_alloc(struct bpf_mem_alloc *ma) * to wait for the pending __free_by_rcu(), and __free_rcu(). RCU Tasks * Trace grace period implies RCU grace period, so all __free_rcu don't * need extra call_rcu() (and thus extra rcu_barrier() here). + * + * __free_rcu() queues itself again unless it sees 'draining'. After + * synchronize_rcu() it either did that already or will not do it, so + * rcu_barrier_tasks_trace() cannot miss it. */ + synchronize_rcu(); rcu_barrier(); /* wait for __free_by_rcu */ rcu_barrier_tasks_trace(); /* wait for __free_rcu */ free_mem_alloc_no_barrier(ma); diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c index 244a939b9d2d..96217b99399d 100644 --- a/kernel/bpf/syscall.c +++ b/kernel/bpf/syscall.c @@ -2448,6 +2448,21 @@ static void __bpf_prog_put_rcu(struct rcu_head *rcu) bpf_prog_free(aux->prog); } +/* + * Progs called from a trampoline can also be reached by a task that was + * preempted in the trampoline before the prog's enter helper took its RCU + * read lock, wait for those first. + */ +static void __bpf_prog_put_rcu_tasks(struct rcu_head *rcu) +{ + struct bpf_prog *prog = container_of(rcu, struct bpf_prog_aux, rcu)->prog; + + if (prog->sleepable) + call_rcu_tasks_trace(rcu, __bpf_prog_put_rcu); + else + call_rcu(rcu, __bpf_prog_put_rcu); +} + static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred) { bpf_prog_kallsyms_del_all(prog); @@ -2461,7 +2476,9 @@ static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred) btf_put(prog->aux->attach_btf); if (deferred) { - if (prog->sleepable) + if (IS_ENABLED(CONFIG_TASKS_RCU) && prog->aux->tramp_linked) + call_rcu_tasks(&prog->aux->rcu, __bpf_prog_put_rcu_tasks); + else if (prog->sleepable) call_rcu_tasks_trace(&prog->aux->rcu, __bpf_prog_put_rcu); else call_rcu(&prog->aux->rcu, __bpf_prog_put_rcu); diff --git a/kernel/bpf/trampoline.c b/kernel/bpf/trampoline.c index 90b70ea0d370..bf4cb3dd444d 100644 --- a/kernel/bpf/trampoline.c +++ b/kernel/bpf/trampoline.c @@ -401,6 +401,7 @@ static struct bpf_trampoline *bpf_trampoline_lookup(u64 key, unsigned long ip) head = &trampoline_ip_table[hash_64(tr->ip, TRAMPOLINE_HASH_BITS)]; hlist_add_head(&tr->hlist_ip, head); refcount_set(&tr->refcnt, 1); + INIT_LIST_HEAD(&tr->images); for (i = 0; i < BPF_TRAMP_MAX; i++) INIT_HLIST_HEAD(&tr->progs_hlist[i]); out: @@ -565,15 +566,22 @@ static void bpf_tramp_image_free(struct bpf_tramp_image *im) arch_free_bpf_trampoline(im->image, im->size); bpf_jit_uncharge_modmem(im->size); percpu_ref_exit(&im->pcref); + kfree(im->skips); kfree_rcu(im, rcu); } static void __bpf_tramp_image_put_deferred(struct work_struct *work) { struct bpf_tramp_image *im; + struct bpf_trampoline *tr; im = container_of(work, struct bpf_tramp_image, work); + tr = im->tr; + trampoline_lock(tr); + list_del(&im->list); + trampoline_unlock(tr); bpf_tramp_image_free(im); + bpf_trampoline_put(tr); } /* callback, fexit step 3 or fentry step 2 */ @@ -601,7 +609,7 @@ static void __bpf_tramp_image_put_rcu_tasks(struct rcu_head *rcu) struct bpf_tramp_image *im; im = container_of(rcu, struct bpf_tramp_image, rcu); - if (im->ip_after_call) + if (im->call_orig) /* the case of fmod_ret/fexit trampoline and CONFIG_PREEMPTION=y */ percpu_ref_kill(&im->pcref); else @@ -621,9 +629,9 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im) * * The trampoline is unreachable before bpf_tramp_image_put(). * - * First, patch the trampoline to avoid calling into fexit progs. - * The progs will be freed even if the original function is still - * executing or sleeping. + * Progs are patched out of the image when they are detached, see + * bpf_trampoline_skip_prog(), so they can be freed even if a task is + * still in the image. * In case of CONFIG_PREEMPT=y use call_rcu_tasks() to wait on * first few asm instructions to execute and call into * __bpf_tramp_enter->percpu_ref_get. @@ -637,11 +645,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im) * percpu_ref_kill will be waiting for. Hence the first * call_rcu_tasks() is not necessary. */ - if (im->ip_after_call) { - int err = bpf_arch_text_poke(im->ip_after_call, BPF_MOD_NOP, - BPF_MOD_JUMP, NULL, - im->ip_epilogue); - WARN_ON(err); + if (im->call_orig) { if (IS_ENABLED(CONFIG_TASKS_RCU)) call_rcu_tasks(&im->rcu, __bpf_tramp_image_put_rcu_tasks); else @@ -658,7 +662,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im) call_rcu_tasks_trace(&im->rcu, __bpf_tramp_image_put_rcu_tasks); } -static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size) +static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size, int nr_progs) { struct bpf_tramp_image *im; struct bpf_ksym *ksym; @@ -669,6 +673,10 @@ static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size) if (!im) goto out; + im->skips = kzalloc_objs(*im->skips, nr_progs); + if (!im->skips) + goto out_free_im; + err = bpf_jit_charge_modmem(size); if (err) goto out_free_im; @@ -695,6 +703,7 @@ out_free_image: out_uncharge: bpf_jit_uncharge_modmem(size); out_free_im: + kfree(im->skips); kfree(im); out: return ERR_PTR(err); @@ -771,11 +780,12 @@ again: goto out; } - im = bpf_tramp_image_alloc(tr->key, size); + im = bpf_tramp_image_alloc(tr->key, size, total); if (IS_ERR(im)) { err = PTR_ERR(im); goto out; } + im->call_orig = tr->flags & BPF_TRAMP_F_CALL_ORIG; err = arch_prepare_bpf_trampoline(im, im->image, im->image + size, &tr->func.model, tr->flags, tnodes, @@ -806,8 +816,14 @@ again: #endif out_free: - if (err) + if (err) { bpf_tramp_image_free(im); + } else { + /* track the image until it is freed, for bpf_trampoline_skip_prog() */ + refcount_inc(&tr->refcnt); + im->tr = tr; + list_add(&im->list, &tr->images); + } out: /* If any error happens, restore previous flags */ if (err) @@ -907,6 +923,7 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr, } hlist_add_head(&node->tramp_hlist, prog_list); + node->link->prog->aux->tramp_linked = true; if (kind == BPF_TRAMP_FSESSION) { tr->progs_cnt[BPF_TRAMP_FENTRY]++; fexit = fsession_exit(node); @@ -920,6 +937,41 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr, return 0; } +/* + * Patch the nop in front of a prog call to a jump over it. A task can be + * preempted anywhere in the image, so archs that need several instructions for + * a jump of any range patch a single near branch here instead. + */ +int __weak arch_bpf_trampoline_skip(void *nop, void *target) +{ + return bpf_arch_text_poke(nop, BPF_MOD_NOP, BPF_MOD_JUMP, NULL, target); +} + +/* + * prog was detached and can be freed, but tasks may still be running in images + * that call it, sleeping in an earlier prog for example. They can be in any + * image that is not freed yet, not only in cur_image, so patch all of them to + * jump over prog. + */ +static void bpf_trampoline_skip_prog(struct bpf_trampoline *tr, struct bpf_prog *prog) +{ + struct bpf_tramp_image *im; + int i, err; + + list_for_each_entry(im, &tr->images, list) { + for (i = 0; i < im->nr_skips; i++) { + struct bpf_tramp_skip *skip = &im->skips[i]; + + if (skip->prog != prog) + continue; + err = arch_bpf_trampoline_skip(skip->nop, skip->target); + WARN_ON_ONCE(err); + /* not a nop anymore, and prog's address can be reused */ + skip->prog = NULL; + } + } +} + static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr, struct bpf_tramp_node *node) { @@ -937,6 +989,7 @@ static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr, } hlist_del_init(&node->tramp_hlist); tr->progs_cnt[kind]--; + bpf_trampoline_skip_prog(tr, node->link->prog); } static int __bpf_trampoline_link_prog(struct bpf_tramp_node *node, @@ -1245,11 +1298,9 @@ void bpf_trampoline_put(struct bpf_trampoline *tr) if (WARN_ON_ONCE(!hlist_empty(&tr->progs_hlist[i]))) goto out; - /* This code will be executed even when the last bpf_tramp_image - * is alive. All progs are detached from the trampoline and the - * trampoline image is patched with jmp into epilogue to skip - * fexit progs. The fentry-only trampoline will be freed via - * multiple rcu callbacks. + /* + * All progs are detached and the last image has been freed, images + * hold a reference on the trampoline until then. */ hlist_del(&tr->hlist_key); hlist_del(&tr->hlist_ip); diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c index 41b49c56e123..5f874979b8d7 100644 --- a/kernel/bpf/verifier.c +++ b/kernel/bpf/verifier.c @@ -14826,7 +14826,15 @@ static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn "Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range."); return -EINVAL; } - reg_bounds_sync(dst_reg); + /* + * A packet pointer that keeps its id or range is checked against a + * range set from the checked pointer's umax, so var_off must not tighten + * its umax. r32 must still match var_off for reg_bounds_sanity_check(). + */ + if (reg_is_pkt_pointer(dst_reg) && (known || dst_reg->range > 0)) + __update_reg32_bounds(dst_reg); + else + reg_bounds_sync(dst_reg); bounds_ret = sanitize_check_bounds(env, insn, dst_reg); if (bounds_ret == -EACCES) return bounds_ret; |
