diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-02 12:59:32 -0700 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-02 12:59:32 -0700 |
| commit | 8f150ccedfbd610aa25509ff42365d70fb20478f (patch) | |
| tree | 1b22e147a9f8464940fcfa0483c96b8842c953c5 | |
| parent | d2dbe503fd806082acb0ca79a9d6641822988c2c (diff) | |
| parent | de020dc8049bfb2b22e3b6d99c031feb2e22d112 (diff) | |
Merge tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf
Pull bpf fixes from Alexei Starovoitov:
- Fix overflow of backward jump offset in constant blinding
(Alexei Starovoitov)
- Fix packet range of packet pointers sharing an id when var_off
tightens umax of one pointer and not the other (Alexei Starovoitov)
- Fix objects stuck in free_by_rcu_ttrace list of bpf memalloc
(Alexei Starovoitov)
- Fix use-after-free of progs detached from busy trampolines: wait for
an RCU tasks grace period before freeing trampoline progs, and patch
detached progs out of trampoline images that are still in use
(Florent Revest)
- Hold map BTF for the memory allocator destructor record to fix UAF in
deferred bpf_mem_alloc destruction (Kumar Kartikeya Dwivedi)
- Fix missing migration protection in resizable hashtab
lookup_and_delete batch operation (Ă–mer Mete Kaya)
* tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf:
bpf: Fix missing migration protection in __rhtab_map_lookup_and_delete_batch()
selftests/bpf: Add a test for objects stuck in free_by_rcu_ttrace
bpf: Fix objects stuck in free_by_rcu_ttrace
bpf: Factor out __do_call_rcu_ttrace()
selftests/bpf: Test packet range of pointers sharing an id
bpf: Fix packet range of pointers sharing an id
selftests/bpf: Detach a trampoline prog while a task sleeps before it
bpf: Skip detached progs in trampoline images that are still in use
bpf: Wait for an RCU tasks grace period before freeing trampoline progs
bpf: Hold map BTF for the memory allocator destructor record
bpf: Fix overflow of jump offset in constant blinding
23 files changed, 975 insertions, 172 deletions
diff --git a/arch/arm64/net/bpf_jit_comp.c b/arch/arm64/net/bpf_jit_comp.c index c5f55d6161fe..c1279fabce47 100644 --- a/arch/arm64/net/bpf_jit_comp.c +++ b/arch/arm64/net/bpf_jit_comp.c @@ -2418,10 +2418,11 @@ bool bpf_jit_supports_subprog_tailcalls(void) return true; } -static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node, - int bargs_off, int retval_off, int run_ctx_off, - bool save_ret) +static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_image *im, + struct bpf_tramp_node *node, int bargs_off, + int retval_off, int run_ctx_off, bool save_ret) { + void *skip; __le32 *branch; u64 enter_prog; u64 exit_prog; @@ -2431,6 +2432,10 @@ static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node, enter_prog = (u64)bpf_trampoline_enter(p); exit_prog = (u64)bpf_trampoline_exit(p); + /* nop, patched to skip this prog when it is detached */ + skip = ctx->ro_image + ctx->idx; + emit(A64_NOP, ctx); + if (node->cookie == 0) { /* if cookie is zero, one instruction is enough to store it */ emit(A64_STR64I(A64_ZR, A64_SP, run_ctx_off + cookie_off), ctx); @@ -2483,11 +2488,13 @@ static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node, emit(A64_ADD_I(1, A64_R(2), A64_SP, run_ctx_off), ctx); emit_call(exit_prog, ctx); + + bpf_tramp_image_add_skip(im, p, skip, ctx->ro_image + ctx->idx); } -static void invoke_bpf_mod_ret(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn, - int bargs_off, int retval_off, int run_ctx_off, - __le32 **branches) +static void invoke_bpf_mod_ret(struct jit_ctx *ctx, struct bpf_tramp_image *im, + struct bpf_tramp_nodes *tn, int bargs_off, + int retval_off, int run_ctx_off, __le32 **branches) { int i; @@ -2496,7 +2503,7 @@ static void invoke_bpf_mod_ret(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn, */ emit(A64_STR64I(A64_ZR, A64_SP, retval_off), ctx); for (i = 0; i < tn->nr_nodes; i++) { - invoke_bpf_prog(ctx, tn->nodes[i], bargs_off, retval_off, + invoke_bpf_prog(ctx, im, tn->nodes[i], bargs_off, retval_off, run_ctx_off, true); /* if (*(u64 *)(sp + retval_off) != 0) * goto do_fexit; @@ -2882,7 +2889,7 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im, store_func_meta(ctx, meta, func_meta_off); cookie_bargs_off--; } - invoke_bpf_prog(ctx, fentry->nodes[i], bargs_off, + invoke_bpf_prog(ctx, im, fentry->nodes[i], bargs_off, retval_off, run_ctx_off, flags & BPF_TRAMP_F_RET_FENTRY_RET); } @@ -2893,7 +2900,7 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im, if (!branches) return -ENOMEM; - invoke_bpf_mod_ret(ctx, fmod_ret, bargs_off, retval_off, + invoke_bpf_mod_ret(ctx, im, fmod_ret, bargs_off, retval_off, run_ctx_off, branches); } @@ -2906,9 +2913,6 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im, emit(A64_RET(A64_R(10)), ctx); /* store return value */ emit(A64_STR64I(A64_R(0), A64_SP, retval_off), ctx); - /* reserve a nop for bpf_tramp_image_put */ - im->ip_after_call = ctx->ro_image + ctx->idx; - emit(A64_NOP, ctx); } /* update the branches saved in invoke_bpf_mod_ret with cbnz */ @@ -2930,12 +2934,11 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im, store_func_meta(ctx, meta, func_meta_off); cookie_bargs_off--; } - invoke_bpf_prog(ctx, fexit->nodes[i], bargs_off, retval_off, + invoke_bpf_prog(ctx, im, fexit->nodes[i], bargs_off, retval_off, run_ctx_off, false); } if (flags & BPF_TRAMP_F_CALL_ORIG) { - im->ip_epilogue = ctx->ro_image + ctx->idx; /* for the first pass, assume the worst case */ if (!ctx->image) ctx->idx += 4; @@ -2994,7 +2997,7 @@ int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, .image = NULL, .idx = 0, }; - struct bpf_tramp_image im; + struct bpf_tramp_image im = {}; struct arg_aux aaux; int ret; @@ -3281,6 +3284,10 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t, * longer reachable, since bpf_tramp_image_put() function already * uses percpu_ref and task-based rcu to do the sync, no need to call * the sync version here, see bpf_tramp_image_put() for details. + * + * 3. when a detached prog is patched out of a trampoline, a CPU that + * still executes the old nop calls the prog before it went through + * a quiescent state, and the prog is freed after grace periods. */ ret = aarch64_insn_patch_text_nosync(ip, new_insn); out: diff --git a/arch/loongarch/net/bpf_jit.c b/arch/loongarch/net/bpf_jit.c index 4da278900938..e78eb582c300 100644 --- a/arch/loongarch/net/bpf_jit.c +++ b/arch/loongarch/net/bpf_jit.c @@ -1578,6 +1578,24 @@ void *bpf_arch_text_copy(void *dst, void *src, size_t len) return ret ? ERR_PTR(-EINVAL) : dst; } +int arch_bpf_trampoline_skip(void *nop, void *target) +{ + u32 old_insn = INSN_NOP; + u32 new_insn = larch_insn_gen_b((unsigned long)nop, (unsigned long)target); + int ret; + + if (memcmp(nop, &old_insn, LOONGARCH_INSN_SIZE)) + return -EFAULT; + + cpus_read_lock(); + mutex_lock(&text_mutex); + ret = larch_insn_text_copy(nop, &new_insn, LOONGARCH_INSN_SIZE); + mutex_unlock(&text_mutex); + cpus_read_unlock(); + + return ret; +} + int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t, enum bpf_text_poke_type new_t, void *old_addr, void *new_addr) @@ -1696,13 +1714,18 @@ static void restore_stk_args(struct jit_ctx *ctx, int nr_stk_args, int args_off, } } -static int invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *n, - int args_off, int retval_off, int run_ctx_off, bool save_ret) +static int invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_image *im, + struct bpf_tramp_node *n, int args_off, int retval_off, + int run_ctx_off, bool save_ret) { int ret; u32 *branch; struct bpf_prog *p = n->link->prog; int cookie_off = offsetof(struct bpf_tramp_run_ctx, bpf_cookie); + void *skip = ctx->ro_image + ctx->idx; + + /* nop, patched to a b over this prog when it is detached */ + emit_insn(ctx, nop); if (n->cookie) emit_store_stack_imm64(ctx, LOONGARCH_GPR_T1, @@ -1755,13 +1778,17 @@ static int invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *n, /* arg3: &run_ctx */ emit_insn(ctx, addid, LOONGARCH_GPR_A2, LOONGARCH_GPR_FP, -run_ctx_off); ret = emit_call(ctx, (const u64)bpf_trampoline_exit(p)); + if (ret) + return ret; - return ret; + bpf_tramp_image_add_skip(im, p, skip, ctx->ro_image + ctx->idx); + return 0; } -static int invoke_bpf(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn, - int args_off, int retval_off, int run_ctx_off, - int func_meta_off, bool save_ret, u64 func_meta, int cookie_off) +static int invoke_bpf(struct jit_ctx *ctx, struct bpf_tramp_image *im, + struct bpf_tramp_nodes *tn, int args_off, int retval_off, + int run_ctx_off, int func_meta_off, bool save_ret, + u64 func_meta, int cookie_off) { int i, cur_cookie = (cookie_off - args_off) / 8; @@ -1774,7 +1801,8 @@ static int invoke_bpf(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn, emit_store_stack_imm64(ctx, LOONGARCH_GPR_T1, -func_meta_off, meta); cur_cookie--; } - err = invoke_bpf_prog(ctx, tn->nodes[i], args_off, retval_off, run_ctx_off, save_ret); + err = invoke_bpf_prog(ctx, im, tn->nodes[i], args_off, retval_off, + run_ctx_off, save_ret); if (err) return err; } @@ -2017,7 +2045,7 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i } if (fentry->nr_nodes) { - ret = invoke_bpf(ctx, fentry, args_off, retval_off, run_ctx_off, func_meta_off, + ret = invoke_bpf(ctx, im, fentry, args_off, retval_off, run_ctx_off, func_meta_off, flags & BPF_TRAMP_F_RET_FENTRY_RET, func_meta, cookie_off); if (ret) return ret; @@ -2029,7 +2057,7 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i emit_insn(ctx, std, LOONGARCH_GPR_ZERO, LOONGARCH_GPR_FP, -retval_off); for (i = 0; i < fmod_ret->nr_nodes; i++) { - ret = invoke_bpf_prog(ctx, fmod_ret->nodes[i], + ret = invoke_bpf_prog(ctx, im, fmod_ret->nodes[i], args_off, retval_off, run_ctx_off, true); if (ret) goto out; @@ -2051,10 +2079,6 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i goto out; emit_insn(ctx, std, LOONGARCH_GPR_A0, LOONGARCH_GPR_FP, -retval_off); emit_insn(ctx, std, regmap[BPF_REG_0], LOONGARCH_GPR_FP, -(retval_off - 8)); - im->ip_after_call = ctx->ro_image + ctx->idx; - /* Reserve space for the move_imm + jirl instruction */ - for (i = 0; i < LOONGARCH_LONG_JUMP_NINSNS; i++) - emit_insn(ctx, nop); } for (i = 0; ctx->image && i < fmod_ret->nr_nodes; i++) { @@ -2068,14 +2092,13 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i emit_store_stack_imm64(ctx, LOONGARCH_GPR_T1, -func_meta_off, func_meta); if (fexit->nr_nodes) { - ret = invoke_bpf(ctx, fexit, args_off, retval_off, run_ctx_off, + ret = invoke_bpf(ctx, im, fexit, args_off, retval_off, run_ctx_off, func_meta_off, false, func_meta, cookie_off); if (ret) goto out; } if (flags & BPF_TRAMP_F_CALL_ORIG) { - im->ip_epilogue = ctx->ro_image + ctx->idx; move_addr(ctx, LOONGARCH_GPR_A0, (const u64)im); ret = emit_call(ctx, (const u64)__bpf_tramp_exit); if (ret) @@ -2177,11 +2200,8 @@ int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *func_addr) { int ret; - struct jit_ctx ctx; - struct bpf_tramp_image im; - - ctx.image = NULL; - ctx.idx = 0; + struct jit_ctx ctx = {}; + struct bpf_tramp_image im = {}; ret = __arch_prepare_bpf_trampoline(&ctx, &im, m, tnodes, func_addr, flags); diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c index 7b07b43575f1..7a612688e7a2 100644 --- a/arch/powerpc/net/bpf_jit_comp.c +++ b/arch/powerpc/net/bpf_jit_comp.c @@ -602,14 +602,18 @@ int arch_protect_bpf_trampoline(void *image, unsigned int size) } static int invoke_bpf_prog(u32 *image, u32 *ro_image, struct codegen_context *ctx, - struct bpf_tramp_node *n, int regs_off, int retval_off, - int run_ctx_off, bool save_ret) + struct bpf_tramp_image *im, struct bpf_tramp_node *n, + int regs_off, int retval_off, int run_ctx_off, bool save_ret) { struct bpf_prog *p = n->link->prog; ppc_inst_t branch_insn; - u32 jmp_idx; + u32 jmp_idx, skip_idx; int ret = 0; + /* nop, patched to skip this prog when it is detached */ + skip_idx = ctx->idx; + EMIT(PPC_RAW_NOP()); + /* Save cookie */ if (IS_ENABLED(CONFIG_PPC64)) { PPC_LI64(_R3, n->cookie); @@ -679,13 +683,17 @@ static int invoke_bpf_prog(u32 *image, u32 *ro_image, struct codegen_context *ct EMIT(PPC_RAW_ADDI(_R5, _R1, run_ctx_off)); ret = bpf_jit_emit_func_call_rel(image, ro_image, ctx, (unsigned long)bpf_trampoline_exit(p)); + if (ret) + return ret; - return ret; + if (ro_image) /* image is NULL for dummy pass */ + bpf_tramp_image_add_skip(im, p, &ro_image[skip_idx], &ro_image[ctx->idx]); + return 0; } static int invoke_bpf_mod_ret(u32 *image, u32 *ro_image, struct codegen_context *ctx, - struct bpf_tramp_nodes *tn, int regs_off, int retval_off, - int run_ctx_off, u32 *branches) + struct bpf_tramp_image *im, struct bpf_tramp_nodes *tn, + int regs_off, int retval_off, int run_ctx_off, u32 *branches) { int i; @@ -696,8 +704,8 @@ static int invoke_bpf_mod_ret(u32 *image, u32 *ro_image, struct codegen_context EMIT(PPC_RAW_LI(_R3, 0)); EMIT(PPC_RAW_STL(_R3, _R1, retval_off)); for (i = 0; i < tn->nr_nodes; i++) { - if (invoke_bpf_prog(image, ro_image, ctx, tn->nodes[i], regs_off, retval_off, - run_ctx_off, true)) + if (invoke_bpf_prog(image, ro_image, ctx, im, tn->nodes[i], regs_off, + retval_off, run_ctx_off, true)) return -EINVAL; /* @@ -1043,8 +1051,8 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im cookie_ctx_off--; } - if (invoke_bpf_prog(image, ro_image, ctx, fentry->nodes[i], regs_off, retval_off, - run_ctx_off, flags & BPF_TRAMP_F_RET_FENTRY_RET)) + if (invoke_bpf_prog(image, ro_image, ctx, im, fentry->nodes[i], regs_off, + retval_off, run_ctx_off, flags & BPF_TRAMP_F_RET_FENTRY_RET)) return -EINVAL; } @@ -1053,7 +1061,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im if (!branches) return -ENOMEM; - if (invoke_bpf_mod_ret(image, ro_image, ctx, fmod_ret, regs_off, retval_off, + if (invoke_bpf_mod_ret(image, ro_image, ctx, im, fmod_ret, regs_off, retval_off, run_ctx_off, branches)) { ret = -EINVAL; goto cleanup; @@ -1090,11 +1098,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im /* Restore updated tail_call_cnt */ if (flags & BPF_TRAMP_F_TAIL_CALL_CTX) bpf_trampoline_restore_tail_call_cnt(image, ctx, bpf_frame_size, r4_off); - - /* Reserve space to patch branch instruction to skip fexit progs */ - if (ro_image) /* image is NULL for dummy pass */ - im->ip_after_call = &((u32 *)ro_image)[ctx->idx]; - EMIT(PPC_RAW_NOP()); } /* Update branches saved in invoke_bpf_mod_ret with address of do_fexit */ @@ -1123,16 +1126,14 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im cookie_ctx_off--; } - if (invoke_bpf_prog(image, ro_image, ctx, fexit->nodes[i], regs_off, retval_off, - run_ctx_off, false)) { + if (invoke_bpf_prog(image, ro_image, ctx, im, fexit->nodes[i], regs_off, + retval_off, run_ctx_off, false)) { ret = -EINVAL; goto cleanup; } } if (flags & BPF_TRAMP_F_CALL_ORIG) { - if (ro_image) /* image is NULL for dummy pass */ - im->ip_epilogue = &((u32 *)ro_image)[ctx->idx]; PPC_LI_ADDR(_R3, im); ret = bpf_jit_emit_func_call_rel(image, ro_image, ctx, (unsigned long)__bpf_tramp_exit); @@ -1192,7 +1193,7 @@ cleanup: int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *func_addr) { - struct bpf_tramp_image im; + struct bpf_tramp_image im = {}; int ret; ret = __arch_prepare_bpf_trampoline(&im, NULL, NULL, NULL, m, flags, tnodes, func_addr); @@ -1320,7 +1321,7 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t, /* * If we are not poking at bpf prog entry, then we are simply patching in/out - * an unconditional branch instruction at im->ip_after_call + * an unconditional branch instruction in a trampoline image */ if (offset) { if (old_t == BPF_MOD_CALL || new_t == BPF_MOD_CALL) { diff --git a/arch/riscv/net/bpf_jit_comp64.c b/arch/riscv/net/bpf_jit_comp64.c index 151031e97a24..01fe66774f02 100644 --- a/arch/riscv/net/bpf_jit_comp64.c +++ b/arch/riscv/net/bpf_jit_comp64.c @@ -822,6 +822,24 @@ static int gen_jump_or_nops(void *target, void *ip, u32 *insns, bool is_call) return emit_jump_and_link(is_call ? RV_REG_T0 : RV_REG_ZERO, rvoff, false, &ctx); } +int arch_bpf_trampoline_skip(void *nop, void *target) +{ + u32 old_insn = rv_nop(); + u32 new_insn = rv_jal(RV_REG_ZERO, (target - nop) >> 1); + int ret; + + if (memcmp(nop, &old_insn, sizeof(old_insn))) + return -EFAULT; + + cpus_read_lock(); + mutex_lock(&text_mutex); + ret = patch_text(nop, &new_insn, sizeof(new_insn)); + mutex_unlock(&text_mutex); + cpus_read_unlock(); + + return ret; +} + int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t, enum bpf_text_poke_type new_t, void *old_addr, void *new_addr) @@ -904,12 +922,17 @@ static void emit_store_stack_imm64(u8 reg, int stack_off, u64 imm64, emit_sd(RV_REG_FP, stack_off, reg, ctx); } -static int invoke_bpf_prog(struct bpf_tramp_node *node, int args_off, int retval_off, - int run_ctx_off, bool save_ret, struct rv_jit_context *ctx) +static int invoke_bpf_prog(struct bpf_tramp_image *im, struct bpf_tramp_node *node, + int args_off, int retval_off, int run_ctx_off, bool save_ret, + struct rv_jit_context *ctx) { int ret, branch_off; struct bpf_prog *p = node->link->prog; int cookie_off = offsetof(struct bpf_tramp_run_ctx, bpf_cookie); + void *skip = ctx->ro_insns + ctx->ninsns; + + /* nop, patched to a jal over this prog when it is detached */ + emit(rv_nop(), ctx); if (node->cookie) emit_store_stack_imm64(RV_REG_T1, -run_ctx_off + cookie_off, node->cookie, ctx); @@ -962,13 +985,17 @@ static int invoke_bpf_prog(struct bpf_tramp_node *node, int args_off, int retval /* arg3: &run_ctx */ emit_addi(RV_REG_A2, RV_REG_FP, -run_ctx_off, ctx); ret = emit_call((const u64)bpf_trampoline_exit(p), true, ctx); + if (ret) + return ret; - return ret; + bpf_tramp_image_add_skip(im, p, skip, ctx->ro_insns + ctx->ninsns); + return 0; } -static int invoke_bpf(struct bpf_tramp_nodes *tn, int args_off, int retval_off, - int run_ctx_off, int func_meta_off, bool save_ret, u64 func_meta, - int cookie_off, struct rv_jit_context *ctx) +static int invoke_bpf(struct bpf_tramp_image *im, struct bpf_tramp_nodes *tn, + int args_off, int retval_off, int run_ctx_off, int func_meta_off, + bool save_ret, u64 func_meta, int cookie_off, + struct rv_jit_context *ctx) { int i, cur_cookie = (cookie_off - args_off) / 8; @@ -981,8 +1008,8 @@ static int invoke_bpf(struct bpf_tramp_nodes *tn, int args_off, int retval_off, emit_store_stack_imm64(RV_REG_T1, -func_meta_off, meta, ctx); cur_cookie--; } - err = invoke_bpf_prog(tn->nodes[i], args_off, retval_off, run_ctx_off, - save_ret, ctx); + err = invoke_bpf_prog(im, tn->nodes[i], args_off, retval_off, + run_ctx_off, save_ret, ctx); if (err) return err; } @@ -1170,7 +1197,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, } if (fentry->nr_nodes) { - ret = invoke_bpf(fentry, args_off, retval_off, run_ctx_off, func_meta_off, + ret = invoke_bpf(im, fentry, args_off, retval_off, run_ctx_off, func_meta_off, flags & BPF_TRAMP_F_RET_FENTRY_RET, func_meta, cookie_off, ctx); if (ret) return ret; @@ -1184,7 +1211,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, /* cleanup to avoid garbage return value confusion */ emit_sd(RV_REG_FP, -retval_off, RV_REG_ZERO, ctx); for (i = 0; i < fmod_ret->nr_nodes; i++) { - ret = invoke_bpf_prog(fmod_ret->nodes[i], args_off, retval_off, + ret = invoke_bpf_prog(im, fmod_ret->nodes[i], args_off, retval_off, run_ctx_off, true, ctx); if (ret) goto out; @@ -1211,10 +1238,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, emit_sd(RV_REG_FP, -tcc_off, RV_REG_TCC, ctx); emit_sd(RV_REG_FP, -retval_off, RV_REG_A0, ctx); emit_sd(RV_REG_FP, -(retval_off - 8), regmap[BPF_REG_0], ctx); - im->ip_after_call = ctx->ro_insns + ctx->ninsns; - /* 2 nops reserved for auipc+jalr pair */ - emit(rv_nop(), ctx); - emit(rv_nop(), ctx); } /* update branches saved in invoke_bpf_mod_ret with bnez */ @@ -1230,14 +1253,13 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, emit_store_stack_imm64(RV_REG_T1, -func_meta_off, func_meta, ctx); if (fexit->nr_nodes) { - ret = invoke_bpf(fexit, args_off, retval_off, run_ctx_off, func_meta_off, + ret = invoke_bpf(im, fexit, args_off, retval_off, run_ctx_off, func_meta_off, false, func_meta, cookie_off, ctx); if (ret) goto out; } if (flags & BPF_TRAMP_F_CALL_ORIG) { - im->ip_epilogue = ctx->ro_insns + ctx->ninsns; emit_imm(RV_REG_A0, ctx->insns ? (const s64)im : RV_MAX_COUNT_IMM, ctx); ret = emit_call((const u64)__bpf_tramp_exit, true, ctx); if (ret) @@ -1299,7 +1321,7 @@ out: int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *func_addr) { - struct bpf_tramp_image im; + struct bpf_tramp_image im = {}; struct rv_jit_context ctx; int ret; diff --git a/arch/s390/net/bpf_jit_comp.c b/arch/s390/net/bpf_jit_comp.c index c4b47070bb59..1b2566012b63 100644 --- a/arch/s390/net/bpf_jit_comp.c +++ b/arch/s390/net/bpf_jit_comp.c @@ -2566,6 +2566,8 @@ struct bpf_tramp_jit { int r14_off; /* Offset of saved %r14, has to be at the * bottom */ int do_fexit; /* do_fexit: label */ + int skip[BPF_MAX_TRAMP_LINKS]; /* skip: labels after each prog */ + int nr_progs; }; static void load_imm64(struct bpf_jit *jit, int dst_reg, u64 val) @@ -2584,6 +2586,7 @@ static void emit_store_stack_imm64(struct bpf_jit *jit, int tmp_reg, int stack_o } static int invoke_bpf_prog(struct bpf_tramp_jit *tjit, + struct bpf_tramp_image *im, const struct btf_func_model *m, struct bpf_tramp_node *node, bool save_ret) { @@ -2591,8 +2594,20 @@ static int invoke_bpf_prog(struct bpf_tramp_jit *tjit, int cookie_off = tjit->run_ctx_off + offsetof(struct bpf_tramp_run_ctx, bpf_cookie); struct bpf_prog *p = node->link->prog; + void *skip = jit->prg_buf + jit->prg; + int idx = tjit->nr_progs++; int patch; + if (idx >= ARRAY_SIZE(tjit->skip)) + return -E2BIG; + + /* + * nop, patched to skip this prog when it is detached + */ + + /* brcl 0,skip */ + EMIT6_PCREL_RILC(0xc0040000, 0, tjit->skip[idx]); + /* * run_ctx.cookie = node->cookie; */ @@ -2652,10 +2667,15 @@ static int invoke_bpf_prog(struct bpf_tramp_jit *tjit, /* brasl %r14,__bpf_prog_exit */ EMIT6_PCREL_RILB_PTR(0xc0050000, REG_14, bpf_trampoline_exit(p)); + /* skip: */ + tjit->skip[idx] = jit->prg; + bpf_tramp_image_add_skip(im, p, skip, jit->prg_buf + jit->prg); + return 0; } static int invoke_bpf(struct bpf_tramp_jit *tjit, + struct bpf_tramp_image *im, const struct btf_func_model *m, struct bpf_tramp_nodes *tn, bool save_ret, u64 func_meta, int cookie_off) @@ -2670,7 +2690,7 @@ static int invoke_bpf(struct bpf_tramp_jit *tjit, emit_store_stack_imm64(jit, REG_0, tjit->func_meta_off, meta); cur_cookie--; } - if (invoke_bpf_prog(tjit, m, tn->nodes[i], save_ret)) + if (invoke_bpf_prog(tjit, im, m, tn->nodes[i], save_ret)) return -EINVAL; } @@ -2712,6 +2732,11 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, u64 func_meta; int i, j; + /* The skip labels are taken from the previous pass. */ + tjit->nr_progs = 0; + if (im) + im->nr_skips = 0; + /* Support as many stack arguments as "mvc" instruction can handle. */ nr_reg_args = min_t(int, m->nr_args, MAX_NR_REG_ARGS); nr_stack_args = m->nr_args - nr_reg_args; @@ -2875,7 +2900,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, emit_store_stack_imm64(jit, REG_0, tjit->retval_off, 0); } - if (invoke_bpf(tjit, m, fentry, flags & BPF_TRAMP_F_RET_FENTRY_RET, + if (invoke_bpf(tjit, im, m, fentry, flags & BPF_TRAMP_F_RET_FENTRY_RET, func_meta, cookie_off)) return -EINVAL; @@ -2889,7 +2914,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, 0xf000 | tjit->retval_off); for (i = 0; i < fmod_ret->nr_nodes; i++) { - if (invoke_bpf_prog(tjit, m, fmod_ret->nodes[i], true)) + if (invoke_bpf_prog(tjit, im, m, fmod_ret->nodes[i], true)) return -EINVAL; /* @@ -2943,15 +2968,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, /* mvc tccnt_off(%r15),tail_call_cnt(4,%r15) */ _EMIT6(0xd203f000 | tjit->tccnt_off, 0xf000 | offsetof(struct prog_frame, tail_call_cnt)); - - im->ip_after_call = jit->prg_buf + jit->prg; - - /* - * The following nop will be patched by bpf_tramp_image_put(). - */ - - /* brcl 0,im->ip_epilogue */ - EMIT6_PCREL_RILC(0xc0040000, 0, (u64)im->ip_epilogue); } /* Set the "is_return" flag for fsession. */ @@ -2962,12 +2978,10 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, /* do_fexit: */ tjit->do_fexit = jit->prg; - if (invoke_bpf(tjit, m, fexit, false, func_meta, cookie_off)) + if (invoke_bpf(tjit, im, m, fexit, false, func_meta, cookie_off)) return -EINVAL; if (flags & BPF_TRAMP_F_CALL_ORIG) { - im->ip_epilogue = jit->prg_buf + jit->prg; - /* * __bpf_tramp_exit(im); */ @@ -3016,7 +3030,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *orig_call) { - struct bpf_tramp_image im; + struct bpf_tramp_image im = {}; struct bpf_tramp_jit tjit; int ret; diff --git a/arch/x86/net/bpf_jit_comp.c b/arch/x86/net/bpf_jit_comp.c index 2853e87797a7..6bca87457e87 100644 --- a/arch/x86/net/bpf_jit_comp.c +++ b/arch/x86/net/bpf_jit_comp.c @@ -3217,16 +3217,21 @@ static void restore_regs(const struct btf_func_model *m, u8 **prog, } static int invoke_bpf_prog(const struct btf_func_model *m, u8 **pprog, + struct bpf_tramp_image *im, struct bpf_tramp_node *node, int stack_size, int run_ctx_off, bool save_ret, void *image, void *rw_image) { u8 *prog = *pprog; - u8 *jmp_insn; + u8 *jmp_insn, *skip; int ctx_cookie_off = offsetof(struct bpf_tramp_run_ctx, bpf_cookie); struct bpf_prog *p = node->link->prog; u64 cookie = node->cookie; + /* nop, patched to skip this prog when it is detached */ + skip = image + (prog - (u8 *)rw_image); + emit_nops(&prog, X86_PATCH_SIZE); + /* mov rdi, cookie */ emit_mov_imm64(&prog, BPF_REG_1, (long) cookie >> 32, (u32) (long) cookie); @@ -3301,6 +3306,8 @@ static int invoke_bpf_prog(const struct btf_func_model *m, u8 **pprog, if (emit_rsb_call(&prog, bpf_trampoline_exit(p), image + (prog - (u8 *)rw_image))) return -EINVAL; + bpf_tramp_image_add_skip(im, p, skip, image + (prog - (u8 *)rw_image)); + *pprog = prog; return 0; } @@ -3332,6 +3339,7 @@ static int emit_cond_near_jump(u8 **pprog, void *func, void *ip, u8 jmp_cond) } static int invoke_bpf(const struct btf_func_model *m, u8 **pprog, + struct bpf_tramp_image *im, struct bpf_tramp_nodes *tl, int stack_size, int run_ctx_off, int func_meta_off, bool save_ret, void *image, void *rw_image, u64 func_meta, @@ -3346,7 +3354,7 @@ static int invoke_bpf(const struct btf_func_model *m, u8 **pprog, func_meta | (cur_cookie << BPF_TRAMP_COOKIE_INDEX_SHIFT)); cur_cookie--; } - if (invoke_bpf_prog(m, &prog, tl->nodes[i], stack_size, + if (invoke_bpf_prog(m, &prog, im, tl->nodes[i], stack_size, run_ctx_off, save_ret, image, rw_image)) return -EINVAL; } @@ -3355,6 +3363,7 @@ static int invoke_bpf(const struct btf_func_model *m, u8 **pprog, } static int invoke_bpf_mod_ret(const struct btf_func_model *m, u8 **pprog, + struct bpf_tramp_image *im, struct bpf_tramp_nodes *tl, int stack_size, int run_ctx_off, u8 **branches, void *image, void *rw_image) @@ -3368,7 +3377,7 @@ static int invoke_bpf_mod_ret(const struct btf_func_model *m, u8 **pprog, emit_mov_imm32(&prog, false, BPF_REG_0, 0); emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_0, -8); for (i = 0; i < tl->nr_nodes; i++) { - if (invoke_bpf_prog(m, &prog, tl->nodes[i], stack_size, run_ctx_off, true, + if (invoke_bpf_prog(m, &prog, im, tl->nodes[i], stack_size, run_ctx_off, true, image, rw_image)) return -EINVAL; @@ -3640,7 +3649,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im } if (fentry->nr_nodes) { - if (invoke_bpf(m, &prog, fentry, regs_off, run_ctx_off, func_meta_off, + if (invoke_bpf(m, &prog, im, fentry, regs_off, run_ctx_off, func_meta_off, flags & BPF_TRAMP_F_RET_FENTRY_RET, image, rw_image, func_meta, cookie_off)) return -EINVAL; @@ -3652,7 +3661,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im if (!branches) return -ENOMEM; - if (invoke_bpf_mod_ret(m, &prog, fmod_ret, regs_off, + if (invoke_bpf_mod_ret(m, &prog, im, fmod_ret, regs_off, run_ctx_off, branches, image, rw_image)) { ret = -EINVAL; goto cleanup; @@ -3682,8 +3691,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im } /* remember return value in a stack for bpf prog to access */ emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_0, -8); - im->ip_after_call = image + (prog - (u8 *)rw_image); - emit_nops(&prog, X86_PATCH_SIZE); } if (fmod_ret->nr_nodes) { @@ -3708,7 +3715,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im emit_store_stack_imm64(&prog, BPF_REG_0, -func_meta_off, func_meta); if (fexit->nr_nodes) { - if (invoke_bpf(m, &prog, fexit, regs_off, run_ctx_off, func_meta_off, + if (invoke_bpf(m, &prog, im, fexit, regs_off, run_ctx_off, func_meta_off, false, image, rw_image, func_meta, cookie_off)) { ret = -EINVAL; goto cleanup; @@ -3723,7 +3730,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im * restored to R0. */ if (flags & BPF_TRAMP_F_CALL_ORIG) { - im->ip_epilogue = image + (prog - (u8 *)rw_image); /* arg1: mov rdi, im */ emit_mov_imm64(&prog, BPF_REG_1, (long) im >> 32, (u32) (long) im); if (emit_rsb_call(&prog, __bpf_tramp_exit, image + (prog - (u8 *)rw_image))) { @@ -3811,7 +3817,7 @@ out: int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *func_addr) { - struct bpf_tramp_image im; + struct bpf_tramp_image im = {}; void *image; int ret; diff --git a/include/linux/bpf.h b/include/linux/bpf.h index 1d2676782d70..0ecb9418dfbd 100644 --- a/include/linux/bpf.h +++ b/include/linux/bpf.h @@ -1258,11 +1258,15 @@ struct btf_func_model { #define BPF_TRAMP_F_INDIRECT BIT(8) /* Each call __bpf_prog_enter + call bpf_func + call __bpf_prog_exit is ~50 - * bytes on x86. + * bytes on x86. The trampoline image has to fit in PAGE_SIZE. */ enum { -#if defined(__s390x__) +#if defined(__s390x__) || defined(__powerpc64__) BPF_MAX_TRAMP_LINKS = 27, +#elif defined(__x86_64__) + BPF_MAX_TRAMP_LINKS = 36, +#elif defined(__aarch64__) + BPF_MAX_TRAMP_LINKS = 37, #else BPF_MAX_TRAMP_LINKS = 38, #endif @@ -1315,6 +1319,7 @@ int arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *image, void *i void *arch_alloc_bpf_trampoline(unsigned int size); void arch_free_bpf_trampoline(void *image, unsigned int size); int __must_check arch_protect_bpf_trampoline(void *image, unsigned int size); +int arch_bpf_trampoline_skip(void *nop, void *target); int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags, struct bpf_tramp_nodes *tnodes, void *func_addr); @@ -1363,19 +1368,48 @@ enum bpf_tramp_prog_type { BPF_TRAMP_FSESSION, }; +/* + * Each prog call in a trampoline image is preceded by a nop. When the prog is + * detached, the nop is patched to a jump to target, right after the call, so + * that tasks still running in the image skip the prog. + */ +struct bpf_tramp_skip { + struct bpf_prog *prog; + void *nop; + void *target; +}; + struct bpf_tramp_image { void *image; int size; struct bpf_ksym ksym; struct percpu_ref pcref; - void *ip_after_call; - void *ip_epilogue; + bool call_orig; + /* entry in tr->images, the image holds a reference on tr */ + struct bpf_trampoline *tr; + struct list_head list; + int nr_skips; + struct bpf_tramp_skip *skips; union { struct rcu_head rcu; struct work_struct work; }; }; +static inline void bpf_tramp_image_add_skip(struct bpf_tramp_image *im, struct bpf_prog *prog, + void *nop, void *target) +{ + struct bpf_tramp_skip *skip; + + /* struct_ops trampolines and arch_bpf_trampoline_size() have no image */ + if (!im || !im->skips) + return; + skip = &im->skips[im->nr_skips++]; + skip->prog = prog; + skip->nop = nop; + skip->target = target; +} + struct bpf_trampoline { /* hlist for trampoline_key_table */ struct hlist_node hlist_key; @@ -1402,6 +1436,8 @@ struct bpf_trampoline { int progs_cnt[BPF_TRAMP_MAX]; /* Executable image of trampoline */ struct bpf_tramp_image *cur_image; + /* Images not freed yet, cur_image and older ones still in use */ + struct list_head images; /* Used as temporary old image storage for multi_attach */ struct { struct bpf_tramp_image *old_image; @@ -1770,6 +1806,7 @@ struct bpf_prog_aux { bool offload_requested; /* Program is bound and offloaded to the netdev. */ bool attach_btf_trace; /* true if attaching to BTF-enabled raw tp */ bool attach_tracing_prog; /* true if tracing another tracing program */ + bool tramp_linked; /* true if it was ever linked to a trampoline */ bool func_proto_unreliable; bool tail_call_reachable; bool xdp_has_frags; diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c index 2e3bf8113ae9..7f11555a5070 100644 --- a/kernel/bpf/core.c +++ b/kernel/bpf/core.c @@ -1362,7 +1362,7 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from, { struct bpf_insn *to = to_buff; u32 imm_rnd = get_random_u32(); - s16 off; + int off; BUILD_BUG_ON(BPF_REG_PARAMS + 2 != MAX_BPF_JIT_REG); BUILD_BUG_ON(BPF_REG_AX + 1 != MAX_BPF_JIT_REG); @@ -1438,6 +1438,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from, off = from->off; if (off < 0) off -= 2; + if (off < S16_MIN) + return -ERANGE; *to++ = BPF_ALU64_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm); *to++ = BPF_ALU64_IMM(BPF_XOR, BPF_REG_AX, imm_rnd); *to++ = BPF_JMP_REG(from->code, from->dst_reg, BPF_REG_AX, off); @@ -1458,6 +1460,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from, off = from->off; if (off < 0) off -= 2; + if (off < S16_MIN) + return -ERANGE; *to++ = BPF_ALU32_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm); *to++ = BPF_ALU32_IMM(BPF_XOR, BPF_REG_AX, imm_rnd); *to++ = BPF_JMP32_REG(from->code, from->dst_reg, BPF_REG_AX, @@ -1606,7 +1610,9 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp if (!rewritten) continue; - if (env) + if (rewritten < 0) + tmp = ERR_PTR(rewritten); + else if (env) tmp = bpf_patch_insn_data(env, i, insn_buff, rewritten); else tmp = bpf_patch_insn_single(clone, i, insn_buff, rewritten); diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c index f9464e566f10..13a2356c84cf 100644 --- a/kernel/bpf/hashtab.c +++ b/kernel/bpf/hashtab.c @@ -128,6 +128,7 @@ struct htab_elem { struct htab_btf_record { struct btf_record *record; + struct btf *btf; u32 key_size; }; @@ -497,8 +498,13 @@ static void htab_dtor_ctx_free(void *ctx) { struct htab_btf_record *hrec = ctx; + /* + * The duplicated record still points into the map BTF, so free it + * before dropping the reference that keeps that BTF alive. + */ btf_record_free(hrec->record); - kfree(ctx); + btf_put(hrec->btf); + kfree(hrec); } static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma, @@ -521,6 +527,15 @@ static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma, kfree(hrec); return err; } + /* + * btf_record_dup() only acquires kernel and module BTF. Fields whose + * types live in the map BTF keep pointing into it: kptrs to local + * types refer to map->btf, and graph roots carry a value record owned + * by its struct meta table. The context can outlive the map when the + * allocator defers its teardown, so hold a reference of our own. + */ + hrec->btf = map->btf; + btf_get(hrec->btf); bpf_mem_alloc_set_dtor(ma, dtor, htab_dtor_ctx_free, hrec); return 0; } @@ -3359,8 +3374,10 @@ static int __rhtab_map_lookup_and_delete_batch(struct bpf_map *map, } if (do_delete) { + migrate_disable(); for (i = 0; i < total; i++) rhtab_delete_elem(rhtab, del_elems[i], NULL, 0); + migrate_enable(); } rcu_read_unlock(); diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c index 8a8f088e83e6..15684d0fc883 100644 --- a/kernel/bpf/memalloc.c +++ b/kernel/bpf/memalloc.c @@ -118,6 +118,11 @@ struct bpf_mem_cache { struct llist_head free_by_rcu_ttrace; struct llist_head waiting_for_gp_ttrace; struct rcu_head rcu_ttrace; + /* + * 0 - idle + * 1 - __free_rcu() is queued + * 2 - __free_rcu() is queued and free_by_rcu_ttrace got more objects since + */ atomic_t call_rcu_ttrace_in_progress; raw_spinlock_t lock; }; @@ -276,6 +281,8 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per return cnt; } +static void __do_call_rcu_ttrace(struct bpf_mem_cache *c); + static void __free_rcu(struct rcu_head *head) { struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace); @@ -285,7 +292,19 @@ static void __free_rcu(struct rcu_head *head) llnode = llist_del_all(&c->waiting_for_gp_ttrace); free_all(c, llnode, !!c->percpu_size); - atomic_set(&c->call_rcu_ttrace_in_progress, 0); + + /* + * do_call_rcu_ttrace() that ran while GP was in flight left its objects + * in free_by_rcu_ttrace. This cache may never free or alloc in bulk + * again, so start the next GP from here. + * 'c' can be freed as soon as call_rcu_ttrace_in_progress is zero. + */ + if (atomic_cmpxchg(&c->call_rcu_ttrace_in_progress, 1, 0) == 1) + return; + + /* Pairs with synchronize_rcu() in free_mem_alloc() */ + guard(rcu)(); + __do_call_rcu_ttrace(c); } static void enque_to_free(struct bpf_mem_cache *c, void *obj) @@ -298,18 +317,15 @@ static void enque_to_free(struct bpf_mem_cache *c, void *obj) llist_add(llnode, &c->free_by_rcu_ttrace); } -static void do_call_rcu_ttrace(struct bpf_mem_cache *c) +static void __do_call_rcu_ttrace(struct bpf_mem_cache *c) { struct llist_node *llnode, *t; - if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) { - if (unlikely(READ_ONCE(c->draining))) { - scoped_guard(raw_spinlock_irqsave, &c->lock) - llnode = llist_del_all(&c->free_by_rcu_ttrace); - free_all(c, llnode, !!c->percpu_size); - } - return; - } + /* + * Must be done before llist_del_all(). Objects that it misses were + * added by do_call_rcu_ttrace() that will set 2 after this store. + */ + atomic_set(&c->call_rcu_ttrace_in_progress, 1); WARN_ON_ONCE(!llist_empty(&c->waiting_for_gp_ttrace)); llist_for_each_safe(llnode, t, llist_del_all(&c->free_by_rcu_ttrace)) @@ -328,6 +344,22 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c) call_rcu_tasks_trace(&c->rcu_ttrace, __free_rcu); } +static void do_call_rcu_ttrace(struct bpf_mem_cache *c) +{ + struct llist_node *llnode; + + if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 2)) { + if (unlikely(READ_ONCE(c->draining))) { + scoped_guard(raw_spinlock_irqsave, &c->lock) + llnode = llist_del_all(&c->free_by_rcu_ttrace); + free_all(c, llnode, !!c->percpu_size); + } + return; + } + + __do_call_rcu_ttrace(c); +} + static void free_bulk(struct bpf_mem_cache *c) { struct bpf_mem_cache *tgt = c->tgt; @@ -700,7 +732,12 @@ static void free_mem_alloc(struct bpf_mem_alloc *ma) * to wait for the pending __free_by_rcu(), and __free_rcu(). RCU Tasks * Trace grace period implies RCU grace period, so all __free_rcu don't * need extra call_rcu() (and thus extra rcu_barrier() here). + * + * __free_rcu() queues itself again unless it sees 'draining'. After + * synchronize_rcu() it either did that already or will not do it, so + * rcu_barrier_tasks_trace() cannot miss it. */ + synchronize_rcu(); rcu_barrier(); /* wait for __free_by_rcu */ rcu_barrier_tasks_trace(); /* wait for __free_rcu */ free_mem_alloc_no_barrier(ma); diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c index 244a939b9d2d..96217b99399d 100644 --- a/kernel/bpf/syscall.c +++ b/kernel/bpf/syscall.c @@ -2448,6 +2448,21 @@ static void __bpf_prog_put_rcu(struct rcu_head *rcu) bpf_prog_free(aux->prog); } +/* + * Progs called from a trampoline can also be reached by a task that was + * preempted in the trampoline before the prog's enter helper took its RCU + * read lock, wait for those first. + */ +static void __bpf_prog_put_rcu_tasks(struct rcu_head *rcu) +{ + struct bpf_prog *prog = container_of(rcu, struct bpf_prog_aux, rcu)->prog; + + if (prog->sleepable) + call_rcu_tasks_trace(rcu, __bpf_prog_put_rcu); + else + call_rcu(rcu, __bpf_prog_put_rcu); +} + static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred) { bpf_prog_kallsyms_del_all(prog); @@ -2461,7 +2476,9 @@ static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred) btf_put(prog->aux->attach_btf); if (deferred) { - if (prog->sleepable) + if (IS_ENABLED(CONFIG_TASKS_RCU) && prog->aux->tramp_linked) + call_rcu_tasks(&prog->aux->rcu, __bpf_prog_put_rcu_tasks); + else if (prog->sleepable) call_rcu_tasks_trace(&prog->aux->rcu, __bpf_prog_put_rcu); else call_rcu(&prog->aux->rcu, __bpf_prog_put_rcu); diff --git a/kernel/bpf/trampoline.c b/kernel/bpf/trampoline.c index 90b70ea0d370..bf4cb3dd444d 100644 --- a/kernel/bpf/trampoline.c +++ b/kernel/bpf/trampoline.c @@ -401,6 +401,7 @@ static struct bpf_trampoline *bpf_trampoline_lookup(u64 key, unsigned long ip) head = &trampoline_ip_table[hash_64(tr->ip, TRAMPOLINE_HASH_BITS)]; hlist_add_head(&tr->hlist_ip, head); refcount_set(&tr->refcnt, 1); + INIT_LIST_HEAD(&tr->images); for (i = 0; i < BPF_TRAMP_MAX; i++) INIT_HLIST_HEAD(&tr->progs_hlist[i]); out: @@ -565,15 +566,22 @@ static void bpf_tramp_image_free(struct bpf_tramp_image *im) arch_free_bpf_trampoline(im->image, im->size); bpf_jit_uncharge_modmem(im->size); percpu_ref_exit(&im->pcref); + kfree(im->skips); kfree_rcu(im, rcu); } static void __bpf_tramp_image_put_deferred(struct work_struct *work) { struct bpf_tramp_image *im; + struct bpf_trampoline *tr; im = container_of(work, struct bpf_tramp_image, work); + tr = im->tr; + trampoline_lock(tr); + list_del(&im->list); + trampoline_unlock(tr); bpf_tramp_image_free(im); + bpf_trampoline_put(tr); } /* callback, fexit step 3 or fentry step 2 */ @@ -601,7 +609,7 @@ static void __bpf_tramp_image_put_rcu_tasks(struct rcu_head *rcu) struct bpf_tramp_image *im; im = container_of(rcu, struct bpf_tramp_image, rcu); - if (im->ip_after_call) + if (im->call_orig) /* the case of fmod_ret/fexit trampoline and CONFIG_PREEMPTION=y */ percpu_ref_kill(&im->pcref); else @@ -621,9 +629,9 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im) * * The trampoline is unreachable before bpf_tramp_image_put(). * - * First, patch the trampoline to avoid calling into fexit progs. - * The progs will be freed even if the original function is still - * executing or sleeping. + * Progs are patched out of the image when they are detached, see + * bpf_trampoline_skip_prog(), so they can be freed even if a task is + * still in the image. * In case of CONFIG_PREEMPT=y use call_rcu_tasks() to wait on * first few asm instructions to execute and call into * __bpf_tramp_enter->percpu_ref_get. @@ -637,11 +645,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im) * percpu_ref_kill will be waiting for. Hence the first * call_rcu_tasks() is not necessary. */ - if (im->ip_after_call) { - int err = bpf_arch_text_poke(im->ip_after_call, BPF_MOD_NOP, - BPF_MOD_JUMP, NULL, - im->ip_epilogue); - WARN_ON(err); + if (im->call_orig) { if (IS_ENABLED(CONFIG_TASKS_RCU)) call_rcu_tasks(&im->rcu, __bpf_tramp_image_put_rcu_tasks); else @@ -658,7 +662,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im) call_rcu_tasks_trace(&im->rcu, __bpf_tramp_image_put_rcu_tasks); } -static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size) +static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size, int nr_progs) { struct bpf_tramp_image *im; struct bpf_ksym *ksym; @@ -669,6 +673,10 @@ static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size) if (!im) goto out; + im->skips = kzalloc_objs(*im->skips, nr_progs); + if (!im->skips) + goto out_free_im; + err = bpf_jit_charge_modmem(size); if (err) goto out_free_im; @@ -695,6 +703,7 @@ out_free_image: out_uncharge: bpf_jit_uncharge_modmem(size); out_free_im: + kfree(im->skips); kfree(im); out: return ERR_PTR(err); @@ -771,11 +780,12 @@ again: goto out; } - im = bpf_tramp_image_alloc(tr->key, size); + im = bpf_tramp_image_alloc(tr->key, size, total); if (IS_ERR(im)) { err = PTR_ERR(im); goto out; } + im->call_orig = tr->flags & BPF_TRAMP_F_CALL_ORIG; err = arch_prepare_bpf_trampoline(im, im->image, im->image + size, &tr->func.model, tr->flags, tnodes, @@ -806,8 +816,14 @@ again: #endif out_free: - if (err) + if (err) { bpf_tramp_image_free(im); + } else { + /* track the image until it is freed, for bpf_trampoline_skip_prog() */ + refcount_inc(&tr->refcnt); + im->tr = tr; + list_add(&im->list, &tr->images); + } out: /* If any error happens, restore previous flags */ if (err) @@ -907,6 +923,7 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr, } hlist_add_head(&node->tramp_hlist, prog_list); + node->link->prog->aux->tramp_linked = true; if (kind == BPF_TRAMP_FSESSION) { tr->progs_cnt[BPF_TRAMP_FENTRY]++; fexit = fsession_exit(node); @@ -920,6 +937,41 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr, return 0; } +/* + * Patch the nop in front of a prog call to a jump over it. A task can be + * preempted anywhere in the image, so archs that need several instructions for + * a jump of any range patch a single near branch here instead. + */ +int __weak arch_bpf_trampoline_skip(void *nop, void *target) +{ + return bpf_arch_text_poke(nop, BPF_MOD_NOP, BPF_MOD_JUMP, NULL, target); +} + +/* + * prog was detached and can be freed, but tasks may still be running in images + * that call it, sleeping in an earlier prog for example. They can be in any + * image that is not freed yet, not only in cur_image, so patch all of them to + * jump over prog. + */ +static void bpf_trampoline_skip_prog(struct bpf_trampoline *tr, struct bpf_prog *prog) +{ + struct bpf_tramp_image *im; + int i, err; + + list_for_each_entry(im, &tr->images, list) { + for (i = 0; i < im->nr_skips; i++) { + struct bpf_tramp_skip *skip = &im->skips[i]; + + if (skip->prog != prog) + continue; + err = arch_bpf_trampoline_skip(skip->nop, skip->target); + WARN_ON_ONCE(err); + /* not a nop anymore, and prog's address can be reused */ + skip->prog = NULL; + } + } +} + static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr, struct bpf_tramp_node *node) { @@ -937,6 +989,7 @@ static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr, } hlist_del_init(&node->tramp_hlist); tr->progs_cnt[kind]--; + bpf_trampoline_skip_prog(tr, node->link->prog); } static int __bpf_trampoline_link_prog(struct bpf_tramp_node *node, @@ -1245,11 +1298,9 @@ void bpf_trampoline_put(struct bpf_trampoline *tr) if (WARN_ON_ONCE(!hlist_empty(&tr->progs_hlist[i]))) goto out; - /* This code will be executed even when the last bpf_tramp_image - * is alive. All progs are detached from the trampoline and the - * trampoline image is patched with jmp into epilogue to skip - * fexit progs. The fentry-only trampoline will be freed via - * multiple rcu callbacks. + /* + * All progs are detached and the last image has been freed, images + * hold a reference on the trampoline until then. */ hlist_del(&tr->hlist_key); hlist_del(&tr->hlist_ip); diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c index 41b49c56e123..5f874979b8d7 100644 --- a/kernel/bpf/verifier.c +++ b/kernel/bpf/verifier.c @@ -14826,7 +14826,15 @@ static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn "Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range."); return -EINVAL; } - reg_bounds_sync(dst_reg); + /* + * A packet pointer that keeps its id or range is checked against a + * range set from the checked pointer's umax, so var_off must not tighten + * its umax. r32 must still match var_off for reg_bounds_sanity_check(). + */ + if (reg_is_pkt_pointer(dst_reg) && (known || dst_reg->range > 0)) + __update_reg32_bounds(dst_reg); + else + reg_bounds_sync(dst_reg); bounds_ret = sanitize_check_bounds(env, insn, dst_reg); if (bounds_ret == -EACCES) return bounds_ret; diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c b/tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c new file mode 100644 index 000000000000..a1d41b109940 --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: GPL-2.0 +#include <test_progs.h> +#include "bpf_ma_ttrace.skel.h" + +#define NR_ELEMS 4096 + +/* + * The first free_bulk() starts RCU tasks trace GP. The rest of the elements are + * deleted while it's in flight. They should be freed without further alloc or + * free from this map. + */ +void test_bpf_ma_ttrace(void) +{ + LIBBPF_OPTS(bpf_test_run_opts, opts); + struct bpf_ma_ttrace *skel; + __u32 cnt = NR_ELEMS; + long *vals = NULL; + int *keys = NULL; + int i, err, fd, nr_cpus; + + skel = bpf_ma_ttrace__open_and_load(); + if (!ASSERT_OK_PTR(skel, "open_and_load")) + return; + nr_cpus = libbpf_num_possible_cpus(); + if (!ASSERT_GT(nr_cpus, 0, "nr_cpus")) + goto out; + skel->bss->nr_cpus = nr_cpus; + + keys = calloc(NR_ELEMS, sizeof(*keys)); + vals = calloc(NR_ELEMS, sizeof(*vals)); + if (!ASSERT_OK_PTR(keys, "keys") || !ASSERT_OK_PTR(vals, "vals")) + goto out; + for (i = 0; i < NR_ELEMS; i++) + keys[i] = i; + + fd = bpf_map__fd(skel->maps.htab); + err = bpf_map_update_batch(fd, keys, vals, &cnt, NULL); + if (!ASSERT_OK(err, "update_batch") || !ASSERT_EQ(cnt, NR_ELEMS, "update_cnt")) + goto out; + err = bpf_map_delete_batch(fd, keys, &cnt, NULL); + if (!ASSERT_OK(err, "delete_batch") || !ASSERT_EQ(cnt, NR_ELEMS, "delete_cnt")) + goto out; + + /* Wait for all __free_rcu() callbacks to finish */ + for (i = 0; i < 300; i++) { + err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.check_ttrace), &opts); + if (!ASSERT_OK(err, "test_run") || !ASSERT_OK(opts.retval, "retval")) + goto out; + if (!skel->bss->in_progress) + break; + usleep(100000); + } + ASSERT_EQ(skel->bss->nr_caches, nr_cpus, "nr_caches"); + ASSERT_EQ(skel->bss->in_progress, 0, "in_progress"); + ASSERT_EQ(skel->bss->not_freed, 0, "not_freed"); +out: + free(keys); + free(vals); + bpf_ma_ttrace__destroy(skel); +} diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c b/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c index ecc3d47919ad..f8497e764beb 100644 --- a/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c +++ b/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c @@ -55,38 +55,6 @@ static void *load_module_thread(void *p) return p; } -static int sys_userfaultfd(int flags) -{ - return syscall(__NR_userfaultfd, flags); -} - -static int test_setup_uffd(void *fault_addr) -{ - struct uffdio_register uffd_register = {}; - struct uffdio_api uffd_api = {}; - int uffd; - - uffd = sys_userfaultfd(O_CLOEXEC); - if (uffd < 0) - return -errno; - - uffd_api.api = UFFD_API; - uffd_api.features = 0; - if (ioctl(uffd, UFFDIO_API, &uffd_api)) { - close(uffd); - return -1; - } - - uffd_register.range.start = (unsigned long)fault_addr; - uffd_register.range.len = getpagesize(); - uffd_register.mode = UFFDIO_REGISTER_MODE_MISSING; - if (ioctl(uffd, UFFDIO_REGISTER, &uffd_register)) { - close(uffd); - return -1; - } - return uffd; -} - static void test_bpf_mod_race_config(const struct test_config *config) { void *fault_addr, *skel_fail; @@ -117,7 +85,7 @@ static void test_bpf_mod_race_config(const struct test_config *config) if (!ASSERT_OK(bpf_mod_race__attach(skel), "bpf_mod_kfunc_race__attach")) goto end_destroy; - uffd = test_setup_uffd(fault_addr); + uffd = uffd_block_page(fault_addr); if (!ASSERT_GE(uffd, 0, "userfaultfd open + register address")) goto end_destroy; diff --git a/tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c b/tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c new file mode 100644 index 000000000000..1eb0d7237605 --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c @@ -0,0 +1,188 @@ +// SPDX-License-Identifier: GPL-2.0 +#include <test_progs.h> +#include <pthread.h> +#include <poll.h> +#include <sys/mman.h> +#include <linux/userfaultfd.h> +#include "tramp_prog_detach.skel.h" +#include "testing_helpers.h" + +/* + * Detach and free progs while a task sleeps in the prog that runs before them + * in the same trampoline image, then let that task continue through the + * image. It must not call into the freed progs. + * + * The task is held in a sleepable prog with userfaultfd, like bpf_mod_race + * does. + */ + +static struct bpf_program *pick_prog(struct tramp_prog_detach *skel, + bool fexit, bool sleepable) +{ + if (fexit) + return sleepable ? skel->progs.fexit_sleepable : + skel->progs.fexit_victim; + return sleepable ? skel->progs.fentry_sleepable : + skel->progs.fentry_victim; +} + +static struct bpf_program *sleepable_prog; + +static void *run_sleepable(void *arg) +{ + LIBBPF_OPTS(bpf_test_run_opts, topts); + + /* calls bpf_fentry_test1() */ + return (void *)(long)bpf_prog_test_run_opts(bpf_program__fd(sleepable_prog), + &topts); +} + +static struct tramp_prog_detach *load_one(bool fexit, bool sleepable) +{ + struct tramp_prog_detach *skel; + int err; + + skel = tramp_prog_detach__open(); + if (!ASSERT_OK_PTR(skel, "open")) + return NULL; + + bpf_program__set_autoload(pick_prog(skel, fexit, sleepable), true); + err = tramp_prog_detach__load(skel); + if (!ASSERT_OK(err, "load")) + goto err; + skel->bss->pid = getpid(); + err = tramp_prog_detach__attach(skel); + if (!ASSERT_OK(err, "attach")) + goto err; + return skel; +err: + tramp_prog_detach__destroy(skel); + return NULL; +} + +/* The .bss map of a destroyed skeleton goes away when its prog is freed */ +static bool wait_for_map_free(__u32 map_id) +{ + int i, fd; + + for (i = 0; i < 100; i++) { + fd = bpf_map_get_fd_by_id(map_id); + if (fd < 0) + return true; + close(fd); + usleep(100 * 1000); + } + return false; +} + +static void test_detach(bool sleepable_fexit, bool victim_fexit) +{ + struct tramp_prog_detach *sleepable = NULL, *victims[2] = {}; + struct pollfd pfd = { .events = POLLIN }; + struct uffdio_copy uffd_copy = {}; + struct bpf_map_info map_info = {}; + __u32 map_info_len = sizeof(map_info); + struct uffd_msg uffd_msg; + void *fault_page, *src_page = MAP_FAILED; + long page_size = getpagesize(); + bool started = false; + void *thread_ret; + pthread_t thread; + int i, uffd = -1; + + fault_page = mmap(NULL, page_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (!ASSERT_NEQ(fault_page, MAP_FAILED, "mmap fault_page")) + return; + src_page = mmap(NULL, page_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (!ASSERT_NEQ(src_page, MAP_FAILED, "mmap src_page")) + goto out; + + /* The most recently attached prog runs first */ + for (i = 0; i < ARRAY_SIZE(victims); i++) { + victims[i] = load_one(victim_fexit, false); + if (!victims[i]) + goto out; + } + sleepable = load_one(sleepable_fexit, true); + if (!sleepable) + goto out; + sleepable_prog = pick_prog(sleepable, sleepable_fexit, true); + + /* Not armed yet so this doesn't block, make sure sleepable runs first */ + if (!ASSERT_OK((long)run_sleepable(NULL), "dry run")) + goto out; + for (i = 0; i < ARRAY_SIZE(victims); i++) + if (!ASSERT_LT(sleepable->bss->ts, victims[i]->bss->ts, "prog order")) + goto out; + + uffd = uffd_block_page(fault_page); + if (!ASSERT_GE(uffd, 0, "userfaultfd open + register address")) + goto out; + sleepable->bss->fault_addr = fault_page; + + if (!ASSERT_OK(pthread_create(&thread, NULL, run_sleepable, NULL), + "pthread_create")) + goto out; + started = true; + + /* Wait for the thread to sleep in bpf_copy_from_user() */ + pfd.fd = uffd; + if (!ASSERT_EQ(poll(&pfd, 1, 10000), 1, "poll uffd")) + goto out; + if (!ASSERT_EQ(read(uffd, &uffd_msg, sizeof(uffd_msg)), sizeof(uffd_msg), + "read uffd")) + goto out; + if (!ASSERT_EQ(uffd_msg.event, UFFD_EVENT_PAGEFAULT, "uffd pagefault")) + goto out; + + /* + * Detach and unload the victim progs and wait for them to be freed. + * After the first one, the task is in an image that isn't the + * trampoline's current one anymore. + */ + for (i = 0; i < ARRAY_SIZE(victims); i++) { + if (!ASSERT_OK(bpf_map_get_info_by_fd(bpf_map__fd(victims[i]->maps.bss), + &map_info, &map_info_len), + "victim bss info")) + goto out; + tramp_prog_detach__destroy(victims[i]); + victims[i] = NULL; + if (!ASSERT_TRUE(wait_for_map_free(map_info.id), "victim freed")) + goto out; + } + +out: + /* Let the thread proceed with the rest of the trampoline */ + if (uffd >= 0) { + uffd_copy.dst = (unsigned long)fault_page; + uffd_copy.src = (unsigned long)src_page; + uffd_copy.len = page_size; + ASSERT_OK(ioctl(uffd, UFFDIO_COPY, &uffd_copy), "uffd copy"); + close(uffd); + } + if (started && + ASSERT_OK(pthread_join(thread, &thread_ret), "pthread_join")) + ASSERT_NULL(thread_ret, "blocking run"); + + for (i = 0; i < ARRAY_SIZE(victims); i++) + tramp_prog_detach__destroy(victims[i]); + tramp_prog_detach__destroy(sleepable); + if (src_page != MAP_FAILED) + munmap(src_page, page_size); + munmap(fault_page, page_size); +} + +void serial_test_tramp_prog_detach(void) +{ + /* a task sleeping before the original function is called */ + if (test__start_subtest("fentry")) + test_detach(false, false); + /* a task sleeping after the original function returned */ + if (test__start_subtest("fexit")) + test_detach(true, true); + /* the original function runs in between and must still be called */ + if (test__start_subtest("fentry_fexit")) + test_detach(false, true); +} diff --git a/tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c b/tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c new file mode 100644 index 000000000000..31b2a962e339 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c @@ -0,0 +1,50 @@ +// SPDX-License-Identifier: GPL-2.0 +#include <vmlinux.h> +#include <bpf/bpf_helpers.h> +#include <bpf/bpf_core_read.h> +#include "bpf_kfuncs.h" + +struct { + __uint(type, BPF_MAP_TYPE_HASH); + __uint(map_flags, BPF_F_NO_PREALLOC); + __uint(max_entries, 4096); + __type(key, int); + __type(value, long); +} htab SEC(".maps"); + +extern const void __per_cpu_offset __ksym; + +int nr_cpus; +int nr_caches; +int in_progress; +int not_freed; + +/* Look at bpf_mem_cache of every cpu that htab allocates its elements from */ +SEC("syscall") +int check_ttrace(void *ctx) +{ + struct bpf_htab *h = bpf_core_cast(&htab, struct bpf_htab); + unsigned long cache = (unsigned long)BPF_CORE_READ(h, ma.cache); + const unsigned long *offsets = &__per_cpu_offset; + struct bpf_mem_cache *c; + unsigned long off; + int cpu; + + nr_caches = 0; + in_progress = 0; + not_freed = 0; + bpf_for(cpu, 0, nr_cpus) { + if (bpf_probe_read_kernel(&off, sizeof(off), offsets + cpu)) + return -1; + c = bpf_core_cast((void *)(cache + off), struct bpf_mem_cache); + if (c->unit_size) + nr_caches++; + if (c->call_rcu_ttrace_in_progress.counter) + in_progress++; + if (c->free_by_rcu_ttrace.first || c->waiting_for_gp_ttrace.first) + not_freed++; + } + return 0; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/tramp_prog_detach.c b/tools/testing/selftests/bpf/progs/tramp_prog_detach.c new file mode 100644 index 000000000000..507d372167ec --- /dev/null +++ b/tools/testing/selftests/bpf/progs/tramp_prog_detach.c @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: GPL-2.0 +#include "vmlinux.h" +#include <bpf/bpf_helpers.h> +#include <bpf/bpf_tracing.h> + +char _license[] SEC("license") = "GPL"; + +int pid; +void *fault_addr; +__u64 ts; + +static int do_sleepable(void) +{ + char dst; + + if (bpf_get_current_pid_tgid() >> 32 != pid) + return 0; + + ts = bpf_ktime_get_ns(); + /* blocks for as long as user space wants when fault_addr is armed */ + bpf_copy_from_user(&dst, sizeof(dst), fault_addr); + return 0; +} + +static int do_victim(void) +{ + if (bpf_get_current_pid_tgid() >> 32 != pid) + return 0; + + ts = bpf_ktime_get_ns(); + return 0; +} + +SEC("?fentry.s/bpf_fentry_test1") +int BPF_PROG(fentry_sleepable, int a) +{ + return do_sleepable(); +} + +SEC("?fentry/bpf_fentry_test1") +int BPF_PROG(fentry_victim, int a) +{ + return do_victim(); +} + +SEC("?fexit.s/bpf_fentry_test1") +int BPF_PROG(fexit_sleepable, int a, int ret) +{ + return do_sleepable(); +} + +SEC("?fexit/bpf_fentry_test1") +int BPF_PROG(fexit_victim, int a, int ret) +{ + return do_victim(); +} diff --git a/tools/testing/selftests/bpf/progs/verifier_align.c b/tools/testing/selftests/bpf/progs/verifier_align.c index 3e52686515ca..0083ef8b8ba7 100644 --- a/tools/testing/selftests/bpf/progs/verifier_align.c +++ b/tools/testing/selftests/bpf/progs/verifier_align.c @@ -284,7 +284,7 @@ __msg("26: {{.*}} R5=pkt(r=8,imm=14)") */ __msg("28: {{.*}} R4={{[^)]*}}var_off=(0x2; 0x7fc){{.*}} R5={{[^)]*}}var_off=(0x2; 0x7fc)") /* Constant is added to R5 again, setting reg->off to 18. */ -__msg("29: {{.*}} R5=pkt(id=3,{{[^)]*}}var_off=(0x2; 0x7fc)") +__msg("29: {{.*}} R5=pkt(id=3,{{[^)]*}}var_off=(0x2; 0xffc)") /* And once more we add a variable; resulting {{[^)]*}}var_off * is still (4n), fixed offset is not changed. * Also, we create a new reg->id. @@ -359,7 +359,7 @@ __msg("7: {{.*}} R6={{[^)]*}}var_off=(0x0; 0x3fc)") __msg("8: {{.*}} R6={{[^)]*}}var_off=(0x2; 0x7fc)") /* Packet pointer has (4n+2) offset */ __msg("11: {{.*}} R5={{[^)]*}}var_off=(0x2; 0x7fc)") -__msg("12: {{.*}} R4={{[^)]*}}var_off=(0x2; 0x7fc)") +__msg("12: {{.*}} R4={{[^)]*}}var_off=(0x2; 0xffc)") /* At the time the word size load is performed from R5, * its total fixed offset is NET_IP_ALIGN + reg->off (0) * which is 2. Then the variable offset is (4n+2), so @@ -375,7 +375,7 @@ __msg("17: {{.*}} R6={{[^)]*}}var_off=(0x0; 0x3fc)") * another (4n+2). */ __msg("19: {{.*}} R5={{[^)]*}}var_off=(0x2; 0xffc)") -__msg("20: {{.*}} R4={{[^)]*}}var_off=(0x2; 0xffc)") +__msg("20: {{.*}} R4={{[^)]*}}var_off=(0x2; 0x1ffc)") /* At the time the word size load is performed from R5, * its total fixed offset is NET_IP_ALIGN + reg->off (0) * which is 2. Then the variable offset is (4n+2), so diff --git a/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c b/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c index 915a9707298b..139ff019d87d 100644 --- a/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c +++ b/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c @@ -920,4 +920,182 @@ l1_%=: r0 = *(u8*)(r9 + 0); \ : __clobber_all); } +SEC("tc") +__description("direct packet access: 8-aligned offset, check p + 8, load 8 bytes at p") +__success __retval(0) __flag(BPF_F_ANY_ALIGNMENT) +__naked void pkt_same_id_check_copy_load_base(void) +{ + asm volatile (" \ + r2 = *(u32*)(r1 + %[__sk_buff_data]); \ + r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \ + r4 = *(u32*)(r1 + %[__sk_buff_mark]); \ + r4 &= 0x38; \ + if r4 > 50 goto l0_%=; \ + /* r4 is a multiple of 8, at most 48 */ \ + r5 = r2; \ + r5 += r4; \ + r6 = r5; \ + r6 += 8; \ + if r6 > r3 goto l0_%=; \ + /* [r5, r5 + 8) is in the packet */ \ + r0 = *(u64*)(r5 + 0); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)), + __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)), + __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)) + : __clobber_all); +} + +SEC("tc") +__description("direct packet access: 8-aligned offset, check p, load past it via p + 8") +__failure __msg("invalid access to packet, off=51 size=1") +__naked void pkt_same_id_check_base_load_via_copy(void) +{ + asm volatile (" \ + r2 = *(u32*)(r1 + %[__sk_buff_data]); \ + r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \ + r4 = *(u32*)(r1 + %[__sk_buff_mark]); \ + r4 &= 0x38; \ + if r4 > 50 goto l0_%=; \ + /* r4 is a multiple of 8, at most 48 */ \ + r5 = r2; \ + r5 += r4; \ + r6 = r5; \ + r6 += 8; \ + if r5 > r3 goto l0_%=; \ + /* r6 - 7 is r5 + 1 */ \ + r0 = *(u8*)(r6 - 7); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)), + __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)), + __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)) + : __clobber_all); +} + +SEC("tc") +__description("direct packet access: 8-aligned offset, check p + 4, 4-byte load at p + 2") +__failure __msg("invalid access to packet, off=52 size=4") +__naked void pkt_same_id_check_base_4_load_via_copy(void) +{ + asm volatile (" \ + r2 = *(u32*)(r1 + %[__sk_buff_data]); \ + r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \ + r4 = *(u32*)(r1 + %[__sk_buff_mark]); \ + r4 &= 0x38; \ + if r4 > 50 goto l0_%=; \ + /* r4 is a multiple of 8, at most 48 */ \ + r5 = r2; \ + r5 += r4; \ + r6 = r5; \ + r6 += 8; \ + r7 = r5; \ + r7 += 4; \ + if r7 > r3 goto l0_%=; \ + /* [r5, r5 + 4) is in the packet, [r5 + 2, r5 + 6) may not be */ \ + r0 = *(u32*)(r6 - 6); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)), + __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)), + __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)) + : __clobber_all); +} + +SEC("tc") +__description("direct packet access: checked pointer minus non-negative unknown keeps range") +__success __retval(0) __flag(BPF_F_ANY_ALIGNMENT) +__naked void pkt_sub_unknown_keeps_range(void) +{ + asm volatile (" \ + r2 = *(u32*)(r1 + %[__sk_buff_data]); \ + r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \ + r4 = *(u32*)(r1 + %[__sk_buff_mark]); \ + r4 &= 0x1f; \ + r4 += 8; \ + r5 = r2; \ + r5 += 40; \ + if r5 > r3 goto l0_%=; \ + /* r5 is 8 to 39 bytes below the checked pointer */ \ + r5 -= r4; \ + r0 = *(u64*)(r5 + 0); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)), + __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)), + __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)) + : __clobber_all); +} + +SEC("tc") +__description("direct packet access: no pruning of a path whose checks prove fewer bytes") +__failure __msg("invalid access to packet, off=255 size=8") +__flag(BPF_F_ANY_ALIGNMENT) __flag(BPF_F_TEST_STATE_FREQ) +__naked void pkt_same_id_pruning(void) +{ + asm volatile (" \ + r2 = *(u32*)(r1 + %[__sk_buff_data]); \ + r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \ + r4 = *(u32*)(r1 + %[__sk_buff_mark]); \ + r0 = *(u32*)(r1 + %[__sk_buff_priority]); \ + r4 &= 0xff; \ + r5 = r2; \ + r5 += r4; \ + if r0 != 0 goto l1_%=; \ + /* this path proves [r5, r5 + 8) */ \ + r6 = r5; \ + r6 += 8; \ + if r6 > r3 goto l0_%=; \ + goto l2_%=; \ +l1_%=: /* this path proves [r5, r5 + 7) */ \ + if r5 > r3 goto l0_%=; \ + r6 = r5; \ + r6 += 7; \ + if r6 > r3 goto l0_%=; \ +l2_%=: r0 = *(u64*)(r5 + 0); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)), + __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)), + __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)), + __imm_const(__sk_buff_priority, offsetof(struct __sk_buff, priority)) + : __clobber_all); +} + +SEC("tc") +__description("direct packet access: spilled copy of checked pointer, load past range") +__failure __msg("invalid access to packet, off=262 size=1, R7(id={{[0-9]+}},off=262,r=262)") +__naked void pkt_spilled_copy_gets_range(void) +{ + asm volatile (" \ + r2 = *(u32*)(r1 + %[__sk_buff_data]); \ + r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \ + r4 = *(u32*)(r1 + %[__sk_buff_mark]); \ + r4 &= 0xff; \ + r5 = r2; \ + r5 += r4; \ + *(u64*)(r10 - 8) = r5; \ + /* proves only bytes before r5 */ \ + if r5 > r3 goto l0_%=; \ + r6 = r5; \ + r6 += 7; \ + if r6 > r3 goto l0_%=; \ + /* [r5, r5 + 7) is in the packet */ \ + r7 = *(u64*)(r10 - 8); \ + r0 = *(u8*)(r7 + 7); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)), + __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)), + __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)) + : __clobber_all); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_meta_access.c b/tools/testing/selftests/bpf/progs/verifier_meta_access.c index 62235f032ffe..c87e0be4b2ff 100644 --- a/tools/testing/selftests/bpf/progs/verifier_meta_access.c +++ b/tools/testing/selftests/bpf/progs/verifier_meta_access.c @@ -281,4 +281,34 @@ l0_%=: r0 = 0; \ : __clobber_all); } +SEC("xdp") +__description("meta access, 8-aligned offset, check p, load a byte at p + 1 via p + 8") +__failure __msg("invalid access to packet, off=51 size=1") +__naked void meta_access_check_base_load_via_copy(void) +{ + asm volatile (" \ + r9 = r1; \ + call %[bpf_get_prandom_u32]; \ + r4 = r0; \ + r4 &= 0x38; \ + if r4 > 50 goto l0_%=; \ + /* r4 is a multiple of 8, at most 48 */ \ + r2 = *(u32*)(r9 + %[xdp_md_data_meta]); \ + r3 = *(u32*)(r9 + %[xdp_md_data]); \ + r5 = r2; \ + r5 += r4; \ + r6 = r5; \ + r6 += 8; \ + if r5 > r3 goto l0_%=; \ + /* r6 - 7 is r5 + 1 */ \ + r0 = *(u8*)(r6 - 7); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm(bpf_get_prandom_u32), + __imm_const(xdp_md_data, offsetof(struct xdp_md, data)), + __imm_const(xdp_md_data_meta, offsetof(struct xdp_md, data_meta)) + : __clobber_all); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/testing_helpers.c b/tools/testing/selftests/bpf/testing_helpers.c index c970e7793dfc..9f672d0c9dda 100644 --- a/tools/testing/selftests/bpf/testing_helpers.c +++ b/tools/testing/selftests/bpf/testing_helpers.c @@ -13,6 +13,7 @@ #include "test_progs.h" #include "testing_helpers.h" #include <linux/membarrier.h> +#include <linux/userfaultfd.h> int parse_num_list(const char *s, bool **num_set, int *num_set_len) { @@ -459,6 +460,33 @@ int kern_sync_rcu(void) return syscall(__NR_membarrier, MEMBARRIER_CMD_SHARED, 0, 0); } +int uffd_block_page(void *fault_addr) +{ + struct uffdio_register uffd_register = {}; + struct uffdio_api uffd_api = {}; + int uffd; + + uffd = syscall(__NR_userfaultfd, O_CLOEXEC); + if (uffd < 0) + return -errno; + + uffd_api.api = UFFD_API; + uffd_api.features = 0; + if (ioctl(uffd, UFFDIO_API, &uffd_api)) { + close(uffd); + return -1; + } + + uffd_register.range.start = (unsigned long)fault_addr; + uffd_register.range.len = getpagesize(); + uffd_register.mode = UFFDIO_REGISTER_MODE_MISSING; + if (ioctl(uffd, UFFDIO_REGISTER, &uffd_register)) { + close(uffd); + return -1; + } + return uffd; +} + int get_xlated_program(int fd_prog, struct bpf_insn **buf, __u32 *cnt) { __u32 buf_element_size = sizeof(struct bpf_insn); diff --git a/tools/testing/selftests/bpf/testing_helpers.h b/tools/testing/selftests/bpf/testing_helpers.h index 2edc6fb7fc52..1f83b6080716 100644 --- a/tools/testing/selftests/bpf/testing_helpers.h +++ b/tools/testing/selftests/bpf/testing_helpers.h @@ -36,6 +36,8 @@ __u64 read_perf_max_sample_freq(void); int load_bpf_testmod(bool verbose); int unload_bpf_testmod(bool verbose); int kern_sync_rcu(void); +/* returns a userfaultfd that makes accesses to fault_addr's page block */ +int uffd_block_page(void *fault_addr); int finit_module(int fd, const char *param_values, int flags); int delete_module(const char *name, int flags); int load_module(const char *path, bool verbose); |
