summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:59:32 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:59:32 -0700
commit8f150ccedfbd610aa25509ff42365d70fb20478f (patch)
tree1b22e147a9f8464940fcfa0483c96b8842c953c5
parentd2dbe503fd806082acb0ca79a9d6641822988c2c (diff)
parentde020dc8049bfb2b22e3b6d99c031feb2e22d112 (diff)
Merge tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf
Pull bpf fixes from Alexei Starovoitov: - Fix overflow of backward jump offset in constant blinding (Alexei Starovoitov) - Fix packet range of packet pointers sharing an id when var_off tightens umax of one pointer and not the other (Alexei Starovoitov) - Fix objects stuck in free_by_rcu_ttrace list of bpf memalloc (Alexei Starovoitov) - Fix use-after-free of progs detached from busy trampolines: wait for an RCU tasks grace period before freeing trampoline progs, and patch detached progs out of trampoline images that are still in use (Florent Revest) - Hold map BTF for the memory allocator destructor record to fix UAF in deferred bpf_mem_alloc destruction (Kumar Kartikeya Dwivedi) - Fix missing migration protection in resizable hashtab lookup_and_delete batch operation (Ă–mer Mete Kaya) * tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf: bpf: Fix missing migration protection in __rhtab_map_lookup_and_delete_batch() selftests/bpf: Add a test for objects stuck in free_by_rcu_ttrace bpf: Fix objects stuck in free_by_rcu_ttrace bpf: Factor out __do_call_rcu_ttrace() selftests/bpf: Test packet range of pointers sharing an id bpf: Fix packet range of pointers sharing an id selftests/bpf: Detach a trampoline prog while a task sleeps before it bpf: Skip detached progs in trampoline images that are still in use bpf: Wait for an RCU tasks grace period before freeing trampoline progs bpf: Hold map BTF for the memory allocator destructor record bpf: Fix overflow of jump offset in constant blinding
-rw-r--r--arch/arm64/net/bpf_jit_comp.c37
-rw-r--r--arch/loongarch/net/bpf_jit.c60
-rw-r--r--arch/powerpc/net/bpf_jit_comp.c45
-rw-r--r--arch/riscv/net/bpf_jit_comp64.c56
-rw-r--r--arch/s390/net/bpf_jit_comp.c46
-rw-r--r--arch/x86/net/bpf_jit_comp.c26
-rw-r--r--include/linux/bpf.h45
-rw-r--r--kernel/bpf/core.c10
-rw-r--r--kernel/bpf/hashtab.c19
-rw-r--r--kernel/bpf/memalloc.c57
-rw-r--r--kernel/bpf/syscall.c19
-rw-r--r--kernel/bpf/trampoline.c85
-rw-r--r--kernel/bpf/verifier.c10
-rw-r--r--tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c60
-rw-r--r--tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c34
-rw-r--r--tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c188
-rw-r--r--tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c50
-rw-r--r--tools/testing/selftests/bpf/progs/tramp_prog_detach.c56
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_align.c6
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c178
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_meta_access.c30
-rw-r--r--tools/testing/selftests/bpf/testing_helpers.c28
-rw-r--r--tools/testing/selftests/bpf/testing_helpers.h2
23 files changed, 975 insertions, 172 deletions
diff --git a/arch/arm64/net/bpf_jit_comp.c b/arch/arm64/net/bpf_jit_comp.c
index c5f55d6161fe..c1279fabce47 100644
--- a/arch/arm64/net/bpf_jit_comp.c
+++ b/arch/arm64/net/bpf_jit_comp.c
@@ -2418,10 +2418,11 @@ bool bpf_jit_supports_subprog_tailcalls(void)
return true;
}
-static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node,
- int bargs_off, int retval_off, int run_ctx_off,
- bool save_ret)
+static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_image *im,
+ struct bpf_tramp_node *node, int bargs_off,
+ int retval_off, int run_ctx_off, bool save_ret)
{
+ void *skip;
__le32 *branch;
u64 enter_prog;
u64 exit_prog;
@@ -2431,6 +2432,10 @@ static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node,
enter_prog = (u64)bpf_trampoline_enter(p);
exit_prog = (u64)bpf_trampoline_exit(p);
+ /* nop, patched to skip this prog when it is detached */
+ skip = ctx->ro_image + ctx->idx;
+ emit(A64_NOP, ctx);
+
if (node->cookie == 0) {
/* if cookie is zero, one instruction is enough to store it */
emit(A64_STR64I(A64_ZR, A64_SP, run_ctx_off + cookie_off), ctx);
@@ -2483,11 +2488,13 @@ static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node,
emit(A64_ADD_I(1, A64_R(2), A64_SP, run_ctx_off), ctx);
emit_call(exit_prog, ctx);
+
+ bpf_tramp_image_add_skip(im, p, skip, ctx->ro_image + ctx->idx);
}
-static void invoke_bpf_mod_ret(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn,
- int bargs_off, int retval_off, int run_ctx_off,
- __le32 **branches)
+static void invoke_bpf_mod_ret(struct jit_ctx *ctx, struct bpf_tramp_image *im,
+ struct bpf_tramp_nodes *tn, int bargs_off,
+ int retval_off, int run_ctx_off, __le32 **branches)
{
int i;
@@ -2496,7 +2503,7 @@ static void invoke_bpf_mod_ret(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn,
*/
emit(A64_STR64I(A64_ZR, A64_SP, retval_off), ctx);
for (i = 0; i < tn->nr_nodes; i++) {
- invoke_bpf_prog(ctx, tn->nodes[i], bargs_off, retval_off,
+ invoke_bpf_prog(ctx, im, tn->nodes[i], bargs_off, retval_off,
run_ctx_off, true);
/* if (*(u64 *)(sp + retval_off) != 0)
* goto do_fexit;
@@ -2882,7 +2889,7 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im,
store_func_meta(ctx, meta, func_meta_off);
cookie_bargs_off--;
}
- invoke_bpf_prog(ctx, fentry->nodes[i], bargs_off,
+ invoke_bpf_prog(ctx, im, fentry->nodes[i], bargs_off,
retval_off, run_ctx_off,
flags & BPF_TRAMP_F_RET_FENTRY_RET);
}
@@ -2893,7 +2900,7 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im,
if (!branches)
return -ENOMEM;
- invoke_bpf_mod_ret(ctx, fmod_ret, bargs_off, retval_off,
+ invoke_bpf_mod_ret(ctx, im, fmod_ret, bargs_off, retval_off,
run_ctx_off, branches);
}
@@ -2906,9 +2913,6 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im,
emit(A64_RET(A64_R(10)), ctx);
/* store return value */
emit(A64_STR64I(A64_R(0), A64_SP, retval_off), ctx);
- /* reserve a nop for bpf_tramp_image_put */
- im->ip_after_call = ctx->ro_image + ctx->idx;
- emit(A64_NOP, ctx);
}
/* update the branches saved in invoke_bpf_mod_ret with cbnz */
@@ -2930,12 +2934,11 @@ static int prepare_trampoline(struct jit_ctx *ctx, struct bpf_tramp_image *im,
store_func_meta(ctx, meta, func_meta_off);
cookie_bargs_off--;
}
- invoke_bpf_prog(ctx, fexit->nodes[i], bargs_off, retval_off,
+ invoke_bpf_prog(ctx, im, fexit->nodes[i], bargs_off, retval_off,
run_ctx_off, false);
}
if (flags & BPF_TRAMP_F_CALL_ORIG) {
- im->ip_epilogue = ctx->ro_image + ctx->idx;
/* for the first pass, assume the worst case */
if (!ctx->image)
ctx->idx += 4;
@@ -2994,7 +2997,7 @@ int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
.image = NULL,
.idx = 0,
};
- struct bpf_tramp_image im;
+ struct bpf_tramp_image im = {};
struct arg_aux aaux;
int ret;
@@ -3281,6 +3284,10 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
* longer reachable, since bpf_tramp_image_put() function already
* uses percpu_ref and task-based rcu to do the sync, no need to call
* the sync version here, see bpf_tramp_image_put() for details.
+ *
+ * 3. when a detached prog is patched out of a trampoline, a CPU that
+ * still executes the old nop calls the prog before it went through
+ * a quiescent state, and the prog is freed after grace periods.
*/
ret = aarch64_insn_patch_text_nosync(ip, new_insn);
out:
diff --git a/arch/loongarch/net/bpf_jit.c b/arch/loongarch/net/bpf_jit.c
index 4da278900938..e78eb582c300 100644
--- a/arch/loongarch/net/bpf_jit.c
+++ b/arch/loongarch/net/bpf_jit.c
@@ -1578,6 +1578,24 @@ void *bpf_arch_text_copy(void *dst, void *src, size_t len)
return ret ? ERR_PTR(-EINVAL) : dst;
}
+int arch_bpf_trampoline_skip(void *nop, void *target)
+{
+ u32 old_insn = INSN_NOP;
+ u32 new_insn = larch_insn_gen_b((unsigned long)nop, (unsigned long)target);
+ int ret;
+
+ if (memcmp(nop, &old_insn, LOONGARCH_INSN_SIZE))
+ return -EFAULT;
+
+ cpus_read_lock();
+ mutex_lock(&text_mutex);
+ ret = larch_insn_text_copy(nop, &new_insn, LOONGARCH_INSN_SIZE);
+ mutex_unlock(&text_mutex);
+ cpus_read_unlock();
+
+ return ret;
+}
+
int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
enum bpf_text_poke_type new_t, void *old_addr,
void *new_addr)
@@ -1696,13 +1714,18 @@ static void restore_stk_args(struct jit_ctx *ctx, int nr_stk_args, int args_off,
}
}
-static int invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *n,
- int args_off, int retval_off, int run_ctx_off, bool save_ret)
+static int invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_image *im,
+ struct bpf_tramp_node *n, int args_off, int retval_off,
+ int run_ctx_off, bool save_ret)
{
int ret;
u32 *branch;
struct bpf_prog *p = n->link->prog;
int cookie_off = offsetof(struct bpf_tramp_run_ctx, bpf_cookie);
+ void *skip = ctx->ro_image + ctx->idx;
+
+ /* nop, patched to a b over this prog when it is detached */
+ emit_insn(ctx, nop);
if (n->cookie)
emit_store_stack_imm64(ctx, LOONGARCH_GPR_T1,
@@ -1755,13 +1778,17 @@ static int invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *n,
/* arg3: &run_ctx */
emit_insn(ctx, addid, LOONGARCH_GPR_A2, LOONGARCH_GPR_FP, -run_ctx_off);
ret = emit_call(ctx, (const u64)bpf_trampoline_exit(p));
+ if (ret)
+ return ret;
- return ret;
+ bpf_tramp_image_add_skip(im, p, skip, ctx->ro_image + ctx->idx);
+ return 0;
}
-static int invoke_bpf(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn,
- int args_off, int retval_off, int run_ctx_off,
- int func_meta_off, bool save_ret, u64 func_meta, int cookie_off)
+static int invoke_bpf(struct jit_ctx *ctx, struct bpf_tramp_image *im,
+ struct bpf_tramp_nodes *tn, int args_off, int retval_off,
+ int run_ctx_off, int func_meta_off, bool save_ret,
+ u64 func_meta, int cookie_off)
{
int i, cur_cookie = (cookie_off - args_off) / 8;
@@ -1774,7 +1801,8 @@ static int invoke_bpf(struct jit_ctx *ctx, struct bpf_tramp_nodes *tn,
emit_store_stack_imm64(ctx, LOONGARCH_GPR_T1, -func_meta_off, meta);
cur_cookie--;
}
- err = invoke_bpf_prog(ctx, tn->nodes[i], args_off, retval_off, run_ctx_off, save_ret);
+ err = invoke_bpf_prog(ctx, im, tn->nodes[i], args_off, retval_off,
+ run_ctx_off, save_ret);
if (err)
return err;
}
@@ -2017,7 +2045,7 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i
}
if (fentry->nr_nodes) {
- ret = invoke_bpf(ctx, fentry, args_off, retval_off, run_ctx_off, func_meta_off,
+ ret = invoke_bpf(ctx, im, fentry, args_off, retval_off, run_ctx_off, func_meta_off,
flags & BPF_TRAMP_F_RET_FENTRY_RET, func_meta, cookie_off);
if (ret)
return ret;
@@ -2029,7 +2057,7 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i
emit_insn(ctx, std, LOONGARCH_GPR_ZERO, LOONGARCH_GPR_FP, -retval_off);
for (i = 0; i < fmod_ret->nr_nodes; i++) {
- ret = invoke_bpf_prog(ctx, fmod_ret->nodes[i],
+ ret = invoke_bpf_prog(ctx, im, fmod_ret->nodes[i],
args_off, retval_off, run_ctx_off, true);
if (ret)
goto out;
@@ -2051,10 +2079,6 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i
goto out;
emit_insn(ctx, std, LOONGARCH_GPR_A0, LOONGARCH_GPR_FP, -retval_off);
emit_insn(ctx, std, regmap[BPF_REG_0], LOONGARCH_GPR_FP, -(retval_off - 8));
- im->ip_after_call = ctx->ro_image + ctx->idx;
- /* Reserve space for the move_imm + jirl instruction */
- for (i = 0; i < LOONGARCH_LONG_JUMP_NINSNS; i++)
- emit_insn(ctx, nop);
}
for (i = 0; ctx->image && i < fmod_ret->nr_nodes; i++) {
@@ -2068,14 +2092,13 @@ static int __arch_prepare_bpf_trampoline(struct jit_ctx *ctx, struct bpf_tramp_i
emit_store_stack_imm64(ctx, LOONGARCH_GPR_T1, -func_meta_off, func_meta);
if (fexit->nr_nodes) {
- ret = invoke_bpf(ctx, fexit, args_off, retval_off, run_ctx_off,
+ ret = invoke_bpf(ctx, im, fexit, args_off, retval_off, run_ctx_off,
func_meta_off, false, func_meta, cookie_off);
if (ret)
goto out;
}
if (flags & BPF_TRAMP_F_CALL_ORIG) {
- im->ip_epilogue = ctx->ro_image + ctx->idx;
move_addr(ctx, LOONGARCH_GPR_A0, (const u64)im);
ret = emit_call(ctx, (const u64)__bpf_tramp_exit);
if (ret)
@@ -2177,11 +2200,8 @@ int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *func_addr)
{
int ret;
- struct jit_ctx ctx;
- struct bpf_tramp_image im;
-
- ctx.image = NULL;
- ctx.idx = 0;
+ struct jit_ctx ctx = {};
+ struct bpf_tramp_image im = {};
ret = __arch_prepare_bpf_trampoline(&ctx, &im, m, tnodes, func_addr, flags);
diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
index 7b07b43575f1..7a612688e7a2 100644
--- a/arch/powerpc/net/bpf_jit_comp.c
+++ b/arch/powerpc/net/bpf_jit_comp.c
@@ -602,14 +602,18 @@ int arch_protect_bpf_trampoline(void *image, unsigned int size)
}
static int invoke_bpf_prog(u32 *image, u32 *ro_image, struct codegen_context *ctx,
- struct bpf_tramp_node *n, int regs_off, int retval_off,
- int run_ctx_off, bool save_ret)
+ struct bpf_tramp_image *im, struct bpf_tramp_node *n,
+ int regs_off, int retval_off, int run_ctx_off, bool save_ret)
{
struct bpf_prog *p = n->link->prog;
ppc_inst_t branch_insn;
- u32 jmp_idx;
+ u32 jmp_idx, skip_idx;
int ret = 0;
+ /* nop, patched to skip this prog when it is detached */
+ skip_idx = ctx->idx;
+ EMIT(PPC_RAW_NOP());
+
/* Save cookie */
if (IS_ENABLED(CONFIG_PPC64)) {
PPC_LI64(_R3, n->cookie);
@@ -679,13 +683,17 @@ static int invoke_bpf_prog(u32 *image, u32 *ro_image, struct codegen_context *ct
EMIT(PPC_RAW_ADDI(_R5, _R1, run_ctx_off));
ret = bpf_jit_emit_func_call_rel(image, ro_image, ctx,
(unsigned long)bpf_trampoline_exit(p));
+ if (ret)
+ return ret;
- return ret;
+ if (ro_image) /* image is NULL for dummy pass */
+ bpf_tramp_image_add_skip(im, p, &ro_image[skip_idx], &ro_image[ctx->idx]);
+ return 0;
}
static int invoke_bpf_mod_ret(u32 *image, u32 *ro_image, struct codegen_context *ctx,
- struct bpf_tramp_nodes *tn, int regs_off, int retval_off,
- int run_ctx_off, u32 *branches)
+ struct bpf_tramp_image *im, struct bpf_tramp_nodes *tn,
+ int regs_off, int retval_off, int run_ctx_off, u32 *branches)
{
int i;
@@ -696,8 +704,8 @@ static int invoke_bpf_mod_ret(u32 *image, u32 *ro_image, struct codegen_context
EMIT(PPC_RAW_LI(_R3, 0));
EMIT(PPC_RAW_STL(_R3, _R1, retval_off));
for (i = 0; i < tn->nr_nodes; i++) {
- if (invoke_bpf_prog(image, ro_image, ctx, tn->nodes[i], regs_off, retval_off,
- run_ctx_off, true))
+ if (invoke_bpf_prog(image, ro_image, ctx, im, tn->nodes[i], regs_off,
+ retval_off, run_ctx_off, true))
return -EINVAL;
/*
@@ -1043,8 +1051,8 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
cookie_ctx_off--;
}
- if (invoke_bpf_prog(image, ro_image, ctx, fentry->nodes[i], regs_off, retval_off,
- run_ctx_off, flags & BPF_TRAMP_F_RET_FENTRY_RET))
+ if (invoke_bpf_prog(image, ro_image, ctx, im, fentry->nodes[i], regs_off,
+ retval_off, run_ctx_off, flags & BPF_TRAMP_F_RET_FENTRY_RET))
return -EINVAL;
}
@@ -1053,7 +1061,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
if (!branches)
return -ENOMEM;
- if (invoke_bpf_mod_ret(image, ro_image, ctx, fmod_ret, regs_off, retval_off,
+ if (invoke_bpf_mod_ret(image, ro_image, ctx, im, fmod_ret, regs_off, retval_off,
run_ctx_off, branches)) {
ret = -EINVAL;
goto cleanup;
@@ -1090,11 +1098,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
/* Restore updated tail_call_cnt */
if (flags & BPF_TRAMP_F_TAIL_CALL_CTX)
bpf_trampoline_restore_tail_call_cnt(image, ctx, bpf_frame_size, r4_off);
-
- /* Reserve space to patch branch instruction to skip fexit progs */
- if (ro_image) /* image is NULL for dummy pass */
- im->ip_after_call = &((u32 *)ro_image)[ctx->idx];
- EMIT(PPC_RAW_NOP());
}
/* Update branches saved in invoke_bpf_mod_ret with address of do_fexit */
@@ -1123,16 +1126,14 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
cookie_ctx_off--;
}
- if (invoke_bpf_prog(image, ro_image, ctx, fexit->nodes[i], regs_off, retval_off,
- run_ctx_off, false)) {
+ if (invoke_bpf_prog(image, ro_image, ctx, im, fexit->nodes[i], regs_off,
+ retval_off, run_ctx_off, false)) {
ret = -EINVAL;
goto cleanup;
}
}
if (flags & BPF_TRAMP_F_CALL_ORIG) {
- if (ro_image) /* image is NULL for dummy pass */
- im->ip_epilogue = &((u32 *)ro_image)[ctx->idx];
PPC_LI_ADDR(_R3, im);
ret = bpf_jit_emit_func_call_rel(image, ro_image, ctx,
(unsigned long)__bpf_tramp_exit);
@@ -1192,7 +1193,7 @@ cleanup:
int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *func_addr)
{
- struct bpf_tramp_image im;
+ struct bpf_tramp_image im = {};
int ret;
ret = __arch_prepare_bpf_trampoline(&im, NULL, NULL, NULL, m, flags, tnodes, func_addr);
@@ -1320,7 +1321,7 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
/*
* If we are not poking at bpf prog entry, then we are simply patching in/out
- * an unconditional branch instruction at im->ip_after_call
+ * an unconditional branch instruction in a trampoline image
*/
if (offset) {
if (old_t == BPF_MOD_CALL || new_t == BPF_MOD_CALL) {
diff --git a/arch/riscv/net/bpf_jit_comp64.c b/arch/riscv/net/bpf_jit_comp64.c
index 151031e97a24..01fe66774f02 100644
--- a/arch/riscv/net/bpf_jit_comp64.c
+++ b/arch/riscv/net/bpf_jit_comp64.c
@@ -822,6 +822,24 @@ static int gen_jump_or_nops(void *target, void *ip, u32 *insns, bool is_call)
return emit_jump_and_link(is_call ? RV_REG_T0 : RV_REG_ZERO, rvoff, false, &ctx);
}
+int arch_bpf_trampoline_skip(void *nop, void *target)
+{
+ u32 old_insn = rv_nop();
+ u32 new_insn = rv_jal(RV_REG_ZERO, (target - nop) >> 1);
+ int ret;
+
+ if (memcmp(nop, &old_insn, sizeof(old_insn)))
+ return -EFAULT;
+
+ cpus_read_lock();
+ mutex_lock(&text_mutex);
+ ret = patch_text(nop, &new_insn, sizeof(new_insn));
+ mutex_unlock(&text_mutex);
+ cpus_read_unlock();
+
+ return ret;
+}
+
int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
enum bpf_text_poke_type new_t, void *old_addr,
void *new_addr)
@@ -904,12 +922,17 @@ static void emit_store_stack_imm64(u8 reg, int stack_off, u64 imm64,
emit_sd(RV_REG_FP, stack_off, reg, ctx);
}
-static int invoke_bpf_prog(struct bpf_tramp_node *node, int args_off, int retval_off,
- int run_ctx_off, bool save_ret, struct rv_jit_context *ctx)
+static int invoke_bpf_prog(struct bpf_tramp_image *im, struct bpf_tramp_node *node,
+ int args_off, int retval_off, int run_ctx_off, bool save_ret,
+ struct rv_jit_context *ctx)
{
int ret, branch_off;
struct bpf_prog *p = node->link->prog;
int cookie_off = offsetof(struct bpf_tramp_run_ctx, bpf_cookie);
+ void *skip = ctx->ro_insns + ctx->ninsns;
+
+ /* nop, patched to a jal over this prog when it is detached */
+ emit(rv_nop(), ctx);
if (node->cookie)
emit_store_stack_imm64(RV_REG_T1, -run_ctx_off + cookie_off, node->cookie, ctx);
@@ -962,13 +985,17 @@ static int invoke_bpf_prog(struct bpf_tramp_node *node, int args_off, int retval
/* arg3: &run_ctx */
emit_addi(RV_REG_A2, RV_REG_FP, -run_ctx_off, ctx);
ret = emit_call((const u64)bpf_trampoline_exit(p), true, ctx);
+ if (ret)
+ return ret;
- return ret;
+ bpf_tramp_image_add_skip(im, p, skip, ctx->ro_insns + ctx->ninsns);
+ return 0;
}
-static int invoke_bpf(struct bpf_tramp_nodes *tn, int args_off, int retval_off,
- int run_ctx_off, int func_meta_off, bool save_ret, u64 func_meta,
- int cookie_off, struct rv_jit_context *ctx)
+static int invoke_bpf(struct bpf_tramp_image *im, struct bpf_tramp_nodes *tn,
+ int args_off, int retval_off, int run_ctx_off, int func_meta_off,
+ bool save_ret, u64 func_meta, int cookie_off,
+ struct rv_jit_context *ctx)
{
int i, cur_cookie = (cookie_off - args_off) / 8;
@@ -981,8 +1008,8 @@ static int invoke_bpf(struct bpf_tramp_nodes *tn, int args_off, int retval_off,
emit_store_stack_imm64(RV_REG_T1, -func_meta_off, meta, ctx);
cur_cookie--;
}
- err = invoke_bpf_prog(tn->nodes[i], args_off, retval_off, run_ctx_off,
- save_ret, ctx);
+ err = invoke_bpf_prog(im, tn->nodes[i], args_off, retval_off,
+ run_ctx_off, save_ret, ctx);
if (err)
return err;
}
@@ -1170,7 +1197,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
}
if (fentry->nr_nodes) {
- ret = invoke_bpf(fentry, args_off, retval_off, run_ctx_off, func_meta_off,
+ ret = invoke_bpf(im, fentry, args_off, retval_off, run_ctx_off, func_meta_off,
flags & BPF_TRAMP_F_RET_FENTRY_RET, func_meta, cookie_off, ctx);
if (ret)
return ret;
@@ -1184,7 +1211,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
/* cleanup to avoid garbage return value confusion */
emit_sd(RV_REG_FP, -retval_off, RV_REG_ZERO, ctx);
for (i = 0; i < fmod_ret->nr_nodes; i++) {
- ret = invoke_bpf_prog(fmod_ret->nodes[i], args_off, retval_off,
+ ret = invoke_bpf_prog(im, fmod_ret->nodes[i], args_off, retval_off,
run_ctx_off, true, ctx);
if (ret)
goto out;
@@ -1211,10 +1238,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
emit_sd(RV_REG_FP, -tcc_off, RV_REG_TCC, ctx);
emit_sd(RV_REG_FP, -retval_off, RV_REG_A0, ctx);
emit_sd(RV_REG_FP, -(retval_off - 8), regmap[BPF_REG_0], ctx);
- im->ip_after_call = ctx->ro_insns + ctx->ninsns;
- /* 2 nops reserved for auipc+jalr pair */
- emit(rv_nop(), ctx);
- emit(rv_nop(), ctx);
}
/* update branches saved in invoke_bpf_mod_ret with bnez */
@@ -1230,14 +1253,13 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
emit_store_stack_imm64(RV_REG_T1, -func_meta_off, func_meta, ctx);
if (fexit->nr_nodes) {
- ret = invoke_bpf(fexit, args_off, retval_off, run_ctx_off, func_meta_off,
+ ret = invoke_bpf(im, fexit, args_off, retval_off, run_ctx_off, func_meta_off,
false, func_meta, cookie_off, ctx);
if (ret)
goto out;
}
if (flags & BPF_TRAMP_F_CALL_ORIG) {
- im->ip_epilogue = ctx->ro_insns + ctx->ninsns;
emit_imm(RV_REG_A0, ctx->insns ? (const s64)im : RV_MAX_COUNT_IMM, ctx);
ret = emit_call((const u64)__bpf_tramp_exit, true, ctx);
if (ret)
@@ -1299,7 +1321,7 @@ out:
int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *func_addr)
{
- struct bpf_tramp_image im;
+ struct bpf_tramp_image im = {};
struct rv_jit_context ctx;
int ret;
diff --git a/arch/s390/net/bpf_jit_comp.c b/arch/s390/net/bpf_jit_comp.c
index c4b47070bb59..1b2566012b63 100644
--- a/arch/s390/net/bpf_jit_comp.c
+++ b/arch/s390/net/bpf_jit_comp.c
@@ -2566,6 +2566,8 @@ struct bpf_tramp_jit {
int r14_off; /* Offset of saved %r14, has to be at the
* bottom */
int do_fexit; /* do_fexit: label */
+ int skip[BPF_MAX_TRAMP_LINKS]; /* skip: labels after each prog */
+ int nr_progs;
};
static void load_imm64(struct bpf_jit *jit, int dst_reg, u64 val)
@@ -2584,6 +2586,7 @@ static void emit_store_stack_imm64(struct bpf_jit *jit, int tmp_reg, int stack_o
}
static int invoke_bpf_prog(struct bpf_tramp_jit *tjit,
+ struct bpf_tramp_image *im,
const struct btf_func_model *m,
struct bpf_tramp_node *node, bool save_ret)
{
@@ -2591,8 +2594,20 @@ static int invoke_bpf_prog(struct bpf_tramp_jit *tjit,
int cookie_off = tjit->run_ctx_off +
offsetof(struct bpf_tramp_run_ctx, bpf_cookie);
struct bpf_prog *p = node->link->prog;
+ void *skip = jit->prg_buf + jit->prg;
+ int idx = tjit->nr_progs++;
int patch;
+ if (idx >= ARRAY_SIZE(tjit->skip))
+ return -E2BIG;
+
+ /*
+ * nop, patched to skip this prog when it is detached
+ */
+
+ /* brcl 0,skip */
+ EMIT6_PCREL_RILC(0xc0040000, 0, tjit->skip[idx]);
+
/*
* run_ctx.cookie = node->cookie;
*/
@@ -2652,10 +2667,15 @@ static int invoke_bpf_prog(struct bpf_tramp_jit *tjit,
/* brasl %r14,__bpf_prog_exit */
EMIT6_PCREL_RILB_PTR(0xc0050000, REG_14, bpf_trampoline_exit(p));
+ /* skip: */
+ tjit->skip[idx] = jit->prg;
+ bpf_tramp_image_add_skip(im, p, skip, jit->prg_buf + jit->prg);
+
return 0;
}
static int invoke_bpf(struct bpf_tramp_jit *tjit,
+ struct bpf_tramp_image *im,
const struct btf_func_model *m,
struct bpf_tramp_nodes *tn, bool save_ret,
u64 func_meta, int cookie_off)
@@ -2670,7 +2690,7 @@ static int invoke_bpf(struct bpf_tramp_jit *tjit,
emit_store_stack_imm64(jit, REG_0, tjit->func_meta_off, meta);
cur_cookie--;
}
- if (invoke_bpf_prog(tjit, m, tn->nodes[i], save_ret))
+ if (invoke_bpf_prog(tjit, im, m, tn->nodes[i], save_ret))
return -EINVAL;
}
@@ -2712,6 +2732,11 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
u64 func_meta;
int i, j;
+ /* The skip labels are taken from the previous pass. */
+ tjit->nr_progs = 0;
+ if (im)
+ im->nr_skips = 0;
+
/* Support as many stack arguments as "mvc" instruction can handle. */
nr_reg_args = min_t(int, m->nr_args, MAX_NR_REG_ARGS);
nr_stack_args = m->nr_args - nr_reg_args;
@@ -2875,7 +2900,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
emit_store_stack_imm64(jit, REG_0, tjit->retval_off, 0);
}
- if (invoke_bpf(tjit, m, fentry, flags & BPF_TRAMP_F_RET_FENTRY_RET,
+ if (invoke_bpf(tjit, im, m, fentry, flags & BPF_TRAMP_F_RET_FENTRY_RET,
func_meta, cookie_off))
return -EINVAL;
@@ -2889,7 +2914,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
0xf000 | tjit->retval_off);
for (i = 0; i < fmod_ret->nr_nodes; i++) {
- if (invoke_bpf_prog(tjit, m, fmod_ret->nodes[i], true))
+ if (invoke_bpf_prog(tjit, im, m, fmod_ret->nodes[i], true))
return -EINVAL;
/*
@@ -2943,15 +2968,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
/* mvc tccnt_off(%r15),tail_call_cnt(4,%r15) */
_EMIT6(0xd203f000 | tjit->tccnt_off,
0xf000 | offsetof(struct prog_frame, tail_call_cnt));
-
- im->ip_after_call = jit->prg_buf + jit->prg;
-
- /*
- * The following nop will be patched by bpf_tramp_image_put().
- */
-
- /* brcl 0,im->ip_epilogue */
- EMIT6_PCREL_RILC(0xc0040000, 0, (u64)im->ip_epilogue);
}
/* Set the "is_return" flag for fsession. */
@@ -2962,12 +2978,10 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
/* do_fexit: */
tjit->do_fexit = jit->prg;
- if (invoke_bpf(tjit, m, fexit, false, func_meta, cookie_off))
+ if (invoke_bpf(tjit, im, m, fexit, false, func_meta, cookie_off))
return -EINVAL;
if (flags & BPF_TRAMP_F_CALL_ORIG) {
- im->ip_epilogue = jit->prg_buf + jit->prg;
-
/*
* __bpf_tramp_exit(im);
*/
@@ -3016,7 +3030,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im,
int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *orig_call)
{
- struct bpf_tramp_image im;
+ struct bpf_tramp_image im = {};
struct bpf_tramp_jit tjit;
int ret;
diff --git a/arch/x86/net/bpf_jit_comp.c b/arch/x86/net/bpf_jit_comp.c
index 2853e87797a7..6bca87457e87 100644
--- a/arch/x86/net/bpf_jit_comp.c
+++ b/arch/x86/net/bpf_jit_comp.c
@@ -3217,16 +3217,21 @@ static void restore_regs(const struct btf_func_model *m, u8 **prog,
}
static int invoke_bpf_prog(const struct btf_func_model *m, u8 **pprog,
+ struct bpf_tramp_image *im,
struct bpf_tramp_node *node, int stack_size,
int run_ctx_off, bool save_ret,
void *image, void *rw_image)
{
u8 *prog = *pprog;
- u8 *jmp_insn;
+ u8 *jmp_insn, *skip;
int ctx_cookie_off = offsetof(struct bpf_tramp_run_ctx, bpf_cookie);
struct bpf_prog *p = node->link->prog;
u64 cookie = node->cookie;
+ /* nop, patched to skip this prog when it is detached */
+ skip = image + (prog - (u8 *)rw_image);
+ emit_nops(&prog, X86_PATCH_SIZE);
+
/* mov rdi, cookie */
emit_mov_imm64(&prog, BPF_REG_1, (long) cookie >> 32, (u32) (long) cookie);
@@ -3301,6 +3306,8 @@ static int invoke_bpf_prog(const struct btf_func_model *m, u8 **pprog,
if (emit_rsb_call(&prog, bpf_trampoline_exit(p), image + (prog - (u8 *)rw_image)))
return -EINVAL;
+ bpf_tramp_image_add_skip(im, p, skip, image + (prog - (u8 *)rw_image));
+
*pprog = prog;
return 0;
}
@@ -3332,6 +3339,7 @@ static int emit_cond_near_jump(u8 **pprog, void *func, void *ip, u8 jmp_cond)
}
static int invoke_bpf(const struct btf_func_model *m, u8 **pprog,
+ struct bpf_tramp_image *im,
struct bpf_tramp_nodes *tl, int stack_size,
int run_ctx_off, int func_meta_off, bool save_ret,
void *image, void *rw_image, u64 func_meta,
@@ -3346,7 +3354,7 @@ static int invoke_bpf(const struct btf_func_model *m, u8 **pprog,
func_meta | (cur_cookie << BPF_TRAMP_COOKIE_INDEX_SHIFT));
cur_cookie--;
}
- if (invoke_bpf_prog(m, &prog, tl->nodes[i], stack_size,
+ if (invoke_bpf_prog(m, &prog, im, tl->nodes[i], stack_size,
run_ctx_off, save_ret, image, rw_image))
return -EINVAL;
}
@@ -3355,6 +3363,7 @@ static int invoke_bpf(const struct btf_func_model *m, u8 **pprog,
}
static int invoke_bpf_mod_ret(const struct btf_func_model *m, u8 **pprog,
+ struct bpf_tramp_image *im,
struct bpf_tramp_nodes *tl, int stack_size,
int run_ctx_off, u8 **branches,
void *image, void *rw_image)
@@ -3368,7 +3377,7 @@ static int invoke_bpf_mod_ret(const struct btf_func_model *m, u8 **pprog,
emit_mov_imm32(&prog, false, BPF_REG_0, 0);
emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_0, -8);
for (i = 0; i < tl->nr_nodes; i++) {
- if (invoke_bpf_prog(m, &prog, tl->nodes[i], stack_size, run_ctx_off, true,
+ if (invoke_bpf_prog(m, &prog, im, tl->nodes[i], stack_size, run_ctx_off, true,
image, rw_image))
return -EINVAL;
@@ -3640,7 +3649,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
}
if (fentry->nr_nodes) {
- if (invoke_bpf(m, &prog, fentry, regs_off, run_ctx_off, func_meta_off,
+ if (invoke_bpf(m, &prog, im, fentry, regs_off, run_ctx_off, func_meta_off,
flags & BPF_TRAMP_F_RET_FENTRY_RET, image, rw_image,
func_meta, cookie_off))
return -EINVAL;
@@ -3652,7 +3661,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
if (!branches)
return -ENOMEM;
- if (invoke_bpf_mod_ret(m, &prog, fmod_ret, regs_off,
+ if (invoke_bpf_mod_ret(m, &prog, im, fmod_ret, regs_off,
run_ctx_off, branches, image, rw_image)) {
ret = -EINVAL;
goto cleanup;
@@ -3682,8 +3691,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
}
/* remember return value in a stack for bpf prog to access */
emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_0, -8);
- im->ip_after_call = image + (prog - (u8 *)rw_image);
- emit_nops(&prog, X86_PATCH_SIZE);
}
if (fmod_ret->nr_nodes) {
@@ -3708,7 +3715,7 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
emit_store_stack_imm64(&prog, BPF_REG_0, -func_meta_off, func_meta);
if (fexit->nr_nodes) {
- if (invoke_bpf(m, &prog, fexit, regs_off, run_ctx_off, func_meta_off,
+ if (invoke_bpf(m, &prog, im, fexit, regs_off, run_ctx_off, func_meta_off,
false, image, rw_image, func_meta, cookie_off)) {
ret = -EINVAL;
goto cleanup;
@@ -3723,7 +3730,6 @@ static int __arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *rw_im
* restored to R0.
*/
if (flags & BPF_TRAMP_F_CALL_ORIG) {
- im->ip_epilogue = image + (prog - (u8 *)rw_image);
/* arg1: mov rdi, im */
emit_mov_imm64(&prog, BPF_REG_1, (long) im >> 32, (u32) (long) im);
if (emit_rsb_call(&prog, __bpf_tramp_exit, image + (prog - (u8 *)rw_image))) {
@@ -3811,7 +3817,7 @@ out:
int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *func_addr)
{
- struct bpf_tramp_image im;
+ struct bpf_tramp_image im = {};
void *image;
int ret;
diff --git a/include/linux/bpf.h b/include/linux/bpf.h
index 1d2676782d70..0ecb9418dfbd 100644
--- a/include/linux/bpf.h
+++ b/include/linux/bpf.h
@@ -1258,11 +1258,15 @@ struct btf_func_model {
#define BPF_TRAMP_F_INDIRECT BIT(8)
/* Each call __bpf_prog_enter + call bpf_func + call __bpf_prog_exit is ~50
- * bytes on x86.
+ * bytes on x86. The trampoline image has to fit in PAGE_SIZE.
*/
enum {
-#if defined(__s390x__)
+#if defined(__s390x__) || defined(__powerpc64__)
BPF_MAX_TRAMP_LINKS = 27,
+#elif defined(__x86_64__)
+ BPF_MAX_TRAMP_LINKS = 36,
+#elif defined(__aarch64__)
+ BPF_MAX_TRAMP_LINKS = 37,
#else
BPF_MAX_TRAMP_LINKS = 38,
#endif
@@ -1315,6 +1319,7 @@ int arch_prepare_bpf_trampoline(struct bpf_tramp_image *im, void *image, void *i
void *arch_alloc_bpf_trampoline(unsigned int size);
void arch_free_bpf_trampoline(void *image, unsigned int size);
int __must_check arch_protect_bpf_trampoline(void *image, unsigned int size);
+int arch_bpf_trampoline_skip(void *nop, void *target);
int arch_bpf_trampoline_size(const struct btf_func_model *m, u32 flags,
struct bpf_tramp_nodes *tnodes, void *func_addr);
@@ -1363,19 +1368,48 @@ enum bpf_tramp_prog_type {
BPF_TRAMP_FSESSION,
};
+/*
+ * Each prog call in a trampoline image is preceded by a nop. When the prog is
+ * detached, the nop is patched to a jump to target, right after the call, so
+ * that tasks still running in the image skip the prog.
+ */
+struct bpf_tramp_skip {
+ struct bpf_prog *prog;
+ void *nop;
+ void *target;
+};
+
struct bpf_tramp_image {
void *image;
int size;
struct bpf_ksym ksym;
struct percpu_ref pcref;
- void *ip_after_call;
- void *ip_epilogue;
+ bool call_orig;
+ /* entry in tr->images, the image holds a reference on tr */
+ struct bpf_trampoline *tr;
+ struct list_head list;
+ int nr_skips;
+ struct bpf_tramp_skip *skips;
union {
struct rcu_head rcu;
struct work_struct work;
};
};
+static inline void bpf_tramp_image_add_skip(struct bpf_tramp_image *im, struct bpf_prog *prog,
+ void *nop, void *target)
+{
+ struct bpf_tramp_skip *skip;
+
+ /* struct_ops trampolines and arch_bpf_trampoline_size() have no image */
+ if (!im || !im->skips)
+ return;
+ skip = &im->skips[im->nr_skips++];
+ skip->prog = prog;
+ skip->nop = nop;
+ skip->target = target;
+}
+
struct bpf_trampoline {
/* hlist for trampoline_key_table */
struct hlist_node hlist_key;
@@ -1402,6 +1436,8 @@ struct bpf_trampoline {
int progs_cnt[BPF_TRAMP_MAX];
/* Executable image of trampoline */
struct bpf_tramp_image *cur_image;
+ /* Images not freed yet, cur_image and older ones still in use */
+ struct list_head images;
/* Used as temporary old image storage for multi_attach */
struct {
struct bpf_tramp_image *old_image;
@@ -1770,6 +1806,7 @@ struct bpf_prog_aux {
bool offload_requested; /* Program is bound and offloaded to the netdev. */
bool attach_btf_trace; /* true if attaching to BTF-enabled raw tp */
bool attach_tracing_prog; /* true if tracing another tracing program */
+ bool tramp_linked; /* true if it was ever linked to a trampoline */
bool func_proto_unreliable;
bool tail_call_reachable;
bool xdp_has_frags;
diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c
index 2e3bf8113ae9..7f11555a5070 100644
--- a/kernel/bpf/core.c
+++ b/kernel/bpf/core.c
@@ -1362,7 +1362,7 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from,
{
struct bpf_insn *to = to_buff;
u32 imm_rnd = get_random_u32();
- s16 off;
+ int off;
BUILD_BUG_ON(BPF_REG_PARAMS + 2 != MAX_BPF_JIT_REG);
BUILD_BUG_ON(BPF_REG_AX + 1 != MAX_BPF_JIT_REG);
@@ -1438,6 +1438,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from,
off = from->off;
if (off < 0)
off -= 2;
+ if (off < S16_MIN)
+ return -ERANGE;
*to++ = BPF_ALU64_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm);
*to++ = BPF_ALU64_IMM(BPF_XOR, BPF_REG_AX, imm_rnd);
*to++ = BPF_JMP_REG(from->code, from->dst_reg, BPF_REG_AX, off);
@@ -1458,6 +1460,8 @@ static int bpf_jit_blind_insn(const struct bpf_insn *from,
off = from->off;
if (off < 0)
off -= 2;
+ if (off < S16_MIN)
+ return -ERANGE;
*to++ = BPF_ALU32_IMM(BPF_MOV, BPF_REG_AX, imm_rnd ^ from->imm);
*to++ = BPF_ALU32_IMM(BPF_XOR, BPF_REG_AX, imm_rnd);
*to++ = BPF_JMP32_REG(from->code, from->dst_reg, BPF_REG_AX,
@@ -1606,7 +1610,9 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp
if (!rewritten)
continue;
- if (env)
+ if (rewritten < 0)
+ tmp = ERR_PTR(rewritten);
+ else if (env)
tmp = bpf_patch_insn_data(env, i, insn_buff, rewritten);
else
tmp = bpf_patch_insn_single(clone, i, insn_buff, rewritten);
diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c
index f9464e566f10..13a2356c84cf 100644
--- a/kernel/bpf/hashtab.c
+++ b/kernel/bpf/hashtab.c
@@ -128,6 +128,7 @@ struct htab_elem {
struct htab_btf_record {
struct btf_record *record;
+ struct btf *btf;
u32 key_size;
};
@@ -497,8 +498,13 @@ static void htab_dtor_ctx_free(void *ctx)
{
struct htab_btf_record *hrec = ctx;
+ /*
+ * The duplicated record still points into the map BTF, so free it
+ * before dropping the reference that keeps that BTF alive.
+ */
btf_record_free(hrec->record);
- kfree(ctx);
+ btf_put(hrec->btf);
+ kfree(hrec);
}
static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma,
@@ -521,6 +527,15 @@ static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma,
kfree(hrec);
return err;
}
+ /*
+ * btf_record_dup() only acquires kernel and module BTF. Fields whose
+ * types live in the map BTF keep pointing into it: kptrs to local
+ * types refer to map->btf, and graph roots carry a value record owned
+ * by its struct meta table. The context can outlive the map when the
+ * allocator defers its teardown, so hold a reference of our own.
+ */
+ hrec->btf = map->btf;
+ btf_get(hrec->btf);
bpf_mem_alloc_set_dtor(ma, dtor, htab_dtor_ctx_free, hrec);
return 0;
}
@@ -3359,8 +3374,10 @@ static int __rhtab_map_lookup_and_delete_batch(struct bpf_map *map,
}
if (do_delete) {
+ migrate_disable();
for (i = 0; i < total; i++)
rhtab_delete_elem(rhtab, del_elems[i], NULL, 0);
+ migrate_enable();
}
rcu_read_unlock();
diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c
index 8a8f088e83e6..15684d0fc883 100644
--- a/kernel/bpf/memalloc.c
+++ b/kernel/bpf/memalloc.c
@@ -118,6 +118,11 @@ struct bpf_mem_cache {
struct llist_head free_by_rcu_ttrace;
struct llist_head waiting_for_gp_ttrace;
struct rcu_head rcu_ttrace;
+ /*
+ * 0 - idle
+ * 1 - __free_rcu() is queued
+ * 2 - __free_rcu() is queued and free_by_rcu_ttrace got more objects since
+ */
atomic_t call_rcu_ttrace_in_progress;
raw_spinlock_t lock;
};
@@ -276,6 +281,8 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per
return cnt;
}
+static void __do_call_rcu_ttrace(struct bpf_mem_cache *c);
+
static void __free_rcu(struct rcu_head *head)
{
struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace);
@@ -285,7 +292,19 @@ static void __free_rcu(struct rcu_head *head)
llnode = llist_del_all(&c->waiting_for_gp_ttrace);
free_all(c, llnode, !!c->percpu_size);
- atomic_set(&c->call_rcu_ttrace_in_progress, 0);
+
+ /*
+ * do_call_rcu_ttrace() that ran while GP was in flight left its objects
+ * in free_by_rcu_ttrace. This cache may never free or alloc in bulk
+ * again, so start the next GP from here.
+ * 'c' can be freed as soon as call_rcu_ttrace_in_progress is zero.
+ */
+ if (atomic_cmpxchg(&c->call_rcu_ttrace_in_progress, 1, 0) == 1)
+ return;
+
+ /* Pairs with synchronize_rcu() in free_mem_alloc() */
+ guard(rcu)();
+ __do_call_rcu_ttrace(c);
}
static void enque_to_free(struct bpf_mem_cache *c, void *obj)
@@ -298,18 +317,15 @@ static void enque_to_free(struct bpf_mem_cache *c, void *obj)
llist_add(llnode, &c->free_by_rcu_ttrace);
}
-static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
+static void __do_call_rcu_ttrace(struct bpf_mem_cache *c)
{
struct llist_node *llnode, *t;
- if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) {
- if (unlikely(READ_ONCE(c->draining))) {
- scoped_guard(raw_spinlock_irqsave, &c->lock)
- llnode = llist_del_all(&c->free_by_rcu_ttrace);
- free_all(c, llnode, !!c->percpu_size);
- }
- return;
- }
+ /*
+ * Must be done before llist_del_all(). Objects that it misses were
+ * added by do_call_rcu_ttrace() that will set 2 after this store.
+ */
+ atomic_set(&c->call_rcu_ttrace_in_progress, 1);
WARN_ON_ONCE(!llist_empty(&c->waiting_for_gp_ttrace));
llist_for_each_safe(llnode, t, llist_del_all(&c->free_by_rcu_ttrace))
@@ -328,6 +344,22 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
call_rcu_tasks_trace(&c->rcu_ttrace, __free_rcu);
}
+static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
+{
+ struct llist_node *llnode;
+
+ if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 2)) {
+ if (unlikely(READ_ONCE(c->draining))) {
+ scoped_guard(raw_spinlock_irqsave, &c->lock)
+ llnode = llist_del_all(&c->free_by_rcu_ttrace);
+ free_all(c, llnode, !!c->percpu_size);
+ }
+ return;
+ }
+
+ __do_call_rcu_ttrace(c);
+}
+
static void free_bulk(struct bpf_mem_cache *c)
{
struct bpf_mem_cache *tgt = c->tgt;
@@ -700,7 +732,12 @@ static void free_mem_alloc(struct bpf_mem_alloc *ma)
* to wait for the pending __free_by_rcu(), and __free_rcu(). RCU Tasks
* Trace grace period implies RCU grace period, so all __free_rcu don't
* need extra call_rcu() (and thus extra rcu_barrier() here).
+ *
+ * __free_rcu() queues itself again unless it sees 'draining'. After
+ * synchronize_rcu() it either did that already or will not do it, so
+ * rcu_barrier_tasks_trace() cannot miss it.
*/
+ synchronize_rcu();
rcu_barrier(); /* wait for __free_by_rcu */
rcu_barrier_tasks_trace(); /* wait for __free_rcu */
free_mem_alloc_no_barrier(ma);
diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c
index 244a939b9d2d..96217b99399d 100644
--- a/kernel/bpf/syscall.c
+++ b/kernel/bpf/syscall.c
@@ -2448,6 +2448,21 @@ static void __bpf_prog_put_rcu(struct rcu_head *rcu)
bpf_prog_free(aux->prog);
}
+/*
+ * Progs called from a trampoline can also be reached by a task that was
+ * preempted in the trampoline before the prog's enter helper took its RCU
+ * read lock, wait for those first.
+ */
+static void __bpf_prog_put_rcu_tasks(struct rcu_head *rcu)
+{
+ struct bpf_prog *prog = container_of(rcu, struct bpf_prog_aux, rcu)->prog;
+
+ if (prog->sleepable)
+ call_rcu_tasks_trace(rcu, __bpf_prog_put_rcu);
+ else
+ call_rcu(rcu, __bpf_prog_put_rcu);
+}
+
static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred)
{
bpf_prog_kallsyms_del_all(prog);
@@ -2461,7 +2476,9 @@ static void __bpf_prog_put_noref(struct bpf_prog *prog, bool deferred)
btf_put(prog->aux->attach_btf);
if (deferred) {
- if (prog->sleepable)
+ if (IS_ENABLED(CONFIG_TASKS_RCU) && prog->aux->tramp_linked)
+ call_rcu_tasks(&prog->aux->rcu, __bpf_prog_put_rcu_tasks);
+ else if (prog->sleepable)
call_rcu_tasks_trace(&prog->aux->rcu, __bpf_prog_put_rcu);
else
call_rcu(&prog->aux->rcu, __bpf_prog_put_rcu);
diff --git a/kernel/bpf/trampoline.c b/kernel/bpf/trampoline.c
index 90b70ea0d370..bf4cb3dd444d 100644
--- a/kernel/bpf/trampoline.c
+++ b/kernel/bpf/trampoline.c
@@ -401,6 +401,7 @@ static struct bpf_trampoline *bpf_trampoline_lookup(u64 key, unsigned long ip)
head = &trampoline_ip_table[hash_64(tr->ip, TRAMPOLINE_HASH_BITS)];
hlist_add_head(&tr->hlist_ip, head);
refcount_set(&tr->refcnt, 1);
+ INIT_LIST_HEAD(&tr->images);
for (i = 0; i < BPF_TRAMP_MAX; i++)
INIT_HLIST_HEAD(&tr->progs_hlist[i]);
out:
@@ -565,15 +566,22 @@ static void bpf_tramp_image_free(struct bpf_tramp_image *im)
arch_free_bpf_trampoline(im->image, im->size);
bpf_jit_uncharge_modmem(im->size);
percpu_ref_exit(&im->pcref);
+ kfree(im->skips);
kfree_rcu(im, rcu);
}
static void __bpf_tramp_image_put_deferred(struct work_struct *work)
{
struct bpf_tramp_image *im;
+ struct bpf_trampoline *tr;
im = container_of(work, struct bpf_tramp_image, work);
+ tr = im->tr;
+ trampoline_lock(tr);
+ list_del(&im->list);
+ trampoline_unlock(tr);
bpf_tramp_image_free(im);
+ bpf_trampoline_put(tr);
}
/* callback, fexit step 3 or fentry step 2 */
@@ -601,7 +609,7 @@ static void __bpf_tramp_image_put_rcu_tasks(struct rcu_head *rcu)
struct bpf_tramp_image *im;
im = container_of(rcu, struct bpf_tramp_image, rcu);
- if (im->ip_after_call)
+ if (im->call_orig)
/* the case of fmod_ret/fexit trampoline and CONFIG_PREEMPTION=y */
percpu_ref_kill(&im->pcref);
else
@@ -621,9 +629,9 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im)
*
* The trampoline is unreachable before bpf_tramp_image_put().
*
- * First, patch the trampoline to avoid calling into fexit progs.
- * The progs will be freed even if the original function is still
- * executing or sleeping.
+ * Progs are patched out of the image when they are detached, see
+ * bpf_trampoline_skip_prog(), so they can be freed even if a task is
+ * still in the image.
* In case of CONFIG_PREEMPT=y use call_rcu_tasks() to wait on
* first few asm instructions to execute and call into
* __bpf_tramp_enter->percpu_ref_get.
@@ -637,11 +645,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im)
* percpu_ref_kill will be waiting for. Hence the first
* call_rcu_tasks() is not necessary.
*/
- if (im->ip_after_call) {
- int err = bpf_arch_text_poke(im->ip_after_call, BPF_MOD_NOP,
- BPF_MOD_JUMP, NULL,
- im->ip_epilogue);
- WARN_ON(err);
+ if (im->call_orig) {
if (IS_ENABLED(CONFIG_TASKS_RCU))
call_rcu_tasks(&im->rcu, __bpf_tramp_image_put_rcu_tasks);
else
@@ -658,7 +662,7 @@ static void bpf_tramp_image_put(struct bpf_tramp_image *im)
call_rcu_tasks_trace(&im->rcu, __bpf_tramp_image_put_rcu_tasks);
}
-static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size)
+static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size, int nr_progs)
{
struct bpf_tramp_image *im;
struct bpf_ksym *ksym;
@@ -669,6 +673,10 @@ static struct bpf_tramp_image *bpf_tramp_image_alloc(u64 key, int size)
if (!im)
goto out;
+ im->skips = kzalloc_objs(*im->skips, nr_progs);
+ if (!im->skips)
+ goto out_free_im;
+
err = bpf_jit_charge_modmem(size);
if (err)
goto out_free_im;
@@ -695,6 +703,7 @@ out_free_image:
out_uncharge:
bpf_jit_uncharge_modmem(size);
out_free_im:
+ kfree(im->skips);
kfree(im);
out:
return ERR_PTR(err);
@@ -771,11 +780,12 @@ again:
goto out;
}
- im = bpf_tramp_image_alloc(tr->key, size);
+ im = bpf_tramp_image_alloc(tr->key, size, total);
if (IS_ERR(im)) {
err = PTR_ERR(im);
goto out;
}
+ im->call_orig = tr->flags & BPF_TRAMP_F_CALL_ORIG;
err = arch_prepare_bpf_trampoline(im, im->image, im->image + size,
&tr->func.model, tr->flags, tnodes,
@@ -806,8 +816,14 @@ again:
#endif
out_free:
- if (err)
+ if (err) {
bpf_tramp_image_free(im);
+ } else {
+ /* track the image until it is freed, for bpf_trampoline_skip_prog() */
+ refcount_inc(&tr->refcnt);
+ im->tr = tr;
+ list_add(&im->list, &tr->images);
+ }
out:
/* If any error happens, restore previous flags */
if (err)
@@ -907,6 +923,7 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr,
}
hlist_add_head(&node->tramp_hlist, prog_list);
+ node->link->prog->aux->tramp_linked = true;
if (kind == BPF_TRAMP_FSESSION) {
tr->progs_cnt[BPF_TRAMP_FENTRY]++;
fexit = fsession_exit(node);
@@ -920,6 +937,41 @@ static int bpf_trampoline_add_prog(struct bpf_trampoline *tr,
return 0;
}
+/*
+ * Patch the nop in front of a prog call to a jump over it. A task can be
+ * preempted anywhere in the image, so archs that need several instructions for
+ * a jump of any range patch a single near branch here instead.
+ */
+int __weak arch_bpf_trampoline_skip(void *nop, void *target)
+{
+ return bpf_arch_text_poke(nop, BPF_MOD_NOP, BPF_MOD_JUMP, NULL, target);
+}
+
+/*
+ * prog was detached and can be freed, but tasks may still be running in images
+ * that call it, sleeping in an earlier prog for example. They can be in any
+ * image that is not freed yet, not only in cur_image, so patch all of them to
+ * jump over prog.
+ */
+static void bpf_trampoline_skip_prog(struct bpf_trampoline *tr, struct bpf_prog *prog)
+{
+ struct bpf_tramp_image *im;
+ int i, err;
+
+ list_for_each_entry(im, &tr->images, list) {
+ for (i = 0; i < im->nr_skips; i++) {
+ struct bpf_tramp_skip *skip = &im->skips[i];
+
+ if (skip->prog != prog)
+ continue;
+ err = arch_bpf_trampoline_skip(skip->nop, skip->target);
+ WARN_ON_ONCE(err);
+ /* not a nop anymore, and prog's address can be reused */
+ skip->prog = NULL;
+ }
+ }
+}
+
static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr,
struct bpf_tramp_node *node)
{
@@ -937,6 +989,7 @@ static void bpf_trampoline_remove_prog(struct bpf_trampoline *tr,
}
hlist_del_init(&node->tramp_hlist);
tr->progs_cnt[kind]--;
+ bpf_trampoline_skip_prog(tr, node->link->prog);
}
static int __bpf_trampoline_link_prog(struct bpf_tramp_node *node,
@@ -1245,11 +1298,9 @@ void bpf_trampoline_put(struct bpf_trampoline *tr)
if (WARN_ON_ONCE(!hlist_empty(&tr->progs_hlist[i])))
goto out;
- /* This code will be executed even when the last bpf_tramp_image
- * is alive. All progs are detached from the trampoline and the
- * trampoline image is patched with jmp into epilogue to skip
- * fexit progs. The fentry-only trampoline will be freed via
- * multiple rcu callbacks.
+ /*
+ * All progs are detached and the last image has been freed, images
+ * hold a reference on the trampoline until then.
*/
hlist_del(&tr->hlist_key);
hlist_del(&tr->hlist_ip);
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 41b49c56e123..5f874979b8d7 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -14826,7 +14826,15 @@ static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn
"Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range.");
return -EINVAL;
}
- reg_bounds_sync(dst_reg);
+ /*
+ * A packet pointer that keeps its id or range is checked against a
+ * range set from the checked pointer's umax, so var_off must not tighten
+ * its umax. r32 must still match var_off for reg_bounds_sanity_check().
+ */
+ if (reg_is_pkt_pointer(dst_reg) && (known || dst_reg->range > 0))
+ __update_reg32_bounds(dst_reg);
+ else
+ reg_bounds_sync(dst_reg);
bounds_ret = sanitize_check_bounds(env, insn, dst_reg);
if (bounds_ret == -EACCES)
return bounds_ret;
diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c b/tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c
new file mode 100644
index 000000000000..a1d41b109940
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c
@@ -0,0 +1,60 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <test_progs.h>
+#include "bpf_ma_ttrace.skel.h"
+
+#define NR_ELEMS 4096
+
+/*
+ * The first free_bulk() starts RCU tasks trace GP. The rest of the elements are
+ * deleted while it's in flight. They should be freed without further alloc or
+ * free from this map.
+ */
+void test_bpf_ma_ttrace(void)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, opts);
+ struct bpf_ma_ttrace *skel;
+ __u32 cnt = NR_ELEMS;
+ long *vals = NULL;
+ int *keys = NULL;
+ int i, err, fd, nr_cpus;
+
+ skel = bpf_ma_ttrace__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ return;
+ nr_cpus = libbpf_num_possible_cpus();
+ if (!ASSERT_GT(nr_cpus, 0, "nr_cpus"))
+ goto out;
+ skel->bss->nr_cpus = nr_cpus;
+
+ keys = calloc(NR_ELEMS, sizeof(*keys));
+ vals = calloc(NR_ELEMS, sizeof(*vals));
+ if (!ASSERT_OK_PTR(keys, "keys") || !ASSERT_OK_PTR(vals, "vals"))
+ goto out;
+ for (i = 0; i < NR_ELEMS; i++)
+ keys[i] = i;
+
+ fd = bpf_map__fd(skel->maps.htab);
+ err = bpf_map_update_batch(fd, keys, vals, &cnt, NULL);
+ if (!ASSERT_OK(err, "update_batch") || !ASSERT_EQ(cnt, NR_ELEMS, "update_cnt"))
+ goto out;
+ err = bpf_map_delete_batch(fd, keys, &cnt, NULL);
+ if (!ASSERT_OK(err, "delete_batch") || !ASSERT_EQ(cnt, NR_ELEMS, "delete_cnt"))
+ goto out;
+
+ /* Wait for all __free_rcu() callbacks to finish */
+ for (i = 0; i < 300; i++) {
+ err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.check_ttrace), &opts);
+ if (!ASSERT_OK(err, "test_run") || !ASSERT_OK(opts.retval, "retval"))
+ goto out;
+ if (!skel->bss->in_progress)
+ break;
+ usleep(100000);
+ }
+ ASSERT_EQ(skel->bss->nr_caches, nr_cpus, "nr_caches");
+ ASSERT_EQ(skel->bss->in_progress, 0, "in_progress");
+ ASSERT_EQ(skel->bss->not_freed, 0, "not_freed");
+out:
+ free(keys);
+ free(vals);
+ bpf_ma_ttrace__destroy(skel);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c b/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c
index ecc3d47919ad..f8497e764beb 100644
--- a/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c
+++ b/tools/testing/selftests/bpf/prog_tests/bpf_mod_race.c
@@ -55,38 +55,6 @@ static void *load_module_thread(void *p)
return p;
}
-static int sys_userfaultfd(int flags)
-{
- return syscall(__NR_userfaultfd, flags);
-}
-
-static int test_setup_uffd(void *fault_addr)
-{
- struct uffdio_register uffd_register = {};
- struct uffdio_api uffd_api = {};
- int uffd;
-
- uffd = sys_userfaultfd(O_CLOEXEC);
- if (uffd < 0)
- return -errno;
-
- uffd_api.api = UFFD_API;
- uffd_api.features = 0;
- if (ioctl(uffd, UFFDIO_API, &uffd_api)) {
- close(uffd);
- return -1;
- }
-
- uffd_register.range.start = (unsigned long)fault_addr;
- uffd_register.range.len = getpagesize();
- uffd_register.mode = UFFDIO_REGISTER_MODE_MISSING;
- if (ioctl(uffd, UFFDIO_REGISTER, &uffd_register)) {
- close(uffd);
- return -1;
- }
- return uffd;
-}
-
static void test_bpf_mod_race_config(const struct test_config *config)
{
void *fault_addr, *skel_fail;
@@ -117,7 +85,7 @@ static void test_bpf_mod_race_config(const struct test_config *config)
if (!ASSERT_OK(bpf_mod_race__attach(skel), "bpf_mod_kfunc_race__attach"))
goto end_destroy;
- uffd = test_setup_uffd(fault_addr);
+ uffd = uffd_block_page(fault_addr);
if (!ASSERT_GE(uffd, 0, "userfaultfd open + register address"))
goto end_destroy;
diff --git a/tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c b/tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c
new file mode 100644
index 000000000000..1eb0d7237605
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c
@@ -0,0 +1,188 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <test_progs.h>
+#include <pthread.h>
+#include <poll.h>
+#include <sys/mman.h>
+#include <linux/userfaultfd.h>
+#include "tramp_prog_detach.skel.h"
+#include "testing_helpers.h"
+
+/*
+ * Detach and free progs while a task sleeps in the prog that runs before them
+ * in the same trampoline image, then let that task continue through the
+ * image. It must not call into the freed progs.
+ *
+ * The task is held in a sleepable prog with userfaultfd, like bpf_mod_race
+ * does.
+ */
+
+static struct bpf_program *pick_prog(struct tramp_prog_detach *skel,
+ bool fexit, bool sleepable)
+{
+ if (fexit)
+ return sleepable ? skel->progs.fexit_sleepable :
+ skel->progs.fexit_victim;
+ return sleepable ? skel->progs.fentry_sleepable :
+ skel->progs.fentry_victim;
+}
+
+static struct bpf_program *sleepable_prog;
+
+static void *run_sleepable(void *arg)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, topts);
+
+ /* calls bpf_fentry_test1() */
+ return (void *)(long)bpf_prog_test_run_opts(bpf_program__fd(sleepable_prog),
+ &topts);
+}
+
+static struct tramp_prog_detach *load_one(bool fexit, bool sleepable)
+{
+ struct tramp_prog_detach *skel;
+ int err;
+
+ skel = tramp_prog_detach__open();
+ if (!ASSERT_OK_PTR(skel, "open"))
+ return NULL;
+
+ bpf_program__set_autoload(pick_prog(skel, fexit, sleepable), true);
+ err = tramp_prog_detach__load(skel);
+ if (!ASSERT_OK(err, "load"))
+ goto err;
+ skel->bss->pid = getpid();
+ err = tramp_prog_detach__attach(skel);
+ if (!ASSERT_OK(err, "attach"))
+ goto err;
+ return skel;
+err:
+ tramp_prog_detach__destroy(skel);
+ return NULL;
+}
+
+/* The .bss map of a destroyed skeleton goes away when its prog is freed */
+static bool wait_for_map_free(__u32 map_id)
+{
+ int i, fd;
+
+ for (i = 0; i < 100; i++) {
+ fd = bpf_map_get_fd_by_id(map_id);
+ if (fd < 0)
+ return true;
+ close(fd);
+ usleep(100 * 1000);
+ }
+ return false;
+}
+
+static void test_detach(bool sleepable_fexit, bool victim_fexit)
+{
+ struct tramp_prog_detach *sleepable = NULL, *victims[2] = {};
+ struct pollfd pfd = { .events = POLLIN };
+ struct uffdio_copy uffd_copy = {};
+ struct bpf_map_info map_info = {};
+ __u32 map_info_len = sizeof(map_info);
+ struct uffd_msg uffd_msg;
+ void *fault_page, *src_page = MAP_FAILED;
+ long page_size = getpagesize();
+ bool started = false;
+ void *thread_ret;
+ pthread_t thread;
+ int i, uffd = -1;
+
+ fault_page = mmap(NULL, page_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (!ASSERT_NEQ(fault_page, MAP_FAILED, "mmap fault_page"))
+ return;
+ src_page = mmap(NULL, page_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (!ASSERT_NEQ(src_page, MAP_FAILED, "mmap src_page"))
+ goto out;
+
+ /* The most recently attached prog runs first */
+ for (i = 0; i < ARRAY_SIZE(victims); i++) {
+ victims[i] = load_one(victim_fexit, false);
+ if (!victims[i])
+ goto out;
+ }
+ sleepable = load_one(sleepable_fexit, true);
+ if (!sleepable)
+ goto out;
+ sleepable_prog = pick_prog(sleepable, sleepable_fexit, true);
+
+ /* Not armed yet so this doesn't block, make sure sleepable runs first */
+ if (!ASSERT_OK((long)run_sleepable(NULL), "dry run"))
+ goto out;
+ for (i = 0; i < ARRAY_SIZE(victims); i++)
+ if (!ASSERT_LT(sleepable->bss->ts, victims[i]->bss->ts, "prog order"))
+ goto out;
+
+ uffd = uffd_block_page(fault_page);
+ if (!ASSERT_GE(uffd, 0, "userfaultfd open + register address"))
+ goto out;
+ sleepable->bss->fault_addr = fault_page;
+
+ if (!ASSERT_OK(pthread_create(&thread, NULL, run_sleepable, NULL),
+ "pthread_create"))
+ goto out;
+ started = true;
+
+ /* Wait for the thread to sleep in bpf_copy_from_user() */
+ pfd.fd = uffd;
+ if (!ASSERT_EQ(poll(&pfd, 1, 10000), 1, "poll uffd"))
+ goto out;
+ if (!ASSERT_EQ(read(uffd, &uffd_msg, sizeof(uffd_msg)), sizeof(uffd_msg),
+ "read uffd"))
+ goto out;
+ if (!ASSERT_EQ(uffd_msg.event, UFFD_EVENT_PAGEFAULT, "uffd pagefault"))
+ goto out;
+
+ /*
+ * Detach and unload the victim progs and wait for them to be freed.
+ * After the first one, the task is in an image that isn't the
+ * trampoline's current one anymore.
+ */
+ for (i = 0; i < ARRAY_SIZE(victims); i++) {
+ if (!ASSERT_OK(bpf_map_get_info_by_fd(bpf_map__fd(victims[i]->maps.bss),
+ &map_info, &map_info_len),
+ "victim bss info"))
+ goto out;
+ tramp_prog_detach__destroy(victims[i]);
+ victims[i] = NULL;
+ if (!ASSERT_TRUE(wait_for_map_free(map_info.id), "victim freed"))
+ goto out;
+ }
+
+out:
+ /* Let the thread proceed with the rest of the trampoline */
+ if (uffd >= 0) {
+ uffd_copy.dst = (unsigned long)fault_page;
+ uffd_copy.src = (unsigned long)src_page;
+ uffd_copy.len = page_size;
+ ASSERT_OK(ioctl(uffd, UFFDIO_COPY, &uffd_copy), "uffd copy");
+ close(uffd);
+ }
+ if (started &&
+ ASSERT_OK(pthread_join(thread, &thread_ret), "pthread_join"))
+ ASSERT_NULL(thread_ret, "blocking run");
+
+ for (i = 0; i < ARRAY_SIZE(victims); i++)
+ tramp_prog_detach__destroy(victims[i]);
+ tramp_prog_detach__destroy(sleepable);
+ if (src_page != MAP_FAILED)
+ munmap(src_page, page_size);
+ munmap(fault_page, page_size);
+}
+
+void serial_test_tramp_prog_detach(void)
+{
+ /* a task sleeping before the original function is called */
+ if (test__start_subtest("fentry"))
+ test_detach(false, false);
+ /* a task sleeping after the original function returned */
+ if (test__start_subtest("fexit"))
+ test_detach(true, true);
+ /* the original function runs in between and must still be called */
+ if (test__start_subtest("fentry_fexit"))
+ test_detach(false, true);
+}
diff --git a/tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c b/tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c
new file mode 100644
index 000000000000..31b2a962e339
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c
@@ -0,0 +1,50 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_core_read.h>
+#include "bpf_kfuncs.h"
+
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __uint(max_entries, 4096);
+ __type(key, int);
+ __type(value, long);
+} htab SEC(".maps");
+
+extern const void __per_cpu_offset __ksym;
+
+int nr_cpus;
+int nr_caches;
+int in_progress;
+int not_freed;
+
+/* Look at bpf_mem_cache of every cpu that htab allocates its elements from */
+SEC("syscall")
+int check_ttrace(void *ctx)
+{
+ struct bpf_htab *h = bpf_core_cast(&htab, struct bpf_htab);
+ unsigned long cache = (unsigned long)BPF_CORE_READ(h, ma.cache);
+ const unsigned long *offsets = &__per_cpu_offset;
+ struct bpf_mem_cache *c;
+ unsigned long off;
+ int cpu;
+
+ nr_caches = 0;
+ in_progress = 0;
+ not_freed = 0;
+ bpf_for(cpu, 0, nr_cpus) {
+ if (bpf_probe_read_kernel(&off, sizeof(off), offsets + cpu))
+ return -1;
+ c = bpf_core_cast((void *)(cache + off), struct bpf_mem_cache);
+ if (c->unit_size)
+ nr_caches++;
+ if (c->call_rcu_ttrace_in_progress.counter)
+ in_progress++;
+ if (c->free_by_rcu_ttrace.first || c->waiting_for_gp_ttrace.first)
+ not_freed++;
+ }
+ return 0;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/tramp_prog_detach.c b/tools/testing/selftests/bpf/progs/tramp_prog_detach.c
new file mode 100644
index 000000000000..507d372167ec
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/tramp_prog_detach.c
@@ -0,0 +1,56 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+int pid;
+void *fault_addr;
+__u64 ts;
+
+static int do_sleepable(void)
+{
+ char dst;
+
+ if (bpf_get_current_pid_tgid() >> 32 != pid)
+ return 0;
+
+ ts = bpf_ktime_get_ns();
+ /* blocks for as long as user space wants when fault_addr is armed */
+ bpf_copy_from_user(&dst, sizeof(dst), fault_addr);
+ return 0;
+}
+
+static int do_victim(void)
+{
+ if (bpf_get_current_pid_tgid() >> 32 != pid)
+ return 0;
+
+ ts = bpf_ktime_get_ns();
+ return 0;
+}
+
+SEC("?fentry.s/bpf_fentry_test1")
+int BPF_PROG(fentry_sleepable, int a)
+{
+ return do_sleepable();
+}
+
+SEC("?fentry/bpf_fentry_test1")
+int BPF_PROG(fentry_victim, int a)
+{
+ return do_victim();
+}
+
+SEC("?fexit.s/bpf_fentry_test1")
+int BPF_PROG(fexit_sleepable, int a, int ret)
+{
+ return do_sleepable();
+}
+
+SEC("?fexit/bpf_fentry_test1")
+int BPF_PROG(fexit_victim, int a, int ret)
+{
+ return do_victim();
+}
diff --git a/tools/testing/selftests/bpf/progs/verifier_align.c b/tools/testing/selftests/bpf/progs/verifier_align.c
index 3e52686515ca..0083ef8b8ba7 100644
--- a/tools/testing/selftests/bpf/progs/verifier_align.c
+++ b/tools/testing/selftests/bpf/progs/verifier_align.c
@@ -284,7 +284,7 @@ __msg("26: {{.*}} R5=pkt(r=8,imm=14)")
*/
__msg("28: {{.*}} R4={{[^)]*}}var_off=(0x2; 0x7fc){{.*}} R5={{[^)]*}}var_off=(0x2; 0x7fc)")
/* Constant is added to R5 again, setting reg->off to 18. */
-__msg("29: {{.*}} R5=pkt(id=3,{{[^)]*}}var_off=(0x2; 0x7fc)")
+__msg("29: {{.*}} R5=pkt(id=3,{{[^)]*}}var_off=(0x2; 0xffc)")
/* And once more we add a variable; resulting {{[^)]*}}var_off
* is still (4n), fixed offset is not changed.
* Also, we create a new reg->id.
@@ -359,7 +359,7 @@ __msg("7: {{.*}} R6={{[^)]*}}var_off=(0x0; 0x3fc)")
__msg("8: {{.*}} R6={{[^)]*}}var_off=(0x2; 0x7fc)")
/* Packet pointer has (4n+2) offset */
__msg("11: {{.*}} R5={{[^)]*}}var_off=(0x2; 0x7fc)")
-__msg("12: {{.*}} R4={{[^)]*}}var_off=(0x2; 0x7fc)")
+__msg("12: {{.*}} R4={{[^)]*}}var_off=(0x2; 0xffc)")
/* At the time the word size load is performed from R5,
* its total fixed offset is NET_IP_ALIGN + reg->off (0)
* which is 2. Then the variable offset is (4n+2), so
@@ -375,7 +375,7 @@ __msg("17: {{.*}} R6={{[^)]*}}var_off=(0x0; 0x3fc)")
* another (4n+2).
*/
__msg("19: {{.*}} R5={{[^)]*}}var_off=(0x2; 0xffc)")
-__msg("20: {{.*}} R4={{[^)]*}}var_off=(0x2; 0xffc)")
+__msg("20: {{.*}} R4={{[^)]*}}var_off=(0x2; 0x1ffc)")
/* At the time the word size load is performed from R5,
* its total fixed offset is NET_IP_ALIGN + reg->off (0)
* which is 2. Then the variable offset is (4n+2), so
diff --git a/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c b/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c
index 915a9707298b..139ff019d87d 100644
--- a/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c
+++ b/tools/testing/selftests/bpf/progs/verifier_direct_packet_access.c
@@ -920,4 +920,182 @@ l1_%=: r0 = *(u8*)(r9 + 0); \
: __clobber_all);
}
+SEC("tc")
+__description("direct packet access: 8-aligned offset, check p + 8, load 8 bytes at p")
+__success __retval(0) __flag(BPF_F_ANY_ALIGNMENT)
+__naked void pkt_same_id_check_copy_load_base(void)
+{
+ asm volatile (" \
+ r2 = *(u32*)(r1 + %[__sk_buff_data]); \
+ r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \
+ r4 = *(u32*)(r1 + %[__sk_buff_mark]); \
+ r4 &= 0x38; \
+ if r4 > 50 goto l0_%=; \
+ /* r4 is a multiple of 8, at most 48 */ \
+ r5 = r2; \
+ r5 += r4; \
+ r6 = r5; \
+ r6 += 8; \
+ if r6 > r3 goto l0_%=; \
+ /* [r5, r5 + 8) is in the packet */ \
+ r0 = *(u64*)(r5 + 0); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)),
+ __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)),
+ __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark))
+ : __clobber_all);
+}
+
+SEC("tc")
+__description("direct packet access: 8-aligned offset, check p, load past it via p + 8")
+__failure __msg("invalid access to packet, off=51 size=1")
+__naked void pkt_same_id_check_base_load_via_copy(void)
+{
+ asm volatile (" \
+ r2 = *(u32*)(r1 + %[__sk_buff_data]); \
+ r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \
+ r4 = *(u32*)(r1 + %[__sk_buff_mark]); \
+ r4 &= 0x38; \
+ if r4 > 50 goto l0_%=; \
+ /* r4 is a multiple of 8, at most 48 */ \
+ r5 = r2; \
+ r5 += r4; \
+ r6 = r5; \
+ r6 += 8; \
+ if r5 > r3 goto l0_%=; \
+ /* r6 - 7 is r5 + 1 */ \
+ r0 = *(u8*)(r6 - 7); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)),
+ __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)),
+ __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark))
+ : __clobber_all);
+}
+
+SEC("tc")
+__description("direct packet access: 8-aligned offset, check p + 4, 4-byte load at p + 2")
+__failure __msg("invalid access to packet, off=52 size=4")
+__naked void pkt_same_id_check_base_4_load_via_copy(void)
+{
+ asm volatile (" \
+ r2 = *(u32*)(r1 + %[__sk_buff_data]); \
+ r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \
+ r4 = *(u32*)(r1 + %[__sk_buff_mark]); \
+ r4 &= 0x38; \
+ if r4 > 50 goto l0_%=; \
+ /* r4 is a multiple of 8, at most 48 */ \
+ r5 = r2; \
+ r5 += r4; \
+ r6 = r5; \
+ r6 += 8; \
+ r7 = r5; \
+ r7 += 4; \
+ if r7 > r3 goto l0_%=; \
+ /* [r5, r5 + 4) is in the packet, [r5 + 2, r5 + 6) may not be */ \
+ r0 = *(u32*)(r6 - 6); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)),
+ __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)),
+ __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark))
+ : __clobber_all);
+}
+
+SEC("tc")
+__description("direct packet access: checked pointer minus non-negative unknown keeps range")
+__success __retval(0) __flag(BPF_F_ANY_ALIGNMENT)
+__naked void pkt_sub_unknown_keeps_range(void)
+{
+ asm volatile (" \
+ r2 = *(u32*)(r1 + %[__sk_buff_data]); \
+ r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \
+ r4 = *(u32*)(r1 + %[__sk_buff_mark]); \
+ r4 &= 0x1f; \
+ r4 += 8; \
+ r5 = r2; \
+ r5 += 40; \
+ if r5 > r3 goto l0_%=; \
+ /* r5 is 8 to 39 bytes below the checked pointer */ \
+ r5 -= r4; \
+ r0 = *(u64*)(r5 + 0); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)),
+ __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)),
+ __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark))
+ : __clobber_all);
+}
+
+SEC("tc")
+__description("direct packet access: no pruning of a path whose checks prove fewer bytes")
+__failure __msg("invalid access to packet, off=255 size=8")
+__flag(BPF_F_ANY_ALIGNMENT) __flag(BPF_F_TEST_STATE_FREQ)
+__naked void pkt_same_id_pruning(void)
+{
+ asm volatile (" \
+ r2 = *(u32*)(r1 + %[__sk_buff_data]); \
+ r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \
+ r4 = *(u32*)(r1 + %[__sk_buff_mark]); \
+ r0 = *(u32*)(r1 + %[__sk_buff_priority]); \
+ r4 &= 0xff; \
+ r5 = r2; \
+ r5 += r4; \
+ if r0 != 0 goto l1_%=; \
+ /* this path proves [r5, r5 + 8) */ \
+ r6 = r5; \
+ r6 += 8; \
+ if r6 > r3 goto l0_%=; \
+ goto l2_%=; \
+l1_%=: /* this path proves [r5, r5 + 7) */ \
+ if r5 > r3 goto l0_%=; \
+ r6 = r5; \
+ r6 += 7; \
+ if r6 > r3 goto l0_%=; \
+l2_%=: r0 = *(u64*)(r5 + 0); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)),
+ __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)),
+ __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark)),
+ __imm_const(__sk_buff_priority, offsetof(struct __sk_buff, priority))
+ : __clobber_all);
+}
+
+SEC("tc")
+__description("direct packet access: spilled copy of checked pointer, load past range")
+__failure __msg("invalid access to packet, off=262 size=1, R7(id={{[0-9]+}},off=262,r=262)")
+__naked void pkt_spilled_copy_gets_range(void)
+{
+ asm volatile (" \
+ r2 = *(u32*)(r1 + %[__sk_buff_data]); \
+ r3 = *(u32*)(r1 + %[__sk_buff_data_end]); \
+ r4 = *(u32*)(r1 + %[__sk_buff_mark]); \
+ r4 &= 0xff; \
+ r5 = r2; \
+ r5 += r4; \
+ *(u64*)(r10 - 8) = r5; \
+ /* proves only bytes before r5 */ \
+ if r5 > r3 goto l0_%=; \
+ r6 = r5; \
+ r6 += 7; \
+ if r6 > r3 goto l0_%=; \
+ /* [r5, r5 + 7) is in the packet */ \
+ r7 = *(u64*)(r10 - 8); \
+ r0 = *(u8*)(r7 + 7); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_data, offsetof(struct __sk_buff, data)),
+ __imm_const(__sk_buff_data_end, offsetof(struct __sk_buff, data_end)),
+ __imm_const(__sk_buff_mark, offsetof(struct __sk_buff, mark))
+ : __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_meta_access.c b/tools/testing/selftests/bpf/progs/verifier_meta_access.c
index 62235f032ffe..c87e0be4b2ff 100644
--- a/tools/testing/selftests/bpf/progs/verifier_meta_access.c
+++ b/tools/testing/selftests/bpf/progs/verifier_meta_access.c
@@ -281,4 +281,34 @@ l0_%=: r0 = 0; \
: __clobber_all);
}
+SEC("xdp")
+__description("meta access, 8-aligned offset, check p, load a byte at p + 1 via p + 8")
+__failure __msg("invalid access to packet, off=51 size=1")
+__naked void meta_access_check_base_load_via_copy(void)
+{
+ asm volatile (" \
+ r9 = r1; \
+ call %[bpf_get_prandom_u32]; \
+ r4 = r0; \
+ r4 &= 0x38; \
+ if r4 > 50 goto l0_%=; \
+ /* r4 is a multiple of 8, at most 48 */ \
+ r2 = *(u32*)(r9 + %[xdp_md_data_meta]); \
+ r3 = *(u32*)(r9 + %[xdp_md_data]); \
+ r5 = r2; \
+ r5 += r4; \
+ r6 = r5; \
+ r6 += 8; \
+ if r5 > r3 goto l0_%=; \
+ /* r6 - 7 is r5 + 1 */ \
+ r0 = *(u8*)(r6 - 7); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_get_prandom_u32),
+ __imm_const(xdp_md_data, offsetof(struct xdp_md, data)),
+ __imm_const(xdp_md_data_meta, offsetof(struct xdp_md, data_meta))
+ : __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/testing_helpers.c b/tools/testing/selftests/bpf/testing_helpers.c
index c970e7793dfc..9f672d0c9dda 100644
--- a/tools/testing/selftests/bpf/testing_helpers.c
+++ b/tools/testing/selftests/bpf/testing_helpers.c
@@ -13,6 +13,7 @@
#include "test_progs.h"
#include "testing_helpers.h"
#include <linux/membarrier.h>
+#include <linux/userfaultfd.h>
int parse_num_list(const char *s, bool **num_set, int *num_set_len)
{
@@ -459,6 +460,33 @@ int kern_sync_rcu(void)
return syscall(__NR_membarrier, MEMBARRIER_CMD_SHARED, 0, 0);
}
+int uffd_block_page(void *fault_addr)
+{
+ struct uffdio_register uffd_register = {};
+ struct uffdio_api uffd_api = {};
+ int uffd;
+
+ uffd = syscall(__NR_userfaultfd, O_CLOEXEC);
+ if (uffd < 0)
+ return -errno;
+
+ uffd_api.api = UFFD_API;
+ uffd_api.features = 0;
+ if (ioctl(uffd, UFFDIO_API, &uffd_api)) {
+ close(uffd);
+ return -1;
+ }
+
+ uffd_register.range.start = (unsigned long)fault_addr;
+ uffd_register.range.len = getpagesize();
+ uffd_register.mode = UFFDIO_REGISTER_MODE_MISSING;
+ if (ioctl(uffd, UFFDIO_REGISTER, &uffd_register)) {
+ close(uffd);
+ return -1;
+ }
+ return uffd;
+}
+
int get_xlated_program(int fd_prog, struct bpf_insn **buf, __u32 *cnt)
{
__u32 buf_element_size = sizeof(struct bpf_insn);
diff --git a/tools/testing/selftests/bpf/testing_helpers.h b/tools/testing/selftests/bpf/testing_helpers.h
index 2edc6fb7fc52..1f83b6080716 100644
--- a/tools/testing/selftests/bpf/testing_helpers.h
+++ b/tools/testing/selftests/bpf/testing_helpers.h
@@ -36,6 +36,8 @@ __u64 read_perf_max_sample_freq(void);
int load_bpf_testmod(bool verbose);
int unload_bpf_testmod(bool verbose);
int kern_sync_rcu(void);
+/* returns a userfaultfd that makes accesses to fault_addr's page block */
+int uffd_block_page(void *fault_addr);
int finit_module(int fd, const char *param_values, int flags);
int delete_module(const char *name, int flags);
int load_module(const char *path, bool verbose);