summaryrefslogtreecommitdiff
path: root/tools/testing/selftests
diff options
context:
space:
mode:
Diffstat (limited to 'tools/testing/selftests')
-rw-r--r--tools/testing/selftests/Makefile5
-rw-r--r--tools/testing/selftests/acct/.gitignore1
-rw-r--r--tools/testing/selftests/acct/Makefile11
-rw-r--r--tools/testing/selftests/acct/cgroupstats.c231
-rw-r--r--tools/testing/selftests/acct/netlink_helper.c116
-rw-r--r--tools/testing/selftests/acct/netlink_helper.h44
-rw-r--r--tools/testing/selftests/acct/taskstats_fill_stats_tgid.c134
-rw-r--r--tools/testing/selftests/alloc_tag/Makefile8
-rw-r--r--tools/testing/selftests/alloc_tag/allocinfo_ioctl_test.c548
-rw-r--r--tools/testing/selftests/alsa/mixer-test.c4
-rw-r--r--tools/testing/selftests/arm64/abi/.gitignore2
-rw-r--r--tools/testing/selftests/arm64/abi/Makefile4
-rw-r--r--tools/testing/selftests/arm64/abi/seccomp_ptrace_x0_bypass.c192
-rw-r--r--tools/testing/selftests/arm64/abi/seccomp_ret_trace_x0_bypass.c201
-rw-r--r--tools/testing/selftests/arm64/config17
-rw-r--r--tools/testing/selftests/arm64/fp/fp-ptrace.c47
-rw-r--r--tools/testing/selftests/arm64/fp/sve-test.S7
-rw-r--r--tools/testing/selftests/arm64/mte/check_buffer_fill.c2
-rw-r--r--tools/testing/selftests/arm64/mte/check_child_memory.c2
-rw-r--r--tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c3
-rw-r--r--tools/testing/selftests/arm64/mte/check_hugetlb_options.c2
-rw-r--r--tools/testing/selftests/arm64/mte/check_ksm_options.c31
-rw-r--r--tools/testing/selftests/arm64/mte/check_mmap_options.c2
-rw-r--r--tools/testing/selftests/arm64/mte/check_prctl.c2
-rw-r--r--tools/testing/selftests/arm64/mte/check_tags_inclusion.c2
-rw-r--r--tools/testing/selftests/arm64/mte/check_user_mem.c2
-rw-r--r--tools/testing/selftests/bpf/Makefile5
-rw-r--r--tools/testing/selftests/bpf/README.rst4
-rw-r--r--tools/testing/selftests/bpf/bench.c20
-rw-r--r--tools/testing/selftests/bpf/benchs/bench_trigger.c38
-rwxr-xr-xtools/testing/selftests/bpf/benchs/run_bench_uprobes.sh2
-rw-r--r--tools/testing/selftests/bpf/bpf_experimental.h10
-rw-r--r--tools/testing/selftests/bpf/libarena/include/bpf_arena_spin_lock.h7
-rw-r--r--tools/testing/selftests/bpf/libarena/src/common.bpf.c7
-rw-r--r--tools/testing/selftests/bpf/network_helpers.c24
-rw-r--r--tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c2
-rw-r--r--tools/testing/selftests/bpf/prog_tests/btf_map_keyless.c59
-rw-r--r--tools/testing/selftests/bpf/prog_tests/cb_refs.c2
-rw-r--r--tools/testing/selftests/bpf/prog_tests/cgroup_iter_memcg.c43
-rw-r--r--tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c192
-rw-r--r--tools/testing/selftests/bpf/prog_tests/dynptr.c10
-rw-r--r--tools/testing/selftests/bpf/prog_tests/exceptions.c1
-rw-r--r--tools/testing/selftests/bpf/prog_tests/global_data_init.c336
-rw-r--r--tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c3
-rw-r--r--tools/testing/selftests/bpf/prog_tests/ksock.c130
-rw-r--r--tools/testing/selftests/bpf/prog_tests/ksock_wq.c45
-rw-r--r--tools/testing/selftests/bpf/prog_tests/linked_list.c4
-rw-r--r--tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c296
-rw-r--r--tools/testing/selftests/bpf/prog_tests/lwt_ip_encap.c11
-rw-r--r--tools/testing/selftests/bpf/prog_tests/percpu_alloc.c105
-rw-r--r--tools/testing/selftests/bpf/prog_tests/rcu_read_lock.c2
-rw-r--r--tools/testing/selftests/bpf/prog_tests/resolve_btfids.c41
-rw-r--r--tools/testing/selftests/bpf/prog_tests/rhash.c6
-rw-r--r--tools/testing/selftests/bpf/prog_tests/rhash_timer.c141
-rw-r--r--tools/testing/selftests/bpf/prog_tests/snprintf_btf.c79
-rw-r--r--tools/testing/selftests/bpf/prog_tests/sock_destroy.c121
-rw-r--r--tools/testing/selftests/bpf/prog_tests/sock_xattr.c67
-rw-r--r--tools/testing/selftests/bpf/prog_tests/sockopt.c22
-rw-r--r--tools/testing/selftests/bpf/prog_tests/spin_lock.c2
-rw-r--r--tools/testing/selftests/bpf/prog_tests/stream.c4
-rw-r--r--tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c125
-rw-r--r--tools/testing/selftests/bpf/prog_tests/test_struct_ops_arena.c10
-rw-r--r--tools/testing/selftests/bpf/prog_tests/test_tc_tunnel.c1
-rw-r--r--tools/testing/selftests/bpf/prog_tests/test_veristat.c113
-rw-r--r--tools/testing/selftests/bpf/prog_tests/test_xsk.c4
-rw-r--r--tools/testing/selftests/bpf/prog_tests/timer_mim.c29
-rw-r--r--tools/testing/selftests/bpf/prog_tests/tp_btf_ids.c132
-rw-r--r--tools/testing/selftests/bpf/prog_tests/uprobe_syscall.c325
-rw-r--r--tools/testing/selftests/bpf/prog_tests/usdt.c74
-rw-r--r--tools/testing/selftests/bpf/prog_tests/verifier.c8
-rw-r--r--tools/testing/selftests/bpf/progs/arena_kfunc.c9
-rw-r--r--tools/testing/selftests/bpf/progs/arena_kfunc_jit.c20
-rw-r--r--tools/testing/selftests/bpf/progs/arena_spin_lock.c7
-rw-r--r--tools/testing/selftests/bpf/progs/bpf_qdisc_fail__untrusted_write.c65
-rw-r--r--tools/testing/selftests/bpf/progs/btf_data.c20
-rw-r--r--tools/testing/selftests/bpf/progs/dynptr_fail.c2
-rw-r--r--tools/testing/selftests/bpf/progs/dynptr_success.c20
-rw-r--r--tools/testing/selftests/bpf/progs/exceptions.c30
-rw-r--r--tools/testing/selftests/bpf/progs/iters.c39
-rw-r--r--tools/testing/selftests/bpf/progs/iters_state_safety.c22
-rw-r--r--tools/testing/selftests/bpf/progs/iters_task_failure.c46
-rw-r--r--tools/testing/selftests/bpf/progs/ksock_common.h78
-rw-r--r--tools/testing/selftests/bpf/progs/ksock_lsm.c72
-rw-r--r--tools/testing/selftests/bpf/progs/ksock_lsm_verifier.c35
-rw-r--r--tools/testing/selftests/bpf/progs/ksock_wq.c62
-rw-r--r--tools/testing/selftests/bpf/progs/map_kptr_fail.c46
-rw-r--r--tools/testing/selftests/bpf/progs/mem_rdonly_untrusted.c234
-rw-r--r--tools/testing/selftests/bpf/progs/percpu_alloc_fail.c59
-rw-r--r--tools/testing/selftests/bpf/progs/preempt_lock.c52
-rw-r--r--tools/testing/selftests/bpf/progs/pyperf.h12
-rw-r--r--tools/testing/selftests/bpf/progs/raw_tp_null_fail.c53
-rw-r--r--tools/testing/selftests/bpf/progs/rbtree_fail.c84
-rw-r--r--tools/testing/selftests/bpf/progs/rcu_read_lock.c76
-rw-r--r--tools/testing/selftests/bpf/progs/refcounted_kptr.c61
-rw-r--r--tools/testing/selftests/bpf/progs/refcounted_kptr_fail.c159
-rw-r--r--tools/testing/selftests/bpf/progs/rhash.c112
-rw-r--r--tools/testing/selftests/bpf/progs/rhash_timer.c98
-rw-r--r--tools/testing/selftests/bpf/progs/snprintf_btf_void.c24
-rw-r--r--tools/testing/selftests/bpf/progs/sock_destroy_prog.c30
-rw-r--r--tools/testing/selftests/bpf/progs/sock_read_xattr.c54
-rw-r--r--tools/testing/selftests/bpf/progs/stream.c102
-rw-r--r--tools/testing/selftests/bpf/progs/struct_ops_arena.c24
-rw-r--r--tools/testing/selftests/bpf/progs/tailcall_callback.c16
-rw-r--r--tools/testing/selftests/bpf/progs/task_local_data.bpf.h3
-rw-r--r--tools/testing/selftests/bpf/progs/test_bpf_nf.c8
-rw-r--r--tools/testing/selftests/bpf/progs/test_global_percpu_data.c89
-rw-r--r--tools/testing/selftests/bpf/progs/test_spin_lock_fail.c67
-rw-r--r--tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c129
-rw-r--r--tools/testing/selftests/bpf/progs/test_tc_tunnel.c91
-rw-r--r--tools/testing/selftests/bpf/progs/test_usdt.c25
-rw-r--r--tools/testing/selftests/bpf/progs/timer_mim_reject.c84
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_arena.c81
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_async_cb_context.c204
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_basic_stack.c6
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_bounds.c41
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_bpf_fastcall.c139
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c56
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_cfg.c14
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_cgroup_storage.c29
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c77
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_global_subprogs.c10
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_gotox.c48
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_iterating_callbacks.c17
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_jeq_infer_not_null.c193
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c75
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_ld_ind.c96
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_loops1.c36
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_map_in_map.c1
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_may_goto_1.c5
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_netfilter_ctx.c78
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_percpu_addr.c72
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_precision.c100
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_private_stack.c15
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_scalar_ids.c13
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_sock.c38
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_spill_fill.c40
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_subprog_insn_stats.c223
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_subprog_precision.c63
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_uninit.c1
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_unpriv.c18
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_var_off.c6
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c35
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_zext.c26
-rw-r--r--tools/testing/selftests/bpf/progs/veristat_bar.c3
-rw-r--r--tools/testing/selftests/bpf/progs/veristat_foo.c31
-rw-r--r--tools/testing/selftests/bpf/test_kmods/bpf_testmod.c15
-rw-r--r--tools/testing/selftests/bpf/test_kmods/bpf_testmod.h8
-rw-r--r--tools/testing/selftests/bpf/test_kmods/bpf_testmod_kfunc.h1
-rw-r--r--tools/testing/selftests/bpf/test_progs.c64
-rw-r--r--tools/testing/selftests/bpf/test_progs.h3
-rw-r--r--tools/testing/selftests/bpf/test_verifier.c2
-rw-r--r--tools/testing/selftests/bpf/usdt.h2
-rw-r--r--tools/testing/selftests/bpf/usdt_2.c15
-rw-r--r--tools/testing/selftests/bpf/verifier/calls.c12
-rw-r--r--tools/testing/selftests/bpf/verifier/pseudo_func.c45
-rw-r--r--tools/testing/selftests/bpf/veristat.c99
-rwxr-xr-xtools/testing/selftests/bpf/vmtest.sh13
-rw-r--r--tools/testing/selftests/cgroup/lib/include/cgroup_util.h2
-rw-r--r--tools/testing/selftests/cgroup/test_core.c14
-rw-r--r--tools/testing/selftests/cgroup/test_cpu.c47
-rw-r--r--tools/testing/selftests/cgroup/test_cpuset.c249
-rwxr-xr-xtools/testing/selftests/cgroup/test_cpuset_prs.sh88
-rwxr-xr-xtools/testing/selftests/cgroup/test_cpuset_v1_hp.sh2
-rw-r--r--tools/testing/selftests/cgroup/test_freezer.c2
-rw-r--r--tools/testing/selftests/cgroup/test_hugetlb_memcg.c10
-rw-r--r--tools/testing/selftests/cgroup/test_kill.c57
-rw-r--r--tools/testing/selftests/cgroup/test_kmem.c4
-rw-r--r--tools/testing/selftests/cgroup/test_memcontrol.c4
-rw-r--r--tools/testing/selftests/cgroup/test_pids.c2
-rw-r--r--tools/testing/selftests/cgroup/test_zswap.c43
-rw-r--r--tools/testing/selftests/clock-helpers.h76
-rwxr-xr-xtools/testing/selftests/cpufreq/governor.sh26
-rwxr-xr-xtools/testing/selftests/cpufreq/special-tests.sh18
-rw-r--r--tools/testing/selftests/damon/Makefile1
-rw-r--r--tools/testing/selftests/damon/_damon_sysfs.py23
-rwxr-xr-xtools/testing/selftests/damon/damos_apply_interval.py2
-rwxr-xr-xtools/testing/selftests/damon/damos_quota_goal.py2
-rwxr-xr-xtools/testing/selftests/damon/damos_tried_regions.py4
-rwxr-xr-xtools/testing/selftests/damon/drgn_dump_damon_status.py3
-rwxr-xr-xtools/testing/selftests/damon/sysfs.py35
-rwxr-xr-xtools/testing/selftests/damon/sysfs.sh79
-rwxr-xr-xtools/testing/selftests/damon/sysfs_refresh.py75
-rwxr-xr-xtools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py2
-rw-r--r--tools/testing/selftests/dax/Makefile6
-rw-r--r--tools/testing/selftests/dax/config4
-rwxr-xr-xtools/testing/selftests/dax/dax-kmem-hotplug.sh317
-rw-r--r--tools/testing/selftests/dax/settings1
-rw-r--r--tools/testing/selftests/dmabuf-heaps/dmabuf-heap.c113
-rw-r--r--tools/testing/selftests/drivers/net/bonding/bond_topo_2d1c.sh10
-rw-r--r--tools/testing/selftests/drivers/net/config6
-rw-r--r--tools/testing/selftests/drivers/net/hw/Makefile6
-rw-r--r--tools/testing/selftests/drivers/net/hw/config2
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/csum.py2
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/devlink_rate_cross_esw.py296
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/devmem.py11
-rw-r--r--tools/testing/selftests/drivers/net/hw/devmem_lib.py114
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/ethtool_rmon.sh2
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/ethtool_std_stats.sh6
-rw-r--r--tools/testing/selftests/drivers/net/hw/ncdevmem.c38
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/nk_devmem.py10
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/rss_ctx.py11
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/toeplitz.py22
-rwxr-xr-xtools/testing/selftests/drivers/net/hw/tso.py2
-rw-r--r--tools/testing/selftests/drivers/net/lib/py/env.py59
-rw-r--r--tools/testing/selftests/drivers/net/netconsole/Makefile1
-rwxr-xr-xtools/testing/selftests/drivers/net/netconsole/netcons_userdata.sh229
-rwxr-xr-xtools/testing/selftests/drivers/net/netdevsim/ethtool-coalesce.sh10
-rwxr-xr-xtools/testing/selftests/drivers/net/psp.py46
-rwxr-xr-xtools/testing/selftests/drivers/net/ring_reconfig.py255
-rwxr-xr-xtools/testing/selftests/drivers/net/shaper.py1065
-rwxr-xr-xtools/testing/selftests/drivers/net/so_txtime.py21
-rwxr-xr-xtools/testing/selftests/drivers/net/xdp.py94
-rw-r--r--tools/testing/selftests/drivers/ntsync/ntsync.c133
-rw-r--r--tools/testing/selftests/exec/.gitignore11
-rw-r--r--tools/testing/selftests/exec/Makefile93
-rw-r--r--tools/testing/selftests/exec/binfmt_bind_interp.c14
-rw-r--r--tools/testing/selftests/exec/binfmt_bpf_app.c12
-rw-r--r--tools/testing/selftests/exec/binfmt_bpf_interp.c15
-rw-r--r--tools/testing/selftests/exec/binfmt_loader_payload.c146
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_bpf.c638
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_common.h315
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_disabled.c172
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_interplimit.c232
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_loader.c372
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_selfpin.c158
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_transparent.c95
-rw-r--r--tools/testing/selftests/exec/binfmt_transparent_interp.c112
-rw-r--r--tools/testing/selftests/exec/bpf_interp.bpf.c61
-rw-r--r--tools/testing/selftests/exec/config10
-rw-r--r--tools/testing/selftests/exec/interp_bind.bpf.c76
-rw-r--r--tools/testing/selftests/exec/loader.bpf.c56
-rw-r--r--tools/testing/selftests/exec/nix_origin.bpf.c224
-rw-r--r--tools/testing/selftests/exec/transparent.bpf.c57
-rw-r--r--tools/testing/selftests/filesystems/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c23
-rw-r--r--tools/testing/selftests/filesystems/failfs/.gitignore2
-rw-r--r--tools/testing/selftests/filesystems/failfs/Makefile5
-rw-r--r--tools/testing/selftests/filesystems/failfs/failfs_test.c585
-rw-r--r--tools/testing/selftests/filesystems/fscontext_ns/Makefile10
-rw-r--r--tools/testing/selftests/filesystems/fscontext_ns/fscontext_ns_test.c239
-rw-r--r--tools/testing/selftests/filesystems/fuse/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/fuse/Makefile3
-rw-r--r--tools/testing/selftests/filesystems/fuse/write_extend_eof_test.c368
-rw-r--r--tools/testing/selftests/filesystems/mntns_cleanup/.gitignore2
-rw-r--r--tools/testing/selftests/filesystems/mntns_cleanup/Makefile6
-rw-r--r--tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c58
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c501
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c16
-rw-r--r--tools/testing/selftests/filesystems/statmount/statmount_test.c5
-rw-r--r--tools/testing/selftests/filesystems/ustat_test.c135
-rwxr-xr-xtools/testing/selftests/ftrace/ftracetest3
-rw-r--r--tools/testing/selftests/ftrace/poll.c2
-rw-r--r--tools/testing/selftests/ftrace/test.d/dynevent/add_remove_uprobe.tc27
-rw-r--r--tools/testing/selftests/ftrace/test.d/dynevent/btf_probe_event.tc51
-rw-r--r--tools/testing/selftests/ftrace/test.d/dynevent/btf_typecast_accepted.tc103
-rw-r--r--tools/testing/selftests/ftrace/test.d/dynevent/eprobes_syntax_errors.tc12
-rw-r--r--tools/testing/selftests/ftrace/test.d/dynevent/fprobe_syntax_errors.tc16
-rw-r--r--tools/testing/selftests/ftrace/test.d/dynevent/tprobe_syntax_errors.tc4
-rw-r--r--tools/testing/selftests/ftrace/test.d/kprobe/kprobe_module.tc6
-rw-r--r--tools/testing/selftests/ftrace/test.d/kprobe/kprobe_syntax_errors.tc16
-rw-r--r--tools/testing/selftests/ftrace/test.d/kprobe/uprobe_syntax_errors.tc5
-rw-r--r--tools/testing/selftests/futex/functional/.gitignore1
-rw-r--r--tools/testing/selftests/futex/functional/Makefile6
-rw-r--r--tools/testing/selftests/futex/functional/futex_lock_pi_exiting.c263
-rw-r--r--tools/testing/selftests/futex/functional/futex_numa_mpol.c127
-rw-r--r--tools/testing/selftests/futex/functional/futex_priv_hash.c163
-rw-r--r--tools/testing/selftests/futex/functional/futex_requeue.c67
-rw-r--r--tools/testing/selftests/futex/functional/futex_requeue_pi.c157
-rw-r--r--tools/testing/selftests/futex/functional/futex_requeue_pi_mismatched_ops.c41
-rw-r--r--tools/testing/selftests/futex/functional/futex_requeue_pi_signal_restart.c74
-rw-r--r--tools/testing/selftests/futex/functional/futex_wait.c127
-rw-r--r--tools/testing/selftests/futex/functional/futex_wait_private_mapped_file.c36
-rw-r--r--tools/testing/selftests/futex/functional/futex_wait_timeout.c108
-rw-r--r--tools/testing/selftests/futex/functional/futex_wait_uninitialized_heap.c35
-rw-r--r--tools/testing/selftests/futex/functional/futex_wait_wouldblock.c34
-rw-r--r--tools/testing/selftests/futex/functional/futex_waitv.c146
-rw-r--r--tools/testing/selftests/futex/functional/robust_list.c183
-rwxr-xr-xtools/testing/selftests/futex/functional/run.sh3
-rw-r--r--tools/testing/selftests/futex/include/futex2test.h10
-rw-r--r--tools/testing/selftests/futex/include/futex_thread.h117
-rw-r--r--tools/testing/selftests/hid/hid_bpf.c142
-rw-r--r--tools/testing/selftests/hid/hid_common.h25
-rw-r--r--tools/testing/selftests/hid/progs/hid.c28
-rw-r--r--tools/testing/selftests/hid/progs/hid_bpf_helpers.h3
-rw-r--r--tools/testing/selftests/ipc/msgque.c2
-rw-r--r--tools/testing/selftests/kselftest_harness.h2
-rw-r--r--tools/testing/selftests/kvm/Makefile.kvm19
-rw-r--r--tools/testing/selftests/kvm/arch_timer.c20
-rw-r--r--tools/testing/selftests/kvm/arm64/arch_timer_edge_cases.c2
-rw-r--r--tools/testing/selftests/kvm/arm64/debug-exceptions.c41
-rw-r--r--tools/testing/selftests/kvm/arm64/get-reg-list.c3
-rw-r--r--tools/testing/selftests/kvm/arm64/stage2_block_transitions.c226
-rw-r--r--tools/testing/selftests/kvm/arm64/vgic_irq.c14
-rw-r--r--tools/testing/selftests/kvm/arm64/vgic_lpi_stress.c8
-rw-r--r--tools/testing/selftests/kvm/demand_paging_test.c2
-rw-r--r--tools/testing/selftests/kvm/dirty_log_perf_test.c4
-rw-r--r--tools/testing/selftests/kvm/dirty_log_test.c15
-rw-r--r--tools/testing/selftests/kvm/hardware_disable_test.c109
-rw-r--r--tools/testing/selftests/kvm/include/kvm_syscalls.h55
-rw-r--r--tools/testing/selftests/kvm/include/kvm_util.h21
-rw-r--r--tools/testing/selftests/kvm/include/loongarch/processor.h18
-rw-r--r--tools/testing/selftests/kvm/include/proc_util.h11
-rw-r--r--tools/testing/selftests/kvm/include/test_util.h26
-rw-r--r--tools/testing/selftests/kvm/include/x86/evmcs.h46
-rw-r--r--tools/testing/selftests/kvm/include/x86/kvm_util_arch.h4
-rw-r--r--tools/testing/selftests/kvm/include/x86/pmu.h6
-rw-r--r--tools/testing/selftests/kvm/include/x86/processor.h64
-rw-r--r--tools/testing/selftests/kvm/include/x86/svm_util.h5
-rw-r--r--tools/testing/selftests/kvm/include/x86/vmx.h73
-rw-r--r--tools/testing/selftests/kvm/irq_test.c362
-rw-r--r--tools/testing/selftests/kvm/irqfd_test.c4
-rw-r--r--tools/testing/selftests/kvm/kvm_page_table_test.c6
-rw-r--r--tools/testing/selftests/kvm/lib/assert.c10
-rw-r--r--tools/testing/selftests/kvm/lib/kvm_util.c52
-rw-r--r--tools/testing/selftests/kvm/lib/loongarch/processor.c11
-rw-r--r--tools/testing/selftests/kvm/lib/memstress.c12
-rw-r--r--tools/testing/selftests/kvm/lib/proc_util.c54
-rw-r--r--tools/testing/selftests/kvm/lib/test_util.c41
-rw-r--r--tools/testing/selftests/kvm/lib/userfaultfd_util.c7
-rw-r--r--tools/testing/selftests/kvm/lib/x86/memstress.c19
-rw-r--r--tools/testing/selftests/kvm/lib/x86/processor.c59
-rw-r--r--tools/testing/selftests/kvm/lib/x86/svm.c68
-rw-r--r--tools/testing/selftests/kvm/lib/x86/ucall.c32
-rw-r--r--tools/testing/selftests/kvm/lib/x86/vmx.c8
-rw-r--r--tools/testing/selftests/kvm/loongarch/fpu_test.c137
-rw-r--r--tools/testing/selftests/kvm/memslot_perf_test.c4
-rw-r--r--tools/testing/selftests/kvm/mmu_stress_test.c19
-rw-r--r--tools/testing/selftests/kvm/pre_fault_memory_test.c4
-rw-r--r--tools/testing/selftests/kvm/riscv/get-reg-list.c90
-rw-r--r--tools/testing/selftests/kvm/riscv/sbi_pmu_test.c4
-rw-r--r--tools/testing/selftests/kvm/rseq_test.c16
-rw-r--r--tools/testing/selftests/kvm/s390/memop.c4
-rw-r--r--tools/testing/selftests/kvm/set_memory_region_test.c6
-rw-r--r--tools/testing/selftests/kvm/steal_time.c39
-rw-r--r--tools/testing/selftests/kvm/vm_types_test.c48
-rw-r--r--tools/testing/selftests/kvm/x86/aperfmperf_test.c9
-rw-r--r--tools/testing/selftests/kvm/x86/evmcs_smm_controls_test.c10
-rw-r--r--tools/testing/selftests/kvm/x86/fix_hypercall_test.c1
-rw-r--r--tools/testing/selftests/kvm/x86/hyperv_clock.c4
-rw-r--r--tools/testing/selftests/kvm/x86/hyperv_evmcs.c8
-rw-r--r--tools/testing/selftests/kvm/x86/hyperv_ipi.c32
-rw-r--r--tools/testing/selftests/kvm/x86/hyperv_svm_test.c6
-rw-r--r--tools/testing/selftests/kvm/x86/hyperv_tlb_flush.c30
-rw-r--r--tools/testing/selftests/kvm/x86/kvm_buslock_test.c9
-rw-r--r--tools/testing/selftests/kvm/x86/nested_close_kvm_test.c12
-rw-r--r--tools/testing/selftests/kvm/x86/nested_dirty_log_test.c8
-rw-r--r--tools/testing/selftests/kvm/x86/nested_emulation_test.c4
-rw-r--r--tools/testing/selftests/kvm/x86/nested_exceptions_test.c9
-rw-r--r--tools/testing/selftests/kvm/x86/nested_invalid_cr3_test.c10
-rw-r--r--tools/testing/selftests/kvm/x86/nested_tdp_fault_test.c9
-rw-r--r--tools/testing/selftests/kvm/x86/nested_tsc_adjust_test.c10
-rw-r--r--tools/testing/selftests/kvm/x86/nested_tsc_scaling_test.c10
-rw-r--r--tools/testing/selftests/kvm/x86/nested_vmsave_vmload_test.c6
-rw-r--r--tools/testing/selftests/kvm/x86/pmu_counters_test.c231
-rw-r--r--tools/testing/selftests/kvm/x86/pmu_event_filter_test.c4
-rw-r--r--tools/testing/selftests/kvm/x86/private_mem_conversions_test.c4
-rw-r--r--tools/testing/selftests/kvm/x86/private_mem_kvm_exits_test.c7
-rw-r--r--tools/testing/selftests/kvm/x86/recalc_apic_map_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/save_restore_pf_stress_test.c288
-rw-r--r--tools/testing/selftests/kvm/x86/set_sregs_test.c83
-rw-r--r--tools/testing/selftests/kvm/x86/sev_dbg_test.c2
-rw-r--r--tools/testing/selftests/kvm/x86/sev_init2_tests.c4
-rw-r--r--tools/testing/selftests/kvm/x86/sev_migrate_tests.c4
-rw-r--r--tools/testing/selftests/kvm/x86/smm_test.c13
-rw-r--r--tools/testing/selftests/kvm/x86/state_test.c11
-rw-r--r--tools/testing/selftests/kvm/x86/svm_int_ctl_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/svm_lbr_nested_state.c6
-rw-r--r--tools/testing/selftests/kvm/x86/svm_nested_clear_efer_svme.c7
-rw-r--r--tools/testing/selftests/kvm/x86/svm_nested_pat_test.c8
-rw-r--r--tools/testing/selftests/kvm/x86/svm_nested_shutdown_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/svm_nested_soft_inject_test.c6
-rw-r--r--tools/testing/selftests/kvm/x86/svm_nested_vmcb12_gpa.c13
-rw-r--r--tools/testing/selftests/kvm/x86/svm_pmu_host_guest_test.c215
-rw-r--r--tools/testing/selftests/kvm/x86/svm_vmcall_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/sync_regs_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/triple_fault_event_test.c9
-rw-r--r--tools/testing/selftests/kvm/x86/tsc_scaling_sync.c4
-rw-r--r--tools/testing/selftests/kvm/x86/vmx_apic_access_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/vmx_apicv_updates_test.c4
-rw-r--r--tools/testing/selftests/kvm/x86/vmx_invalid_nested_guest_state.c126
-rw-r--r--tools/testing/selftests/kvm/x86/vmx_nested_la57_state_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/vmx_preemption_timer_test.c5
-rw-r--r--tools/testing/selftests/kvm/x86/xapic_ipi_test.c33
-rw-r--r--tools/testing/selftests/kvm/x86/xen_shinfo_test.c10
-rw-r--r--tools/testing/selftests/landlock/audit.h35
-rw-r--r--tools/testing/selftests/landlock/base_test.c104
-rw-r--r--tools/testing/selftests/landlock/common.h47
-rw-r--r--tools/testing/selftests/landlock/config2
-rw-r--r--tools/testing/selftests/landlock/fs_test.c738
-rw-r--r--tools/testing/selftests/landlock/net_test.c611
-rw-r--r--tools/testing/selftests/landlock/ptrace_test.c402
-rw-r--r--tools/testing/selftests/landlock/scoped_abstract_unix_test.c296
-rw-r--r--tools/testing/selftests/landlock/scoped_signal_test.c404
-rw-r--r--tools/testing/selftests/landlock/trace.h638
-rw-r--r--tools/testing/selftests/landlock/trace_fs_test.c732
-rw-r--r--tools/testing/selftests/landlock/trace_test.c1629
-rw-r--r--tools/testing/selftests/landlock/true.c10
-rw-r--r--tools/testing/selftests/landlock/tsync_test.c96
-rw-r--r--tools/testing/selftests/livepatch/functions.sh14
-rwxr-xr-xtools/testing/selftests/livepatch/test-ftrace.sh45
-rw-r--r--tools/testing/selftests/liveupdate/.gitignore2
-rw-r--r--tools/testing/selftests/liveupdate/Makefile14
-rw-r--r--tools/testing/selftests/liveupdate/config1
-rw-r--r--tools/testing/selftests/liveupdate/config.aarch642
-rw-r--r--tools/testing/selftests/liveupdate/config.x86_642
-rw-r--r--tools/testing/selftests/liveupdate/init.c179
-rw-r--r--tools/testing/selftests/liveupdate/lib/include/libliveupdate.h (renamed from tools/testing/selftests/liveupdate/luo_test_utils.h)11
-rw-r--r--tools/testing/selftests/liveupdate/lib/libliveupdate.mk20
-rw-r--r--tools/testing/selftests/liveupdate/lib/lu_utils.c (renamed from tools/testing/selftests/liveupdate/luo_test_utils.c)78
-rw-r--r--tools/testing/selftests/liveupdate/liveupdate.c106
-rw-r--r--tools/testing/selftests/liveupdate/luo_kexec_simple.c2
-rw-r--r--tools/testing/selftests/liveupdate/luo_multi_session.c2
-rw-r--r--tools/testing/selftests/liveupdate/luo_stress_files.c3
-rw-r--r--tools/testing/selftests/liveupdate/luo_stress_sessions.c3
-rwxr-xr-xtools/testing/selftests/liveupdate/run-vmtests.sh97
-rwxr-xr-xtools/testing/selftests/liveupdate/vmtest.sh263
-rw-r--r--tools/testing/selftests/lsm/common.c4
-rw-r--r--tools/testing/selftests/mincore/mincore_selftest.c12
-rw-r--r--tools/testing/selftests/mm/.gitignore76
-rw-r--r--tools/testing/selftests/mm/Makefile8
-rwxr-xr-xtools/testing/selftests/mm/check_config.sh2
-rw-r--r--tools/testing/selftests/mm/compaction_test.c38
-rw-r--r--tools/testing/selftests/mm/cow.c11
-rw-r--r--tools/testing/selftests/mm/folio_split_race_test.c2
-rw-r--r--tools/testing/selftests/mm/guard-regions.c10
-rw-r--r--tools/testing/selftests/mm/gup_longterm.c4
-rw-r--r--tools/testing/selftests/mm/hmm-tests.c792
-rw-r--r--tools/testing/selftests/mm/hugepage_settings.c2
-rw-r--r--tools/testing/selftests/mm/hugetlb-vmemmap.c132
-rw-r--r--tools/testing/selftests/mm/khugepaged.c204
-rwxr-xr-xtools/testing/selftests/mm/ksft_kmemleak_confirm.sh130
-rw-r--r--tools/testing/selftests/mm/ksm_tests.c28
-rw-r--r--tools/testing/selftests/mm/memfd_secret.c30
-rw-r--r--tools/testing/selftests/mm/memory-failure.c58
-rw-r--r--tools/testing/selftests/mm/merge.c57
-rw-r--r--tools/testing/selftests/mm/migration.c37
-rw-r--r--tools/testing/selftests/mm/mremap_dontunmap.c2
-rw-r--r--tools/testing/selftests/mm/mseal_test.c109
-rw-r--r--tools/testing/selftests/mm/pagemap_ioctl.c116
-rw-r--r--tools/testing/selftests/mm/pkey-helpers.h4
-rw-r--r--tools/testing/selftests/mm/pkey-powerpc.h2
-rw-r--r--tools/testing/selftests/mm/pkey_sighandler_tests.c115
-rw-r--r--tools/testing/selftests/mm/pkey_util.c90
-rw-r--r--tools/testing/selftests/mm/prctl_thp_disable.c2
-rw-r--r--tools/testing/selftests/mm/protection_keys.c99
-rw-r--r--tools/testing/selftests/mm/rmap.c81
-rwxr-xr-xtools/testing/selftests/mm/run_vmtests.sh5
-rw-r--r--tools/testing/selftests/mm/soft-dirty.c5
-rw-r--r--tools/testing/selftests/mm/split_huge_page_test.c139
-rw-r--r--tools/testing/selftests/mm/thuge-gen.c15
-rw-r--r--tools/testing/selftests/mm/uffd-common.c13
-rw-r--r--tools/testing/selftests/mm/uffd-unit-tests.c919
-rw-r--r--tools/testing/selftests/mm/vm_util.c233
-rw-r--r--tools/testing/selftests/mm/vm_util.h9
-rw-r--r--tools/testing/selftests/namespaces/nsid_test.c4
-rw-r--r--tools/testing/selftests/nci/nci_dev.c45
-rw-r--r--tools/testing/selftests/net/Makefile7
-rw-r--r--tools/testing/selftests/net/af_unix/.gitignore2
-rw-r--r--tools/testing/selftests/net/af_unix/Makefile8
-rw-r--r--tools/testing/selftests/net/af_unix/config7
-rw-r--r--tools/testing/selftests/net/af_unix/msg_oob.c67
-rw-r--r--tools/testing/selftests/net/af_unix/scm_rights.c17
-rw-r--r--tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.bpf.c36
-rw-r--r--tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.c292
-rwxr-xr-xtools/testing/selftests/net/big_tcp_tunnels.sh218
-rw-r--r--tools/testing/selftests/net/config8
-rwxr-xr-xtools/testing/selftests/net/cork_fragsize.py187
-rwxr-xr-xtools/testing/selftests/net/exception_cache.sh521
-rwxr-xr-xtools/testing/selftests/net/fib_nexthops.sh111
-rw-r--r--tools/testing/selftests/net/fin_ack_lat.c7
-rwxr-xr-xtools/testing/selftests/net/fin_ack_lat.sh2
-rw-r--r--tools/testing/selftests/net/forwarding/lib.sh15
-rwxr-xr-xtools/testing/selftests/net/fou_mcast_encap.sh172
-rw-r--r--tools/testing/selftests/net/getsockopt_iter.c336
-rw-r--r--tools/testing/selftests/net/hsr/Makefile1
-rwxr-xr-xtools/testing/selftests/net/hsr/hsr_prp_redbox.sh99
-rwxr-xr-xtools/testing/selftests/net/ipv6_flowlabel.sh3
-rw-r--r--tools/testing/selftests/net/ipv6_flowlabel_mgr.c584
-rw-r--r--tools/testing/selftests/net/lib/py/__init__.py4
-rw-r--r--tools/testing/selftests/net/lib/py/ynl.py7
-rw-r--r--tools/testing/selftests/net/mptcp/mptcp_connect.c8
-rwxr-xr-xtools/testing/selftests/net/mptcp/mptcp_connect.sh8
-rw-r--r--tools/testing/selftests/net/mptcp/mptcp_diag.c3
-rwxr-xr-xtools/testing/selftests/net/mptcp/mptcp_join.sh95
-rw-r--r--tools/testing/selftests/net/mptcp/mptcp_lib.sh26
-rwxr-xr-xtools/testing/selftests/net/mptcp/simult_flows.sh11
-rw-r--r--tools/testing/selftests/net/netfilter/conntrack_dump_flush.c31
-rwxr-xr-xtools/testing/selftests/net/nl_nlctrl.py116
-rw-r--r--tools/testing/selftests/net/openvswitch/config4
-rwxr-xr-xtools/testing/selftests/net/openvswitch/openvswitch.sh296
-rw-r--r--tools/testing/selftests/net/openvswitch/ovs-dpctl.py124
-rw-r--r--tools/testing/selftests/net/ovpn/common.sh13
-rw-r--r--tools/testing/selftests/net/ovpn/ovpn-cli.c54
-rwxr-xr-xtools/testing/selftests/net/ovpn/test.sh75
-rw-r--r--tools/testing/selftests/net/packetdrill/config2
-rw-r--r--tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv4.pkt67
-rw-r--r--tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv6.pkt67
-rw-r--r--tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt29
-rw-r--r--tools/testing/selftests/net/packetdrill/tcp_urg_ptr_retransmit.pkt65
-rwxr-xr-xtools/testing/selftests/net/pmtu.sh2
-rw-r--r--tools/testing/selftests/net/psock_snd.c237
-rwxr-xr-xtools/testing/selftests/net/psock_snd.sh15
-rw-r--r--tools/testing/selftests/net/reuseport_bpf_numa.c24
-rwxr-xr-xtools/testing/selftests/net/rtnetlink.py286
-rwxr-xr-xtools/testing/selftests/net/srv6_encap_lookup_l3vpn_test.sh1027
-rwxr-xr-xtools/testing/selftests/net/srv6_end_dt46_l3vpn_test.sh10
-rw-r--r--tools/testing/selftests/net/tcp_ao/key-management.c67
-rw-r--r--tools/testing/selftests/net/tcp_ao/lib/aolib.h6
-rw-r--r--tools/testing/selftests/net/tcp_ao/lib/sock.c4
-rw-r--r--tools/testing/selftests/net/tcp_mmap.c10
-rwxr-xr-xtools/testing/selftests/net/test_vxlan_mdb.sh6
-rwxr-xr-xtools/testing/selftests/net/test_vxlan_nh.sh40
-rw-r--r--tools/testing/selftests/net/tls.c75
-rw-r--r--tools/testing/selftests/nolibc/Makefile.include2
-rw-r--r--tools/testing/selftests/nolibc/Makefile.nolibc2
-rw-r--r--tools/testing/selftests/nolibc/nolibc-test.c57
-rwxr-xr-xtools/testing/selftests/nolibc/run-tests.sh3
-rw-r--r--tools/testing/selftests/power_supply/helpers.sh8
-rwxr-xr-xtools/testing/selftests/power_supply/test_power_supply_properties.sh3
-rw-r--r--tools/testing/selftests/powerpc/pmu/ebb/lost_exception_test.c11
-rw-r--r--tools/testing/selftests/powerpc/tm/tm-signal-context-force-tm.c4
-rw-r--r--tools/testing/selftests/powerpc/tm/tm.h8
-rw-r--r--tools/testing/selftests/prctl/set-anon-vma-name-test.c21
-rw-r--r--tools/testing/selftests/proc/proc-maps-race.c3
-rw-r--r--tools/testing/selftests/proc/proc-pidns.c1
-rw-r--r--tools/testing/selftests/rdma/Makefile3
-rw-r--r--tools/testing/selftests/rdma/config2
-rwxr-xr-xtools/testing/selftests/rdma/rxe_netns_names.sh334
-rw-r--r--tools/testing/selftests/riscv/cfi/cfi_rv_test.h2
-rw-r--r--tools/testing/selftests/riscv/hwprobe/hwprobe.c20
-rw-r--r--tools/testing/selftests/rseq/.gitignore2
-rw-r--r--tools/testing/selftests/rseq/rseq-x86-thread-pointer.h4
-rw-r--r--tools/testing/selftests/rseq/rseq.c2
-rw-r--r--tools/testing/selftests/syscall_user_dispatch/config2
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json115
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/actions/ct.json40
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/actions/gact-rollback.json78
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/filters/route.json210
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/filters/u32.json71
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/infra/qdiscs.json164
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json72
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/qdiscs/ets.json4
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json72
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json34
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json128
-rw-r--r--tools/testing/selftests/tc-testing/tc-tests/qdiscs/pfifo_fast.json204
-rw-r--r--tools/testing/selftests/timers/Makefile2
-rw-r--r--tools/testing/selftests/timers/adjtick.c2
-rw-r--r--tools/testing/selftests/timers/alarmtimer-suspend.c37
-rw-r--r--tools/testing/selftests/timers/inconsistency-check.c37
-rw-r--r--tools/testing/selftests/timers/leap-a-day.c21
-rw-r--r--tools/testing/selftests/timers/mqueue-lat.c2
-rw-r--r--tools/testing/selftests/timers/nanosleep.c144
-rw-r--r--tools/testing/selftests/timers/nsleep-lat.c101
-rw-r--r--tools/testing/selftests/timers/posix_timers.c34
-rw-r--r--tools/testing/selftests/timers/raw_skew.c2
-rw-r--r--tools/testing/selftests/timers/set-2038.c2
-rw-r--r--tools/testing/selftests/timers/set-timer-lat.c40
-rw-r--r--tools/testing/selftests/timers/valid-adjtimex.c2
-rw-r--r--tools/testing/selftests/ublk/Makefile6
-rw-r--r--tools/testing/selftests/ublk/batch.c2
-rw-r--r--tools/testing/selftests/ublk/kublk.c210
-rw-r--r--tools/testing/selftests/ublk/kublk.h43
-rwxr-xr-xtools/testing/selftests/ublk/test_batch_04.sh44
-rwxr-xr-xtools/testing/selftests/ublk/test_loop_08.sh25
-rwxr-xr-xtools/testing/selftests/ublk/test_params_01.sh114
-rwxr-xr-xtools/testing/selftests/ublk/test_recover_03.sh5
-rw-r--r--tools/testing/selftests/verification/test.d/rv_deadline.tc23
-rw-r--r--tools/testing/selftests/verification/test.d/rv_monitor_enable_disable.tc10
-rw-r--r--tools/testing/selftests/verification/test.d/rv_monitor_reactor.tc4
-rw-r--r--tools/testing/selftests/verification/test.d/rv_stall.tc33
-rw-r--r--tools/testing/selftests/verification/test.d/rv_wwnr_printk.tc32
l---------tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h1
l---------tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h1
l---------tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h1
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/igb/igb.c585
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h352
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c783
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/assert.h18
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h29
-rw-r--r--tools/testing/selftests/vfio/lib/iommu.c12
-rw-r--r--tools/testing/selftests/vfio/lib/iova_allocator.c4
-rw-r--r--tools/testing/selftests/vfio/lib/libvfio.mk3
-rw-r--r--tools/testing/selftests/vfio/lib/sysfs.c3
-rw-r--r--tools/testing/selftests/vfio/lib/vfio_pci_device.c74
-rw-r--r--tools/testing/selftests/vfio/lib/vfio_pci_driver.c22
-rw-r--r--tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c4
-rw-r--r--tools/testing/selftests/vfio/vfio_pci_driver_test.c57
-rw-r--r--tools/testing/selftests/wireguard/qemu/kernel.config2
-rw-r--r--tools/testing/selftests/x86/Makefile2
-rw-r--r--tools/testing/selftests/x86/int_signal.c311
-rwxr-xr-xtools/testing/selftests/zram/zram_lib.sh2
595 files changed, 39186 insertions, 4327 deletions
diff --git a/tools/testing/selftests/Makefile b/tools/testing/selftests/Makefile
index 8d4db2241cc2..2d960626750e 100644
--- a/tools/testing/selftests/Makefile
+++ b/tools/testing/selftests/Makefile
@@ -1,5 +1,6 @@
# SPDX-License-Identifier: GPL-2.0
TARGETS += acct
+TARGETS += alloc_tag
TARGETS += alsa
TARGETS += amd-pstate
TARGETS += arm64
@@ -14,6 +15,7 @@ TARGETS += core
TARGETS += cpufreq
TARGETS += cpu-hotplug
TARGETS += damon
+TARGETS += dax
TARGETS += devices/error_logs
TARGETS += devices/probe
TARGETS += dmabuf-heaps
@@ -33,6 +35,7 @@ TARGETS += fchmodat2
TARGETS += filesystems
TARGETS += filesystems/binderfs
TARGETS += filesystems/epoll
+TARGETS += filesystems/failfs
TARGETS += filesystems/fat
TARGETS += filesystems/overlayfs
TARGETS += filesystems/statmount
@@ -42,6 +45,8 @@ TARGETS += filesystems/fuse
TARGETS += filesystems/move_mount
TARGETS += filesystems/empty_mntns
TARGETS += filesystems/fsmount_ns
+TARGETS += filesystems/fscontext_ns
+TARGETS += filesystems/mntns_cleanup
TARGETS += firmware
TARGETS += fpu
TARGETS += ftrace
diff --git a/tools/testing/selftests/acct/.gitignore b/tools/testing/selftests/acct/.gitignore
index 9e9c61c5bfd6..fe0896f54e15 100644
--- a/tools/testing/selftests/acct/.gitignore
+++ b/tools/testing/selftests/acct/.gitignore
@@ -1,4 +1,5 @@
acct_syscall
taskstats_fill_stats_tgid
+cgroupstats
config
process_log
diff --git a/tools/testing/selftests/acct/Makefile b/tools/testing/selftests/acct/Makefile
index 083cab5ddb72..93a11a28a636 100644
--- a/tools/testing/selftests/acct/Makefile
+++ b/tools/testing/selftests/acct/Makefile
@@ -1,8 +1,19 @@
# SPDX-License-Identifier: GPL-2.0
TEST_GEN_PROGS := acct_syscall
TEST_GEN_PROGS += taskstats_fill_stats_tgid
+TEST_GEN_PROGS += cgroupstats
+
+NETLINK_HELPER_PROGS := cgroupstats taskstats_fill_stats_tgid
CFLAGS += -Wall
LDLIBS += -lpthread
include ../lib.mk
+
+$(NETLINK_HELPER_PROGS): %: %.c netlink_helper.c netlink_helper.h
+ $(call msg,CC,,$@)
+ $(Q)$(LINK.c) $< netlink_helper.c $(LDLIBS) -o $@
+
+$(addprefix $(OUTPUT)/,$(NETLINK_HELPER_PROGS)): $(OUTPUT)/%: %.c netlink_helper.c netlink_helper.h
+ $(call msg,CC,,$@)
+ $(Q)$(LINK.c) $< netlink_helper.c $(LDLIBS) -o $@
diff --git a/tools/testing/selftests/acct/cgroupstats.c b/tools/testing/selftests/acct/cgroupstats.c
new file mode 100644
index 000000000000..0b421a4ca72b
--- /dev/null
+++ b/tools/testing/selftests/acct/cgroupstats.c
@@ -0,0 +1,231 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+
+#include <errno.h>
+#include <fcntl.h>
+#include <linux/cgroupstats.h>
+#include <linux/genetlink.h>
+#include <linux/netlink.h>
+#include <stdbool.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/socket.h>
+#include <sys/types.h>
+#include <unistd.h>
+
+#include "netlink_helper.h"
+#include "kselftest.h"
+
+static int send_cgroupstats_cmd(int fd, int family_id, uint32_t cgroup_fd,
+ int flags)
+{
+ struct {
+ struct nlmsghdr nlh;
+ struct genlmsghdr genl;
+ char buf[256];
+ } req = { 0 };
+ struct nlattr *na;
+
+ req.nlh.nlmsg_len = NLMSG_LENGTH(GENL_HDRLEN);
+ req.nlh.nlmsg_type = family_id;
+ req.nlh.nlmsg_flags = NLM_F_REQUEST | flags;
+ req.nlh.nlmsg_seq = 2;
+ req.nlh.nlmsg_pid = getpid();
+
+ req.genl.cmd = CGROUPSTATS_CMD_GET;
+ req.genl.version = 1;
+
+ na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len));
+ na->nla_type = CGROUPSTATS_CMD_ATTR_FD;
+ na->nla_len = NLA_HDRLEN + sizeof(cgroup_fd);
+ memcpy(nla_data(na), &cgroup_fd, sizeof(cgroup_fd));
+ req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len);
+
+ return send_request(fd, &req, req.nlh.nlmsg_len);
+}
+
+/*
+ * Receive and decode a cgroupstats response.
+ *
+ * Returns:
+ * 0 — success, stats filled from CGROUPSTATS_CMD_NEW reply
+ * <0 — NLMSG_ERROR errno (e.g. -EBADF, -EINVAL)
+ */
+static int recv_cgroupstats_response(int fd, struct cgroupstats *stats)
+{
+ char resp[8192];
+ struct nlmsghdr *nlh;
+ struct genlmsghdr *genl;
+ struct nlattr *na;
+ int len;
+ int rem;
+
+ memset(stats, 0, sizeof(*stats));
+
+ len = recv(fd, resp, sizeof(resp), 0);
+ if (len < 0)
+ return -errno;
+
+ for (nlh = (struct nlmsghdr *)resp; NLMSG_OK(nlh, len);
+ nlh = NLMSG_NEXT(nlh, len)) {
+ if (nlh->nlmsg_type == NLMSG_ERROR) {
+ struct nlmsgerr *err = NLMSG_DATA(nlh);
+
+ return err->error;
+ }
+
+ genl = (struct genlmsghdr *)NLMSG_DATA(nlh);
+ if (genl->cmd != CGROUPSTATS_CMD_NEW)
+ continue;
+
+ rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN;
+ na = (struct nlattr *)((char *)genl + GENL_HDRLEN);
+ while (nla_ok(na, rem)) {
+ if (na->nla_type == CGROUPSTATS_TYPE_CGROUP_STATS) {
+ memcpy(stats, nla_data(na), sizeof(*stats));
+ return 0;
+ }
+ na = nla_next(na, &rem);
+ }
+ }
+
+ return -EIO;
+}
+
+/* mkdtemp() modifies the template in place, so this cannot be const. */
+static char cg_mountpoint[32];
+static bool cg_mounted;
+
+static int setup_cgroup_v1(void)
+{
+ strcpy(cg_mountpoint, "/tmp/cgstats_test_XXXXXX");
+
+ if (!mkdtemp(cg_mountpoint))
+ return -errno;
+
+ if (mount("cgstats_test", cg_mountpoint, "cgroup", 0,
+ "none,name=cgstats_test") < 0) {
+ int ret = -errno;
+
+ rmdir(cg_mountpoint);
+ return ret;
+ }
+
+ cg_mounted = true;
+ return 0;
+}
+
+static void cleanup_cgroup_v1(void)
+{
+ if (!cg_mounted)
+ return;
+ umount2(cg_mountpoint, MNT_DETACH);
+ rmdir(cg_mountpoint);
+ cg_mounted = false;
+}
+
+int main(void)
+{
+ struct cgroupstats stats;
+ uint64_t total_tasks;
+ int family_id;
+ int nl_fd;
+ int cg_fd;
+ int ret;
+
+ ksft_print_header();
+
+ nl_fd = netlink_open();
+ if (nl_fd < 0)
+ ksft_exit_skip("failed to open generic netlink socket: %s\n",
+ strerror(-nl_fd));
+
+ family_id = get_family_id(nl_fd, TASKSTATS_GENL_NAME);
+ if (family_id < 0)
+ ksft_exit_skip("taskstats generic netlink family unavailable: %s\n",
+ strerror(-family_id));
+
+ ksft_set_plan(3);
+
+ /*
+ * Test 1: mount a private cgroup v1 hierarchy, query it, and
+ * verify the response contains sane task counts. If the test
+ * environment cannot create a private cgroup v1 mount, skip this
+ * case and continue with the unprivileged regression checks below.
+ */
+ ret = setup_cgroup_v1();
+ if (ret) {
+ ksft_test_result_skip("cgroupstats query: cannot mount cgroup v1: %s\n",
+ strerror(-ret));
+ } else {
+ cg_fd = open(cg_mountpoint, O_RDONLY | O_DIRECTORY);
+ if (cg_fd < 0) {
+ ksft_test_result_fail("cgroupstats query: open mountpoint: %s\n",
+ strerror(errno));
+ } else {
+ ret = send_cgroupstats_cmd(nl_fd, family_id,
+ (uint32_t)cg_fd, 0);
+ if (ret) {
+ ksft_test_result_fail("cgroupstats query: send: %s\n",
+ strerror(-ret));
+ } else {
+ ret = recv_cgroupstats_response(nl_fd, &stats);
+ if (ret < 0) {
+ ksft_test_result_fail("cgroupstats query: %s\n",
+ strerror(-ret));
+ } else {
+ total_tasks = (uint64_t)stats.nr_sleeping +
+ (uint64_t)stats.nr_running +
+ (uint64_t)stats.nr_stopped +
+ (uint64_t)stats.nr_uninterruptible +
+ (uint64_t)stats.nr_io_wait;
+
+ ksft_print_msg("cgroupstats query: total_tasks=%llu\n",
+ (unsigned long long)total_tasks);
+
+ ksft_test_result(total_tasks > 0,
+ "cgroupstats query returns valid stats\n");
+ }
+ }
+ close(cg_fd);
+ }
+ }
+ cleanup_cgroup_v1();
+
+ /*
+ * Test 2: invalid fd without NLM_F_ACK. The kernel should
+ * return -EBADF via NLMSG_ERROR regardless of whether the
+ * client requested an explicit ACK.
+ */
+ ret = send_cgroupstats_cmd(nl_fd, family_id, 0xFFFFFFFF, 0);
+ if (ret)
+ ksft_exit_fail_msg("send test 2 failed: %s\n", strerror(-ret));
+
+ ret = recv_cgroupstats_response(nl_fd, &stats);
+ ksft_print_msg("bad fd (no ACK): response=%d (%s)\n",
+ ret, ret < 0 ? strerror(-ret) : "unexpected success");
+ ksft_test_result(ret == -EBADF,
+ "cgroupstats rejects bad fd without NLM_F_ACK\n");
+
+ /*
+ * Test 3: invalid fd with NLM_F_ACK. Same expectation as
+ * test 2, but exercised through a different netlink flag
+ * path in the kernel's ack/error handling.
+ */
+ ret = send_cgroupstats_cmd(nl_fd, family_id, 0xFFFFFFFF, NLM_F_ACK);
+ if (ret)
+ ksft_exit_fail_msg("send test 3 failed: %s\n", strerror(-ret));
+
+ ret = recv_cgroupstats_response(nl_fd, &stats);
+ ksft_print_msg("bad fd (with ACK): response=%d (%s)\n",
+ ret, ret < 0 ? strerror(-ret) : "unexpected success");
+ ksft_test_result(ret == -EBADF,
+ "cgroupstats rejects bad fd with NLM_F_ACK\n");
+
+ close(nl_fd);
+ ksft_finished();
+ return ksft_get_fail_cnt() ? KSFT_FAIL : KSFT_PASS;
+}
diff --git a/tools/testing/selftests/acct/netlink_helper.c b/tools/testing/selftests/acct/netlink_helper.c
new file mode 100644
index 000000000000..3ed834f0e770
--- /dev/null
+++ b/tools/testing/selftests/acct/netlink_helper.c
@@ -0,0 +1,116 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <errno.h>
+#include <stdint.h>
+#include <string.h>
+#include <sys/socket.h>
+#include <sys/time.h>
+#include <unistd.h>
+#include <linux/genetlink.h>
+
+#include "netlink_helper.h"
+
+int netlink_open(void)
+{
+ struct timeval tv = { .tv_sec = ACCT_RCV_TIMEOUT_SEC };
+ struct sockaddr_nl addr = {
+ .nl_family = AF_NETLINK,
+ .nl_pid = getpid(),
+ };
+ int fd;
+
+ fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_GENERIC);
+ if (fd < 0)
+ return -errno;
+
+ if (setsockopt(fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)) < 0) {
+ int err = -errno;
+
+ close(fd);
+ return err;
+ }
+
+ if (bind(fd, (struct sockaddr *)&addr, sizeof(addr)) < 0) {
+ int err = -errno;
+
+ close(fd);
+ return err;
+ }
+
+ return fd;
+}
+
+int send_request(int fd, void *buf, size_t len)
+{
+ struct sockaddr_nl addr = {
+ .nl_family = AF_NETLINK,
+ };
+
+ if (sendto(fd, buf, len, 0, (struct sockaddr *)&addr, sizeof(addr)) < 0)
+ return -errno;
+
+ return 0;
+}
+
+/*
+ * Resolve the generic netlink family ID for @name.
+ * Returns the family ID (>= 0) on success, negative errno on failure.
+ */
+int get_family_id(int fd, const char *name)
+{
+ struct {
+ struct nlmsghdr nlh;
+ struct genlmsghdr genl;
+ char buf[256];
+ } req = { 0 };
+ char resp[8192];
+ struct nlmsghdr *nlh;
+ struct genlmsghdr *genl;
+ struct nlattr *na;
+ int len;
+ int rem;
+ int ret;
+
+ req.nlh.nlmsg_len = NLMSG_LENGTH(GENL_HDRLEN);
+ req.nlh.nlmsg_type = GENL_ID_CTRL;
+ req.nlh.nlmsg_flags = NLM_F_REQUEST;
+ req.nlh.nlmsg_seq = 1;
+ req.nlh.nlmsg_pid = getpid();
+
+ req.genl.cmd = CTRL_CMD_GETFAMILY;
+ req.genl.version = 1;
+
+ na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len));
+ na->nla_type = CTRL_ATTR_FAMILY_NAME;
+ na->nla_len = NLA_HDRLEN + strlen(name) + 1;
+ memcpy(nla_data(na), name, strlen(name) + 1);
+ req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len);
+
+ ret = send_request(fd, &req, req.nlh.nlmsg_len);
+ if (ret)
+ return ret;
+
+ len = recv(fd, resp, sizeof(resp), 0);
+ if (len < 0)
+ return -errno;
+
+ for (nlh = (struct nlmsghdr *)resp; NLMSG_OK(nlh, len);
+ nlh = NLMSG_NEXT(nlh, len)) {
+ if (nlh->nlmsg_type == NLMSG_ERROR) {
+ struct nlmsgerr *err = NLMSG_DATA(nlh);
+
+ return err->error ? err->error : -ENOENT;
+ }
+
+ genl = (struct genlmsghdr *)NLMSG_DATA(nlh);
+ rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN;
+ na = (struct nlattr *)((char *)genl + GENL_HDRLEN);
+ while (nla_ok(na, rem)) {
+ if (na->nla_type == CTRL_ATTR_FAMILY_ID)
+ return *(uint16_t *)nla_data(na);
+ na = nla_next(na, &rem);
+ }
+ }
+
+ return -ENOENT;
+}
diff --git a/tools/testing/selftests/acct/netlink_helper.h b/tools/testing/selftests/acct/netlink_helper.h
new file mode 100644
index 000000000000..0320729c4c06
--- /dev/null
+++ b/tools/testing/selftests/acct/netlink_helper.h
@@ -0,0 +1,44 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Shared generic netlink helpers for the acct selftests.
+ */
+#ifndef ACSELFTESTS_ACCT_NETLINK_HELPER_H
+#define ACSELFTESTS_ACCT_NETLINK_HELPER_H
+
+#include <stdbool.h>
+#include <linux/netlink.h>
+
+#ifndef NLA_ALIGNTO
+#define NLA_ALIGNTO 4
+#define NLA_ALIGN(len) (((len) + NLA_ALIGNTO - 1) & ~(NLA_ALIGNTO - 1))
+#define NLA_HDRLEN ((int)NLA_ALIGN(sizeof(struct nlattr)))
+#endif
+
+/* Fail an individual test case instead of hanging the whole binary. */
+#define ACCT_RCV_TIMEOUT_SEC 2
+
+static inline void *nla_data(const struct nlattr *na)
+{
+ return (void *)((char *)na + NLA_HDRLEN);
+}
+
+static inline bool nla_ok(const struct nlattr *na, int remaining)
+{
+ return remaining >= (int)sizeof(*na) &&
+ na->nla_len >= sizeof(*na) &&
+ na->nla_len <= remaining;
+}
+
+static inline struct nlattr *nla_next(const struct nlattr *na, int *remaining)
+{
+ int aligned_len = NLA_ALIGN(na->nla_len);
+
+ *remaining -= aligned_len;
+ return (struct nlattr *)((char *)na + aligned_len);
+}
+
+int netlink_open(void);
+int send_request(int fd, void *buf, size_t len);
+int get_family_id(int fd, const char *name);
+
+#endif /* ACSELFTESTS_ACCT_NETLINK_HELPER_H */
diff --git a/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c b/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c
index d6cab4ae26f2..9a4c1554dee3 100644
--- a/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c
+++ b/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c
@@ -16,14 +16,9 @@
#include <time.h>
#include <unistd.h>
+#include "netlink_helper.h"
#include "kselftest.h"
-#ifndef NLA_ALIGN
-#define NLA_ALIGNTO 4
-#define NLA_ALIGN(len) (((len) + NLA_ALIGNTO - 1) & ~(NLA_ALIGNTO - 1))
-#define NLA_HDRLEN ((int)NLA_ALIGN(sizeof(struct nlattr)))
-#endif
-
#define BUSY_NS (200ULL * 1000 * 1000)
struct worker_ctx {
@@ -35,26 +30,6 @@ struct worker_ctx {
static unsigned long busy_sink;
-static void *taskstats_nla_data(const struct nlattr *na)
-{
- return (void *)((char *)na + NLA_HDRLEN);
-}
-
-static bool taskstats_nla_ok(const struct nlattr *na, int remaining)
-{
- return remaining >= (int)sizeof(*na) &&
- na->nla_len >= sizeof(*na) &&
- na->nla_len <= remaining;
-}
-
-static struct nlattr *taskstats_nla_next(const struct nlattr *na, int *remaining)
-{
- int aligned_len = NLA_ALIGN(na->nla_len);
-
- *remaining -= aligned_len;
- return (struct nlattr *)((char *)na + aligned_len);
-}
-
static uint64_t timespec_diff_ns(const struct timespec *start,
const struct timespec *end)
{
@@ -84,99 +59,6 @@ static void burn_cpu_for_ns(uint64_t runtime_ns)
busy_sink = acc;
}
-static int netlink_open(void)
-{
- struct sockaddr_nl addr = {
- .nl_family = AF_NETLINK,
- .nl_pid = getpid(),
- };
- int fd;
-
- fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_GENERIC);
- if (fd < 0)
- return -errno;
-
- if (bind(fd, (struct sockaddr *)&addr, sizeof(addr)) < 0) {
- int err = -errno;
-
- close(fd);
- return err;
- }
-
- return fd;
-}
-
-static int send_request(int fd, void *buf, size_t len)
-{
- struct sockaddr_nl addr = {
- .nl_family = AF_NETLINK,
- };
-
- if (sendto(fd, buf, len, 0, (struct sockaddr *)&addr, sizeof(addr)) < 0)
- return -errno;
-
- return 0;
-}
-
-static int get_family_id(int fd, const char *name)
-{
- struct {
- struct nlmsghdr nlh;
- struct genlmsghdr genl;
- char buf[256];
- } req = { 0 };
- char resp[8192];
- struct nlmsghdr *nlh;
- struct genlmsghdr *genl;
- struct nlattr *na;
- int len;
- int rem;
- int ret;
-
- req.nlh.nlmsg_len = NLMSG_LENGTH(GENL_HDRLEN);
- req.nlh.nlmsg_type = GENL_ID_CTRL;
- req.nlh.nlmsg_flags = NLM_F_REQUEST;
- req.nlh.nlmsg_seq = 1;
- req.nlh.nlmsg_pid = getpid();
-
- req.genl.cmd = CTRL_CMD_GETFAMILY;
- req.genl.version = 1;
-
- na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len));
- na->nla_type = CTRL_ATTR_FAMILY_NAME;
- na->nla_len = NLA_HDRLEN + strlen(name) + 1;
- memcpy(taskstats_nla_data(na), name, strlen(name) + 1);
- req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len);
-
- ret = send_request(fd, &req, req.nlh.nlmsg_len);
- if (ret)
- return ret;
-
- len = recv(fd, resp, sizeof(resp), 0);
- if (len < 0)
- return -errno;
-
- for (nlh = (struct nlmsghdr *)resp; NLMSG_OK(nlh, len);
- nlh = NLMSG_NEXT(nlh, len)) {
- if (nlh->nlmsg_type == NLMSG_ERROR) {
- struct nlmsgerr *err = NLMSG_DATA(nlh);
-
- return err->error ? err->error : -ENOENT;
- }
-
- genl = (struct genlmsghdr *)NLMSG_DATA(nlh);
- rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN;
- na = (struct nlattr *)((char *)genl + GENL_HDRLEN);
- while (taskstats_nla_ok(na, rem)) {
- if (na->nla_type == CTRL_ATTR_FAMILY_ID)
- return *(uint16_t *)taskstats_nla_data(na);
- na = taskstats_nla_next(na, &rem);
- }
- }
-
- return -ENOENT;
-}
-
static int get_taskstats(int fd, int family_id, uint16_t attr_type, uint32_t id,
struct taskstats *stats)
{
@@ -209,7 +91,7 @@ static int get_taskstats(int fd, int family_id, uint16_t attr_type, uint32_t id,
na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len));
na->nla_type = attr_type;
na->nla_len = NLA_HDRLEN + sizeof(id);
- memcpy(taskstats_nla_data(na), &id, sizeof(id));
+ memcpy(nla_data(na), &id, sizeof(id));
req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len);
ret = send_request(fd, &req, req.nlh.nlmsg_len);
@@ -231,21 +113,21 @@ static int get_taskstats(int fd, int family_id, uint16_t attr_type, uint32_t id,
genl = (struct genlmsghdr *)NLMSG_DATA(nlh);
rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN;
na = (struct nlattr *)((char *)genl + GENL_HDRLEN);
- while (taskstats_nla_ok(na, rem)) {
+ while (nla_ok(na, rem)) {
if (na->nla_type == TASKSTATS_TYPE_AGGR_PID ||
na->nla_type == TASKSTATS_TYPE_AGGR_TGID) {
- nested = (struct nlattr *)taskstats_nla_data(na);
+ nested = (struct nlattr *)nla_data(na);
nrem = na->nla_len - NLA_HDRLEN;
- while (taskstats_nla_ok(nested, nrem)) {
+ while (nla_ok(nested, nrem)) {
if (nested->nla_type == TASKSTATS_TYPE_STATS) {
- memcpy(stats, taskstats_nla_data(nested),
+ memcpy(stats, nla_data(nested),
sizeof(*stats));
return 0;
}
- nested = taskstats_nla_next(nested, &nrem);
+ nested = nla_next(nested, &nrem);
}
}
- na = taskstats_nla_next(na, &rem);
+ na = nla_next(na, &rem);
}
}
diff --git a/tools/testing/selftests/alloc_tag/Makefile b/tools/testing/selftests/alloc_tag/Makefile
new file mode 100644
index 000000000000..c4637f69e9c2
--- /dev/null
+++ b/tools/testing/selftests/alloc_tag/Makefile
@@ -0,0 +1,8 @@
+# SPDX-License-Identifier: GPL-2.0
+
+TEST_GEN_PROGS := allocinfo_ioctl_test
+
+CFLAGS += -Wall
+CFLAGS += $(KHDR_INCLUDES)
+
+include ../lib.mk
diff --git a/tools/testing/selftests/alloc_tag/allocinfo_ioctl_test.c b/tools/testing/selftests/alloc_tag/allocinfo_ioctl_test.c
new file mode 100644
index 000000000000..74fd64b2370c
--- /dev/null
+++ b/tools/testing/selftests/alloc_tag/allocinfo_ioctl_test.c
@@ -0,0 +1,548 @@
+// SPDX-License-Identifier: GPL-2.0-only
+
+/* kselftest for allocinfo ioctl
+ * allocinfo ioctl retrieves allocinfo data through ioctl
+ * Copyright (C) 2026 Google, Inc.
+ */
+
+#include <errno.h>
+#include <fcntl.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <stdbool.h>
+#include <unistd.h>
+#include <sys/ioctl.h>
+#include <linux/types.h>
+#include <linux/alloc_tag.h>
+#include "../kselftest.h"
+
+#define MAX_LINE_LEN 512
+#define ALLOCINFO_PROC "/proc/allocinfo"
+
+enum ioctl_ret {
+ IOCTL_SUCCESS = 0,
+ IOCTL_FAILURE = 1,
+ IOCTL_INVALID_DATA = 2,
+};
+
+#define VEC_MAX_ENTRIES 32
+
+struct allocinfo_tag_data_vec {
+ struct allocinfo_tag_data tag[VEC_MAX_ENTRIES];
+ __u64 count;
+};
+
+static inline int __allocinfo_get_content_id(int dev_fd, struct allocinfo_content_id *params)
+{
+ return ioctl(dev_fd, ALLOCINFO_IOC_CONTENT_ID, params);
+}
+
+static inline int __allocinfo_get_at(int dev_fd, struct allocinfo_get_at *params)
+{
+ return ioctl(dev_fd, ALLOCINFO_IOC_GET_AT, params);
+}
+
+static inline int __allocinfo_get_next(int dev_fd, struct allocinfo_tag_data *params)
+{
+ return ioctl(dev_fd, ALLOCINFO_IOC_GET_NEXT, params);
+}
+
+static bool match_entry(const struct allocinfo_tag_data *procfs_entry,
+ const struct allocinfo_tag_data *tag_data,
+ bool match_bytes, bool match_calls, bool match_lineno,
+ bool match_function, bool match_filename)
+{
+ if (match_bytes && tag_data->counter.bytes != procfs_entry->counter.bytes) {
+ ksft_print_msg("size retrieved through ioctl does not match procfs\n");
+ return false;
+ }
+
+ if (match_calls && tag_data->counter.calls != procfs_entry->counter.calls) {
+ ksft_print_msg("call count retrieved through ioctl does not match procfs\n");
+ return false;
+ }
+
+ if (match_lineno && tag_data->tag.lineno != procfs_entry->tag.lineno) {
+ ksft_print_msg("lineno retrieved through ioctl does not match procfs\n");
+ return false;
+ }
+
+ if (match_function &&
+ strncmp(tag_data->tag.function, procfs_entry->tag.function, ALLOCINFO_STR_SIZE)) {
+ ksft_print_msg("function retrieved through ioctl does not match procfs\n");
+ return false;
+ }
+
+ if (match_filename &&
+ strncmp(tag_data->tag.filename, procfs_entry->tag.filename, ALLOCINFO_STR_SIZE)) {
+ ksft_print_msg("filename retrieved through ioctl does not match procfs\n");
+ return false;
+ }
+ return true;
+}
+
+static bool match_entries(const struct allocinfo_tag_data_vec *procfs_entries,
+ const struct allocinfo_tag_data_vec *tags,
+ bool match_bytes, bool match_calls, bool match_lineno,
+ bool match_function, bool match_filename)
+{
+ __u64 i;
+
+ if (procfs_entries->count != tags->count) {
+ ksft_print_msg("Entry count mismatch. ioctl entries: %llu, proc entries: %llu\n",
+ tags->count, procfs_entries->count);
+ return false;
+ }
+ for (i = 0; i < procfs_entries->count; i++) {
+ if (!match_entry(&procfs_entries->tag[i], &tags->tag[i],
+ match_bytes, match_calls, match_lineno,
+ match_function, match_filename)) {
+ ksft_print_msg("%lluth entry does not match.\n", i);
+ return false;
+ }
+ }
+ return true;
+}
+
+static const char *allocinfo_str(const char *str)
+{
+ size_t len = strlen(str);
+
+ if (len >= ALLOCINFO_STR_SIZE)
+ str += (len - ALLOCINFO_STR_SIZE) + 1;
+ return str;
+}
+
+static void allocinfo_copy_str(char *dest, const char *src)
+{
+ strncpy(dest, allocinfo_str(src), ALLOCINFO_STR_SIZE - 1);
+ dest[ALLOCINFO_STR_SIZE - 1] = '\0';
+}
+
+static int get_filtered_procfs_entries(struct allocinfo_tag_data_vec *procfs_entries,
+ const struct allocinfo_filter *filter)
+{
+ FILE *fp = fopen(ALLOCINFO_PROC, "r");
+ char line[MAX_LINE_LEN];
+ int matches;
+ struct allocinfo_tag_data procfs_entry;
+
+ if (!fp) {
+ ksft_print_msg("Failed to open " ALLOCINFO_PROC " for reading\n");
+ return 1;
+ }
+ memset(procfs_entries, 0, sizeof(*procfs_entries));
+ while (fgets(line, sizeof(line), fp) && procfs_entries->count < VEC_MAX_ENTRIES) {
+ char filename[MAX_LINE_LEN];
+ char function[MAX_LINE_LEN];
+
+ memset(&procfs_entry, 0, sizeof(procfs_entry));
+ matches = sscanf(line, "%llu %llu %[^:]:%llu func:%s",
+ &procfs_entry.counter.bytes,
+ &procfs_entry.counter.calls,
+ filename,
+ &procfs_entry.tag.lineno,
+ function);
+
+ if (matches != 5)
+ continue;
+
+ allocinfo_copy_str(procfs_entry.tag.filename, filename);
+ allocinfo_copy_str(procfs_entry.tag.function, function);
+
+ if (filter->mask & ALLOCINFO_FILTER_MASK_FILENAME) {
+ if (strncmp(procfs_entry.tag.filename,
+ filter->fields.filename, ALLOCINFO_STR_SIZE))
+ continue;
+ }
+ if (filter->mask & ALLOCINFO_FILTER_MASK_FUNCTION) {
+ if (strncmp(procfs_entry.tag.function,
+ filter->fields.function, ALLOCINFO_STR_SIZE))
+ continue;
+ }
+ if (filter->mask & ALLOCINFO_FILTER_MASK_LINENO) {
+ if (procfs_entry.tag.lineno != filter->fields.lineno)
+ continue;
+ }
+ if (filter->mask & ALLOCINFO_FILTER_MASK_MIN_SIZE) {
+ if (procfs_entry.counter.bytes < filter->min_size)
+ continue;
+ }
+ if (filter->mask & ALLOCINFO_FILTER_MASK_MAX_SIZE) {
+ if (procfs_entry.counter.bytes > filter->max_size)
+ continue;
+ }
+
+ memcpy(&procfs_entries->tag[procfs_entries->count++], &procfs_entry,
+ sizeof(procfs_entry));
+ }
+ fclose(fp);
+ return 0;
+}
+
+static enum ioctl_ret get_filtered_ioctl_entries(struct allocinfo_tag_data_vec *tags,
+ const struct allocinfo_filter *filter,
+ __u64 start_pos)
+{
+ int fd = open(ALLOCINFO_PROC, O_RDONLY);
+
+ if (fd < 0) {
+ ksft_print_msg("Failed to open " ALLOCINFO_PROC " for IOCTL\n");
+ return IOCTL_FAILURE;
+ }
+
+ struct allocinfo_content_id start_cont_id, end_cont_id;
+ struct allocinfo_get_at get_at_params;
+ const int max_retries = 10;
+ int retry_count = 0;
+ int status;
+
+ /*
+ * __allocinfo_get_content_id may return different values if a kernel module was loaded
+ * between the two calls. If that happens, the data gathered cannot be considered consistent
+ * and hence needs to be fetched again to avoid flakiness.
+ */
+ do {
+ if (__allocinfo_get_content_id(fd, &start_cont_id)) {
+ ksft_print_msg("allocinfo_get_content_id failed\n");
+ status = IOCTL_FAILURE;
+ break;
+ }
+
+ memset(tags, 0, sizeof(*tags));
+ memset(&get_at_params, 0, sizeof(get_at_params));
+ memcpy(&get_at_params.filter, filter, sizeof(*filter));
+ get_at_params.pos = start_pos;
+ if (__allocinfo_get_at(fd, &get_at_params)) {
+ ksft_print_msg("allocinfo_get_at failed\n");
+ status = IOCTL_FAILURE;
+ break;
+ }
+ memcpy(&tags->tag[tags->count++], &get_at_params.data, sizeof(get_at_params.data));
+
+ while (tags->count < VEC_MAX_ENTRIES &&
+ __allocinfo_get_next(fd, &tags->tag[tags->count]) == 0)
+ tags->count++;
+
+ if (__allocinfo_get_content_id(fd, &end_cont_id)) {
+ ksft_print_msg("allocinfo_get_content_id failed\n");
+ status = IOCTL_FAILURE;
+ break;
+ }
+
+ if (start_cont_id.id == end_cont_id.id) {
+ status = IOCTL_SUCCESS;
+ } else {
+ ksft_print_msg("allocinfo_get_content_id mismatch, retrying...\n");
+ status = IOCTL_INVALID_DATA;
+ }
+ } while (status == IOCTL_INVALID_DATA && retry_count++ < max_retries);
+
+ close(fd);
+ return status;
+}
+
+static int run_filter_test(const struct allocinfo_filter *filter)
+{
+ struct allocinfo_tag_data_vec *tags = malloc(sizeof(*tags));
+ struct allocinfo_tag_data_vec *procfs_entries = malloc(sizeof(*procfs_entries));
+ int ioctl_status;
+ int ret = KSFT_PASS;
+
+ if (!tags || !procfs_entries) {
+ ksft_print_msg("Memory allocation failed.\n");
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ if (get_filtered_procfs_entries(procfs_entries, filter)) {
+ ksft_print_msg("Error retrieving entries from " ALLOCINFO_PROC "\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+
+ if (procfs_entries->count == 0) {
+ ksft_print_msg("No entries found in " ALLOCINFO_PROC ", skipping test\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+
+ ioctl_status = get_filtered_ioctl_entries(tags, filter, 0);
+ if (ioctl_status == IOCTL_INVALID_DATA) {
+ ksft_print_msg("Trouble retrieving valid IOCTL entries, skipping.\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+ if (ioctl_status == IOCTL_FAILURE) {
+ ksft_print_msg("Error retrieving IOCTL entries.\n");
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ if (!match_entries(procfs_entries, tags, false, false, true, true, true))
+ ret = KSFT_FAIL;
+
+exit:
+ free(tags);
+ free(procfs_entries);
+ return ret;
+}
+
+static int test_filename_filter(void)
+{
+ struct allocinfo_filter filter;
+ const char *target_filename = "mm/memory.c";
+
+ memset(&filter, 0, sizeof(filter));
+ filter.mask |= ALLOCINFO_FILTER_MASK_FILENAME;
+ strncpy(filter.fields.filename, target_filename, ALLOCINFO_STR_SIZE);
+
+ return run_filter_test(&filter);
+}
+
+static int test_function_filter(void)
+{
+ struct allocinfo_filter filter;
+ const char *target_function = "dup_mm";
+
+ memset(&filter, 0, sizeof(filter));
+ filter.mask |= ALLOCINFO_FILTER_MASK_FUNCTION;
+ strncpy(filter.fields.function, target_function, ALLOCINFO_STR_SIZE);
+
+ return run_filter_test(&filter);
+}
+
+static int test_size_filter(void)
+{
+ int fd;
+ struct allocinfo_tag_data_vec *tags = malloc(sizeof(*tags));
+ struct allocinfo_tag_data_vec *procfs_entries = malloc(sizeof(*procfs_entries));
+ struct allocinfo_filter filter;
+ int ret = KSFT_PASS;
+ __u64 target_size, i, pos;
+ struct allocinfo_tag_data *found_tag = NULL;
+ const char *target_function = "do_init_module";
+ struct allocinfo_content_id start_cont_id, end_cont_id;
+ int retry = 0;
+ const int max_retries = 10;
+
+ if (!tags || !procfs_entries) {
+ ksft_print_msg("Memory allocation failed.\n");
+ ret = KSFT_FAIL;
+ goto freemem;
+ }
+
+ fd = open(ALLOCINFO_PROC, O_RDONLY);
+ if (fd < 0) {
+ ksft_print_msg("Failed to open " ALLOCINFO_PROC ": %s\n", strerror(errno));
+ ret = KSFT_SKIP;
+ goto freemem;
+ }
+
+ do {
+ found_tag = NULL;
+ pos = 0;
+
+ if (__allocinfo_get_content_id(fd, &start_cont_id)) {
+ ksft_print_msg("allocinfo_get_content_id failed\n");
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ memset(&filter, 0, sizeof(filter));
+ filter.mask |= ALLOCINFO_FILTER_MASK_FUNCTION;
+ strncpy(filter.fields.function, target_function, ALLOCINFO_STR_SIZE);
+
+ if (get_filtered_procfs_entries(procfs_entries, &filter)) {
+ ksft_print_msg("Error retrieving entries from " ALLOCINFO_PROC "\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+
+ if (procfs_entries->count == 0) {
+ ksft_print_msg("Function %s not found in procfs\n", target_function);
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+
+ target_size = procfs_entries->tag[0].counter.bytes;
+
+ memset(&filter, 0, sizeof(filter));
+ filter.mask |= ALLOCINFO_FILTER_MASK_MIN_SIZE | ALLOCINFO_FILTER_MASK_MAX_SIZE;
+ filter.min_size = target_size;
+ filter.max_size = target_size;
+
+ while (1) {
+ struct allocinfo_get_at get_at_params;
+
+ memset(&get_at_params, 0, sizeof(get_at_params));
+ memcpy(&get_at_params.filter, &filter, sizeof(filter));
+ get_at_params.pos = pos;
+
+ if (__allocinfo_get_at(fd, &get_at_params))
+ break;
+
+ tags->count = 0;
+ memcpy(&tags->tag[tags->count++], &get_at_params.data,
+ sizeof(get_at_params.data));
+
+ while (tags->count < VEC_MAX_ENTRIES &&
+ __allocinfo_get_next(fd, &tags->tag[tags->count]) == 0)
+ tags->count++;
+
+ for (i = 0; i < tags->count; i++) {
+ if (strcmp(tags->tag[i].tag.function, target_function) == 0) {
+ found_tag = &tags->tag[i];
+ break;
+ }
+ }
+
+ if (found_tag || tags->count < VEC_MAX_ENTRIES)
+ break;
+
+ pos += tags->count;
+ }
+
+ if (__allocinfo_get_content_id(fd, &end_cont_id)) {
+ ksft_print_msg("allocinfo_get_content_id failed\n");
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ if (start_cont_id.id == end_cont_id.id)
+ break;
+
+ ksft_print_msg("Module load detected during size verification, retrying...\n");
+ } while (retry++ < max_retries);
+
+ if (start_cont_id.id == end_cont_id.id && !found_tag) {
+ ksft_print_msg("Entry with function %s not found in IOCTL results\n",
+ target_function);
+ ret = KSFT_FAIL;
+ } else if (start_cont_id.id != end_cont_id.id) {
+ ksft_print_msg("Failed to match content_ids for procfs and IOCTL, skipping...\n");
+ ret = KSFT_SKIP;
+ } else if (found_tag && found_tag->counter.bytes != target_size) {
+ ksft_print_msg("IOCTL entry size %llu does not match target size %llu\n",
+ found_tag->counter.bytes, target_size);
+ ret = KSFT_FAIL;
+ }
+
+exit:
+ close(fd);
+freemem:
+ free(tags);
+ free(procfs_entries);
+ return ret;
+}
+
+static int test_lineno_filter(void)
+{
+ struct allocinfo_tag_data_vec *tags = malloc(sizeof(*tags));
+ struct allocinfo_tag_data_vec *procfs_entries = malloc(sizeof(*procfs_entries));
+ struct allocinfo_filter filter;
+ enum ioctl_ret ioctl_status;
+ int ret = KSFT_PASS;
+ __u64 target_lineno, i;
+ struct allocinfo_tag_data *target_tag;
+ bool found = false;
+
+ if (!tags || !procfs_entries) {
+ ksft_print_msg("Memory allocation failed.\n");
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ memset(&filter, 0, sizeof(filter));
+
+ if (get_filtered_procfs_entries(procfs_entries, &filter)) {
+ ksft_print_msg("Error retrieving entries from " ALLOCINFO_PROC "\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+ if (procfs_entries->count == 0) {
+ ksft_print_msg("Could not retrieve procfs entries\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+ /*
+ * We depend on the procfs results to determine the line number for the filter before
+ * making the ioctl query. Hence, we cannot reuse run_filter_test here.
+ */
+ target_tag = &procfs_entries->tag[0];
+ target_lineno = target_tag->tag.lineno;
+
+ filter.mask |= ALLOCINFO_FILTER_MASK_LINENO;
+ filter.fields.lineno = target_lineno;
+
+ ioctl_status = get_filtered_ioctl_entries(tags, &filter, 0);
+ if (ioctl_status == IOCTL_INVALID_DATA) {
+ ksft_print_msg("Trouble retrieving valid IOCTL entries, skipping.\n");
+ ret = KSFT_SKIP;
+ goto exit;
+ }
+ if (ioctl_status == IOCTL_FAILURE) {
+ ksft_print_msg("Error retrieving IOCTL entries.\n");
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ for (i = 0; i < tags->count; i++) {
+ if (tags->tag[i].tag.lineno != target_lineno) {
+ ksft_print_msg("IOCTL entry %llu has incorrect lineno %llu.\n",
+ i, tags->tag[i].tag.lineno);
+ ret = KSFT_FAIL;
+ goto exit;
+ }
+
+ if (strncmp(tags->tag[i].tag.function, target_tag->tag.function,
+ ALLOCINFO_STR_SIZE) == 0 &&
+ strncmp(tags->tag[i].tag.filename, target_tag->tag.filename,
+ ALLOCINFO_STR_SIZE) == 0)
+ found = true;
+ }
+
+ if (!found) {
+ ksft_print_msg("Original procfs entry not found in IOCTL lineno filter results.\n");
+ ret = KSFT_FAIL;
+ }
+
+exit:
+ free(tags);
+ free(procfs_entries);
+ return ret;
+}
+
+int main(int argc, char *argv[])
+{
+ int ret;
+
+ ksft_set_plan(4);
+
+ ret = test_filename_filter();
+ if (ret == KSFT_SKIP)
+ ksft_test_result_skip("Skipping test_filename_filter\n");
+ else
+ ksft_test_result(ret == KSFT_PASS, "test_filename_filter\n");
+
+ ret = test_function_filter();
+ if (ret == KSFT_SKIP)
+ ksft_test_result_skip("Skipping test_function_filter\n");
+ else
+ ksft_test_result(ret == KSFT_PASS, "test_function_filter\n");
+
+ ret = test_size_filter();
+ if (ret == KSFT_SKIP)
+ ksft_test_result_skip("Skipping test_size_filter\n");
+ else
+ ksft_test_result(ret == KSFT_PASS, "test_size_filter\n");
+
+ ret = test_lineno_filter();
+ if (ret == KSFT_SKIP)
+ ksft_test_result_skip("Skipping test_lineno_filter\n");
+ else
+ ksft_test_result(ret == KSFT_PASS, "test_lineno_filter\n");
+
+ ksft_finished();
+}
diff --git a/tools/testing/selftests/alsa/mixer-test.c b/tools/testing/selftests/alsa/mixer-test.c
index a329f901c5ed..0857d64c322a 100644
--- a/tools/testing/selftests/alsa/mixer-test.c
+++ b/tools/testing/selftests/alsa/mixer-test.c
@@ -319,8 +319,8 @@ static bool ctl_value_index_valid(struct ctl_data *ctl,
/* Only check step size if there is one and we're in bounds */
if (snd_ctl_elem_info_get_step(ctl->info) &&
- (int_val - snd_ctl_elem_info_get_min(ctl->info) %
- snd_ctl_elem_info_get_step(ctl->info))) {
+ (int_val - snd_ctl_elem_info_get_min(ctl->info)) %
+ snd_ctl_elem_info_get_step(ctl->info)) {
ksft_print_msg("%s.%d value %ld invalid for step %ld minimum %ld\n",
ctl->name, index, int_val,
snd_ctl_elem_info_get_step(ctl->info),
diff --git a/tools/testing/selftests/arm64/abi/.gitignore b/tools/testing/selftests/arm64/abi/.gitignore
index 44f8b80f37e3..491a80db9dff 100644
--- a/tools/testing/selftests/arm64/abi/.gitignore
+++ b/tools/testing/selftests/arm64/abi/.gitignore
@@ -1,4 +1,6 @@
hwcap
ptrace
+seccomp_ptrace_x0_bypass
+seccomp_ret_trace_x0_bypass
syscall-abi
tpidr2
diff --git a/tools/testing/selftests/arm64/abi/Makefile b/tools/testing/selftests/arm64/abi/Makefile
index 483488f8c2ad..e91d4cdf17ad 100644
--- a/tools/testing/selftests/arm64/abi/Makefile
+++ b/tools/testing/selftests/arm64/abi/Makefile
@@ -1,7 +1,9 @@
# SPDX-License-Identifier: GPL-2.0
# Copyright (C) 2021 ARM Limited
-TEST_GEN_PROGS := hwcap ptrace syscall-abi tpidr2
+CFLAGS += $(KHDR_INCLUDES) -I$(top_srcdir)/tools/include
+
+TEST_GEN_PROGS := hwcap ptrace syscall-abi tpidr2 seccomp_ptrace_x0_bypass seccomp_ret_trace_x0_bypass
include ../../lib.mk
diff --git a/tools/testing/selftests/arm64/abi/seccomp_ptrace_x0_bypass.c b/tools/testing/selftests/arm64/abi/seccomp_ptrace_x0_bypass.c
new file mode 100644
index 000000000000..4ee8e5aaad6f
--- /dev/null
+++ b/tools/testing/selftests/arm64/abi/seccomp_ptrace_x0_bypass.c
@@ -0,0 +1,192 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test that seccomp, tracepoints and audit observe the correct syscall
+ * arguments after a ptracer has modified them at syscall-enter-stop.
+ *
+ * On arm64, both the first argument and the return value of a syscall
+ * are passed in register x0. The original x0 is saved in
+ * pt_regs::orig_x0 during syscall entry and returned as the first
+ * argument by syscall_get_arguments(). Because ptrace modifications
+ * to x0 are not automatically reflected in orig_x0, seccomp, tracepoints
+ * and audit may see a stale value unless orig_x0 is explicitly
+ * re-synchronised after a ptrace stop.
+ *
+ * This test sets up a seccomp filter that allows write(2, ...) but kills
+ * the task for any other fd. A ptracer changes the fd argument from 2
+ * to 1 at the syscall-enter stop. If the orig_x0 re-sync works, seccomp
+ * sees the modified argument (fd=1) and kills the child with SIGSYS
+ * (test passes). If orig_x0 is not re-synced, seccomp sees the original
+ * fd=2, the write succeeds and the child exits normally (test fails,
+ * vulnerability present).
+ */
+#include <errno.h>
+#include <stdbool.h>
+#include <stddef.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/prctl.h>
+#include <sys/ptrace.h>
+#include <sys/uio.h>
+#include <sys/wait.h>
+#include <asm/ptrace.h>
+#include <linux/elf.h>
+#include <linux/filter.h>
+#include <linux/seccomp.h>
+#include <asm/unistd.h>
+
+#include "kselftest.h"
+
+#define EXPECTED_TESTS 1
+
+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+#define ARG0_OFFSET (offsetof(struct seccomp_data, args))
+#else
+#define ARG0_OFFSET (offsetof(struct seccomp_data, args) + 4)
+#endif
+
+static int do_child(void)
+{
+ if (ptrace(PTRACE_TRACEME, 0, NULL, NULL))
+ ksft_exit_fail_perror("PTRACE_TRACEME");
+
+ if (raise(SIGSTOP))
+ ksft_exit_fail_perror("raise(SIGSTOP)");
+
+ /*
+ * Seccomp filter:
+ * If syscall is not write -> ALLOW
+ * If syscall is write:
+ * - If args[0] (fd) == 2 -> ALLOW
+ * - Otherwise -> KILL
+ */
+ struct sock_filter filter[] = {
+ BPF_STMT(BPF_LD | BPF_W | BPF_ABS, offsetof(struct seccomp_data, nr)), /* nr */
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_write, 0, 3),
+ BPF_STMT(BPF_LD | BPF_W | BPF_ABS, ARG0_OFFSET), /* args[0] */
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, 2, 1, 0),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_KILL),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_ALLOW),
+ };
+ struct sock_fprog prog = {
+ .len = ARRAY_SIZE(filter),
+ .filter = filter,
+ };
+
+ if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0))
+ ksft_exit_fail_perror("prctl NO_NEW_PRIVS");
+
+ if (prctl(PR_SET_SECCOMP, SECCOMP_MODE_FILTER, &prog))
+ ksft_exit_fail_perror("prctl SECCOMP");
+
+ /*
+ * Invoke write(2, ...) while the tracer will change the first
+ * argument (fd) from 2 to 1 at syscall entry.
+ */
+ syscall(__NR_write, 2, NULL, 0);
+ _exit(0);
+}
+
+static int do_parent(pid_t child)
+{
+ bool bypass = false;
+ int status;
+
+ /* Wait for the initial SIGSTOP */
+ if (waitpid(child, &status, 0) != child)
+ ksft_exit_fail_msg("waitpid failed");
+
+ if (!WIFSTOPPED(status) || WSTOPSIG(status) != SIGSTOP)
+ ksft_exit_fail_msg("unexpected stop status");
+
+ if (ptrace(PTRACE_SETOPTIONS, child, 0, PTRACE_O_TRACESYSGOOD | PTRACE_O_EXITKILL))
+ ksft_exit_fail_perror("PTRACE_SETOPTIONS");
+
+ if (ptrace(PTRACE_SYSCALL, child, 0, 0))
+ ksft_exit_fail_perror("PTRACE_SYSCALL");
+
+ while (1) {
+ int sig;
+
+ if (waitpid(child, &status, 0) != child)
+ ksft_exit_fail_msg("waitpid lost child");
+
+ if (WIFEXITED(status)) {
+ /* Child exited normally – bypass succeeded */
+ bypass = true;
+ break;
+ }
+
+ if (WIFSIGNALED(status)) {
+ sig = WTERMSIG(status);
+ if (sig == SIGSYS)
+ break;
+ ksft_exit_fail_msg("child died unexpectedly from signal %d (%s)",
+ sig, strsignal(sig));
+ }
+
+ if (!WIFSTOPPED(status))
+ ksft_exit_fail_msg("unexpected wait status");
+
+ sig = WSTOPSIG(status);
+
+ if (sig == (SIGTRAP | 0x80)) {
+ struct user_regs_struct regs;
+ struct iovec iov = {
+ .iov_base = &regs,
+ .iov_len = sizeof(regs),
+ };
+
+ if (ptrace(PTRACE_GETREGSET, child, NT_PRSTATUS, &iov))
+ ksft_exit_fail_perror("PTRACE_GETREGSET");
+
+ unsigned long syscall_nr = regs.regs[8];
+ unsigned long x0 = regs.regs[0];
+
+ /* Modify fd from 2 to 1 at write entry */
+ if (syscall_nr == __NR_write && x0 == 2) {
+ regs.regs[0] = 1;
+ if (ptrace(PTRACE_SETREGSET, child, NT_PRSTATUS, &iov))
+ ksft_exit_fail_perror("PTRACE_SETREGSET");
+ }
+
+ if (ptrace(PTRACE_SYSCALL, child, 0, 0))
+ ksft_exit_fail_perror("PTRACE_SYSCALL");
+ } else {
+ /* Forward other signals */
+ if (ptrace(PTRACE_SYSCALL, child, 0, sig))
+ ksft_exit_fail_perror("PTRACE_SYSCALL");
+ }
+ }
+
+ /* bypass == true means vulnerability exists -> test fails */
+ return bypass ? EXIT_FAILURE : EXIT_SUCCESS;
+}
+
+int main(void)
+{
+ pid_t child;
+
+ ksft_print_header();
+ ksft_set_plan(EXPECTED_TESTS);
+
+ child = fork();
+ if (child < 0)
+ ksft_exit_fail_msg("fork failed: %s", strerror(errno));
+
+ if (!child)
+ return do_child();
+
+ /*
+ * do_parent() returns EXIT_SUCCESS if the child was killed by
+ * SIGSYS (i.e. seccomp correctly saw the modified argument),
+ * and EXIT_FAILURE if the child exited normally (bypass).
+ */
+ int result = do_parent(child);
+
+ ksft_test_result(result == EXIT_SUCCESS, "seccomp_ptrace_x0_bypass\n");
+
+ ksft_print_cnts();
+ return result;
+}
diff --git a/tools/testing/selftests/arm64/abi/seccomp_ret_trace_x0_bypass.c b/tools/testing/selftests/arm64/abi/seccomp_ret_trace_x0_bypass.c
new file mode 100644
index 000000000000..a23081763328
--- /dev/null
+++ b/tools/testing/selftests/arm64/abi/seccomp_ret_trace_x0_bypass.c
@@ -0,0 +1,201 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test for SECCOMP_RET_TRACE argument modification bypass
+ * via stale orig_x0 during filter re-evaluation.
+ *
+ * On arm64, syscall_get_arguments() reads the first argument from
+ * regs->orig_x0. When a seccomp filter returns SECCOMP_RET_TRACE,
+ * ptrace may modify regs->regs[0] while orig_x0 remains unchanged.
+ * The kernel then re-evaluates the filter; if it sees the stale
+ * orig_x0, it may incorrectly allow a syscall that the tracer intended
+ * to block.
+ *
+ * This test installs a filter that:
+ * - TRACEs write() when fd == 2
+ * - returns ERRNO(EPERM) when fd == 1
+ * - allows all other syscalls
+ *
+ * The child calls write(2, ...). The parent catches the SECCOMP stop,
+ * changes x0 (fd) from 2 to 1, and resumes the child.
+ *
+ * If re-evaluation sees the old fd=2 (stale orig_x0), the filter
+ * returns TRACE again; because recheck_after_trace is true, the kernel
+ * allows the syscall to proceed. write(1, ...) succeeds, child exits 0.
+ * -> test FAIL (bypass detected).
+ *
+ * If re-evaluation sees the new fd=1 (synced orig_x0), the filter
+ * returns ERRNO(EPERM), write fails, child exits 1.
+ * -> test PASS (no bypass).
+ *
+ * No special privileges required beyond CAP_SYS_PTRACE.
+ */
+#include <errno.h>
+#include <signal.h>
+#include <stddef.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <unistd.h>
+#include <errno.h>
+#include <sys/prctl.h>
+#include <sys/ptrace.h>
+#include <sys/uio.h>
+#include <sys/wait.h>
+#include <linux/elf.h>
+#include <linux/filter.h>
+#include <linux/seccomp.h>
+#include <linux/ptrace.h>
+#include <asm/unistd.h>
+
+#include "kselftest.h"
+
+#define PTRACE_EVENT_MASK(status) ((status) >> 16)
+
+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+#define ARG0_OFFSET (offsetof(struct seccomp_data, args))
+#else
+#define ARG0_OFFSET (offsetof(struct seccomp_data, args) + 4)
+#endif
+
+static int do_child(void)
+{
+ long ret;
+
+ if (ptrace(PTRACE_TRACEME, 0, NULL, NULL))
+ _exit(2);
+
+ raise(SIGSTOP); /* synchronize with parent */
+
+ /*
+ * Filter:
+ * if syscall == write:
+ * if fd == 2 -> TRACE
+ * if fd == 1 -> ERRNO(EPERM)
+ * else -> ALLOW
+ * else -> ALLOW
+ */
+ struct sock_filter filter[] = {
+ /* Load syscall number */
+ BPF_STMT(BPF_LD | BPF_W | BPF_ABS, offsetof(struct seccomp_data, nr)),
+ /* If not write, allow */
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_write, 0, 5),
+ /* Load first argument (fd) */
+ BPF_STMT(BPF_LD | BPF_W | BPF_ABS, ARG0_OFFSET),
+ /* fd == 2 ? */
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, 2, 0, 1),
+ /* Yes: TRACE */
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_TRACE),
+ /* fd == 1 ? */
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, 1, 0, 1),
+ /* Yes: ERRNO(EPERM) */
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_ERRNO | (EPERM & SECCOMP_RET_DATA)),
+ /* Other fd: ALLOW */
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_ALLOW),
+ };
+
+ struct sock_fprog prog = {
+ .len = ARRAY_SIZE(filter),
+ .filter = filter,
+ };
+
+ if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0))
+ _exit(3);
+ if (prctl(PR_SET_SECCOMP, SECCOMP_MODE_FILTER, &prog))
+ _exit(4);
+
+ /*
+ * write(2, ...) triggers TRACE, parent changes fd to 1.
+ * If re-eval sees fd=1 -> ERRNO -> write fails, ret = -EPERM.
+ * If re-eval sees fd=2 -> TRACE again -> allowed -> write succeeds.
+ */
+ ret = syscall(__NR_write, 2, "", 0);
+ _exit(ret == 0 ? 0 : 1);
+}
+
+int main(void)
+{
+ struct user_pt_regs regs;
+ struct iovec iov = { .iov_base = &regs, .iov_len = sizeof(regs) };
+ pid_t child;
+ int status;
+
+ ksft_print_header();
+ ksft_set_plan(1);
+
+ child = fork();
+ if (child < 0)
+ ksft_exit_fail_msg("fork failed: %s", strerror(errno));
+
+ if (!child)
+ return do_child();
+
+ /* 1. Wait for initial SIGSTOP */
+ if (waitpid(child, &status, 0) != child)
+ ksft_exit_fail_msg("waitpid SIGSTOP");
+ if (!WIFSTOPPED(status) || WSTOPSIG(status) != SIGSTOP)
+ ksft_exit_fail_msg("unexpected initial stop");
+
+ /* 2. Enable SECCOMP ptrace events */
+ if (ptrace(PTRACE_SETOPTIONS, child, 0, PTRACE_O_TRACESECCOMP))
+ ksft_exit_fail_msg("PTRACE_SETOPTIONS");
+
+ /* 3. Continue child to hit SECCOMP stop */
+ if (ptrace(PTRACE_CONT, child, 0, 0))
+ ksft_exit_fail_msg("PTRACE_CONT");
+
+ /* 4. Wait for SECCOMP stop */
+ while (1) {
+ if (waitpid(child, &status, 0) != child)
+ ksft_exit_fail_msg("waitpid SECCOMP");
+ if (WIFEXITED(status)) {
+ ksft_test_result_fail("child exited before SECCOMP stop\n");
+ goto out;
+ }
+ if (WIFSIGNALED(status)) {
+ ksft_test_result_fail("child killed unexpectedly\n");
+ goto out;
+ }
+ if (WIFSTOPPED(status) &&
+ WSTOPSIG(status) == SIGTRAP &&
+ PTRACE_EVENT_MASK(status) == PTRACE_EVENT_SECCOMP)
+ break;
+ ptrace(PTRACE_CONT, child, 0, WSTOPSIG(status));
+ }
+
+ /* 5. Modify x0 (fd) from 2 to 1 */
+ if (ptrace(PTRACE_GETREGSET, child, NT_PRSTATUS, &iov))
+ ksft_exit_fail_perror("GETREGSET");
+ if (regs.regs[8] != __NR_write || regs.regs[0] != 2) {
+ ksft_test_result_fail("unexpected regs: syscall=%llu, x0=%llu\n",
+ regs.regs[8], regs.regs[0]);
+ goto out;
+ }
+ regs.regs[0] = 1;
+ if (ptrace(PTRACE_SETREGSET, child, NT_PRSTATUS, &iov))
+ ksft_exit_fail_perror("SETREGSET");
+
+ /* 6. Resume child */
+ if (ptrace(PTRACE_CONT, child, 0, 0))
+ ksft_exit_fail_perror("PTRACE_CONT");
+
+ /* 7. Reap child – must exit normally */
+ if (waitpid(child, &status, 0) != child)
+ ksft_exit_fail_msg("final waitpid");
+
+ if (!WIFEXITED(status)) {
+ ksft_test_result_fail("child did not exit normally\n");
+ goto out;
+ }
+
+ if (WEXITSTATUS(status) != 0)
+ ksft_test_result_pass("seccomp correctly denied modified syscall\n");
+ else
+ ksft_test_result_fail("write succeeded, orig_x0 bypass likely\n");
+
+out:
+ if (child > 0) {
+ kill(child, SIGKILL);
+ waitpid(child, NULL, 0);
+ }
+ ksft_print_cnts();
+ return ksft_get_fail_cnt() ? EXIT_FAILURE : EXIT_SUCCESS;
+}
diff --git a/tools/testing/selftests/arm64/config b/tools/testing/selftests/arm64/config
new file mode 100644
index 000000000000..0fa975585392
--- /dev/null
+++ b/tools/testing/selftests/arm64/config
@@ -0,0 +1,17 @@
+CONFIG_ARM64_BTI=y
+CONFIG_ARM64_GCS=y
+CONFIG_ARM64_MTE=y
+CONFIG_ARM64_POE=y
+CONFIG_ARM64_PTR_AUTH=y
+CONFIG_ARM64_SME=y
+CONFIG_ARM64_SVE=y
+CONFIG_ARM64_TAGGED_ADDR_ABI=y
+CONFIG_HUGETLBFS=y
+CONFIG_KSM=y
+CONFIG_PROC_FS=y
+CONFIG_SECCOMP=y
+CONFIG_SECCOMP_FILTER=y
+CONFIG_SHMEM=y
+CONFIG_SYSCTL=y
+CONFIG_SYSFS=y
+CONFIG_TMPFS=y
diff --git a/tools/testing/selftests/arm64/fp/fp-ptrace.c b/tools/testing/selftests/arm64/fp/fp-ptrace.c
index 22c584b78be5..b435837c8c0e 100644
--- a/tools/testing/selftests/arm64/fp/fp-ptrace.c
+++ b/tools/testing/selftests/arm64/fp/fp-ptrace.c
@@ -65,6 +65,9 @@
/* VL 128..2048 in powers of 2 */
#define MAX_NUM_VLS 5
+/* Sentinel for detecting buffer bytes the kernel did not write */
+#define REGSET_SENTINEL 0xa5
+
/*
* FPMR bits we can set without doing feature checks to see if values
* are valid.
@@ -181,6 +184,20 @@ static bool compare_buffer(const char *name, void *out,
return false;
}
+static bool buffer_is_filled(const void *buffer, size_t size,
+ unsigned char value)
+{
+ const unsigned char *bytes = buffer;
+ size_t i;
+
+ for (i = 0; i < size; i++) {
+ if (bytes[i] != value)
+ return false;
+ }
+
+ return true;
+}
+
struct test_config {
int sve_vl_in;
int sve_vl_expected;
@@ -401,6 +418,7 @@ static bool check_ptrace_values_sve(pid_t child, struct test_config *config)
struct user_sve_header *sve;
struct user_fpsimd_state *fpsimd;
struct iovec iov;
+ size_t buf_size;
int ret, vq;
bool pass = true;
@@ -409,14 +427,16 @@ static bool check_ptrace_values_sve(pid_t child, struct test_config *config)
vq = __sve_vq_from_vl(config->sve_vl_in);
- iov.iov_len = SVE_PT_SVE_OFFSET + SVE_PT_SVE_SIZE(vq, SVE_PT_REGS_SVE);
- iov.iov_base = malloc(iov.iov_len);
+ buf_size = SVE_PT_SVE_OFFSET + SVE_PT_SVE_SIZE(vq, SVE_PT_REGS_SVE);
+ iov.iov_len = buf_size;
+ iov.iov_base = malloc(buf_size);
if (!iov.iov_base) {
ksft_print_msg("OOM allocating %lu byte SVE buffer\n",
iov.iov_len);
return false;
}
+ memset(iov.iov_base, REGSET_SENTINEL, buf_size);
ret = ptrace(PTRACE_GETREGSET, child, NT_ARM_SVE, &iov);
if (ret != 0) {
ksft_print_msg("Failed to read initial SVE: %s (%d)\n",
@@ -440,10 +460,16 @@ static bool check_ptrace_values_sve(pid_t child, struct test_config *config)
}
if (svcr_in & SVCR_SM) {
- if (sve->size != sizeof(sve)) {
+ if (sve->size != sizeof(*sve)) {
ksft_print_msg("NT_ARM_SVE reports data with PSTATE.SM\n");
pass = false;
}
+ if (!buffer_is_filled(iov.iov_base + sizeof(*sve),
+ buf_size - sizeof(*sve), REGSET_SENTINEL)) {
+ ksft_print_msg("NT_ARM_SVE wrote beyond its header with PSTATE.SM\n");
+ pass = false;
+ }
+ goto out;
} else {
if (sve->size != SVE_PT_SIZE(vq, sve->flags)) {
ksft_print_msg("Mismatch in SVE header size: %d != %lu\n",
@@ -485,6 +511,7 @@ static bool check_ptrace_values_ssve(pid_t child, struct test_config *config)
struct user_sve_header *sve;
struct user_fpsimd_state *fpsimd;
struct iovec iov;
+ size_t buf_size;
int ret, vq;
bool pass = true;
@@ -493,14 +520,16 @@ static bool check_ptrace_values_ssve(pid_t child, struct test_config *config)
vq = __sve_vq_from_vl(config->sme_vl_in);
- iov.iov_len = SVE_PT_SVE_OFFSET + SVE_PT_SVE_SIZE(vq, SVE_PT_REGS_SVE);
- iov.iov_base = malloc(iov.iov_len);
+ buf_size = SVE_PT_SVE_OFFSET + SVE_PT_SVE_SIZE(vq, SVE_PT_REGS_SVE);
+ iov.iov_len = buf_size;
+ iov.iov_base = malloc(buf_size);
if (!iov.iov_base) {
ksft_print_msg("OOM allocating %lu byte SSVE buffer\n",
iov.iov_len);
return false;
}
+ memset(iov.iov_base, REGSET_SENTINEL, buf_size);
ret = ptrace(PTRACE_GETREGSET, child, NT_ARM_SSVE, &iov);
if (ret != 0) {
ksft_print_msg("Failed to read initial SSVE: %s (%d)\n",
@@ -523,10 +552,16 @@ static bool check_ptrace_values_ssve(pid_t child, struct test_config *config)
}
if (!(svcr_in & SVCR_SM)) {
- if (sve->size != sizeof(sve)) {
+ if (sve->size != sizeof(*sve)) {
ksft_print_msg("NT_ARM_SSVE reports data without PSTATE.SM\n");
pass = false;
}
+ if (!buffer_is_filled(iov.iov_base + sizeof(*sve),
+ buf_size - sizeof(*sve), REGSET_SENTINEL)) {
+ ksft_print_msg("NT_ARM_SSVE wrote beyond its header without PSTATE.SM\n");
+ pass = false;
+ }
+ goto out;
} else {
if (sve->size != SVE_PT_SIZE(vq, sve->flags)) {
ksft_print_msg("Mismatch in SSVE header size: %d != %lu\n",
diff --git a/tools/testing/selftests/arm64/fp/sve-test.S b/tools/testing/selftests/arm64/fp/sve-test.S
index 80e072f221cd..7ef7835389e7 100644
--- a/tools/testing/selftests/arm64/fp/sve-test.S
+++ b/tools/testing/selftests/arm64/fp/sve-test.S
@@ -298,15 +298,20 @@ function irritator_handler
add x0, x0, #1
str x0, [x2, #ucontext_regs + 8 * 23]
+#ifndef SSVE
// Corrupt some random Z-regs
movi v0.8b, #1
movi v9.16b, #2
movi v31.8b, #3
// And P0
ptrue p0.d
-#ifndef SSVE
// And FFR
wrffr p15.b
+#else
+ // Enter and exit streaming mode, will reset all of the V, Z, P
+ // and FFR registers that the system has.
+ smstart_sm
+ smstop
#endif
ret
diff --git a/tools/testing/selftests/arm64/mte/check_buffer_fill.c b/tools/testing/selftests/arm64/mte/check_buffer_fill.c
index ff4e07503349..039b1d7d8566 100644
--- a/tools/testing/selftests/arm64/mte/check_buffer_fill.c
+++ b/tools/testing/selftests/arm64/mte/check_buffer_fill.c
@@ -406,6 +406,8 @@ int main(int argc, char *argv[])
size_t page_size = getpagesize();
int item = ARRAY_SIZE(sizes);
+ ksft_print_header();
+
sizes[item - 3] = page_size - 1;
sizes[item - 2] = page_size;
sizes[item - 1] = page_size + 1;
diff --git a/tools/testing/selftests/arm64/mte/check_child_memory.c b/tools/testing/selftests/arm64/mte/check_child_memory.c
index 5e97ee792e4d..e6a8acca2a94 100644
--- a/tools/testing/selftests/arm64/mte/check_child_memory.c
+++ b/tools/testing/selftests/arm64/mte/check_child_memory.c
@@ -146,6 +146,8 @@ int main(int argc, char *argv[])
int err;
int item = ARRAY_SIZE(sizes);
+ ksft_print_header();
+
page_size = getpagesize();
if (!page_size) {
ksft_print_msg("ERR: Unable to get page size\n");
diff --git a/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c b/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c
index 325bca0de0f6..5d9dc8bcfbf5 100644
--- a/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c
+++ b/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c
@@ -69,7 +69,7 @@ fail:
int execute_test(pid_t pid)
{
pthread_t thread_id[MAX_THREADS];
- int thread_data[MAX_THREADS];
+ intptr_t thread_data[MAX_THREADS];
for (int i = 0; i < MAX_THREADS; i++)
pthread_create(&thread_id[i], NULL,
@@ -131,6 +131,7 @@ int main(int argc, char *argv[])
if (err)
return err;
+ ksft_print_header();
ksft_set_plan(1);
evaluate_test(mte_gcr_fork_test(),
diff --git a/tools/testing/selftests/arm64/mte/check_hugetlb_options.c b/tools/testing/selftests/arm64/mte/check_hugetlb_options.c
index aad1234c7e0f..23e4a7a9950c 100644
--- a/tools/testing/selftests/arm64/mte/check_hugetlb_options.c
+++ b/tools/testing/selftests/arm64/mte/check_hugetlb_options.c
@@ -230,6 +230,8 @@ int main(int argc, char *argv[])
void *map_ptr;
unsigned long map_size;
+ ksft_print_header();
+
err = mte_default_setup();
if (err)
return err;
diff --git a/tools/testing/selftests/arm64/mte/check_ksm_options.c b/tools/testing/selftests/arm64/mte/check_ksm_options.c
index 0cf5faef1724..4855b737d550 100644
--- a/tools/testing/selftests/arm64/mte/check_ksm_options.c
+++ b/tools/testing/selftests/arm64/mte/check_ksm_options.c
@@ -6,6 +6,7 @@
#include <errno.h>
#include <fcntl.h>
#include <signal.h>
+#include <stdbool.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
@@ -22,6 +23,20 @@
static size_t page_sz;
static unsigned long ksm_sysfs[5];
+static bool has_merge_across_nodes;
+
+static bool merge_across_nodes_available(void)
+{
+ const char *path = PATH_KSM "merge_across_nodes";
+
+ if (!access(path, R_OK | W_OK))
+ return true;
+ if (errno == ENOENT)
+ return false;
+
+ ksft_exit_skip("Unable to read and write %s: %s\n", path,
+ strerror(errno));
+}
static unsigned long read_sysfs(char *str)
{
@@ -56,8 +71,10 @@ static void write_sysfs(char *str, unsigned long val)
static void mte_ksm_setup(void)
{
- ksm_sysfs[0] = read_sysfs(PATH_KSM "merge_across_nodes");
- write_sysfs(PATH_KSM "merge_across_nodes", 1);
+ if (has_merge_across_nodes) {
+ ksm_sysfs[0] = read_sysfs(PATH_KSM "merge_across_nodes");
+ write_sysfs(PATH_KSM "merge_across_nodes", 1);
+ }
ksm_sysfs[1] = read_sysfs(PATH_KSM "sleep_millisecs");
write_sysfs(PATH_KSM "sleep_millisecs", 0);
ksm_sysfs[2] = read_sysfs(PATH_KSM "run");
@@ -70,7 +87,8 @@ static void mte_ksm_setup(void)
static void mte_ksm_restore(void)
{
- write_sysfs(PATH_KSM "merge_across_nodes", ksm_sysfs[0]);
+ if (has_merge_across_nodes)
+ write_sysfs(PATH_KSM "merge_across_nodes", ksm_sysfs[0]);
write_sysfs(PATH_KSM "sleep_millisecs", ksm_sysfs[1]);
write_sysfs(PATH_KSM "run", ksm_sysfs[2]);
write_sysfs(PATH_KSM "max_page_sharing", ksm_sysfs[3]);
@@ -132,9 +150,16 @@ int main(int argc, char *argv[])
{
int err;
+ ksft_print_header();
+
err = mte_default_setup();
if (err)
return err;
+
+ if (geteuid() != 0)
+ ksft_exit_skip("Please run the test as root\n");
+
+ has_merge_across_nodes = merge_across_nodes_available();
page_sz = getpagesize();
if (!page_sz) {
ksft_print_msg("ERR: Unable to get page size\n");
diff --git a/tools/testing/selftests/arm64/mte/check_mmap_options.c b/tools/testing/selftests/arm64/mte/check_mmap_options.c
index c100af3012cb..492f2cd41f43 100644
--- a/tools/testing/selftests/arm64/mte/check_mmap_options.c
+++ b/tools/testing/selftests/arm64/mte/check_mmap_options.c
@@ -945,6 +945,8 @@ int main(int argc, char *argv[])
},
};
+ ksft_print_header();
+
err = mte_default_setup();
if (err)
return err;
diff --git a/tools/testing/selftests/arm64/mte/check_prctl.c b/tools/testing/selftests/arm64/mte/check_prctl.c
index f7f320defa7b..d16a91117eef 100644
--- a/tools/testing/selftests/arm64/mte/check_prctl.c
+++ b/tools/testing/selftests/arm64/mte/check_prctl.c
@@ -119,7 +119,7 @@ int main(void)
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(mte_modes));
+ ksft_set_plan(ARRAY_SIZE(mte_modes) + 1);
check_basic_read();
for (i = 0; i < ARRAY_SIZE(mte_modes); i++)
diff --git a/tools/testing/selftests/arm64/mte/check_tags_inclusion.c b/tools/testing/selftests/arm64/mte/check_tags_inclusion.c
index 4b764f2a8185..6b4fa6705d7c 100644
--- a/tools/testing/selftests/arm64/mte/check_tags_inclusion.c
+++ b/tools/testing/selftests/arm64/mte/check_tags_inclusion.c
@@ -175,6 +175,8 @@ int main(int argc, char *argv[])
{
int err;
+ ksft_print_header();
+
err = mte_default_setup();
if (err)
return err;
diff --git a/tools/testing/selftests/arm64/mte/check_user_mem.c b/tools/testing/selftests/arm64/mte/check_user_mem.c
index fb7936c4e097..af343aa61732 100644
--- a/tools/testing/selftests/arm64/mte/check_user_mem.c
+++ b/tools/testing/selftests/arm64/mte/check_user_mem.c
@@ -201,6 +201,8 @@ int main(int argc, char *argv[])
int tag_offsets[] = {page_sz, MT_GRANULE_SIZE};
char test_name[TEST_NAME_MAX];
+ ksft_print_header();
+
page_sz = getpagesize();
if (!page_sz) {
ksft_print_msg("ERR: Unable to get page size\n");
diff --git a/tools/testing/selftests/bpf/Makefile b/tools/testing/selftests/bpf/Makefile
index d3655a706482..93c707116fad 100644
--- a/tools/testing/selftests/bpf/Makefile
+++ b/tools/testing/selftests/bpf/Makefile
@@ -3,7 +3,6 @@ include ../../../build/Build.include
include ../../../scripts/Makefile.arch
include ../../../scripts/Makefile.include
-CXX ?= $(CROSS_COMPILE)g++
OBJCOPY ?= $(CROSS_COMPILE)objcopy
CURDIR := $(abspath .)
@@ -464,7 +463,7 @@ endif
CLANG_SYS_INCLUDES = $(call get_sys_includes,$(CLANG),$(CLANG_TARGET_ARCH))
BPF_CFLAGS = -g -Wall -Werror -D__TARGET_ARCH_$(SRCARCH) $(MENDIAN) \
-I$(INCLUDE_DIR) -I$(CURDIR) -I$(APIDIR) \
- -I$(CURDIR)/libarena/include \
+ -I$(TOOLSINCDIR) -I$(CURDIR)/libarena/include \
-I$(abspath $(OUTPUT)/../usr/include) \
-std=gnu11 \
-fno-strict-aliasing \
@@ -531,7 +530,7 @@ LSKELS_SIGNED := fentry_test.c fexit_test.c atomics.c
# Generate both light skeleton and libbpf skeleton for these
LSKELS_EXTRA := test_ksyms_module.c test_ksyms_weak.c kfunc_call_test.c \
- kfunc_call_test_subprog.c
+ kfunc_call_test_subprog.c test_global_percpu_data.c
SKEL_BLACKLIST += $$(LSKELS) $$(LSKELS_SIGNED)
test_static_linked.skel.h-deps := test_static_linked1.bpf.o test_static_linked2.bpf.o
diff --git a/tools/testing/selftests/bpf/README.rst b/tools/testing/selftests/bpf/README.rst
index 37164322a102..07c834433b38 100644
--- a/tools/testing/selftests/bpf/README.rst
+++ b/tools/testing/selftests/bpf/README.rst
@@ -107,12 +107,12 @@ Docker container and local rootfs image. The overall steps are as follows:
tools/testing/selftests/bpf/vmtest.sh \
-l <path of local rootfs image> -- \
./test_progs -d \
- \"$(cat tools/testing/selftests/bpf/DENYLIST.riscv64 \
+ "$(cat tools/testing/selftests/bpf/DENYLIST.riscv64 \
| cut -d'#' -f1 \
| sed -e 's/^[[:space:]]*//' \
-e 's/[[:space:]]*$//' \
| tr -s '\n' ',' \
- )\"
+ )"
Link: https://github.com/pulehui/riscv-bpf-vmtest.git [0]
Link: https://github.com/libbpf/ci/blob/main/rootfs/mkrootfs_debian.sh [1]
diff --git a/tools/testing/selftests/bpf/bench.c b/tools/testing/selftests/bpf/bench.c
index b86b73456d3c..465233ea5f79 100644
--- a/tools/testing/selftests/bpf/bench.c
+++ b/tools/testing/selftests/bpf/bench.c
@@ -541,12 +541,12 @@ extern const struct bench bench_trig_uretprobe_multi_push;
extern const struct bench bench_trig_uprobe_multi_ret;
extern const struct bench bench_trig_uretprobe_multi_ret;
#ifdef __x86_64__
-extern const struct bench bench_trig_uprobe_nop5;
-extern const struct bench bench_trig_uretprobe_nop5;
-extern const struct bench bench_trig_uprobe_multi_nop5;
-extern const struct bench bench_trig_uretprobe_multi_nop5;
+extern const struct bench bench_trig_uprobe_nop10;
+extern const struct bench bench_trig_uretprobe_nop10;
+extern const struct bench bench_trig_uprobe_multi_nop10;
+extern const struct bench bench_trig_uretprobe_multi_nop10;
extern const struct bench bench_trig_usdt_nop;
-extern const struct bench bench_trig_usdt_nop5;
+extern const struct bench bench_trig_usdt_nop10;
#endif
extern const struct bench bench_rb_libbpf;
@@ -625,12 +625,12 @@ static const struct bench *benchs[] = {
&bench_trig_uprobe_multi_ret,
&bench_trig_uretprobe_multi_ret,
#ifdef __x86_64__
- &bench_trig_uprobe_nop5,
- &bench_trig_uretprobe_nop5,
- &bench_trig_uprobe_multi_nop5,
- &bench_trig_uretprobe_multi_nop5,
+ &bench_trig_uprobe_nop10,
+ &bench_trig_uretprobe_nop10,
+ &bench_trig_uprobe_multi_nop10,
+ &bench_trig_uretprobe_multi_nop10,
&bench_trig_usdt_nop,
- &bench_trig_usdt_nop5,
+ &bench_trig_usdt_nop10,
#endif
/* ringbuf/perfbuf benchmarks */
&bench_rb_libbpf,
diff --git a/tools/testing/selftests/bpf/benchs/bench_trigger.c b/tools/testing/selftests/bpf/benchs/bench_trigger.c
index 2f22ec61667b..61513efc167a 100644
--- a/tools/testing/selftests/bpf/benchs/bench_trigger.c
+++ b/tools/testing/selftests/bpf/benchs/bench_trigger.c
@@ -396,15 +396,15 @@ static void *uprobe_producer_ret(void *input)
}
#ifdef __x86_64__
-__nocf_check __weak void uprobe_target_nop5(void)
+__nocf_check __weak void uprobe_target_nop10(void)
{
- asm volatile (".byte 0x0f, 0x1f, 0x44, 0x00, 0x00");
+ asm volatile (".byte 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00");
}
-static void *uprobe_producer_nop5(void *input)
+static void *uprobe_producer_nop10(void *input)
{
while (true)
- uprobe_target_nop5();
+ uprobe_target_nop10();
return NULL;
}
@@ -418,7 +418,7 @@ static void *uprobe_producer_usdt_nop(void *input)
return NULL;
}
-static void *uprobe_producer_usdt_nop5(void *input)
+static void *uprobe_producer_usdt_nop10(void *input)
{
while (true)
usdt_2();
@@ -542,24 +542,24 @@ static void uretprobe_multi_ret_setup(void)
}
#ifdef __x86_64__
-static void uprobe_nop5_setup(void)
+static void uprobe_nop10_setup(void)
{
- usetup(false, false /* !use_multi */, &uprobe_target_nop5);
+ usetup(false, false /* !use_multi */, &uprobe_target_nop10);
}
-static void uretprobe_nop5_setup(void)
+static void uretprobe_nop10_setup(void)
{
- usetup(true, false /* !use_multi */, &uprobe_target_nop5);
+ usetup(true, false /* !use_multi */, &uprobe_target_nop10);
}
-static void uprobe_multi_nop5_setup(void)
+static void uprobe_multi_nop10_setup(void)
{
- usetup(false, true /* use_multi */, &uprobe_target_nop5);
+ usetup(false, true /* use_multi */, &uprobe_target_nop10);
}
-static void uretprobe_multi_nop5_setup(void)
+static void uretprobe_multi_nop10_setup(void)
{
- usetup(true, true /* use_multi */, &uprobe_target_nop5);
+ usetup(true, true /* use_multi */, &uprobe_target_nop10);
}
static void usdt_setup(const char *name)
@@ -598,7 +598,7 @@ static void usdt_nop_setup(void)
usdt_setup("usdt_1");
}
-static void usdt_nop5_setup(void)
+static void usdt_nop10_setup(void)
{
usdt_setup("usdt_2");
}
@@ -665,10 +665,10 @@ BENCH_TRIG_USERMODE(uretprobe_multi_nop, nop, "uretprobe-multi-nop");
BENCH_TRIG_USERMODE(uretprobe_multi_push, push, "uretprobe-multi-push");
BENCH_TRIG_USERMODE(uretprobe_multi_ret, ret, "uretprobe-multi-ret");
#ifdef __x86_64__
-BENCH_TRIG_USERMODE(uprobe_nop5, nop5, "uprobe-nop5");
-BENCH_TRIG_USERMODE(uretprobe_nop5, nop5, "uretprobe-nop5");
-BENCH_TRIG_USERMODE(uprobe_multi_nop5, nop5, "uprobe-multi-nop5");
-BENCH_TRIG_USERMODE(uretprobe_multi_nop5, nop5, "uretprobe-multi-nop5");
+BENCH_TRIG_USERMODE(uprobe_nop10, nop10, "uprobe-nop10");
+BENCH_TRIG_USERMODE(uretprobe_nop10, nop10, "uretprobe-nop10");
+BENCH_TRIG_USERMODE(uprobe_multi_nop10, nop10, "uprobe-multi-nop10");
+BENCH_TRIG_USERMODE(uretprobe_multi_nop10, nop10, "uretprobe-multi-nop10");
BENCH_TRIG_USERMODE(usdt_nop, usdt_nop, "usdt-nop");
-BENCH_TRIG_USERMODE(usdt_nop5, usdt_nop5, "usdt-nop5");
+BENCH_TRIG_USERMODE(usdt_nop10, usdt_nop10, "usdt-nop10");
#endif
diff --git a/tools/testing/selftests/bpf/benchs/run_bench_uprobes.sh b/tools/testing/selftests/bpf/benchs/run_bench_uprobes.sh
index 9ec59423b949..e490b337e960 100755
--- a/tools/testing/selftests/bpf/benchs/run_bench_uprobes.sh
+++ b/tools/testing/selftests/bpf/benchs/run_bench_uprobes.sh
@@ -2,7 +2,7 @@
set -eufo pipefail
-for i in usermode-count syscall-count {uprobe,uretprobe}-{nop,push,ret,nop5} usdt-nop usdt-nop5
+for i in usermode-count syscall-count {uprobe,uretprobe}-{nop,push,ret,nop10} usdt-nop usdt-nop10
do
summary=$(sudo ./bench -w2 -d5 -a trig-$i | tail -n1 | cut -d'(' -f1 | cut -d' ' -f3-)
printf "%-15s: %s\n" $i "$summary"
diff --git a/tools/testing/selftests/bpf/bpf_experimental.h b/tools/testing/selftests/bpf/bpf_experimental.h
index ff37ae5a113d..2893bf06ff25 100644
--- a/tools/testing/selftests/bpf/bpf_experimental.h
+++ b/tools/testing/selftests/bpf/bpf_experimental.h
@@ -364,19 +364,25 @@ extern void bpf_iter_dmabuf_destroy(struct bpf_iter_dmabuf *it) __weak __ksym;
extern int bpf_cgroup_read_xattr(struct cgroup *cgroup, const char *name__str,
struct bpf_dynptr *value_p) __weak __ksym;
+extern int bpf_sock_read_xattr(struct socket *sock, const char *name__str,
+ struct bpf_dynptr *value_p) __weak __ksym;
+
#define PREEMPT_BITS 8
#define SOFTIRQ_BITS 8
+#define HARDIRQ_DISABLE_BITS 8
#define HARDIRQ_BITS 4
-#define NMI_BITS 4
+#define NMI_BITS 1
#define PREEMPT_SHIFT 0
#define SOFTIRQ_SHIFT (PREEMPT_SHIFT + PREEMPT_BITS)
-#define HARDIRQ_SHIFT (SOFTIRQ_SHIFT + SOFTIRQ_BITS)
+#define HARDIRQ_DISABLE_SHIFT (SOFTIRQ_SHIFT + SOFTIRQ_BITS)
+#define HARDIRQ_SHIFT (HARDIRQ_DISABLE_SHIFT + HARDIRQ_DISABLE_BITS)
#define NMI_SHIFT (HARDIRQ_SHIFT + HARDIRQ_BITS)
#define __IRQ_MASK(x) ((1UL << (x))-1)
#define SOFTIRQ_MASK (__IRQ_MASK(SOFTIRQ_BITS) << SOFTIRQ_SHIFT)
+#define HARDIRQ_DISABLE_MASK (__IRQ_MASK(HARDIRQ_DISABLE_BITS) << HARDIRQ_DISABLE_SHIFT)
#define HARDIRQ_MASK (__IRQ_MASK(HARDIRQ_BITS) << HARDIRQ_SHIFT)
#define NMI_MASK (__IRQ_MASK(NMI_BITS) << NMI_SHIFT)
diff --git a/tools/testing/selftests/bpf/libarena/include/bpf_arena_spin_lock.h b/tools/testing/selftests/bpf/libarena/include/bpf_arena_spin_lock.h
index ae6b72d15bb6..71d9db610263 100644
--- a/tools/testing/selftests/bpf/libarena/include/bpf_arena_spin_lock.h
+++ b/tools/testing/selftests/bpf/libarena/include/bpf_arena_spin_lock.h
@@ -103,12 +103,7 @@ struct arena_qnode {
#define _Q_LOCKED_VAL (1U << _Q_LOCKED_OFFSET)
#define _Q_PENDING_VAL (1U << _Q_PENDING_OFFSET)
-/*
- * The qnodes are marked __weak so we can define them in the header
- * while still ensuring all compilation units use the same struct
- * instance.
- */
-struct arena_qnode __weak __arena __hidden qnodes[_Q_MAX_CPUS][_Q_MAX_NODES];
+extern struct arena_qnode __arena __hidden qnodes[_Q_MAX_CPUS][_Q_MAX_NODES];
static inline u32 encode_tail(int cpu, int idx)
{
diff --git a/tools/testing/selftests/bpf/libarena/src/common.bpf.c b/tools/testing/selftests/bpf/libarena/src/common.bpf.c
index 569f0f64d518..41b1de3452fe 100644
--- a/tools/testing/selftests/bpf/libarena/src/common.bpf.c
+++ b/tools/testing/selftests/bpf/libarena/src/common.bpf.c
@@ -7,6 +7,13 @@
struct buddy __arena buddy;
volatile u32 zero = 0;
+/*
+ * Storage for the queue nodes declared by bpf_arena_spin_lock.h. Each program
+ * linking the arena spinlock provides exactly one definition, so that the array
+ * is emitted once rather than once per translation unit.
+ */
+struct arena_qnode __arena __hidden qnodes[_Q_MAX_CPUS][_Q_MAX_NODES];
+
int arena_fls(__u64 word)
{
if (!word)
diff --git a/tools/testing/selftests/bpf/network_helpers.c b/tools/testing/selftests/bpf/network_helpers.c
index db935a9d9fc1..cdf2d7d3ab32 100644
--- a/tools/testing/selftests/bpf/network_helpers.c
+++ b/tools/testing/selftests/bpf/network_helpers.c
@@ -424,7 +424,8 @@ int make_sockaddr(int family, const char *addr_str, __u16 port,
*len = sizeof(*sin6);
return 0;
} else if (family == AF_UNIX) {
- /* Note that we always use abstract unix sockets to avoid having
+ /*
+ * Note that we always use abstract unix sockets to avoid having
* to clean up leftover files.
*/
struct sockaddr_un *sun = (void *)addr;
@@ -865,7 +866,8 @@ static bool is_ethernet(const u_char *packet)
memcpy(&arphdr_type, packet + 8, 2);
arphdr_type = ntohs(arphdr_type);
- /* Except the following cases, the protocol type contains the
+ /*
+ * Except the following cases, the protocol type contains the
* Ethernet protocol type for the packet.
*
* https://www.tcpdump.org/linktypes/LINKTYPE_LINUX_SLL2.html
@@ -1033,19 +1035,22 @@ static void *traffic_monitor_thread(void *arg)
if (!packet)
continue;
- /* According to the man page of pcap_dump(), first argument
+ /*
+ * According to the man page of pcap_dump(), first argument
* is the pcap_dumper_t pointer even it's argument type is
* u_char *.
*/
pcap_dump((u_char *)dumper, &header, packet);
- /* Not sure what other types of packets look like. Here, we
+ /*
+ * Not sure what other types of packets look like. Here, we
* parse only Ethernet and compatible packets.
*/
if (!is_ethernet(packet))
continue;
- /* Skip SLL2 header
+ /*
+ * Skip SLL2 header
* https://www.tcpdump.org/linktypes/LINKTYPE_LINUX_SLL2.html
*
* Although the document doesn't mention that, the payload
@@ -1079,7 +1084,8 @@ static void *traffic_monitor_thread(void *arg)
return NULL;
}
-/* Prepare the pcap handle to capture packets.
+/*
+ * Prepare the pcap handle to capture packets.
*
* This pcap is non-blocking and immediate mode is enabled to receive
* captured packets as soon as possible. The snaplen is set to 1024 bytes
@@ -1150,7 +1156,8 @@ static void encode_test_name(char *buf, size_t len, const char *test_name, const
#define PCAP_DIR "/tmp/tmon_pcap"
-/* Start to monitor the network traffic in the given network namespace.
+/*
+ * Start to monitor the network traffic in the given network namespace.
*
* netns: the name of the network namespace to monitor. If NULL, the
* current network namespace is monitored.
@@ -1255,7 +1262,8 @@ static void traffic_monitor_release(struct tmonitor_ctx *ctx)
free(ctx);
}
-/* Stop the network traffic monitor.
+/*
+ * Stop the network traffic monitor.
*
* ctx: the context returned by traffic_monitor_start()
*/
diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c b/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c
index 77f1c0550c9b..6dbd1487343c 100644
--- a/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c
+++ b/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c
@@ -11,6 +11,7 @@
#include "bpf_qdisc_fail__invalid_dynptr.skel.h"
#include "bpf_qdisc_fail__invalid_dynptr_slice.skel.h"
#include "bpf_qdisc_fail__invalid_dynptr_cross_frame.skel.h"
+#include "bpf_qdisc_fail__untrusted_write.skel.h"
#include "bpf_qdisc_dynptr_use_after_invalidate_clone.skel.h"
#define LO_IFINDEX 1
@@ -230,6 +231,7 @@ void test_ns_bpf_qdisc(void)
RUN_TESTS(bpf_qdisc_fail__invalid_dynptr);
RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_cross_frame);
RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_slice);
+ RUN_TESTS(bpf_qdisc_fail__untrusted_write);
RUN_TESTS(bpf_qdisc_dynptr_use_after_invalidate_clone);
}
diff --git a/tools/testing/selftests/bpf/prog_tests/btf_map_keyless.c b/tools/testing/selftests/bpf/prog_tests/btf_map_keyless.c
new file mode 100644
index 000000000000..3248bccc3557
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/btf_map_keyless.c
@@ -0,0 +1,59 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <test_progs.h>
+#include <bpf/btf.h>
+
+/*
+ * A hash map with a key-less BTF (btf_key_type_id == 0) used to be accepted
+ * and then NULL-deref in btf_type_show() when dumped through bpffs. A fixed
+ * kernel rejects it at creation; verify that rejection, with a keyed positive
+ * control so the -EINVAL is about the missing key type and not some unrelated
+ * failure.
+ */
+static void check_keyless(int map_type, __u32 map_flags, int btf_fd, int val_id)
+{
+ LIBBPF_OPTS(bpf_map_create_opts, opts);
+ int map_fd;
+
+ opts.map_flags = map_flags;
+ opts.btf_fd = btf_fd;
+ opts.btf_value_type_id = val_id;
+
+ /* Positive control: the same map with a real key type is accepted. */
+ opts.btf_key_type_id = val_id;
+ map_fd = bpf_map_create(map_type, "keyed_map", 4, 4, 8, &opts);
+ if (!ASSERT_GE(map_fd, 0, "keyed create is accepted"))
+ return;
+ close(map_fd);
+
+ /* A key-less BTF must be rejected. */
+ opts.btf_key_type_id = 0;
+ map_fd = bpf_map_create(map_type, "keyless_map", 4, 4, 8, &opts);
+ ASSERT_EQ(map_fd, -EINVAL, "key-less create is rejected");
+ if (map_fd >= 0)
+ close(map_fd);
+}
+
+void test_btf_map_keyless(void)
+{
+ int btf_fd, val_id;
+ struct btf *btf;
+
+ btf = btf__new_empty();
+ if (!ASSERT_OK_PTR(btf, "btf__new_empty"))
+ return;
+
+ val_id = btf__add_int(btf, "int", 4, BTF_INT_SIGNED);
+ if (!ASSERT_GT(val_id, 0, "btf__add_int"))
+ goto out;
+
+ if (!ASSERT_OK(btf__load_into_kernel(btf), "btf__load_into_kernel"))
+ goto out;
+ btf_fd = btf__fd(btf);
+
+ if (test__start_subtest("hash"))
+ check_keyless(BPF_MAP_TYPE_HASH, 0, btf_fd, val_id);
+ if (test__start_subtest("rhash"))
+ check_keyless(BPF_MAP_TYPE_RHASH, BPF_F_NO_PREALLOC, btf_fd, val_id);
+out:
+ btf__free(btf);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/cb_refs.c b/tools/testing/selftests/bpf/prog_tests/cb_refs.c
index 78566b817fd7..490e15e7126d 100644
--- a/tools/testing/selftests/bpf/prog_tests/cb_refs.c
+++ b/tools/testing/selftests/bpf/prog_tests/cb_refs.c
@@ -13,7 +13,7 @@ struct {
} cb_refs_tests[] = {
{ "underflow_prog", "release kfunc bpf_kfunc_call_test_release expects referenced PTR_TO_BTF_ID passed to R1" },
{ "leak_prog", "Possibly NULL pointer passed to helper R2" },
- { "nested_cb", "Unreleased reference id=4 alloc_insn=2" }, /* alloc_insn=2{4,5} */
+ { "nested_cb", "Unreleased reference id=5 alloc_insn=2" }, /* alloc_insn=2{4,5} */
{ "non_cb_transfer_ref", "Unreleased reference id=4 alloc_insn=1" }, /* alloc_insn=1{1,2} */
};
diff --git a/tools/testing/selftests/bpf/prog_tests/cgroup_iter_memcg.c b/tools/testing/selftests/bpf/prog_tests/cgroup_iter_memcg.c
index b7c18d590b99..5a1e08d39a06 100644
--- a/tools/testing/selftests/bpf/prog_tests/cgroup_iter_memcg.c
+++ b/tools/testing/selftests/bpf/prog_tests/cgroup_iter_memcg.c
@@ -10,6 +10,17 @@
#include "cgroup_iter_memcg.h"
#include "cgroup_iter_memcg.skel.h"
+/*
+ * memcg stats are cached per-cpu and only become visible once the periodic
+ * flusher runs (FLUSH_TIME, 2s), or once pending updates cross
+ * MEMCG_CHARGE_BATCH * num_online_cpus(). That threshold grows with the CPU
+ * count, so on a large machine a single pass does not reach it and
+ * bpf_mem_cgroup_flush_stats() returns without flushing anything. Retry for
+ * long enough to cover a flusher cycle.
+ */
+#define MEMCG_STAT_RETRIES 16
+#define MEMCG_STAT_RETRY_DELAY_US (250 * 1000)
+
static int read_stats(struct bpf_link *link)
{
int fd, ret = 0;
@@ -35,11 +46,13 @@ static int read_stats(struct bpf_link *link)
static void test_anon(struct bpf_link *link, struct memcg_query *memcg_query)
{
+ int retries = 0;
void *map;
size_t len;
len = sysconf(_SC_PAGESIZE) * 1024;
+retry:
/*
* Increase memcg anon usage by mapping and writing
* to a new anon region.
@@ -53,6 +66,12 @@ static void test_anon(struct bpf_link *link, struct memcg_query *memcg_query)
if (!ASSERT_OK(read_stats(link), "read stats"))
goto cleanup;
+ if (!memcg_query->nr_anon_mapped && ++retries < MEMCG_STAT_RETRIES) {
+ usleep(MEMCG_STAT_RETRY_DELAY_US);
+ munmap(map, len);
+ goto retry;
+ }
+
ASSERT_GT(memcg_query->nr_anon_mapped, 0, "final anon mapped val");
cleanup:
@@ -61,6 +80,7 @@ cleanup:
static void test_file(struct bpf_link *link, struct memcg_query *memcg_query)
{
+ int retries = 0;
void *map;
size_t len;
char *path;
@@ -76,6 +96,7 @@ static void test_file(struct bpf_link *link, struct memcg_query *memcg_query)
fd = open(path, O_CREAT | O_RDWR, 0644);
if (!ASSERT_OK_FD(fd, "open fd"))
return;
+retry:
if (!ASSERT_OK(ftruncate(fd, len), "ftruncate"))
goto cleanup_fd;
@@ -88,6 +109,13 @@ static void test_file(struct bpf_link *link, struct memcg_query *memcg_query)
if (!ASSERT_OK(read_stats(link), "read stats"))
goto cleanup_map;
+ if ((!memcg_query->nr_file_pages || !memcg_query->nr_file_mapped) &&
+ ++retries < MEMCG_STAT_RETRIES) {
+ usleep(MEMCG_STAT_RETRY_DELAY_US);
+ munmap(map, len);
+ goto retry;
+ }
+
ASSERT_GT(memcg_query->nr_file_pages, 0, "final file value");
ASSERT_GT(memcg_query->nr_file_mapped, 0, "final file mapped value");
@@ -100,6 +128,7 @@ cleanup_fd:
static void test_shmem(struct bpf_link *link, struct memcg_query *memcg_query)
{
+ int retries = 0;
size_t len;
int fd;
@@ -113,12 +142,18 @@ static void test_shmem(struct bpf_link *link, struct memcg_query *memcg_query)
if (!ASSERT_OK_FD(fd, "memfd_create"))
return;
+retry:
if (!ASSERT_OK(fallocate(fd, 0, 0, len), "fallocate"))
goto cleanup;
if (!ASSERT_OK(read_stats(link), "read stats"))
goto cleanup;
+ if (!memcg_query->nr_shmem && ++retries < MEMCG_STAT_RETRIES) {
+ usleep(MEMCG_STAT_RETRY_DELAY_US);
+ goto retry;
+ }
+
ASSERT_GT(memcg_query->nr_shmem, 0, "final shmem value");
cleanup:
@@ -127,11 +162,13 @@ cleanup:
static void test_pgfault(struct bpf_link *link, struct memcg_query *memcg_query)
{
+ int retries = 0;
void *map;
size_t len;
len = sysconf(_SC_PAGESIZE) * 1024;
+retry:
/* Create region to use for triggering a page fault. */
map = mmap(NULL, len, PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
if (!ASSERT_NEQ(map, MAP_FAILED, "mmap anon"))
@@ -143,6 +180,12 @@ static void test_pgfault(struct bpf_link *link, struct memcg_query *memcg_query)
if (!ASSERT_OK(read_stats(link), "read stats"))
goto cleanup;
+ if (!memcg_query->pgfault && ++retries < MEMCG_STAT_RETRIES) {
+ usleep(MEMCG_STAT_RETRY_DELAY_US);
+ munmap(map, len);
+ goto retry;
+ }
+
ASSERT_GT(memcg_query->pgfault, 0, "final pgfault val");
cleanup:
diff --git a/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c b/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c
index a18d3680fb16..51f42b02a267 100644
--- a/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c
+++ b/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c
@@ -14,6 +14,197 @@
static char log[16 * 1024];
+static int load_core_relo_insns(int btf_fd, struct bpf_insn *insns, int insn_cnt,
+ struct bpf_func_info *funcs, int func_cnt,
+ int enum_id, int access_str_off, int insn_idx,
+ bool relocate)
+{
+ struct bpf_core_relo relo = {
+ .insn_off = insn_idx * sizeof(struct bpf_insn),
+ .type_id = enum_id,
+ .access_str_off = access_str_off,
+ .kind = BPF_CORE_ENUMVAL_VALUE,
+ };
+ union bpf_attr attr = {
+ .prog_type = BPF_PROG_TYPE_SOCKET_FILTER,
+ .insn_cnt = insn_cnt,
+ .insns = (__u64)insns,
+ .license = (__u64)"GPL",
+ .log_buf = (__u64)log,
+ .log_size = sizeof(log),
+ .log_level = 2,
+ .prog_btf_fd = btf_fd,
+ .func_info_rec_size = sizeof(struct bpf_func_info),
+ .func_info = (__u64)funcs,
+ .func_info_cnt = func_cnt,
+ };
+
+ if (relocate) {
+ attr.core_relo_cnt = 1;
+ attr.core_relos = (__u64)&relo;
+ attr.core_relo_rec_size = sizeof(relo);
+ }
+ memset(log, 0, sizeof(log));
+ return sys_bpf_prog_load(&attr, sizeof(attr), 1);
+}
+
+static void test_early_core_relo(void)
+{
+ static const char unrecognized[] = "trying to relocate unrecognized insn #2";
+ static const struct {
+ const char *name;
+ struct bpf_insn insns[2];
+ const char *err_msg;
+ } tests[] = {
+ { "poison_exit", { BPF_EXIT_INSN() }, unrecognized },
+ { "poison_ja", { BPF_JMP_A(1) }, unrecognized },
+ { "poison_jmp", { BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized },
+ { "poison_jmp32", { BPF_JMP32_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized },
+ { "poison_call", { BPF_EMIT_CALL(BPF_FUNC_get_prandom_u32) }, unrecognized },
+ { "poison_alu_reg", { BPF_MOV32_REG(BPF_REG_0, BPF_REG_1) }, unrecognized },
+ { "poison_alu64_reg", { BPF_MOV64_REG(BPF_REG_0, BPF_REG_1) }, unrecognized },
+ { "poison_ld_abs", { BPF_LD_ABS(BPF_W, 0) },
+ "insn #2 (LDIMM64) has unexpected form" },
+ { "poison_alu_imm", { BPF_MOV32_IMM(BPF_REG_0, 0) } },
+ { "poison_alu64_imm", { BPF_MOV64_IMM(BPF_REG_0, 0) } },
+ { "poison_ldx", { BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0) } },
+ { "poison_st", { BPF_ST_MEM(BPF_W, BPF_REG_10, -4, 0) } },
+ { "poison_stx", { BPF_STX_MEM(BPF_W, BPF_REG_10, BPF_REG_0, -4) } },
+ { "poison_ldimm64", { BPF_LD_IMM64(BPF_REG_0, 0) } },
+ };
+ struct test_btf {
+ struct btf_header hdr;
+ __u32 types[18];
+ char strings[64];
+ } raw_btf = {
+ .hdr = {
+ .magic = BTF_MAGIC,
+ .version = BTF_VERSION,
+ .hdr_len = sizeof(struct btf_header),
+ .type_off = 0,
+ .type_len = sizeof(raw_btf.types),
+ .str_off = offsetof(struct test_btf, strings) -
+ offsetof(struct test_btf, types),
+ .str_len = sizeof(raw_btf.strings),
+ },
+ .types = {
+ BTF_TYPE_INT_ENC(1, BTF_INT_SIGNED, 0, 32, 4), /* [1] int */
+ BTF_FUNC_PROTO_ENC(1, 0), /* [2] int (*)(void) */
+ BTF_FUNC_ENC(5, 2), /* [3] main_fn */
+ BTF_FUNC_ENC(13, 2), /* [4] sub_fn */
+ BTF_TYPE_ENC(20, BTF_INFO_ENC(BTF_KIND_ENUM, 0, 1), 4), /* [5] enum */
+ BTF_ENUM_ENC(45, 0), /* value = 0 */
+ },
+ .strings = "\0int\0main_fn\0sub_fn\0core_relo_poison_missing\0value\0" "0",
+ };
+ struct bpf_func_info funcs[] = {
+ { .insn_off = 0, .type_id = 3 },
+ { .insn_off = 3, .type_id = 4 },
+ };
+ struct bpf_insn core_only[] = {
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1),
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_EXIT_INSN(),
+ };
+ struct bpf_insn subprog[] = {
+ BPF_CALL_REL(2),
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_EXIT_INSN(),
+ };
+ struct bpf_insn truncated_ldimm64[] = {
+ BPF_RAW_INSN(BPF_LD | BPF_IMM | BPF_DW, 0, 0, 0, 0),
+ };
+ int access_str_off = 51; /* offset of "0" */
+ int enum_id = 5;
+ int btf_fd, prog_fd = -1, i;
+
+ btf_fd = bpf_btf_load(&raw_btf, sizeof(raw_btf), NULL);
+ if (!ASSERT_GE(btf_fd, 0, "btf_load"))
+ goto cleanup;
+
+ if (test__start_subtest("without_func_info")) {
+ prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0,
+ enum_id, access_str_off, 2, false);
+ if (!ASSERT_GE(prog_fd, 0, "control_load"))
+ goto cleanup;
+ close(prog_fd);
+ prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0,
+ enum_id, access_str_off, 2, true);
+ if (!ASSERT_GE(prog_fd, 0, "poisoned_load"))
+ goto cleanup;
+ ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log");
+ close(prog_fd);
+ prog_fd = -1;
+ }
+
+ if (test__start_subtest("before_subprog_validation")) {
+ prog_fd = load_core_relo_insns(btf_fd, subprog, ARRAY_SIZE(subprog), funcs, 2,
+ enum_id, access_str_off, 2, true);
+ if (!ASSERT_LT(prog_fd, 0, "poisoned_load"))
+ goto cleanup;
+ ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log");
+ ASSERT_HAS_SUBSTR(log, "last insn is not an exit or jmp", "poisoned_load_log");
+ }
+
+ if (test__start_subtest("truncated_ldimm64")) {
+ prog_fd = load_core_relo_insns(btf_fd, truncated_ldimm64,
+ ARRAY_SIZE(truncated_ldimm64), NULL, 0,
+ enum_id, access_str_off, 0, true);
+ if (!ASSERT_LT(prog_fd, 0, "truncated_load"))
+ goto cleanup;
+ ASSERT_HAS_SUBSTR(log, "invalid bpf_ld_imm64 insn", "truncated_load_log");
+ }
+
+ for (i = 0; i < ARRAY_SIZE(tests); i++) {
+ struct bpf_insn insns[] = {
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1),
+ tests[i].insns[0],
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_EXIT_INSN(),
+ };
+ bool is_ldimm64 = insns[2].code == (BPF_LD | BPF_DW | BPF_IMM);
+
+ if (!test__start_subtest(tests[i].name))
+ continue;
+ if (is_ldimm64) {
+ insns[1].off = 2;
+ insns[3] = tests[i].insns[1];
+ }
+ prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1,
+ enum_id, access_str_off, 2, false);
+ if (!ASSERT_GE(prog_fd, 0, "control_load"))
+ goto cleanup;
+ close(prog_fd);
+ prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1,
+ enum_id, access_str_off, 2, true);
+ if (!tests[i].err_msg) {
+ ASSERT_GE(prog_fd, 0, "dead_poison_load");
+ ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log");
+ if (is_ldimm64)
+ ASSERT_HAS_SUBSTR(log, "substituting insn #3", "poison_ldimm64_log");
+ } else {
+ ASSERT_LT(prog_fd, 0, "invalid_poison_load");
+ ASSERT_HAS_SUBSTR(log, tests[i].err_msg, "invalid_poison_log");
+ ASSERT_NULL(strstr(log, "substituting insn"), "invalid_poison_substitution");
+ }
+ close(prog_fd);
+ prog_fd = -1;
+ }
+
+cleanup:
+ if (env.verbosity > VERBOSE_NORMAL && log[0]) {
+ printf("-------- program load log start --------\n");
+ printf("%s", log);
+ printf("-------- program load log end ----------\n");
+ }
+ close(prog_fd);
+ close(btf_fd);
+}
+
/* Check that verifier rejects BPF program containing relocation
* pointing to non-existent BTF type.
*/
@@ -120,6 +311,7 @@ out:
void test_core_reloc_raw(void)
{
+ test_early_core_relo();
if (test__start_subtest("bad_local_id"))
test_bad_local_id();
}
diff --git a/tools/testing/selftests/bpf/prog_tests/dynptr.c b/tools/testing/selftests/bpf/prog_tests/dynptr.c
index 5fda11590708..4396560365e8 100644
--- a/tools/testing/selftests/bpf/prog_tests/dynptr.c
+++ b/tools/testing/selftests/bpf/prog_tests/dynptr.c
@@ -9,6 +9,7 @@
enum test_setup_type {
SETUP_SYSCALL_SLEEP,
SETUP_SKB_PROG,
+ SETUP_SKB_PROG_NONLINEAR,
SETUP_SKB_PROG_TP,
SETUP_XDP_PROG,
};
@@ -32,6 +33,7 @@ static struct {
{"test_ringbuf", SETUP_SYSCALL_SLEEP},
{"test_skb_readonly", SETUP_SKB_PROG},
{"test_dynptr_skb_data", SETUP_SKB_PROG},
+ {"test_dynptr_skb_slice_non_linear", SETUP_SKB_PROG_NONLINEAR},
{"test_dynptr_skb_meta_data", SETUP_SKB_PROG},
{"test_dynptr_skb_meta_flags", SETUP_SKB_PROG},
{"test_adjust", SETUP_SYSCALL_SLEEP},
@@ -94,7 +96,9 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ
bpf_link__destroy(link);
break;
case SETUP_SKB_PROG:
+ case SETUP_SKB_PROG_NONLINEAR:
{
+ struct __sk_buff ctx = {};
int prog_fd;
char buf[64];
@@ -106,6 +110,12 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ
.repeat = 1,
);
+ if (setup_type == SETUP_SKB_PROG_NONLINEAR) {
+ ctx.data_end = ETH_HLEN + sizeof(struct iphdr);
+ topts.ctx_in = &ctx;
+ topts.ctx_size_in = sizeof(ctx);
+ }
+
prog_fd = bpf_program__fd(prog);
if (!ASSERT_GE(prog_fd, 0, "prog_fd"))
goto cleanup;
diff --git a/tools/testing/selftests/bpf/prog_tests/exceptions.c b/tools/testing/selftests/bpf/prog_tests/exceptions.c
index 3588d6f97fd4..639866ce09a9 100644
--- a/tools/testing/selftests/bpf/prog_tests/exceptions.c
+++ b/tools/testing/selftests/bpf/prog_tests/exceptions.c
@@ -55,6 +55,7 @@ static void test_exceptions_success(void)
RUN_SUCCESS(exception_ext, 0);
RUN_SUCCESS(exception_ext_mod_cb_runtime, 35);
RUN_SUCCESS(exception_throw_subprog, 1);
+ RUN_SUCCESS(exception_throw_subprog_stack_cb, 0x1234);
RUN_SUCCESS(exception_assert_nz_gfunc, 1);
RUN_SUCCESS(exception_assert_zero_gfunc, 1);
RUN_SUCCESS(exception_assert_neg_gfunc, 1);
diff --git a/tools/testing/selftests/bpf/prog_tests/global_data_init.c b/tools/testing/selftests/bpf/prog_tests/global_data_init.c
index 8466332d7406..5671c31085cd 100644
--- a/tools/testing/selftests/bpf/prog_tests/global_data_init.c
+++ b/tools/testing/selftests/bpf/prog_tests/global_data_init.c
@@ -1,5 +1,8 @@
// SPDX-License-Identifier: GPL-2.0
#include <test_progs.h>
+#include "bpf/libbpf_internal.h"
+#include "test_global_percpu_data.skel.h"
+#include "test_global_percpu_data.lskel.h"
void test_global_data_init(void)
{
@@ -60,3 +63,336 @@ out:
free(newval);
bpf_object__close(obj);
}
+
+static void test_percpu_data_on_cpus(struct bpf_map *map, int map_fd, int prog_fd, int *runp)
+{
+ struct test_global_percpu_data__percpu *data = NULL;
+ int i, err, key = 0, num_online, run = 0;
+ __u64 args[2] = {0x1234ULL, 0x5678ULL};
+ size_t data_sz;
+ bool *online;
+ LIBBPF_OPTS(bpf_test_run_opts, topts,
+ .ctx_in = args,
+ .ctx_size_in = sizeof(args),
+ .flags = BPF_F_TEST_RUN_ON_CPU,
+ );
+
+ err = parse_cpu_mask_file("/sys/devices/system/cpu/online", &online, &num_online);
+ if (!ASSERT_OK(err, "parse_cpu_mask_file"))
+ return;
+
+ data_sz = map ? bpf_map__value_size(map) : sizeof(*data);
+ data = calloc(1, data_sz);
+ if (!ASSERT_OK_PTR(data, "calloc percpu data"))
+ goto out;
+
+ /* run on every online-CPU */
+ for (i = 0; i < num_online; i++) {
+ __u64 flags;
+
+ if (!online[i])
+ continue;
+
+ topts.cpu = i;
+ topts.retval = -1;
+ err = bpf_prog_test_run_opts(prog_fd, &topts);
+ ASSERT_OK(err, "bpf_prog_test_run_opts");
+ ASSERT_EQ(topts.retval, 0, "bpf_prog_test_run_opts retval");
+
+ memset(data, 0, data_sz);
+ flags = ((__u64) i << 32) | BPF_F_CPU;
+ if (map)
+ err = bpf_map__lookup_elem(map, &key, sizeof(key), data, data_sz, flags);
+ else
+ err = bpf_map_lookup_elem_flags(map_fd, &key, data, flags);
+ if (!ASSERT_OK(err, "lookup_elem on cpu"))
+ break;
+
+ ASSERT_EQ(*runp, ++run, "run");
+ ASSERT_EQ(data->cpu_id[0], i, "cpu_id");
+ ASSERT_EQ(data->data, 1, "data");
+ ASSERT_TRUE(data->set, "set");
+ ASSERT_EQ(data->nums[6], 0xc0de, "nums[6]");
+ ASSERT_EQ(data->struct_data.i, 1, "struct_data.i");
+ ASSERT_TRUE(data->struct_data.set, "struct_data.set");
+ ASSERT_EQ(data->struct_data.nums[6], 0xc0de, "struct_data.nums[6]");
+ }
+
+out:
+ free(data);
+ free(online);
+}
+
+static void test_global_percpu_data_init(void)
+{
+ struct test_global_percpu_data__percpu init_value = {};
+ struct test_global_percpu_data__percpu *init_data;
+ const __u32 desired_sz = sysconf(_SC_PAGE_SIZE);
+ struct test_global_percpu_data *skel = NULL;
+ size_t init_data_sz;
+ struct bpf_map *map;
+ int prog_fd, err;
+
+ skel = test_global_percpu_data__open();
+ if (!ASSERT_OK_PTR(skel, "test_global_percpu_data__open"))
+ goto out;
+ if (!ASSERT_OK_PTR(skel->percpu, "skel->percpu"))
+ goto out;
+ if (!ASSERT_OK_PTR(skel->data_percpu, "skel->data_percpu"))
+ goto out;
+ if (!ASSERT_OK_PTR(skel->percpu_data, "skel->percpu_data"))
+ goto out;
+ if (!ASSERT_OK_PTR(skel->percpu_looooooooong, "skel->percpu_looooooooong"))
+ goto out;
+
+ ASSERT_STREQ(bpf_map__name(skel->maps.percpu_data), ".percpu.data",
+ ".percpu.data map name");
+ ASSERT_STREQ(bpf_map__name(skel->maps.data_percpu), ".data.percpu",
+ ".data.percpu map name");
+ ASSERT_STREQ(bpf_map__name(skel->maps.percpu_looooooooong), ".percpu.looooooooong",
+ "long map name");
+ ASSERT_STREQ(bpf_map__name(skel->maps.percpu), ".percpu", "map name");
+ ASSERT_EQ(skel->percpu->data, -1, "skel->percpu->data");
+ ASSERT_FALSE(skel->percpu->set, "skel->percpu->set");
+ ASSERT_EQ(skel->percpu->nums[6], 0, "skel->percpu->nums[6]");
+ ASSERT_EQ(skel->percpu->struct_data.i, -1, "struct_data.i");
+ ASSERT_FALSE(skel->percpu->struct_data.set, "struct_data.set");
+ ASSERT_EQ(skel->percpu->struct_data.nums[6], 0, "struct_data.nums[6]");
+
+ map = skel->maps.percpu;
+ if (!ASSERT_EQ(bpf_map__type(map), BPF_MAP_TYPE_PERCPU_ARRAY, "bpf_map__type"))
+ goto out;
+
+ init_value.data = 2;
+ init_value.nums[6] = -1;
+ init_value.struct_data.i = 2;
+ init_value.struct_data.nums[6] = -1;
+ err = bpf_map__set_initial_value(map, &init_value, sizeof(init_value));
+ if (!ASSERT_OK(err, "bpf_map__set_initial_value"))
+ goto out;
+
+ init_data = bpf_map__initial_value(map, &init_data_sz);
+ if (!ASSERT_OK_PTR(init_data, "bpf_map__initial_value"))
+ goto out;
+
+ ASSERT_EQ(init_data->data, init_value.data, "init_value data");
+ ASSERT_EQ(init_data->set, init_value.set, "init_value set");
+ ASSERT_EQ(init_data->struct_data.i, init_value.struct_data.i, "init_value struct_data.i");
+ ASSERT_EQ(init_data->struct_data.nums[6], init_value.struct_data.nums[6],
+ "init_value struct_data.nums[6]");
+ ASSERT_EQ(init_data_sz, sizeof(init_value), "init_value size");
+ ASSERT_EQ((void *) init_data, (void *) skel->percpu, "skel->percpu eq init_data");
+ ASSERT_EQ(skel->percpu->data, init_value.data, "skel->percpu->data");
+ ASSERT_EQ(skel->percpu->set, init_value.set, "skel->percpu->set");
+ ASSERT_EQ(skel->percpu->struct_data.i, init_value.struct_data.i,
+ "skel->percpu->struct_data.i");
+ ASSERT_EQ(skel->percpu->struct_data.nums[6], init_value.struct_data.nums[6],
+ "skel->percpu->struct_data.nums[6]");
+
+ ASSERT_GT(desired_sz, sizeof(init_value), "desired_sz");
+ err = bpf_map__set_value_size(map, desired_sz);
+ if (!ASSERT_OK(err, "bpf_map__set_value_size"))
+ goto out;
+ if (!ASSERT_EQ(bpf_map__value_size(map), desired_sz, "percpu value size"))
+ goto out;
+ if (!ASSERT_NEQ(bpf_map__btf_value_type_id(map), 0, "percpu BTF value type"))
+ goto out;
+
+ init_data = bpf_map__initial_value(map, &init_data_sz);
+ if (!ASSERT_OK_PTR(init_data, "resized bpf_map__initial_value"))
+ goto out;
+ if (!ASSERT_EQ(init_data_sz, desired_sz, "resized initial value size"))
+ goto out;
+ if (!ASSERT_EQ(init_data->data, init_value.data, "resized initial value data"))
+ goto out;
+
+ err = test_global_percpu_data__load(skel);
+ if (!ASSERT_OK(err, "test_global_percpu_data__load"))
+ goto out;
+
+ ASSERT_OK_PTR(skel->percpu, "skel->percpu");
+
+ prog_fd = bpf_program__fd(skel->progs.update_percpu_data);
+ test_percpu_data_on_cpus(map, bpf_map__fd(map), prog_fd, &skel->bss->run);
+
+out:
+ test_global_percpu_data__destroy(skel);
+}
+
+static void test_global_percpu_data_lskel(void)
+{
+ struct test_global_percpu_data_lskel *lskel = NULL;
+ int prog_fd, map_fd;
+
+ lskel = test_global_percpu_data_lskel__open_and_load();
+ if (!ASSERT_OK_PTR(lskel, "test_global_percpu_data_lskel__open_and_load"))
+ goto out;
+
+ map_fd = lskel->maps.percpu.map_fd;
+ prog_fd = lskel->progs.update_percpu_data.prog_fd;
+ test_percpu_data_on_cpus(NULL, map_fd, prog_fd, &lskel->bss->run);
+
+out:
+ test_global_percpu_data_lskel__destroy(lskel);
+}
+
+static int create_rdonly_percpu_array(void)
+{
+ LIBBPF_OPTS(bpf_map_create_opts, map_opts,
+ .map_flags = BPF_F_RDONLY_PROG,
+ );
+ int key = 0, map_fd, err;
+ __u64 value = 0;
+
+ map_fd = bpf_map_create(BPF_MAP_TYPE_PERCPU_ARRAY, "percpu_ro_map", sizeof(int),
+ sizeof(__u64), 1, &map_opts);
+ if (!ASSERT_GE(map_fd, 0, "bpf_map_create"))
+ return -1;
+
+ err = bpf_map_update_elem(map_fd, &key, &value, BPF_F_ALL_CPUS);
+ if (!ASSERT_OK(err, "bpf_map_update_elem"))
+ goto out;
+
+ err = bpf_map_freeze(map_fd);
+ if (!ASSERT_OK(err, "bpf_map_freeze"))
+ goto out;
+
+ return map_fd;
+
+out:
+ close(map_fd);
+ return -1;
+}
+
+static void test_global_percpu_data_rdonly_direct_read(void)
+{
+ /*
+ * Raw instructions with manually prepared rdonly percpu_array map
+ * for testing direct-read global percpu data, because libbpf
+ * doesn't have rdonly internal percpu_array map support for
+ * global percpu data.
+ */
+ struct bpf_insn insns[] = {
+ BPF_LD_MAP_VALUE(BPF_REG_1, 0, 0),
+ BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, 0),
+ BPF_EXIT_INSN(),
+ };
+ int map_fd, prog_fd;
+
+ map_fd = create_rdonly_percpu_array();
+ if (map_fd < 0)
+ return;
+
+ insns[0].imm = map_fd;
+ prog_fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, "percpu_ro_prog", "GPL", insns,
+ ARRAY_SIZE(insns), NULL);
+ if (ASSERT_GE(prog_fd, 0, "bpf_prog_load"))
+ close(prog_fd);
+ close(map_fd);
+}
+
+static void test_global_percpu_data_rdonly_direct_write(void)
+{
+ LIBBPF_OPTS(bpf_prog_load_opts, prog_opts);
+ /* See the comment in test_global_percpu_data_rdonly_direct_read() */
+ struct bpf_insn insns[] = {
+ BPF_LD_MAP_VALUE(BPF_REG_1, 0, 0),
+ BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, 0),
+ BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0),
+ BPF_EXIT_INSN(),
+ };
+ char log_buf[256] = {};
+ int map_fd, prog_fd;
+
+ prog_opts.log_buf = log_buf;
+ prog_opts.log_size = sizeof(log_buf);
+ prog_opts.log_level = 1;
+
+ map_fd = create_rdonly_percpu_array();
+ if (map_fd < 0)
+ return;
+
+ insns[0].imm = map_fd;
+ prog_fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, "percpu_ro_prog", "GPL", insns,
+ ARRAY_SIZE(insns), &prog_opts);
+ if (!ASSERT_LT(prog_fd, 0, "bpf_prog_load"))
+ close(prog_fd);
+ else
+ ASSERT_HAS_SUBSTR(log_buf, "write into map forbidden", "verifier log");
+ close(map_fd);
+}
+
+static void test_global_percpu_data_verifier_log(void)
+{
+ RUN_TESTS(test_global_percpu_data);
+}
+
+static void test_global_percpu_data_iter(void)
+{
+ DECLARE_LIBBPF_OPTS(bpf_iter_attach_opts, opts);
+ struct test_global_percpu_data *skel;
+ union bpf_iter_link_info linfo = {};
+ struct bpf_link *link = NULL;
+ int fd, num_cpus, len, err;
+ char buf[16];
+
+ num_cpus = libbpf_num_possible_cpus();
+ if (!ASSERT_GT(num_cpus, 0, "libbpf_num_possible_cpus"))
+ return;
+
+ skel = test_global_percpu_data__open();
+ if (!ASSERT_OK_PTR(skel, "test_global_percpu_data__open"))
+ return;
+
+ skel->rodata->num_cpus = num_cpus;
+ skel->rodata->num_off = offsetof(struct test_global_percpu_data__percpu,
+ struct_data.nums[6]);
+ skel->rodata->elem_sz = roundup(sizeof(struct test_global_percpu_data__percpu), 8);
+ skel->percpu->struct_data.nums[6] = 0xc0de;
+
+ err = test_global_percpu_data__load(skel);
+ if (!ASSERT_OK(err, "test_global_percpu_data__load"))
+ goto out;
+
+ linfo.map.map_fd = bpf_map__fd(skel->maps.percpu);
+ opts.link_info = &linfo;
+ opts.link_info_len = sizeof(linfo);
+ link = bpf_program__attach_iter(skel->progs.dump_percpu_data, &opts);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_iter"))
+ goto out;
+
+ fd = bpf_iter_create(bpf_link__fd(link));
+ if (!ASSERT_GE(fd, 0, "bpf_iter_create"))
+ goto out;
+
+ while ((len = read(fd, buf, sizeof(buf))) > 0)
+ do { } while (0);
+ ASSERT_EQ(len, 0, "read iter");
+ ASSERT_TRUE(skel->bss->run_iter, "run_iter");
+ ASSERT_EQ(skel->bss->sum, 0xc0de * num_cpus, "sum");
+
+ close(fd);
+out:
+ bpf_link__destroy(link);
+ test_global_percpu_data__destroy(skel);
+}
+
+void test_global_percpu_data(void)
+{
+ if (!feat_supported(NULL, FEAT_PERCPU_DATA)) {
+ test__skip();
+ return;
+ }
+
+ if (test__start_subtest("init"))
+ test_global_percpu_data_init();
+ if (test__start_subtest("lskel"))
+ test_global_percpu_data_lskel();
+ if (test__start_subtest("rdonly_direct_read"))
+ test_global_percpu_data_rdonly_direct_read();
+ if (test__start_subtest("rdonly_direct_write"))
+ test_global_percpu_data_rdonly_direct_write();
+ test_global_percpu_data_verifier_log();
+ if (test__start_subtest("iter"))
+ test_global_percpu_data_iter();
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c b/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c
index 7def158da9eb..1215d6edd590 100644
--- a/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c
+++ b/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c
@@ -14,7 +14,8 @@ void test_kptr_xchg_inline(void)
int err;
#if !(defined(__x86_64__) || defined(__aarch64__) || \
- (defined(__riscv) && __riscv_xlen == 64))
+ (defined(__riscv) && __riscv_xlen == 64) || \
+ (defined(__loongarch__) && __loongarch_grlen == 64))
test__skip();
return;
#endif
diff --git a/tools/testing/selftests/bpf/prog_tests/ksock.c b/tools/testing/selftests/bpf/prog_tests/ksock.c
new file mode 100644
index 000000000000..dd6b167623d9
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/ksock.c
@@ -0,0 +1,130 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Isovalent */
+
+#include <arpa/inet.h>
+
+#include "test_progs.h"
+#include "network_helpers.h"
+#include "ksock_lsm.skel.h"
+#include "ksock_lsm_verifier.skel.h"
+
+#define NS_TEST "ksock_lsm_ns"
+#define RECV_PORT 7777
+#define RECV_TIMEOUT_SEC 5
+
+struct ksock_test_env {
+ struct nstoken *nstoken;
+ int rfd;
+};
+
+static bool ksock_test_env_setup(struct ksock_test_env *env)
+{
+ struct sockaddr_in addr = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ .sin_port = htons(RECV_PORT),
+ };
+ struct timeval tv = { .tv_sec = RECV_TIMEOUT_SEC };
+ int err;
+
+ memset(env, 0, sizeof(*env));
+ env->rfd = -1;
+
+ if (!ASSERT_OK(make_netns(NS_TEST), "make_netns"))
+ goto fail;
+
+ env->nstoken = open_netns(NS_TEST);
+ if (!ASSERT_OK_PTR(env->nstoken, "open_netns"))
+ goto fail;
+
+ env->rfd = socket(AF_INET, SOCK_DGRAM, IPPROTO_UDP);
+ if (!ASSERT_OK_FD(env->rfd, "receiver socket"))
+ goto fail;
+
+ err = bind(env->rfd, (struct sockaddr *)&addr, sizeof(addr));
+ if (!ASSERT_OK(err, "bind receiver"))
+ goto fail;
+
+ err = setsockopt(env->rfd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv));
+ if (!ASSERT_OK(err, "set rcvtimeo"))
+ goto fail;
+
+ return true;
+
+fail:
+ return false;
+}
+
+void test_ksock_lsm(void)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, opts);
+ struct ksock_test_env env;
+ struct sockaddr_in trigger_addr = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ };
+ struct ksock_lsm *skel;
+ char recv_data[sizeof(skel->data->send_data)] = {};
+ ssize_t n;
+ int tfd = -1;
+ int err;
+
+ skel = ksock_lsm__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "skel open_and_load"))
+ return;
+
+ if (!ksock_test_env_setup(&env))
+ goto fail;
+
+ /* Step 1: Run the setup SYSCALL prog to create the ksock */
+ skel->bss->ipv4_remote = htonl(INADDR_LOOPBACK);
+ skel->bss->remote_port = RECV_PORT;
+ err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.ksock_setup),
+ &opts);
+ if (!ASSERT_OK(err, "ksock_setup run"))
+ goto fail;
+ if (!ASSERT_OK(opts.retval, "ksock_setup retval"))
+ goto fail;
+
+ /* Step 2: Attach LSM prog and trigger socket_bind from userspace */
+ skel->links.ksock_socket_bind =
+ bpf_program__attach_lsm(skel->progs.ksock_socket_bind);
+ if (!ASSERT_OK_PTR(skel->links.ksock_socket_bind,
+ "attach socket_bind lsm"))
+ goto fail;
+
+ tfd = socket(AF_INET, SOCK_DGRAM, IPPROTO_UDP);
+ if (!ASSERT_OK_FD(tfd, "trigger socket"))
+ goto fail;
+
+ skel->bss->target_pid = getpid();
+ err = bind(tfd, (struct sockaddr *)&trigger_addr, sizeof(trigger_addr));
+ skel->bss->target_pid = 0;
+ if (!ASSERT_OK(err, "trigger bind"))
+ goto fail;
+
+ /* Step 3: Verify the LSM hook sent the notification */
+ if (!ASSERT_EQ(skel->data->send_ret, sizeof(skel->data->send_data),
+ "LSM send bytes"))
+ goto fail;
+
+ n = recvfrom(env.rfd, recv_data, sizeof(recv_data), 0, NULL, NULL);
+ if (ASSERT_EQ(n, sizeof(recv_data), "recvfrom len"))
+ ASSERT_MEMEQ(recv_data, skel->data->send_data, sizeof(recv_data),
+ "payload match");
+
+fail:
+ if (tfd >= 0)
+ close(tfd);
+ if (env.rfd >= 0)
+ close(env.rfd);
+ if (env.nstoken)
+ close_netns(env.nstoken);
+ remove_netns(NS_TEST);
+ ksock_lsm__destroy(skel);
+}
+
+void test_ksock_lsm_verifier(void)
+{
+ RUN_TESTS(ksock_lsm_verifier);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/ksock_wq.c b/tools/testing/selftests/bpf/prog_tests/ksock_wq.c
new file mode 100644
index 000000000000..d6dc20b8f95b
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/ksock_wq.c
@@ -0,0 +1,45 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Isovalent */
+
+#include <unistd.h>
+
+#include "test_progs.h"
+#include "ksock_wq.skel.h"
+
+#define CALLBACK_WAIT_RETRIES 1000
+#define CALLBACK_WAIT_US 1000
+
+void test_ksock_wq(void)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, opts);
+ struct ksock_wq *skel;
+ u32 callback_done;
+ int err, i;
+
+ skel = ksock_wq__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "ksock_wq open and load"))
+ return;
+
+ err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.ksock_wq_start),
+ &opts);
+ if (!ASSERT_OK(err, "run ksock_wq_start"))
+ goto out;
+ if (!ASSERT_OK(opts.retval, "ksock_wq_start retval"))
+ goto out;
+
+ for (i = 0; i < CALLBACK_WAIT_RETRIES; i++) {
+ if (__atomic_load_n(&skel->bss->callback_done, __ATOMIC_ACQUIRE))
+ break;
+ usleep(CALLBACK_WAIT_US);
+ }
+ callback_done = __atomic_load_n(&skel->bss->callback_done,
+ __ATOMIC_ACQUIRE);
+ if (!ASSERT_EQ(callback_done, 1, "workqueue callback completed"))
+ goto out;
+
+ ASSERT_EQ(skel->bss->create_err, -EOPNOTSUPP,
+ "workqueue create rejected");
+
+out:
+ ksock_wq__destroy(skel);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/linked_list.c b/tools/testing/selftests/bpf/prog_tests/linked_list.c
index c3d133c6a00d..52fabbee3dd5 100644
--- a/tools/testing/selftests/bpf/prog_tests/linked_list.c
+++ b/tools/testing/selftests/bpf/prog_tests/linked_list.c
@@ -714,7 +714,7 @@ static void test_btf(void)
break;
err = btf__load_into_kernel(btf);
- ASSERT_EQ(err, -ELOOP, "check btf");
+ ASSERT_EQ(err, 0, "check btf");
btf__free(btf);
break;
}
@@ -773,7 +773,7 @@ static void test_btf(void)
break;
err = btf__load_into_kernel(btf);
- ASSERT_EQ(err, -ELOOP, "check btf");
+ ASSERT_EQ(err, 0, "check btf");
btf__free(btf);
break;
}
diff --git a/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c b/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c
new file mode 100644
index 000000000000..a487aa68f2ee
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c
@@ -0,0 +1,296 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */
+
+#include <bpf/btf.h>
+#include <linux/btf.h>
+#include <test_progs.h>
+
+#define SPIN_LOCK 2
+#define LIST_HEAD 3
+#define LIST_NODE 4
+/* Keep in sync with BTF_MAX_OWNERSHIP_DEPTH. */
+#define MAX_OWNERSHIP_DEPTH 8
+
+static struct btf *init_btf(void)
+{
+ struct btf *btf;
+ int id;
+
+ btf = btf__new_empty();
+ if (!ASSERT_OK_PTR(btf, "btf__new_empty"))
+ return NULL;
+ id = btf__add_int(btf, "int", 4, BTF_INT_SIGNED);
+ if (!ASSERT_EQ(id, 1, "btf__add_int"))
+ goto err_out;
+ id = btf__add_struct(btf, "bpf_spin_lock", 4);
+ if (!ASSERT_EQ(id, SPIN_LOCK, "btf__add_struct bpf_spin_lock"))
+ goto err_out;
+ id = btf__add_struct(btf, "bpf_list_head", 16);
+ if (!ASSERT_EQ(id, LIST_HEAD, "btf__add_struct bpf_list_head"))
+ goto err_out;
+ id = btf__add_struct(btf, "bpf_list_node", 24);
+ if (!ASSERT_EQ(id, LIST_NODE, "btf__add_struct bpf_list_node"))
+ goto err_out;
+ return btf;
+
+err_out:
+ btf__free(btf);
+ return NULL;
+}
+
+static int add_local_kptr(struct btf *btf, int pointee_id, const char *tag)
+{
+ int id;
+
+ id = btf__add_type_tag(btf, tag, pointee_id);
+ if (!ASSERT_GT(id, 0, "btf__add_type_tag"))
+ return id;
+ id = btf__add_ptr(btf, id);
+ ASSERT_GT(id, 0, "btf__add_ptr");
+ return id;
+}
+
+static void test_self_cycle(const char *tag, int expected_err)
+{
+ struct btf *btf;
+ int id, err;
+
+ btf = init_btf();
+ if (!ASSERT_OK_PTR(btf, "init_btf"))
+ return;
+ id = add_local_kptr(btf, 7, tag);
+ if (id <= 0)
+ goto out;
+ id = btf__add_struct(btf, "self_cycle", 8);
+ if (!ASSERT_EQ(id, 7, "btf__add_struct self_cycle"))
+ goto out;
+ err = btf__add_field(btf, "next", 6, 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field self_cycle::next"))
+ goto out;
+
+ err = btf__load_into_kernel(btf);
+ ASSERT_EQ(err, expected_err, "check btf");
+out:
+ btf__free(btf);
+}
+
+static void test_aba_cycle(void)
+{
+ struct btf *btf;
+ int id, err;
+
+ btf = init_btf();
+ if (!ASSERT_OK_PTR(btf, "init_btf"))
+ return;
+ id = add_local_kptr(btf, 10, "kptr");
+ if (id <= 0)
+ goto out;
+ id = add_local_kptr(btf, 9, "kptr");
+ if (id <= 0)
+ goto out;
+ id = btf__add_struct(btf, "cycle_a", 8);
+ if (!ASSERT_EQ(id, 9, "btf__add_struct cycle_a"))
+ goto out;
+ err = btf__add_field(btf, "b", 6, 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field cycle_a::b"))
+ goto out;
+ id = btf__add_struct(btf, "cycle_b", 8);
+ if (!ASSERT_EQ(id, 10, "btf__add_struct cycle_b"))
+ goto out;
+ err = btf__add_field(btf, "a", 8, 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field cycle_b::a"))
+ goto out;
+
+ err = btf__load_into_kernel(btf);
+ ASSERT_EQ(err, -ELOOP, "check btf");
+out:
+ btf__free(btf);
+}
+
+static void test_mixed_cycle(void)
+{
+ struct btf *btf;
+ int id, err;
+
+ btf = init_btf();
+ if (!ASSERT_OK_PTR(btf, "init_btf"))
+ return;
+ id = add_local_kptr(btf, 7, "kptr");
+ if (id <= 0)
+ goto out;
+ id = btf__add_struct(btf, "mixed_owner", 20);
+ if (!ASSERT_EQ(id, 7, "btf__add_struct mixed_owner"))
+ goto out;
+ err = btf__add_field(btf, "root", LIST_HEAD, 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field mixed_owner::root"))
+ goto out;
+ err = btf__add_field(btf, "lock", SPIN_LOCK, 128, 0);
+ if (!ASSERT_OK(err, "btf__add_field mixed_owner::lock"))
+ goto out;
+ id = btf__add_decl_tag(btf, "contains:mixed_node:node", 7, 0);
+ if (!ASSERT_EQ(id, 8, "btf__add_decl_tag mixed_owner"))
+ goto out;
+ id = btf__add_struct(btf, "mixed_node", 32);
+ if (!ASSERT_EQ(id, 9, "btf__add_struct mixed_node"))
+ goto out;
+ err = btf__add_field(btf, "node", LIST_NODE, 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field mixed_node::node"))
+ goto out;
+ err = btf__add_field(btf, "owner", 6, 192, 0);
+ if (!ASSERT_OK(err, "btf__add_field mixed_node::owner"))
+ goto out;
+
+ err = btf__load_into_kernel(btf);
+ ASSERT_EQ(err, -ELOOP, "check btf");
+out:
+ btf__free(btf);
+}
+
+static void test_acyclic_depth(int depth, bool child_first, bool shared_suffix, int expected_err)
+{
+ int ptr_id[MAX_OWNERSHIP_DEPTH + 1];
+ int first_struct_id;
+ struct btf *btf;
+ int id, err, i, n, pointee_id;
+
+ btf = init_btf();
+ if (!ASSERT_OK_PTR(btf, "init_btf"))
+ return;
+ first_struct_id = 5 + 2 * depth;
+ for (i = 0; i < depth; i++) {
+ if (i == depth - 1)
+ pointee_id = first_struct_id + depth;
+ else
+ pointee_id = first_struct_id + (child_first ? depth - 2 - i : i + 1);
+ ptr_id[i] = add_local_kptr(btf, pointee_id, "kptr");
+ if (ptr_id[i] <= 0)
+ goto out;
+ }
+ for (n = 0; n < depth; n++) {
+ char name[32];
+ int offset = 0;
+
+ i = child_first ? depth - 1 - n : n;
+ snprintf(name, sizeof(name), "owner_%d", i);
+ id = btf__add_struct(btf, name, shared_suffix && !i ? 16 : 8);
+ if (!ASSERT_EQ(id, first_struct_id + n, "btf__add_struct owner"))
+ goto out;
+ if (shared_suffix && !i) {
+ /*
+ * Visit the shared suffix through the shorter path before
+ * reaching it again with less remaining depth.
+ */
+ err = btf__add_field(btf, "suffix", ptr_id[1], 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field owner::suffix"))
+ goto out;
+ offset = 64;
+ }
+ err = btf__add_field(btf, "next", ptr_id[i], offset, 0);
+ if (!ASSERT_OK(err, "btf__add_field owner::next"))
+ goto out;
+ }
+ id = btf__add_struct(btf, "plain_leaf", 4);
+ if (!ASSERT_EQ(id, first_struct_id + depth, "btf__add_struct plain_leaf"))
+ goto out;
+
+ err = btf__load_into_kernel(btf);
+ ASSERT_EQ(err, expected_err, "check btf");
+out:
+ btf__free(btf);
+}
+
+static void test_graph_depth(bool rbtree, int depth, int expected_err)
+{
+ int root_type = LIST_HEAD, node_type = LIST_NODE, node_size = 24;
+ int id, err, i, lock_off, root_off, size;
+ struct btf *btf;
+
+ btf = init_btf();
+ if (!ASSERT_OK_PTR(btf, "init_btf"))
+ return;
+ if (rbtree) {
+ root_type = btf__add_struct(btf, "bpf_rb_root", 16);
+ if (!ASSERT_GT(root_type, 0, "btf__add_struct bpf_rb_root"))
+ goto out;
+ node_type = btf__add_struct(btf, "bpf_rb_node", 32);
+ if (!ASSERT_GT(node_type, 0, "btf__add_struct bpf_rb_node"))
+ goto out;
+ node_size = 32;
+ }
+
+ for (i = 0; i < depth; i++) {
+ char name[32], tag[64];
+
+ lock_off = i ? node_size : 0;
+ root_off = lock_off + 8;
+ size = i == depth - 1 ? node_size : root_off + 16;
+ snprintf(name, sizeof(name), "graph_owner_%d", i);
+ id = btf__add_struct(btf, name, size);
+ if (!ASSERT_GT(id, 0, "btf__add_struct graph_owner"))
+ goto out;
+ if (i) {
+ err = btf__add_field(btf, "node", node_type, 0, 0);
+ if (!ASSERT_OK(err, "btf__add_field graph_owner::node"))
+ goto out;
+ }
+ if (i == depth - 1)
+ continue;
+ err = btf__add_field(btf, "lock", SPIN_LOCK, lock_off * 8, 0);
+ if (!ASSERT_OK(err, "btf__add_field graph_owner::lock"))
+ goto out;
+ err = btf__add_field(btf, "root", root_type, root_off * 8, 0);
+ if (!ASSERT_OK(err, "btf__add_field graph_owner::root"))
+ goto out;
+ snprintf(tag, sizeof(tag), "contains:graph_owner_%d:node", i + 1);
+ err = btf__add_decl_tag(btf, tag, id, i ? 2 : 1);
+ if (!ASSERT_GT(err, 0, "btf__add_decl_tag graph_owner"))
+ goto out;
+ }
+
+ err = btf__load_into_kernel(btf);
+ ASSERT_EQ(err, expected_err, "check btf");
+out:
+ btf__free(btf);
+}
+
+void test_local_kptr_ownership(void)
+{
+ if (test__start_subtest("self_cycle"))
+ test_self_cycle("kptr", -ELOOP);
+ if (test__start_subtest("untrusted_self_cycle"))
+ test_self_cycle("kptr_untrusted", 0);
+ if (test__start_subtest("percpu_self_cycle"))
+ test_self_cycle("percpu_kptr", -ELOOP);
+ if (test__start_subtest("ABA_cycle"))
+ test_aba_cycle();
+ if (test__start_subtest("mixed_graph_root_cycle"))
+ test_mixed_cycle();
+ if (test__start_subtest("max_acyclic"))
+ test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, false, 0);
+ if (test__start_subtest("too_deep_acyclic"))
+ test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, false, -ELOOP);
+ if (test__start_subtest("max_acyclic_child_first"))
+ test_acyclic_depth(MAX_OWNERSHIP_DEPTH, true, false, 0);
+ if (test__start_subtest("too_deep_acyclic_child_first"))
+ test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, true, false, -ELOOP);
+ if (test__start_subtest("max_acyclic_shared_suffix"))
+ test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, true, 0);
+ if (test__start_subtest("too_deep_acyclic_shared_suffix"))
+ test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, true, -ELOOP);
+ if (test__start_subtest("list_three_types"))
+ test_graph_depth(false, 3, 0);
+ if (test__start_subtest("list_four_types"))
+ test_graph_depth(false, 4, 0);
+ if (test__start_subtest("list_max_depth"))
+ test_graph_depth(false, MAX_OWNERSHIP_DEPTH, 0);
+ if (test__start_subtest("list_too_deep"))
+ test_graph_depth(false, MAX_OWNERSHIP_DEPTH + 1, -ELOOP);
+ if (test__start_subtest("rbtree_three_types"))
+ test_graph_depth(true, 3, 0);
+ if (test__start_subtest("rbtree_four_types"))
+ test_graph_depth(true, 4, 0);
+ if (test__start_subtest("rbtree_max_depth"))
+ test_graph_depth(true, MAX_OWNERSHIP_DEPTH, 0);
+ if (test__start_subtest("rbtree_too_deep"))
+ test_graph_depth(true, MAX_OWNERSHIP_DEPTH + 1, -ELOOP);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/lwt_ip_encap.c b/tools/testing/selftests/bpf/prog_tests/lwt_ip_encap.c
index 6606f0ed9a9a..39e8a3b8b6af 100644
--- a/tools/testing/selftests/bpf/prog_tests/lwt_ip_encap.c
+++ b/tools/testing/selftests/bpf/prog_tests/lwt_ip_encap.c
@@ -410,7 +410,8 @@ close_netns:
static int check_ping_ok(const char *ns1)
{
SYS(fail, "ip netns exec %s ping -c 1 -W1 -I veth1 %s > /dev/null", ns1, IP4_ADDR_DST);
- SYS(fail, "ip netns exec %s ping6 -c 1 -W1 -I veth1 %s > /dev/null", ns1, IP6_ADDR_DST);
+ SYS(fail, "ip netns exec %s %s -c 1 -W1 -I veth1 %s > /dev/null", ns1,
+ ping_command(AF_INET6), IP6_ADDR_DST);
return 0;
fail:
return -1;
@@ -424,7 +425,8 @@ static int check_ping_fails(const char *ns1)
if (!ret)
return -1;
- ret = SYS_NOFAIL("ip netns exec %s ping6 -c 1 -W1 -I veth1 %s", ns1, IP6_ADDR_DST);
+ ret = SYS_NOFAIL("ip netns exec %s %s -c 1 -W1 -I veth1 %s", ns1,
+ ping_command(AF_INET6), IP6_ADDR_DST);
if (!ret)
return -1;
@@ -657,9 +659,10 @@ static void lwt_ip_encap_vxlan(bool ipv4_encap)
skel->bss->fexit_triggered = false;
if (ipv4_encap)
- SYS(out, "ip netns exec %s ping -c 1 -W1 %s", ns1, IP4_ADDR_DST);
+ SYS(out, "ip netns exec %s ping -c 1 -W1 %s", ns1, IP4_ADDR_DST);
else
- SYS(out, "ip netns exec %s ping6 -c 1 -W1 %s", ns1, IP6_ADDR_DST);
+ SYS(out, "ip netns exec %s %s -c 1 -W1 %s", ns1,
+ ping_command(AF_INET6), IP6_ADDR_DST);
if (!ASSERT_TRUE(skel->bss->fexit_triggered, "fexit_triggered"))
goto out;
diff --git a/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c b/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c
index a72ae0b29f6e..7b4a1e24363b 100644
--- a/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c
+++ b/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c
@@ -1,4 +1,6 @@
// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <sched.h>
#include <test_progs.h>
#include "cgroup_helpers.h"
#include "percpu_alloc_array.skel.h"
@@ -350,6 +352,103 @@ static void test_lru_percpu_hash_cpu_flag(void)
test_percpu_map_cpu_flag(BPF_MAP_TYPE_LRU_PERCPU_HASH);
}
+/*
+ * A BPF_F_CPU update that creates an element must zero the value on the other
+ * cpus, rather than leave them holding whatever the recycled element last
+ * contained. max_entries is 1 so the second key can only reuse the element
+ * the first one released.
+ */
+static void test_percpu_map_cpu_flag_create(enum bpf_map_type map_type, __u32 map_flags)
+{
+ LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = map_flags);
+ const u32 stale = 0xDEADC0DE, fresh = 0xC0FFEE;
+ int nr_cpus, cpu, map_fd, err, key;
+ int pinned_cpu, value_cpu;
+ cpu_set_t old_mask, new_mask;
+ bool restore_mask = false;
+ u32 value;
+ u64 flags;
+
+ nr_cpus = libbpf_num_possible_cpus();
+ if (!ASSERT_GT(nr_cpus, 0, "libbpf_num_possible_cpus"))
+ return;
+
+ if (nr_cpus < 2) {
+ test__skip();
+ return;
+ }
+
+ map_fd = bpf_map_create(map_type, "cpu_flag_create", sizeof(key), sizeof(value), 1, &opts);
+ if (!ASSERT_GE(map_fd, 0, "bpf_map_create"))
+ return;
+
+ /* NO_PREALLOC recycles per cpu, so keep the delete and the create on one cpu. */
+ err = sched_getaffinity(0, sizeof(old_mask), &old_mask);
+ if (!ASSERT_OK(err, "sched_getaffinity"))
+ goto out;
+
+ pinned_cpu = sched_getcpu();
+ if (!ASSERT_GE(pinned_cpu, 0, "sched_getcpu"))
+ goto out;
+
+ CPU_ZERO(&new_mask);
+ CPU_SET(pinned_cpu, &new_mask);
+ err = sched_setaffinity(0, sizeof(new_mask), &new_mask);
+ if (!ASSERT_OK(err, "sched_setaffinity"))
+ goto out;
+ restore_mask = true;
+
+ value_cpu = pinned_cpu ? 0 : 1;
+
+ key = 1;
+ value = stale;
+ err = bpf_map_update_elem(map_fd, &key, &value, BPF_F_ALL_CPUS);
+ if (!ASSERT_OK(err, "bpf_map_update_elem all_cpus"))
+ goto out;
+
+ err = bpf_map_delete_elem(map_fd, &key);
+ if (!ASSERT_OK(err, "bpf_map_delete_elem"))
+ goto out;
+
+ key = 2;
+ value = fresh;
+ flags = (u64)value_cpu << 32 | BPF_F_CPU;
+ err = bpf_map_update_elem(map_fd, &key, &value, flags);
+ if (!ASSERT_OK(err, "bpf_map_update_elem specified cpu"))
+ goto out;
+
+ for (cpu = 0; cpu < nr_cpus; cpu++) {
+ value = 0;
+ flags = (u64)cpu << 32 | BPF_F_CPU;
+ err = bpf_map_lookup_elem_flags(map_fd, &key, &value, flags);
+ if (!ASSERT_OK(err, "bpf_map_lookup_elem_flags specified cpu"))
+ goto out;
+ if (!ASSERT_EQ(value, cpu == value_cpu ? fresh : 0, "value on specified cpu"))
+ goto out;
+ }
+
+out:
+ if (restore_mask)
+ sched_setaffinity(0, sizeof(old_mask), &old_mask);
+ close(map_fd);
+}
+
+static void test_percpu_hash_cpu_flag_create(void)
+{
+ test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, 0);
+}
+
+static void test_percpu_hash_cpu_flag_create_malloc(void)
+{
+ test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, BPF_F_NO_PREALLOC);
+}
+
+static void test_lru_percpu_hash_cpu_flag_create(void)
+{
+ /* lru without prealloc is -ENOTSUPP, so there is no malloc variant */
+ test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_LRU_PERCPU_HASH, 0);
+}
+
static void test_percpu_cgroup_storage_cpu_flag(void)
{
struct percpu_alloc_array *skel = NULL;
@@ -454,6 +553,12 @@ void test_percpu_alloc(void)
test_percpu_hash_cpu_flag();
if (test__start_subtest("cpu_flag_lru_percpu_hash"))
test_lru_percpu_hash_cpu_flag();
+ if (test__start_subtest("cpu_flag_create_percpu_hash"))
+ test_percpu_hash_cpu_flag_create();
+ if (test__start_subtest("cpu_flag_create_percpu_hash_malloc"))
+ test_percpu_hash_cpu_flag_create_malloc();
+ if (test__start_subtest("cpu_flag_create_lru_percpu_hash"))
+ test_lru_percpu_hash_cpu_flag_create();
if (test__start_subtest("cpu_flag_percpu_cgroup_storage"))
test_percpu_cgroup_storage_cpu_flag();
if (test__start_subtest("cpu_flag_array"))
diff --git a/tools/testing/selftests/bpf/prog_tests/rcu_read_lock.c b/tools/testing/selftests/bpf/prog_tests/rcu_read_lock.c
index 246eb259c08a..6a07b2b418d1 100644
--- a/tools/testing/selftests/bpf/prog_tests/rcu_read_lock.c
+++ b/tools/testing/selftests/bpf/prog_tests/rcu_read_lock.c
@@ -34,6 +34,8 @@ static void test_success(void)
bpf_program__set_autoload(skel->progs.rcu_read_lock_global_subprog, true);
bpf_program__set_autoload(skel->progs.rcu_read_lock_subprog_lock, true);
bpf_program__set_autoload(skel->progs.rcu_read_lock_subprog_unlock, true);
+ bpf_program__set_autoload(skel->progs.non_own_ref_untrusted_ld, true);
+ bpf_program__set_autoload(skel->progs.rcu_untrusted_union_ld, true);
err = rcu_read_lock__load(skel);
if (!ASSERT_OK(err, "skel_load"))
goto out;
diff --git a/tools/testing/selftests/bpf/prog_tests/resolve_btfids.c b/tools/testing/selftests/bpf/prog_tests/resolve_btfids.c
index 732cfed35e1c..3f9949e8227d 100644
--- a/tools/testing/selftests/bpf/prog_tests/resolve_btfids.c
+++ b/tools/testing/selftests/bpf/prog_tests/resolve_btfids.c
@@ -15,6 +15,7 @@
#define DECL_TAG_FASTCALL "bpf_fastcall"
#define DECL_TAG_KFUNC "bpf_kfunc"
#define TYPE_ATTR_ARENA "address_space(1)"
+#define ARENA_ARG(n) (1U << (n))
#ifndef KF_FASTCALL
#define KF_FASTCALL (1 << 12)
@@ -49,13 +50,20 @@ struct kfunc_symbol {
const char *name;
s32 id;
u32 flags;
+ u32 arena_args;
+ bool arena_ret;
};
static struct kfunc_symbol kfunc_symbols[] = {
- { "kfunc_a", -1, 0 },
- { "kfunc_b", -1, KF_FASTCALL },
- { "kfunc_c", -1, KF_ARENA_RET | KF_ARENA_ARG1 | KF_ARENA_ARG2 },
- { "kfunc_d", -1, KF_ARENA_ARG2 },
+ { "kfunc_a", -1, 0, 0, false },
+ { "kfunc_b", -1, KF_FASTCALL, 0, false },
+ { "kfunc_c", -1, KF_ARENA_RET | KF_ARENA_ARG1 | KF_ARENA_ARG2,
+ ARENA_ARG(0) | ARENA_ARG(1), true },
+ { "kfunc_d", -1, KF_ARENA_ARG2, ARENA_ARG(1), false },
+ { "kfunc_e", -1, 0, ARENA_ARG(0) | ARENA_ARG(1) | ARENA_ARG(2) |
+ ARENA_ARG(3) | ARENA_ARG(4), false },
+ { "kfunc_f", -1, 0, ARENA_ARG(1), false },
+ { "kfunc_g", -1, KF_ARENA_RET, ARENA_ARG(0) | ARENA_ARG(1), true },
};
/* Align the .BTF_ids section to 4 bytes */
@@ -105,6 +113,9 @@ BTF_ID_FLAGS(func, kfunc_a)
BTF_ID_FLAGS(func, kfunc_b, KF_FASTCALL)
BTF_ID_FLAGS(func, kfunc_c, KF_ARENA_RET | KF_ARENA_ARG1 | KF_ARENA_ARG2)
BTF_ID_FLAGS(func, kfunc_d, KF_ARENA_ARG2)
+BTF_ID_FLAGS(func, kfunc_e)
+BTF_ID_FLAGS(func, kfunc_f)
+BTF_ID_FLAGS(func, kfunc_g, KF_ARENA_RET)
BTF_KFUNCS_END(test_kfunc_set)
/*
@@ -112,6 +123,9 @@ BTF_KFUNCS_END(test_kfunc_set)
* actually sort at least one of the two sets.
*/
BTF_KFUNCS_START(test_kfunc_set_rev)
+BTF_ID_FLAGS(func, kfunc_g, KF_ARENA_RET)
+BTF_ID_FLAGS(func, kfunc_f)
+BTF_ID_FLAGS(func, kfunc_e)
BTF_ID_FLAGS(func, kfunc_d, KF_ARENA_ARG2)
BTF_ID_FLAGS(func, kfunc_c, KF_ARENA_RET | KF_ARENA_ARG1 | KF_ARENA_ARG2)
BTF_ID_FLAGS(func, kfunc_b, KF_FASTCALL)
@@ -301,15 +315,15 @@ void test_resolve_btfids(void)
}
/*
- * Check resolve_btfids wrapped exactly the arena-flagged return/args
- * with the address_space(1) type attribute, and left other
+ * Check resolve_btfids wrapped exactly the arena-flagged or suffixed
+ * return/args with the address_space(1) type attribute, and left other
* pointers/returns untouched.
*/
for (i = 0; i < ARRAY_SIZE(kfunc_symbols); i++) {
const struct btf_type *fn, *proto;
const struct btf_param *params;
const char *name = kfunc_symbols[i].name;
- u32 fl = kfunc_symbols[i].flags;
+ u32 arena_args = kfunc_symbols[i].arena_args;
__u32 nr;
fn = btf__type_by_id(btf, kfunc_symbols[i].id);
@@ -322,15 +336,10 @@ void test_resolve_btfids(void)
nr = btf_vlen(proto);
ASSERT_EQ(is_arena_tagged_ptr(btf, proto->type),
- !!(fl & KF_ARENA_RET), name);
- if (nr > 0) {
- ASSERT_EQ(is_arena_tagged_ptr(btf, params[0].type),
- !!(fl & KF_ARENA_ARG1), name);
- }
- if (nr > 1) {
- ASSERT_EQ(is_arena_tagged_ptr(btf, params[1].type),
- !!(fl & KF_ARENA_ARG2), name);
- }
+ kfunc_symbols[i].arena_ret, name);
+ for (j = 0; j < nr; j++)
+ ASSERT_EQ(is_arena_tagged_ptr(btf, params[j].type),
+ !!(arena_args & ARENA_ARG(j)), name);
}
out:
diff --git a/tools/testing/selftests/bpf/prog_tests/rhash.c b/tools/testing/selftests/bpf/prog_tests/rhash.c
index 98bb66907b7f..0641bd5b0a9e 100644
--- a/tools/testing/selftests/bpf/prog_tests/rhash.c
+++ b/tools/testing/selftests/bpf/prog_tests/rhash.c
@@ -172,6 +172,12 @@ void test_rhash(void)
if (test__start_subtest("test_rhash_delete_nonexistent"))
rhash_run("test_rhash_delete_nonexistent");
+ if (test__start_subtest("test_rhash_kptr_update"))
+ rhash_run("test_rhash_kptr_update");
+
+ if (test__start_subtest("test_rhash_kptr_delete"))
+ rhash_run("test_rhash_kptr_delete");
+
if (test__start_subtest("test_rhash_map_extra_presize"))
rhash_map_extra_presize();
diff --git a/tools/testing/selftests/bpf/prog_tests/rhash_timer.c b/tools/testing/selftests/bpf/prog_tests/rhash_timer.c
new file mode 100644
index 000000000000..3aad9fc02e06
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/rhash_timer.c
@@ -0,0 +1,141 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#define _GNU_SOURCE
+#include <sched.h>
+
+#include <test_progs.h>
+#include "rhash_timer.skel.h"
+
+#define MAX_ATTEMPTS 256
+#define RCU_SYNC_INTERVAL 64
+
+static int pin_to_first_cpu(cpu_set_t *old_mask)
+{
+ cpu_set_t new_mask;
+ int cpu;
+
+ if (sched_getaffinity(0, sizeof(*old_mask), old_mask))
+ return -errno;
+
+ for (cpu = 0; cpu < CPU_SETSIZE; cpu++)
+ if (CPU_ISSET(cpu, old_mask))
+ break;
+ if (cpu == CPU_SETSIZE)
+ return -EINVAL;
+
+ CPU_ZERO(&new_mask);
+ CPU_SET(cpu, &new_mask);
+ if (sched_setaffinity(0, sizeof(new_mask), &new_mask))
+ return -errno;
+ return 0;
+}
+
+static int update_timer_map(int map_fd, __u64 key)
+{
+ __u64 value[3] = {};
+
+ return bpf_map_update_elem(map_fd, &key, value, BPF_NOEXIST);
+}
+
+static int run_prog(int prog_fd, struct bpf_test_run_opts *opts)
+{
+ int err;
+
+ err = bpf_prog_test_run_opts(prog_fd, opts);
+ if (err)
+ return err;
+ return opts->retval;
+}
+
+void test_rhash_timer(void)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, opts);
+ struct rhash_timer *skel = NULL;
+ cpu_set_t old_mask;
+ int map_fd = -1, arm_fd, cancel_fd;
+ bool affinity_set = false;
+ __u64 key = 1;
+ int attempt, err;
+
+ err = pin_to_first_cpu(&old_mask);
+ if (!ASSERT_OK(err, "pin_to_first_cpu"))
+ return;
+ affinity_set = true;
+
+ skel = rhash_timer__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ goto out;
+
+ map_fd = bpf_map__fd(skel->maps.timer_map);
+ if (!ASSERT_GE(map_fd, 0, "timer_map fd"))
+ goto out;
+ arm_fd = bpf_program__fd(skel->progs.arm_deleted_timer);
+ if (!ASSERT_GE(arm_fd, 0, "arm_deleted_timer fd"))
+ goto out;
+ cancel_fd = bpf_program__fd(skel->progs.cancel_recycled_timer);
+ if (!ASSERT_GE(cancel_fd, 0, "cancel_recycled_timer fd"))
+ goto out;
+
+ err = update_timer_map(map_fd, key);
+ if (!ASSERT_OK(err, "seed_timer_map"))
+ goto out;
+
+ for (attempt = 0; attempt < MAX_ATTEMPTS; attempt++) {
+ err = run_prog(arm_fd, &opts);
+ if (err) {
+ ASSERT_OK(err, "arm_deleted_timer");
+ goto out;
+ }
+ if (skel->bss->armed != attempt + 1) {
+ ASSERT_EQ(skel->bss->armed, attempt + 1, "armed");
+ goto out;
+ }
+ if (skel->bss->timer_init_err) {
+ ASSERT_OK(skel->bss->timer_init_err, "timer_init_err");
+ goto out;
+ }
+ if (skel->bss->timer_set_callback_err) {
+ ASSERT_OK(skel->bss->timer_set_callback_err,
+ "timer_set_callback_err");
+ goto out;
+ }
+ if (skel->bss->timer_start_err) {
+ ASSERT_OK(skel->bss->timer_start_err, "timer_start_err");
+ goto out;
+ }
+
+ if ((attempt + 1) % RCU_SYNC_INTERVAL == 0) {
+ err = kern_sync_rcu();
+ if (err) {
+ ASSERT_OK(err, "kern_sync_rcu");
+ goto out;
+ }
+ }
+
+ err = update_timer_map(map_fd, ++key);
+ if (err) {
+ ASSERT_OK(err, "replace_timer_map");
+ goto out;
+ }
+
+ err = run_prog(cancel_fd, &opts);
+ if (err) {
+ ASSERT_OK(err, "cancel_recycled_timer");
+ goto out;
+ }
+ if (skel->bss->timer_cancel_err) {
+ ASSERT_OK(skel->bss->timer_cancel_err, "timer_cancel_err");
+ goto out;
+ }
+ if (skel->bss->cancelled)
+ break;
+ }
+
+ ASSERT_GT(skel->bss->cancelled, 0, "preserved timer");
+out:
+ if (map_fd >= 0)
+ bpf_map_delete_elem(map_fd, &key);
+ rhash_timer__destroy(skel);
+ if (affinity_set)
+ sched_setaffinity(0, sizeof(old_mask), &old_mask);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/snprintf_btf.c b/tools/testing/selftests/bpf/prog_tests/snprintf_btf.c
index dd41b826be30..edce9c1b54fb 100644
--- a/tools/testing/selftests/bpf/prog_tests/snprintf_btf.c
+++ b/tools/testing/selftests/bpf/prog_tests/snprintf_btf.c
@@ -1,7 +1,9 @@
// SPDX-License-Identifier: GPL-2.0
#include <test_progs.h>
#include <linux/btf.h>
+#include <bpf/btf.h>
#include "netif_receive_skb.skel.h"
+#include "snprintf_btf_void.skel.h"
/* Demonstrate that bpf_snprintf_btf succeeds and that various data types
* are formatted correctly.
@@ -58,3 +60,80 @@ void serial_test_snprintf_btf(void)
cleanup:
netif_receive_skb__destroy(skel);
}
+
+/*
+ * bpf_snprintf_btf() renders a type_id taken straight from the vmlinux BTF.
+ * Two such type_ids used to NULL-deref in the BTF show path:
+ * - a "const void" (a modifier resolving to void) in btf_modifier_show()
+ * - a BTF_KIND_VAR in btf_var_show() (base BTF has no resolved_ids)
+ * A fixed kernel renders both without crashing.
+ */
+static long run(struct snprintf_btf_void *skel, __u32 type_id)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, topts);
+ char ctx[8] = {};
+
+ skel->bss->type_id = type_id;
+ topts.ctx_in = ctx;
+ topts.ctx_size_in = sizeof(ctx);
+ if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(skel->progs.dump_type),
+ &topts), "test_run"))
+ return -1;
+ return skel->bss->ret;
+}
+
+void test_snprintf_btf_void(void)
+{
+ const struct btf_type *t;
+ struct snprintf_btf_void *skel;
+ int i, n, cv = 0, var = 0;
+ struct btf *btf;
+
+ btf = btf__parse("/sys/kernel/btf/vmlinux", NULL);
+ if (!btf) {
+ test__skip();
+ return;
+ }
+
+ skel = snprintf_btf_void__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "skel_open_and_load"))
+ goto out_btf;
+
+ n = btf__type_cnt(btf);
+ for (i = 1; i < n && !(cv && var); i++) {
+ t = btf__type_by_id(btf, i);
+ if (!cv && btf_kind(t) == BTF_KIND_CONST && t->type == 0)
+ cv = i;
+ /* Pick a VAR small enough to render from the program's buffer. */
+ if (!var && btf_kind(t) == BTF_KIND_VAR) {
+ long sz = btf__resolve_size(btf, t->type);
+
+ if (sz > 0 && sz <= (long)sizeof(skel->bss->obj))
+ var = i;
+ }
+ }
+
+ /* "const void" renders the "<unsupported kind:0>" placeholder. */
+ if (test__start_subtest("const_void")) {
+ if (cv) {
+ ASSERT_EQ(run(skel, cv),
+ sizeof("<unsupported kind:0>") - 1, "ret");
+ ASSERT_STREQ(skel->bss->out, "<unsupported kind:0>",
+ "placeholder");
+ } else {
+ test__skip();
+ }
+ }
+
+ /* A BTF_KIND_VAR must resolve and render without error. */
+ if (test__start_subtest("var")) {
+ if (var)
+ ASSERT_GT(run(skel, var), 0, "ret");
+ else
+ test__skip();
+ }
+
+ snprintf_btf_void__destroy(skel);
+out_btf:
+ btf__free(btf);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/sock_destroy.c b/tools/testing/selftests/bpf/prog_tests/sock_destroy.c
index 9c11938fe597..78d642a02bdb 100644
--- a/tools/testing/selftests/bpf/prog_tests/sock_destroy.c
+++ b/tools/testing/selftests/bpf/prog_tests/sock_destroy.c
@@ -1,4 +1,5 @@
// SPDX-License-Identifier: GPL-2.0
+#include <poll.h>
#include <test_progs.h>
#include <bpf/bpf_endian.h>
@@ -110,6 +111,122 @@ cleanup:
close(serv);
}
+static void test_tcp_listen_pending(struct sock_destroy_prog *skel)
+{
+ int serv = -1, clien = -1, accept_serv = -1, n, serv_port;
+ struct pollfd pfd = { .events = POLLIN };
+ char buf[1];
+
+ serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0);
+ if (!ASSERT_GE(serv, 0, "start_server"))
+ goto cleanup;
+ serv_port = get_socket_local_port(serv);
+ if (!ASSERT_GE(serv_port, 0, "get_sock_local_port"))
+ goto cleanup;
+ skel->bss->serv_port = (__be16)serv_port;
+
+ /*
+ * Connect but never accept, so the child sits in the accept queue
+ * of the listener. Wait until it's actually there.
+ */
+ clien = connect_to_fd(serv, 0);
+ if (!ASSERT_GE(clien, 0, "connect_to_fd"))
+ goto cleanup;
+ pfd.fd = serv;
+ if (!ASSERT_EQ(poll(&pfd, 1, -1), 1, "poll listener"))
+ goto cleanup;
+
+ /* Run iterator program that destroys server sockets. */
+ start_iter_sockets(skel->progs.iter_tcp6_server);
+
+ accept_serv = accept(serv, NULL, NULL);
+ if (!ASSERT_LT(accept_serv, 0, "accept on destroyed listener"))
+ goto cleanup;
+ ASSERT_EQ(errno, EINVAL, "error code on destroyed listener");
+
+ /* The unaccepted child was reset along with the listener. */
+ n = recv(clien, buf, sizeof(buf), 0);
+ if (!ASSERT_LT(n, 0, "client recv on reset child"))
+ goto cleanup;
+ ASSERT_EQ(errno, ECONNRESET, "error code on reset child");
+
+cleanup:
+ if (clien != -1)
+ close(clien);
+ if (accept_serv != -1)
+ close(accept_serv);
+ if (serv != -1)
+ close(serv);
+}
+
+static void test_tcp_timewait(struct sock_destroy_prog *skel)
+{
+ int serv = -1, clien = -1, accept_serv = -1, n;
+ struct timeval tv = {};
+ char buf[1];
+
+ serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0);
+ if (!ASSERT_GE(serv, 0, "start_server"))
+ goto cleanup;
+
+ clien = connect_to_fd(serv, 0);
+ if (!ASSERT_GE(clien, 0, "connect_to_fd"))
+ goto cleanup;
+
+ accept_serv = accept(serv, NULL, NULL);
+ if (!ASSERT_GE(accept_serv, 0, "serv accept"))
+ goto cleanup;
+
+ /*
+ * Active close from the client, then close the server side. Once
+ * recv() sees EOF the server FIN has been processed and the client
+ * sock is in TIME_WAIT. Block without timeout so a loaded CI box
+ * can't race us.
+ */
+ if (!ASSERT_OK(setsockopt(clien, SOL_SOCKET, SO_RCVTIMEO, &tv,
+ sizeof(tv)), "clear rcvtimeo"))
+ goto cleanup;
+ if (!ASSERT_OK(shutdown(clien, SHUT_WR), "client shutdown"))
+ goto cleanup;
+
+ /*
+ * Make sure the server has seen the client FIN before it closes,
+ * so the two FINs never cross.
+ */
+ n = recv(accept_serv, buf, sizeof(buf), 0);
+ if (!ASSERT_EQ(n, 0, "server recv EOF"))
+ goto cleanup;
+
+ close(accept_serv);
+ accept_serv = -1;
+
+ /* block until return EOF */
+ n = recv(clien, buf, sizeof(buf), 0);
+ if (!ASSERT_EQ(n, 0, "client recv EOF"))
+ goto cleanup;
+
+ /* Run iterator program that destroys the timewait client sock. */
+ skel->bss->tw_found = 0;
+ start_iter_sockets(skel->progs.iter_tcp6_timewait);
+ if (!ASSERT_EQ(skel->bss->tw_found, 1, "timewait sock found"))
+ goto cleanup;
+
+ ASSERT_OK(skel->bss->tw_destroy_err, "destroy timewait sock");
+
+ /* The destroyed timewait sock must be gone. */
+ skel->bss->tw_found = 0;
+ start_iter_sockets(skel->progs.iter_tcp6_timewait);
+ ASSERT_EQ(skel->bss->tw_found, 0, "timewait sock destroyed");
+
+cleanup:
+ if (clien != -1)
+ close(clien);
+ if (accept_serv != -1)
+ close(accept_serv);
+ if (serv != -1)
+ close(serv);
+}
+
static void test_udp_client(struct sock_destroy_prog *skel)
{
int serv = -1, clien = -1, n = 0;
@@ -204,6 +321,10 @@ void test_sock_destroy(void)
test_tcp_client(skel);
if (test__start_subtest("tcp_server"))
test_tcp_server(skel);
+ if (test__start_subtest("tcp_listen_pending"))
+ test_tcp_listen_pending(skel);
+ if (test__start_subtest("tcp_timewait"))
+ test_tcp_timewait(skel);
if (test__start_subtest("udp_client"))
test_udp_client(skel);
if (test__start_subtest("udp_server"))
diff --git a/tools/testing/selftests/bpf/prog_tests/sock_xattr.c b/tools/testing/selftests/bpf/prog_tests/sock_xattr.c
new file mode 100644
index 000000000000..b5816e90f01a
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/sock_xattr.c
@@ -0,0 +1,67 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/* Copyright (c) 2026 Christian Brauner */
+
+#include <errno.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/xattr.h>
+#include <sys/socket.h>
+#include <netinet/in.h>
+#include <test_progs.h>
+
+#include "sock_read_xattr.skel.h"
+
+static const char xattr_value[] = "bpf_sock_value";
+static const char xattr_name[] = "user.bpf_test";
+
+static void test_read_sock_xattr(void)
+{
+ struct sockaddr_in addr = {};
+ struct sock_read_xattr *skel = NULL;
+ struct bpf_link *link = NULL;
+ int sock_fd = -1, err;
+
+ sock_fd = socket(AF_INET, SOCK_STREAM, 0);
+ if (!ASSERT_OK_FD(sock_fd, "socket"))
+ return;
+
+ err = fsetxattr(sock_fd, xattr_name, xattr_value, sizeof(xattr_value), 0);
+ if (!ASSERT_OK(err, "fsetxattr"))
+ goto out;
+
+ skel = sock_read_xattr__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "sock_read_xattr__open_and_load"))
+ goto out;
+
+ skel->bss->monitored_pid = sys_gettid();
+
+ /* Only attach the functional program; the verifier-only programs
+ * above are not pid-gated and would clobber the shared globals.
+ */
+ link = bpf_program__attach(skel->progs.read_sock_xattr);
+ if (!ASSERT_OK_PTR(link, "attach read_sock_xattr"))
+ goto out;
+
+ addr.sin_family = AF_INET;
+ addr.sin_port = htons(1234);
+ addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ /* Only the lsm/socket_connect hook matters; the connect may fail. */
+ connect(sock_fd, (struct sockaddr *)&addr, sizeof(addr));
+
+ ASSERT_EQ(skel->data->read_ret, sizeof(xattr_value), "read_ret");
+ ASSERT_STREQ(skel->bss->value, xattr_value, "value");
+
+out:
+ bpf_link__destroy(link);
+ if (sock_fd >= 0)
+ close(sock_fd);
+ sock_read_xattr__destroy(skel);
+}
+
+void test_sock_xattr(void)
+{
+ RUN_TESTS(sock_read_xattr);
+
+ if (test__start_subtest("read_sock_xattr"))
+ test_read_sock_xattr();
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/sockopt.c b/tools/testing/selftests/bpf/prog_tests/sockopt.c
index eaac83a7f388..6c96f2d9fccf 100644
--- a/tools/testing/selftests/bpf/prog_tests/sockopt.c
+++ b/tools/testing/selftests/bpf/prog_tests/sockopt.c
@@ -1,4 +1,5 @@
// SPDX-License-Identifier: GPL-2.0
+#include <linux/tcp.h>
#include <test_progs.h>
#include <io_uring/mini_liburing.h>
#include "cgroup_helpers.h"
@@ -284,6 +285,27 @@ static struct sockopt_test {
.io_uring_support = true,
},
{
+ .descr = "getsockopt: deny negative ctx->optlen in TCP_ZEROCOPY_RECEIVE",
+ .insns = {
+ /* ctx->optlen = -1 */
+ BPF_MOV64_IMM(BPF_REG_0, -1),
+ BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_0,
+ offsetof(struct bpf_sockopt, optlen)),
+
+ /* return 1 */
+ BPF_MOV64_IMM(BPF_REG_0, 1),
+ BPF_EXIT_INSN(),
+ },
+ .attach_type = BPF_CGROUP_GETSOCKOPT,
+ .expected_attach_type = BPF_CGROUP_GETSOCKOPT,
+
+ .get_level = IPPROTO_TCP,
+ .get_optname = TCP_ZEROCOPY_RECEIVE,
+ .get_optlen = sizeof(struct tcp_zerocopy_receive),
+
+ .error = EFAULT_GETSOCKOPT,
+ },
+ {
.descr = "getsockopt: ignore >PAGE_SIZE optlen",
.insns = {
/* write 0xFF to the first optval byte */
diff --git a/tools/testing/selftests/bpf/prog_tests/spin_lock.c b/tools/testing/selftests/bpf/prog_tests/spin_lock.c
index 5c3579438427..e368370262c8 100644
--- a/tools/testing/selftests/bpf/prog_tests/spin_lock.c
+++ b/tools/testing/selftests/bpf/prog_tests/spin_lock.c
@@ -54,6 +54,8 @@ static struct {
{ "lock_global_sleepable_helper_subprog", "global function calls are not allowed while holding a lock" },
{ "lock_global_sleepable_kfunc_subprog", "global function calls are not allowed while holding a lock" },
{ "lock_global_sleepable_subprog_indirect", "global function calls are not allowed while holding a lock" },
+ { "callback_value_lock_identity", "bpf_spin_unlock of different lock" },
+ { "callback_inner_map_value_lock_identity", "bpf_spin_unlock of different lock" },
};
static int match_regex(const char *pattern, const char *string)
diff --git a/tools/testing/selftests/bpf/prog_tests/stream.c b/tools/testing/selftests/bpf/prog_tests/stream.c
index 15dd3ae2a84b..e4e9374309e2 100644
--- a/tools/testing/selftests/bpf/prog_tests/stream.c
+++ b/tools/testing/selftests/bpf/prog_tests/stream.c
@@ -105,6 +105,10 @@ void test_stream_arena_fault_address(void)
test_address(skel->progs.stream_arena_write_fault, &skel->bss->fault_addr);
if (test__start_subtest("load_acquire_fault"))
test_address(skel->progs.stream_arena_load_acquire_fault, &skel->bss->fault_addr);
+ if (test__start_subtest("xchg_fault"))
+ test_address(skel->progs.stream_arena_xchg_fault, &skel->bss->fault_addr);
+ if (test__start_subtest("cmpxchg_fault"))
+ test_address(skel->progs.stream_arena_cmpxchg_fault, &skel->bss->fault_addr);
stream__destroy(skel);
}
diff --git a/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c b/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c
new file mode 100644
index 000000000000..7acdbd5757a9
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c
@@ -0,0 +1,125 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <netinet/tcp.h>
+
+#include "test_progs.h"
+#include "network_helpers.h"
+#include "test_tc_change_tail_pmtu.skel.h"
+
+#define CLIENT_NS "tc-change-tail-cli-ns"
+#define SERVER_NS "tc-change-tail-srv-ns"
+#define CLIENT_IP "192.168.1.1"
+#define SERVER_IP "192.168.1.2"
+
+#define TEST_PMTU 1000
+#define TEST_MSS_MAX (TEST_PMTU - 20 - 20)
+#define TIMEOUT_MS 3000
+#define XFER_BYTES 8192
+
+void test_tc_change_tail_pmtu(void)
+{
+ LIBBPF_OPTS(bpf_tcx_opts, tcx_opts);
+ int mss_before = 0, mss_after = 0, ifindex, port;
+ int srv_fd = -1, srv_conn_fd = -1, cli_fd = -1;
+ struct test_tc_change_tail_pmtu *skel = NULL;
+ struct nstoken *nstoken = NULL;
+ static char buf[XFER_BYTES];
+ socklen_t optlen;
+ ssize_t bytes;
+ size_t total;
+
+ if (!ASSERT_OK(make_netns(CLIENT_NS), "make client ns"))
+ return;
+ if (!ASSERT_OK(make_netns(SERVER_NS), "make server ns"))
+ goto out_client_ns;
+
+ nstoken = open_netns(CLIENT_NS);
+ if (!ASSERT_OK_PTR(nstoken, "open client ns"))
+ goto out;
+ SYS(out, "ip link add veth1 type veth peer name veth2 netns " SERVER_NS);
+ SYS(out, "ip -4 addr add " CLIENT_IP "/24 dev veth1");
+ SYS(out, "ip link set veth1 up");
+ ifindex = if_nametoindex("veth1");
+ if (!ASSERT_NEQ(ifindex, 0, "if_nametoindex"))
+ goto out;
+ close_netns(nstoken);
+ nstoken = NULL;
+
+ nstoken = open_netns(SERVER_NS);
+ if (!ASSERT_OK_PTR(nstoken, "open server ns"))
+ goto out;
+ SYS(out, "ip -4 addr add " SERVER_IP "/24 dev veth2");
+ SYS(out, "ip link set veth2 up");
+ srv_fd = start_server(AF_INET, SOCK_STREAM, SERVER_IP, 0, TIMEOUT_MS);
+ if (!ASSERT_OK_FD(srv_fd, "start server"))
+ goto out;
+ close_netns(nstoken);
+ nstoken = NULL;
+
+ skel = test_tc_change_tail_pmtu__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open and load skeleton"))
+ goto out;
+
+ port = get_socket_local_port(srv_fd);
+ if (!ASSERT_GE(port, 0, "get server port"))
+ goto out;
+
+ skel->bss->server_port = port;
+ skel->bss->pmtu = TEST_PMTU;
+
+ nstoken = open_netns(CLIENT_NS);
+ if (!ASSERT_OK_PTR(nstoken, "open client ns"))
+ goto out;
+
+ skel->links.change_tail_icmp =
+ bpf_program__attach_tcx(skel->progs.change_tail_icmp, ifindex,
+ &tcx_opts);
+ if (!ASSERT_OK_PTR(skel->links.change_tail_icmp, "attach tcx"))
+ goto out;
+
+ cli_fd = connect_to_fd(srv_fd, TIMEOUT_MS);
+ if (!ASSERT_OK_FD(cli_fd, "connect to server"))
+ goto out;
+ srv_conn_fd = accept(srv_fd, NULL, NULL);
+ if (!ASSERT_OK_FD(srv_conn_fd, "accept connection"))
+ goto out;
+ if (!ASSERT_OK(settimeo(srv_conn_fd, TIMEOUT_MS), "set server timeout"))
+ goto out;
+
+ optlen = sizeof(mss_before);
+ if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_before,
+ &optlen), "get mss before"))
+ goto out;
+
+ bytes = send(cli_fd, buf, sizeof(buf), 0);
+ if (!ASSERT_EQ(bytes, (ssize_t)sizeof(buf), "send data"))
+ goto out;
+
+ for (total = 0; total < sizeof(buf); total += bytes) {
+ bytes = recv(srv_conn_fd, buf, sizeof(buf), 0);
+ if (bytes <= 0)
+ break;
+ }
+
+ ASSERT_EQ(total, sizeof(buf), "receive data");
+ ASSERT_OK(skel->data->change_tail_ret, "change tail");
+ ASSERT_OK(skel->bss->adjust_room_ret, "adjust room");
+ ASSERT_TRUE(skel->bss->icmp_sent, "icmp sent");
+
+ optlen = sizeof(mss_after);
+ if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_after,
+ &optlen), "get mss after"))
+ goto out;
+
+ ASSERT_LT(mss_after, mss_before, "mss reduced");
+ ASSERT_LE(mss_after, TEST_MSS_MAX, "mss below pmtu");
+out:
+ close(srv_conn_fd);
+ close(cli_fd);
+ close(srv_fd);
+ test_tc_change_tail_pmtu__destroy(skel);
+ close_netns(nstoken);
+ remove_netns(SERVER_NS);
+out_client_ns:
+ remove_netns(CLIENT_NS);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/test_struct_ops_arena.c b/tools/testing/selftests/bpf/prog_tests/test_struct_ops_arena.c
index 940ec2cda0d5..7f9f54ba3fbe 100644
--- a/tools/testing/selftests/bpf/prog_tests/test_struct_ops_arena.c
+++ b/tools/testing/selftests/bpf/prog_tests/test_struct_ops_arena.c
@@ -6,7 +6,7 @@
#include "struct_ops_arena_attach.skel.h"
#include "struct_ops_arena_fail.skel.h"
-#if defined(__x86_64__)
+#if defined(__x86_64__) || defined(__aarch64__)
/*
* Attach callbacks with __arena and __arena__nullable arguments and drive
* them through the bpf_testmod_ops3_call_test_arena*() kfuncs.
@@ -111,11 +111,11 @@ static void arena_arg_attach(void)
void serial_test_struct_ops_arena(void)
{
/*
- * Arena struct_ops arguments need JIT support, currently x86-64 only.
- * Elsewhere verification fails with "JIT does not support arena
- * arguments", so the programs cannot even load.
+ * Arena struct_ops arguments need JIT support, currently x86-64 and
+ * arm64 only. Elsewhere verification fails with "JIT does not support
+ * arena arguments", so the programs cannot even load.
*/
-#if defined(__x86_64__)
+#if defined(__x86_64__) || defined(__aarch64__)
if (test__start_subtest("arena_arg"))
arena_arg();
if (test__start_subtest("arena_arg_fail"))
diff --git a/tools/testing/selftests/bpf/prog_tests/test_tc_tunnel.c b/tools/testing/selftests/bpf/prog_tests/test_tc_tunnel.c
index 1aa7c9463980..67ba27d69347 100644
--- a/tools/testing/selftests/bpf/prog_tests/test_tc_tunnel.c
+++ b/tools/testing/selftests/bpf/prog_tests/test_tc_tunnel.c
@@ -438,7 +438,6 @@ static int setup(void)
SYS(fail_close_ns_client, "ip link add %s type veth peer name %s",
"veth1 mtu 1500 netns " CLIENT_NS " address " MAC_ADDR_VETH1,
"veth2 mtu 1500 netns " SERVER_NS " address " MAC_ADDR_VETH2);
- SYS(fail_close_ns_client, "ethtool -K veth1 tso off");
SYS(fail_close_ns_client, "ip link set veth1 up");
nstoken_server = open_netns(SERVER_NS);
if (!ASSERT_OK_PTR(nstoken_server, "open server ns"))
diff --git a/tools/testing/selftests/bpf/prog_tests/test_veristat.c b/tools/testing/selftests/bpf/prog_tests/test_veristat.c
index 9aff08ac55c0..11f3de2b66ad 100644
--- a/tools/testing/selftests/bpf/prog_tests/test_veristat.c
+++ b/tools/testing/selftests/bpf/prog_tests/test_veristat.c
@@ -37,6 +37,14 @@ static struct fixture *init_fixture(void)
return fix;
}
+static void read_output(struct fixture *fix)
+{
+ ssize_t len = pread(fix->fd, fix->output, fix->sz - 1, 0);
+
+ fix->output[len < 0 ? 0 : len] = 0;
+ ASSERT_GE(len, 0, "pread");
+}
+
static void teardown_fixture(struct fixture *fix)
{
free(fix->output);
@@ -74,7 +82,7 @@ static void test_set_global_vars_succeeds(void)
" -G \"struct11 [ 7 ] [ 5 ] .struct2[0][1].u.mat[3][0] = 175\" " \
" -vl2 > %s", fix->veristat, fix->tmpfile);
- read(fix->fd, fix->output, fix->sz);
+ read_output(fix);
__CHECK_STR("=0xf000000000000001 ", "var_s64 = 0xf000000000000001");
__CHECK_STR("=0xfedcba9876543210 ", "var_u64 = 0xfedcba9876543210");
__CHECK_STR("=0x80000000 ", "var_s32 = -0x80000000");
@@ -116,7 +124,7 @@ static void test_set_global_vars_from_file_succeeds(void)
syncfs(fd);
SYS(out, "%s set_global_vars.bpf.o -G \"@%s\" -vl2 > %s",
fix->veristat, input_file, fix->tmpfile);
- read(fix->fd, fix->output, fix->sz);
+ read_output(fix);
__CHECK_STR("=0x8000 ", "var_s16 = -32768");
__CHECK_STR("=0xecec ", "var_u16 = 60652");
@@ -134,7 +142,7 @@ static void test_set_global_vars_out_of_range(void)
"%s set_global_vars.bpf.o -G \"var_s32 = 2147483648\" -vl2 2> %s",
fix->veristat, fix->tmpfile);
- read(fix->fd, fix->output, fix->sz);
+ read_output(fix);
__CHECK_STR("is out of range [-2147483648; 2147483647]", "out of range");
out:
@@ -149,7 +157,7 @@ static void test_unsupported_ptr_array_type(void)
"%s set_global_vars.bpf.o -G \"ptr_arr[0] = 0\" -vl2 2> %s",
fix->veristat, fix->tmpfile);
- read(fix->fd, fix->output, fix->sz);
+ read_output(fix);
__CHECK_STR("Can't set ptr_arr[0]. Only ints and enums are supported", "ptr_arr");
out:
@@ -164,7 +172,7 @@ static void test_array_out_of_bounds(void)
"%s set_global_vars.bpf.o -G \"arr[99] = 0\" -vl2 2> %s",
fix->veristat, fix->tmpfile);
- read(fix->fd, fix->output, fix->sz);
+ read_output(fix);
__CHECK_STR("Array index 99 is out of bounds", "arr[99]");
out:
@@ -179,7 +187,7 @@ static void test_array_index_not_found(void)
"%s set_global_vars.bpf.o -G \"arr[EG2] = 0\" -vl2 2> %s",
fix->veristat, fix->tmpfile);
- read(fix->fd, fix->output, fix->sz);
+ read_output(fix);
__CHECK_STR("Can't resolve enum value EG2", "arr[EG2]");
out:
@@ -230,6 +238,97 @@ out:
teardown_fixture(fix);
}
+/*
+ * Name filter tests below run veristat on veristat_foo.bpf.o and
+ * veristat_bar.bpf.o, both defining programs 'foo', 'bar' and 'buz'.
+ * Every entry describes a single (filters, file, prog) combination and
+ * tells whether that program is expected in the veristat output:
+ * 'true' if it is, 'false' if it is not and -1 if veristat is expected
+ * to reject the filter.
+ */
+#define FILTER_OBJS "veristat_foo.bpf.o veristat_bar.bpf.o"
+
+static const struct name_filter_case {
+ const char *filters;
+ const char *file;
+ const char *prog;
+ int included;
+} name_filter_cases[] = {
+ /* no filters, every program is processed */
+ { "", "foo", "foo", true },
+ { "", "foo", "bar", true },
+ { "", "foo", "buz", true },
+ { "", "bar", "foo", true },
+ { "", "bar", "bar", true },
+ { "", "bar", "buz", true },
+ /* deny filters */
+ { "-f '!*foo*'", "foo", "bar", false },
+ { "-f '!*foo*'", "bar", "foo", false },
+ { "-f '!*foo*'", "bar", "bar", true },
+ { "-f '!*foo*/bar'", "foo", "bar", false },
+ { "-f '!*foo*/bar'", "foo", "buz", true },
+ { "-f '!*foo*/bar'", "bar", "bar", true },
+ { "-f '!*foo*/'", "foo", "bar", false },
+ { "-f '!*foo*/'", "bar", "bar", true },
+ { "-f '!/bar'", "foo", "bar", false },
+ { "-f '!/bar'", "foo", "foo", true },
+ { "-f '!/'", "foo", "bar", -1 },
+ { "-f '!'", "foo", "bar", -1 },
+ /* allow filters */
+ { "-f '*foo*'", "foo", "bar", true },
+ { "-f '*foo*'", "bar", "foo", true },
+ { "-f '*foo*'", "bar", "bar", false },
+ { "-f '*foo*/bar'", "foo", "bar", true },
+ { "-f '*foo*/bar'", "foo", "buz", false },
+ { "-f '*foo*/bar'", "bar", "bar", false },
+ { "-f '*foo*/'", "foo", "bar", true },
+ { "-f '*foo*/'", "bar", "bar", false },
+ { "-f '/bar'", "foo", "bar", true },
+ { "-f '/bar'", "foo", "foo", false },
+ { "-f '/'", "foo", "bar", -1 },
+ { "-f ''", "foo", "bar", -1 },
+ /* allow and deny filters combined */
+ { "-f '*foo*/' -f '!/bar'", "foo", "foo", true },
+ { "-f '*foo*/' -f '!/bar'", "foo", "bar", false },
+ { "-f '*foo*/' -f '!/bar'", "bar", "foo", false },
+};
+
+static void test_name_filters(void)
+{
+ struct fixture *fix = init_fixture();
+ const struct name_filter_case *t;
+ char cmd[512], row[64], name[128];
+ int i, err;
+
+ for (i = 0; i < ARRAY_SIZE(name_filter_cases); i++) {
+ t = &name_filter_cases[i];
+ /* stderr is merged with stdout in order to catch error messages */
+ snprintf(cmd, sizeof(cmd), "%s " FILTER_OBJS " -q -o csv -e file,prog %s > %s 2>&1",
+ fix->veristat, t->filters, fix->tmpfile);
+ err = system(cmd);
+ read_output(fix);
+
+ snprintf(row, sizeof(row), "veristat_%s.bpf.o,%s", t->file, t->prog);
+ snprintf(name, sizeof(name), "veristat %s: %s", t->filters, row);
+ switch (t->included) {
+ case true:
+ ASSERT_OK(err, name);
+ ASSERT_HAS_SUBSTR(fix->output, row, name);
+ break;
+ case false:
+ ASSERT_OK(err, name);
+ ASSERT_FALSE(!!strstr(fix->output, row), name);
+ break;
+ case -1:
+ ASSERT_NEQ(err, 0, name);
+ ASSERT_HAS_SUBSTR(fix->output, "Invalid filter", name);
+ break;
+ }
+ }
+
+ teardown_fixture(fix);
+}
+
void test_veristat(void)
{
if (test__start_subtest("set_global_vars_succeeds"))
@@ -256,6 +355,8 @@ void test_veristat(void)
if (test__start_subtest("test_no_array_index_for_array"))
test_no_array_index_for_array();
+ if (test__start_subtest("name_filters"))
+ test_name_filters();
}
#undef __CHECK_STR
diff --git a/tools/testing/selftests/bpf/prog_tests/test_xsk.c b/tools/testing/selftests/bpf/prog_tests/test_xsk.c
index 38ce6060b8fa..4549358cc8c2 100644
--- a/tools/testing/selftests/bpf/prog_tests/test_xsk.c
+++ b/tools/testing/selftests/bpf/prog_tests/test_xsk.c
@@ -1164,8 +1164,8 @@ static int __send_pkts(struct ifobject *ifobject, struct xsk_socket_info *xsk,
bool test_timeout)
{
u32 i, idx = 0, valid_pkts = 0, valid_frags = 0, buffer_len;
+ struct xsk_umem_info *umem = ifobject->xsk_arr[0].umem_real;
struct pkt_stream *pkt_stream = xsk->pkt_stream;
- struct xsk_umem_info *umem = xsk->umem;
bool use_poll = ifobject->use_poll;
struct pollfd fds = { };
int ret;
@@ -1513,7 +1513,7 @@ static int thread_common_ops_tx(struct test_spec *test, struct ifobject *ifobjec
umem_tx->base_addr = 0;
umem_tx->next_buffer = 0;
- ret = xsk_configure(test, ifobject, umem_tx, true);
+ ret = xsk_configure(test, ifobject, umem_rx, true);
if (ret)
return ret;
ifobject->xsk = &ifobject->xsk_arr[0];
diff --git a/tools/testing/selftests/bpf/prog_tests/timer_mim.c b/tools/testing/selftests/bpf/prog_tests/timer_mim.c
index c930c7d7105b..fa7bb769ca31 100644
--- a/tools/testing/selftests/bpf/prog_tests/timer_mim.c
+++ b/tools/testing/selftests/bpf/prog_tests/timer_mim.c
@@ -59,10 +59,32 @@ void serial_test_timer_mim(void)
int err;
old_print_fn = libbpf_set_print(NULL);
- timer_reject_skel = timer_mim_reject__open_and_load();
- libbpf_set_print(old_print_fn);
- if (!ASSERT_ERR_PTR(timer_reject_skel, "timer_reject_skel_load"))
+ timer_reject_skel = timer_mim_reject__open();
+ if (!ASSERT_OK_PTR(timer_reject_skel, "timer_reject_skel_open"))
+ goto cleanup;
+ bpf_program__set_autoload(timer_reject_skel->progs.test1, true);
+ err = timer_mim_reject__load(timer_reject_skel);
+ ASSERT_ERR(err, "timer_reject_skel_load");
+ timer_mim_reject__destroy(timer_reject_skel);
+
+ timer_reject_skel = timer_mim_reject__open();
+ if (!ASSERT_OK_PTR(timer_reject_skel, "callback_reject_skel_open"))
goto cleanup;
+ bpf_program__set_autoload(timer_reject_skel->progs.callback_map_uid_mismatch, true);
+ err = timer_mim_reject__load(timer_reject_skel);
+ ASSERT_ERR(err, "callback_reject_skel_load");
+ timer_mim_reject__destroy(timer_reject_skel);
+
+ timer_reject_skel = timer_mim_reject__open();
+ if (!ASSERT_OK_PTR(timer_reject_skel, "callback_accept_skel_open"))
+ goto cleanup;
+ bpf_program__set_autoload(timer_reject_skel->progs.callback_map_uid_match, true);
+ err = timer_mim_reject__load(timer_reject_skel);
+ if (!ASSERT_OK(err, "callback_accept_skel_load"))
+ goto cleanup;
+ timer_mim_reject__destroy(timer_reject_skel);
+ timer_reject_skel = NULL;
+ libbpf_set_print(old_print_fn);
timer_skel = timer_mim__open_and_load();
if (!timer_skel && errno == EOPNOTSUPP) {
@@ -75,6 +97,7 @@ void serial_test_timer_mim(void)
err = timer_mim(timer_skel);
ASSERT_OK(err, "timer_mim");
cleanup:
+ libbpf_set_print(old_print_fn);
timer_mim__destroy(timer_skel);
timer_mim_reject__destroy(timer_reject_skel);
}
diff --git a/tools/testing/selftests/bpf/prog_tests/tp_btf_ids.c b/tools/testing/selftests/bpf/prog_tests/tp_btf_ids.c
new file mode 100644
index 000000000000..c0e7e11e71b8
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/tp_btf_ids.c
@@ -0,0 +1,132 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <test_progs.h>
+#include <bpf/btf.h>
+
+#define TRACEFS "/sys/kernel/tracing"
+#define DEBUGFS_TRACING "/sys/kernel/debug/tracing"
+#define EVENT_SUBPATH "events/bpf_testmod/bpf_testmod_test_read/btf_ids"
+
+struct btf_ids_info {
+ __u32 obj_id;
+ __u32 raw_id;
+ __u32 tp_id;
+};
+
+static const char *btf_ids_path(char *buf, size_t sz)
+{
+ if (access(TRACEFS "/trace", F_OK) == 0)
+ snprintf(buf, sz, "%s/%s", TRACEFS, EVENT_SUBPATH);
+ else
+ snprintf(buf, sz, "%s/%s", DEBUGFS_TRACING, EVENT_SUBPATH);
+ return buf;
+}
+
+static int read_btf_ids(struct btf_ids_info *info)
+{
+ char path[256], buf[256];
+ int fd, n;
+
+ fd = open(btf_ids_path(path, sizeof(path)), O_RDONLY);
+ if (fd < 0)
+ return -errno;
+
+ n = read(fd, buf, sizeof(buf) - 1);
+ close(fd);
+ if (n <= 0)
+ return -EIO;
+ buf[n] = '\0';
+
+ if (sscanf(buf,
+ "btf_obj_id: %u\nraw_btf_id: %u\ntp_btf_id: %u\n",
+ &info->obj_id, &info->raw_id, &info->tp_id) != 3)
+ return -EINVAL;
+ return 0;
+}
+
+static const char *param_name(struct btf *btf, const struct btf_param *p)
+{
+ return btf__name_by_offset(btf, p->name_off);
+}
+
+static const char *member_name(struct btf *btf, const struct btf_member *m)
+{
+ return btf__name_by_offset(btf, m->name_off);
+}
+
+void test_tp_btf_ids(void)
+{
+ const struct btf_type *proto_t, *rec_t;
+ const struct btf_param *params;
+ const struct btf_member *members;
+ struct btf_ids_info info;
+ struct btf *vmlinux_btf, *btf;
+ const char *name;
+ int err;
+
+ if (!env.has_testmod) {
+ test__skip();
+ return;
+ }
+
+ err = read_btf_ids(&info);
+ if (!ASSERT_OK(err, "read btf_ids"))
+ return;
+
+ ASSERT_GT(info.obj_id, 0, "obj_id non-zero");
+ ASSERT_GT(info.raw_id, 0, "raw_id non-zero");
+ ASSERT_GT(info.tp_id, 0, "tp_id non-zero");
+
+ vmlinux_btf = btf__load_vmlinux_btf();
+ if (!ASSERT_OK_PTR(vmlinux_btf, "load vmlinux BTF"))
+ return;
+
+ /* Module BTF is split BTF; load with vmlinux as base. */
+ btf = btf__load_from_kernel_by_id_split(info.obj_id, vmlinux_btf);
+ if (!ASSERT_OK_PTR(btf, "load module BTF")) {
+ btf__free(vmlinux_btf);
+ return;
+ }
+
+ /*
+ * raw_btf_id should be the FUNC_PROTO of __bpf_trace_<call>:
+ * void *__data, struct task_struct *task,
+ * struct bpf_testmod_test_read_ctx *ctx
+ */
+ proto_t = btf__type_by_id(btf, info.raw_id);
+ if (!ASSERT_OK_PTR(proto_t, "raw type_by_id"))
+ goto out;
+ if (!ASSERT_TRUE(btf_is_func_proto(proto_t), "raw is FUNC_PROTO"))
+ goto out;
+ if (!ASSERT_EQ(btf_vlen(proto_t), 3, "func_proto arg count"))
+ goto out;
+
+ params = btf_params(proto_t);
+ ASSERT_STREQ(param_name(btf, &params[0]), "__data", "arg0 name");
+ ASSERT_STREQ(param_name(btf, &params[1]), "task", "arg1 name");
+ ASSERT_STREQ(param_name(btf, &params[2]), "ctx", "arg2 name");
+
+ /*
+ * tp_btf_id should be STRUCT trace_event_raw_<call> with the
+ * fields declared by TP_STRUCT__entry plus the common header.
+ */
+ rec_t = btf__type_by_id(btf, info.tp_id);
+ if (!ASSERT_OK_PTR(rec_t, "tp type_by_id"))
+ goto out;
+ if (!ASSERT_TRUE(btf_is_struct(rec_t), "tp is STRUCT"))
+ goto out;
+ name = btf__name_by_offset(btf, rec_t->name_off);
+ ASSERT_STREQ(name, "trace_event_raw_bpf_testmod_test_read",
+ "tp struct name");
+ if (!ASSERT_GE(btf_vlen(rec_t), 5, "tp struct field count"))
+ goto out;
+
+ members = btf_members(rec_t);
+ ASSERT_STREQ(member_name(btf, &members[0]), "ent", "field0 name");
+ ASSERT_STREQ(member_name(btf, &members[1]), "pid", "field1 name");
+ ASSERT_STREQ(member_name(btf, &members[2]), "comm", "field2 name");
+ ASSERT_STREQ(member_name(btf, &members[3]), "off", "field3 name");
+ ASSERT_STREQ(member_name(btf, &members[4]), "len", "field4 name");
+out:
+ btf__free(btf);
+ btf__free(vmlinux_btf);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/uprobe_syscall.c b/tools/testing/selftests/bpf/prog_tests/uprobe_syscall.c
index 955a37751b52..e193206fc5d2 100644
--- a/tools/testing/selftests/bpf/prog_tests/uprobe_syscall.c
+++ b/tools/testing/selftests/bpf/prog_tests/uprobe_syscall.c
@@ -4,6 +4,8 @@
#ifdef __x86_64__
+#define _GNU_SOURCE
+#include <sched.h>
#include <unistd.h>
#include <asm/ptrace.h>
#include <linux/compiler.h>
@@ -13,11 +15,12 @@
#include <sys/syscall.h>
#include <sys/prctl.h>
#include <asm/prctl.h>
+#include <stdnoreturn.h>
#include "uprobe_syscall.skel.h"
#include "uprobe_syscall_executed.skel.h"
#include "bpf/libbpf_internal.h"
-#define USDT_NOP .byte 0x0f, 0x1f, 0x44, 0x00, 0x00
+#define USDT_NOP .byte 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00
#include "usdt.h"
#pragma GCC diagnostic ignored "-Wattributes"
@@ -26,7 +29,7 @@ __attribute__((aligned(16)))
__nocf_check __weak __naked unsigned long uprobe_regs_trigger(void)
{
asm volatile (
- ".byte 0x0f, 0x1f, 0x44, 0x00, 0x00\n" /* nop5 */
+ ".byte 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00\n" /* nop10 */
"movq $0xdeadbeef, %rax\n"
"ret\n"
);
@@ -345,9 +348,9 @@ cleanup:
__attribute__((aligned(16)))
__nocf_check __weak __naked void uprobe_test(void)
{
- asm volatile (" \n"
- ".byte 0x0f, 0x1f, 0x44, 0x00, 0x00 \n"
- "ret \n"
+ asm volatile (
+ ".byte 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00\n" /* nop10 */
+ "ret\n"
);
}
@@ -357,6 +360,50 @@ __nocf_check __weak void usdt_test(void)
USDT(optimized_uprobe, usdt);
}
+/*
+ * Assembly-level red zone clobbering test. Stores known values in the
+ * red zone (below RSP), executes a nop10 (uprobe site), and checks that
+ * the values survived. Returns 0 if intact, 1 if clobbered.
+ *
+ * The nop5 optimization used CALL (which pushes a return address to
+ * [rsp-8]), the value at -8(%rsp) was overwritten. The nop10 optimization
+ * should escape that by moving stackpointer below the redzone before
+ * doing the CALL.
+ *
+ * Align the code at 64 bytes, to make sure nop10 is not on page boundary.
+ */
+__attribute__((aligned(64)))
+__nocf_check __weak __naked unsigned long uprobe_red_zone_test(void)
+{
+ asm volatile (
+ "movabs $0x1111111111111111, %%rax\n"
+ "movq %%rax, -8(%%rsp)\n"
+ "movabs $0x2222222222222222, %%rax\n"
+ "movq %%rax, -16(%%rsp)\n"
+ "movabs $0x3333333333333333, %%rax\n"
+ "movq %%rax, -24(%%rsp)\n"
+
+ ".byte 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00\n" /* nop10: uprobe site */
+
+ "movabs $0x1111111111111111, %%rax\n"
+ "cmpq %%rax, -8(%%rsp)\n"
+ "jne 1f\n"
+ "movabs $0x2222222222222222, %%rax\n"
+ "cmpq %%rax, -16(%%rsp)\n"
+ "jne 1f\n"
+ "movabs $0x3333333333333333, %%rax\n"
+ "cmpq %%rax, -24(%%rsp)\n"
+ "jne 1f\n"
+
+ "xorl %%eax, %%eax\n"
+ "retq\n"
+ "1:\n"
+ "movl $1, %%eax\n"
+ "retq\n"
+ ::: "rax", "memory"
+ );
+}
+
static int find_uprobes_trampoline(void *tramp_addr)
{
void *start, *end;
@@ -388,14 +435,15 @@ static int find_uprobes_trampoline(void *tramp_addr)
return ret;
}
-static unsigned char nop5[5] = { 0x0f, 0x1f, 0x44, 0x00, 0x00 };
+static unsigned char nop10[10] = { 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00 };
+static unsigned char lea_rsp[5] = { 0x48, 0x8d, 0x64, 0x24, 0x80 };
-static void *find_nop5(void *fn)
+static void *find_nop10(void *fn)
{
int i;
- for (i = 0; i < 10; i++) {
- if (!memcmp(nop5, fn + i, 5))
+ for (i = 0; i < 128; i++) {
+ if (!memcmp(nop10, fn + i, 10))
return fn + i;
}
return NULL;
@@ -403,6 +451,16 @@ static void *find_nop5(void *fn)
typedef void (__attribute__((nocf_check)) *trigger_t)(void);
+static void check_attach_notrigger(struct uprobe_syscall_executed *skel,
+ void *addr, int executed)
+{
+ unsigned char *op = addr;
+
+ /* Make sure bpf program was not executed. */
+ ASSERT_EQ(skel->bss->executed, executed, "executed");
+ ASSERT_EQ(*op, 0xcc, "int3");
+}
+
static void *check_attach(struct uprobe_syscall_executed *skel, trigger_t trigger,
void *addr, int executed)
{
@@ -420,7 +478,8 @@ static void *check_attach(struct uprobe_syscall_executed *skel, trigger_t trigge
ASSERT_EQ(skel->bss->executed, executed, "executed");
/* .. and check the trampoline is as expected. */
- call = (struct __arch_relative_insn *) addr;
+ ASSERT_OK(memcmp(addr, lea_rsp, 5), "lea_rsp");
+ call = (struct __arch_relative_insn *)(addr + 5);
tramp = (void *) (call + 1) + call->raddr;
ASSERT_EQ(call->op, 0xe8, "call");
ASSERT_OK(find_uprobes_trampoline(tramp), "uprobes_trampoline");
@@ -428,21 +487,26 @@ static void *check_attach(struct uprobe_syscall_executed *skel, trigger_t trigge
return tramp;
}
-static void check_detach(void *addr, void *tramp)
+static bool check_detach(void *addr, void *tramp)
{
+ static const unsigned char nop10_prefix[] = { 0x66, 0x2e, 0x0f, 0x1f, 0x84 };
+ bool ok = true;
+
/* [uprobes_trampoline] stays after detach */
- ASSERT_OK(find_uprobes_trampoline(tramp), "uprobes_trampoline");
- ASSERT_OK(memcmp(addr, nop5, 5), "nop5");
+ ok &= ASSERT_OK(find_uprobes_trampoline(tramp), "uprobes_trampoline");
+ ok &= ASSERT_OK(memcmp(addr, nop10_prefix, 5), "nop10_prefix");
+ return ok;
}
-static void check(struct uprobe_syscall_executed *skel, struct bpf_link *link,
- trigger_t trigger, void *addr, int executed)
+static void *check(struct uprobe_syscall_executed *skel, struct bpf_link *link,
+ trigger_t trigger, void *addr, int executed)
{
void *tramp;
tramp = check_attach(skel, trigger, addr, executed);
bpf_link__destroy(link);
check_detach(addr, tramp);
+ return tramp;
}
static void test_uprobe_legacy(void)
@@ -453,6 +517,7 @@ static void test_uprobe_legacy(void)
);
struct bpf_link *link;
unsigned long offset;
+ void *tramp;
offset = get_uprobe_offset(&uprobe_test);
if (!ASSERT_GE(offset, 0, "get_uprobe_offset"))
@@ -470,7 +535,30 @@ static void test_uprobe_legacy(void)
if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_opts"))
goto cleanup;
- check(skel, link, uprobe_test, uprobe_test, 2);
+ tramp = check(skel, link, uprobe_test, uprobe_test, 2);
+
+ /* reattach and detach without triggering optimization */
+ link = bpf_program__attach_uprobe_opts(skel->progs.test_uprobe,
+ 0, "/proc/self/exe", offset, NULL);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_opts"))
+ goto cleanup;
+
+ check_attach_notrigger(skel, uprobe_test, 2);
+
+ bpf_link__destroy(link);
+ if (!check_detach(uprobe_test, tramp))
+ goto cleanup;
+
+ uprobe_test();
+ ASSERT_EQ(skel->bss->executed, 2, "executed_no_probe");
+
+ /* reattach with triggering optimization */
+ link = bpf_program__attach_uprobe_opts(skel->progs.test_uprobe,
+ 0, "/proc/self/exe", offset, NULL);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_opts"))
+ goto cleanup;
+
+ check(skel, link, uprobe_test, uprobe_test, 4);
/* uretprobe */
skel->bss->executed = 0;
@@ -492,6 +580,7 @@ static void test_uprobe_multi(void)
LIBBPF_OPTS(bpf_uprobe_multi_opts, opts);
struct bpf_link *link;
unsigned long offset;
+ void *tramp;
offset = get_uprobe_offset(&uprobe_test);
if (!ASSERT_GE(offset, 0, "get_uprobe_offset"))
@@ -512,7 +601,30 @@ static void test_uprobe_multi(void)
if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_multi"))
goto cleanup;
- check(skel, link, uprobe_test, uprobe_test, 2);
+ tramp = check(skel, link, uprobe_test, uprobe_test, 2);
+
+ /* reattach and detach without triggering optimization */
+ link = bpf_program__attach_uprobe_multi(skel->progs.test_uprobe_multi,
+ 0, "/proc/self/exe", NULL, &opts);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_multi"))
+ goto cleanup;
+
+ check_attach_notrigger(skel, uprobe_test, 2);
+
+ bpf_link__destroy(link);
+ if (!check_detach(uprobe_test, tramp))
+ goto cleanup;
+
+ uprobe_test();
+ ASSERT_EQ(skel->bss->executed, 2, "executed_no_probe");
+
+ /* reattach with triggering optimization */
+ link = bpf_program__attach_uprobe_multi(skel->progs.test_uprobe_multi,
+ 0, "/proc/self/exe", NULL, &opts);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_multi"))
+ goto cleanup;
+
+ check(skel, link, uprobe_test, uprobe_test, 4);
/* uretprobe.multi */
skel->bss->executed = 0;
@@ -536,6 +648,7 @@ static void test_uprobe_session(void)
);
struct bpf_link *link;
unsigned long offset;
+ void *tramp;
offset = get_uprobe_offset(&uprobe_test);
if (!ASSERT_GE(offset, 0, "get_uprobe_offset"))
@@ -555,7 +668,30 @@ static void test_uprobe_session(void)
if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_multi"))
goto cleanup;
- check(skel, link, uprobe_test, uprobe_test, 4);
+ tramp = check(skel, link, uprobe_test, uprobe_test, 4);
+
+ /* reattach and detach without triggering optimization */
+ link = bpf_program__attach_uprobe_multi(skel->progs.test_uprobe_session,
+ 0, "/proc/self/exe", NULL, &opts);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_multi"))
+ goto cleanup;
+
+ check_attach_notrigger(skel, uprobe_test, 4);
+
+ bpf_link__destroy(link);
+ if (!check_detach(uprobe_test, tramp))
+ goto cleanup;
+
+ uprobe_test();
+ ASSERT_EQ(skel->bss->executed, 4, "executed_no_probe");
+
+ /* reattach with triggering optimization */
+ link = bpf_program__attach_uprobe_multi(skel->progs.test_uprobe_session,
+ 0, "/proc/self/exe", NULL, &opts);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_uprobe_multi"))
+ goto cleanup;
+
+ check(skel, link, uprobe_test, uprobe_test, 8);
cleanup:
uprobe_syscall_executed__destroy(skel);
@@ -565,11 +701,11 @@ static void test_uprobe_usdt(void)
{
struct uprobe_syscall_executed *skel;
struct bpf_link *link;
- void *addr;
+ void *addr, *tramp;
errno = 0;
- addr = find_nop5(usdt_test);
- if (!ASSERT_OK_PTR(addr, "find_nop5"))
+ addr = find_nop10(usdt_test);
+ if (!ASSERT_OK_PTR(addr, "find_nop10"))
return;
skel = uprobe_syscall_executed__open_and_load();
@@ -584,7 +720,32 @@ static void test_uprobe_usdt(void)
if (!ASSERT_OK_PTR(link, "bpf_program__attach_usdt"))
goto cleanup;
- check(skel, link, usdt_test, addr, 2);
+ tramp = check(skel, link, usdt_test, addr, 2);
+
+ /* reattach and detach without triggering optimization */
+ link = bpf_program__attach_usdt(skel->progs.test_usdt,
+ -1 /* all PIDs */, "/proc/self/exe",
+ "optimized_uprobe", "usdt", NULL);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_usdt"))
+ goto cleanup;
+
+ check_attach_notrigger(skel, addr, 2);
+
+ bpf_link__destroy(link);
+ if (!check_detach(addr, tramp))
+ goto cleanup;
+
+ usdt_test();
+ ASSERT_EQ(skel->bss->executed, 2, "executed_no_probe");
+
+ /* reattach with triggering optimization */
+ link = bpf_program__attach_usdt(skel->progs.test_usdt,
+ -1 /* all PIDs */, "/proc/self/exe",
+ "optimized_uprobe", "usdt", NULL);
+ if (!ASSERT_OK_PTR(link, "bpf_program__attach_usdt"))
+ goto cleanup;
+
+ check(skel, link, usdt_test, addr, 4);
cleanup:
uprobe_syscall_executed__destroy(skel);
@@ -757,12 +918,124 @@ cleanup:
#define __NR_uprobe 336
#endif
+static void test_uprobe_red_zone(void)
+{
+ struct uprobe_syscall_executed *skel;
+ struct bpf_link *link;
+ void *nop10_addr;
+ size_t offset;
+ int i;
+
+ nop10_addr = find_nop10(uprobe_red_zone_test);
+ if (!ASSERT_NEQ(nop10_addr, NULL, "find_nop10"))
+ return;
+
+ skel = uprobe_syscall_executed__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ return;
+
+ offset = get_uprobe_offset(nop10_addr);
+ link = bpf_program__attach_uprobe_opts(skel->progs.test_uprobe,
+ 0, "/proc/self/exe", offset, NULL);
+ if (!ASSERT_OK_PTR(link, "attach_uprobe"))
+ goto cleanup;
+
+ for (i = 0; i < 10; i++)
+ ASSERT_EQ(uprobe_red_zone_test(), 0, "red_zone_intact");
+
+ bpf_link__destroy(link);
+
+cleanup:
+ uprobe_syscall_executed__destroy(skel);
+}
+
static void test_uprobe_error(void)
{
long err = syscall(__NR_uprobe);
ASSERT_EQ(err, -1, "error");
- ASSERT_EQ(errno, ENXIO, "errno");
+ ASSERT_EQ(errno, EPROTO, "errno");
+}
+
+__attribute__((aligned(16)))
+__nocf_check __weak __naked void uprobe_fork_test(void)
+{
+ asm volatile (
+ ".byte 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00\n" /* nop10 */
+ "ret\n"
+ );
+}
+
+static noreturn int child_func(void *arg)
+{
+ struct uprobe_syscall_executed *skel = arg;
+
+ /* Make sure the child's probe is still there and optimized.. */
+ if (memcmp(uprobe_fork_test, lea_rsp, sizeof(lea_rsp)))
+ _exit(1);
+
+ skel->bss->pid = getpid();
+
+ /* .. and it executes properly. */
+ uprobe_fork_test();
+
+ if (skel->bss->executed != 3)
+ _exit(2);
+
+ _exit(0);
+}
+
+static void test_uprobe_fork_optimized(bool clone_vm)
+{
+ struct uprobe_syscall_executed *skel = NULL;
+ unsigned long offset;
+ int pid, status, err;
+ char stack[65535];
+
+ offset = get_uprobe_offset(&uprobe_fork_test);
+ if (!ASSERT_GE(offset, 0, "get_uprobe_offset"))
+ return;
+
+ skel = uprobe_syscall_executed__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ goto cleanup;
+
+ skel->links.test_uprobe = bpf_program__attach_uprobe_opts(skel->progs.test_uprobe,
+ -1, "/proc/self/exe", offset, NULL);
+ if (!ASSERT_OK_PTR(skel->links.test_uprobe, "attach_uprobe"))
+ goto cleanup;
+
+ skel->bss->pid = getpid();
+
+ /* Trigger optimization of uprobe in uprobe_fork_test. */
+ uprobe_fork_test();
+ uprobe_fork_test();
+
+ /* Make sure it got optimied. */
+ if (!ASSERT_OK(memcmp(uprobe_fork_test, lea_rsp, sizeof(lea_rsp)), "optimized"))
+ goto cleanup;
+
+ if (clone_vm) {
+ pid = clone(child_func, stack + sizeof(stack), CLONE_VM|SIGCHLD, skel);
+ if (!ASSERT_GT(pid, 0, "clone"))
+ goto cleanup;
+ } else {
+ pid = fork();
+ if (!ASSERT_GE(pid, 0, "fork"))
+ goto cleanup;
+ if (pid == 0)
+ child_func(skel);
+ }
+
+ /* Wait for the child and verify it exited properly with 0. */
+ err = waitpid(pid, &status, 0);
+ if (ASSERT_EQ(err, pid, "waitpid")) {
+ ASSERT_EQ(WIFEXITED(status), 1, "child_exited");
+ ASSERT_EQ(WEXITSTATUS(status), 0, "child_exit_code");
+ }
+
+cleanup:
+ uprobe_syscall_executed__destroy(skel);
}
static void __test_uprobe_syscall(void)
@@ -783,6 +1056,12 @@ static void __test_uprobe_syscall(void)
test_uprobe_usdt();
if (test__start_subtest("uprobe_race"))
test_uprobe_race();
+ if (test__start_subtest("uprobe_red_zone"))
+ test_uprobe_red_zone();
+ if (test__start_subtest("uprobe_optimized_fork"))
+ test_uprobe_fork_optimized(false);
+ if (test__start_subtest("uprobe_optimized_clone_vm"))
+ test_uprobe_fork_optimized(true);
if (test__start_subtest("uprobe_error"))
test_uprobe_error();
if (test__start_subtest("uprobe_regs_equal"))
diff --git a/tools/testing/selftests/bpf/prog_tests/usdt.c b/tools/testing/selftests/bpf/prog_tests/usdt.c
index 69759b27794d..8004c9568ffa 100644
--- a/tools/testing/selftests/bpf/prog_tests/usdt.c
+++ b/tools/testing/selftests/bpf/prog_tests/usdt.c
@@ -250,9 +250,10 @@ cleanup:
#ifdef __x86_64__
extern void usdt_1(void);
extern void usdt_2(void);
+extern void usdt_red_zone_trigger(void);
static unsigned char nop1[1] = { 0x90 };
-static unsigned char nop1_nop5_combo[6] = { 0x90, 0x0f, 0x1f, 0x44, 0x00, 0x00 };
+static unsigned char nop1_nop10_combo[11] = { 0x90, 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00 };
static void *find_instr(void *fn, unsigned char *instr, size_t cnt)
{
@@ -271,17 +272,17 @@ static void subtest_optimized_attach(void)
__u8 *addr_1, *addr_2;
/* usdt_1 USDT probe has single nop instruction */
- addr_1 = find_instr(usdt_1, nop1_nop5_combo, 6);
- if (!ASSERT_NULL(addr_1, "usdt_1_find_nop1_nop5_combo"))
+ addr_1 = find_instr(usdt_1, nop1_nop10_combo, 11);
+ if (!ASSERT_NULL(addr_1, "usdt_1_find_nop1_nop10_combo"))
return;
addr_1 = find_instr(usdt_1, nop1, 1);
if (!ASSERT_OK_PTR(addr_1, "usdt_1_find_nop1"))
return;
- /* usdt_2 USDT probe has nop,nop5 instructions combo */
- addr_2 = find_instr(usdt_2, nop1_nop5_combo, 6);
- if (!ASSERT_OK_PTR(addr_2, "usdt_2_find_nop1_nop5_combo"))
+ /* usdt_2 USDT probe has nop,nop10 instructions combo */
+ addr_2 = find_instr(usdt_2, nop1_nop10_combo, 11);
+ if (!ASSERT_OK_PTR(addr_2, "usdt_2_find_nop1_nop10_combo"))
return;
skel = test_usdt__open_and_load();
@@ -309,12 +310,12 @@ static void subtest_optimized_attach(void)
bpf_link__destroy(skel->links.usdt_executed);
- /* we expect the nop5 ip */
+ /* we expect the nop10 ip */
skel->bss->expected_ip = (unsigned long) addr_2 + 1;
/*
* Attach program on top of usdt_2 which is probe defined on top
- * of nop1,nop5 combo, so the probe gets optimized on top of nop5.
+ * of nop1,nop10 combo, so the probe gets optimized on top of nop10.
*/
skel->links.usdt_executed = bpf_program__attach_usdt(skel->progs.usdt_executed,
0 /*self*/, "/proc/self/exe",
@@ -328,13 +329,64 @@ static void subtest_optimized_attach(void)
/* nop stays on addr_2 address */
ASSERT_EQ(*addr_2, 0x90, "nop");
- /* call is on addr_2 + 1 address */
- ASSERT_EQ(*(addr_2 + 1), 0xe8, "call");
+ /*
+ * lea -0x80(%rsp), %rsp
+ * call ...
+ */
+ static unsigned char expected[] = { 0x48, 0x8d, 0x64, 0x24, 0x80, 0xe8 };
+
+ ASSERT_MEMEQ(addr_2 + 1, expected, sizeof(expected), "lea_and_call");
ASSERT_EQ(skel->bss->executed, 4, "executed");
cleanup:
test_usdt__destroy(skel);
}
+
+/*
+ * Test that USDT arguments survive nop10 optimization in a function where
+ * the compiler places operands in the red zone.
+ *
+ * Signal handlers are prone to having the compiler place USDT argument
+ * operands in the red zone (below rsp).
+ *
+ * The nop5 optimization used CALL (which pushes a return address to
+ * [rsp-8]), the value at -8(%rsp) was overwritten. The nop10 optimization
+ * should escape that by moving stackpointer below the redzone before
+ * doing the CALL.
+ */
+static void subtest_optimized_red_zone(void)
+{
+ struct test_usdt *skel;
+ int i;
+
+ skel = test_usdt__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ return;
+
+ skel->bss->expected_arg[0] = 0xDEADBEEF;
+ skel->bss->expected_arg[1] = 0xCAFEBABE;
+ skel->bss->expected_arg[2] = 0xFEEDFACE;
+ skel->bss->expected_pid = getpid();
+
+ skel->links.usdt_check_arg = bpf_program__attach_usdt(
+ skel->progs.usdt_check_arg, 0, "/proc/self/exe",
+ "optimized_attach", "usdt_red_zone", NULL);
+ if (!ASSERT_OK_PTR(skel->links.usdt_check_arg, "attach_usdt_red_zone"))
+ goto cleanup;
+
+ for (i = 0; i < 10; i++)
+ usdt_red_zone_trigger();
+
+ ASSERT_EQ(skel->bss->arg_total, 10, "arg_total");
+ ASSERT_EQ(skel->bss->arg_bad, 0, "arg_bad");
+ ASSERT_EQ(skel->bss->arg_last[0], 0xDEADBEEF, "arg_last_1");
+ ASSERT_EQ(skel->bss->arg_last[1], 0xCAFEBABE, "arg_last_2");
+ ASSERT_EQ(skel->bss->arg_last[2], 0xFEEDFACE, "arg_last_3");
+
+cleanup:
+ test_usdt__destroy(skel);
+}
+
#endif
unsigned short test_usdt_100_semaphore SEC(".probes");
@@ -608,6 +660,8 @@ void test_usdt(void)
subtest_basic_usdt(true);
if (test__start_subtest("optimized_attach"))
subtest_optimized_attach();
+ if (test__start_subtest("optimized_red_zone"))
+ subtest_optimized_red_zone();
#endif
if (test__start_subtest("multispec"))
subtest_multispec_usdt();
diff --git a/tools/testing/selftests/bpf/prog_tests/verifier.c b/tools/testing/selftests/bpf/prog_tests/verifier.c
index 5b265af3b1d5..8b439e194bcc 100644
--- a/tools/testing/selftests/bpf/prog_tests/verifier.c
+++ b/tools/testing/selftests/bpf/prog_tests/verifier.c
@@ -23,6 +23,7 @@
#include "verifier_bpf_trap.skel.h"
#include "verifier_bswap.skel.h"
#include "verifier_btf_ctx_access.skel.h"
+#include "verifier_btf_flex_array.skel.h"
#include "verifier_btf_unreliable_prog.skel.h"
#include "verifier_call_large_imm.skel.h"
#include "verifier_cfg.skel.h"
@@ -53,6 +54,7 @@
#include "verifier_iterating_callbacks.skel.h"
#include "verifier_jeq_infer_not_null.skel.h"
#include "verifier_jit_convergence.skel.h"
+#include "verifier_kfunc_perfmon.skel.h"
#include "verifier_ld_ind.skel.h"
#include "verifier_ldsx.skel.h"
#include "verifier_leak_ptr.skel.h"
@@ -79,6 +81,7 @@
#include "verifier_netfilter_retcode.skel.h"
#include "verifier_bpf_fastcall.skel.h"
#include "verifier_or_jmp32_k.skel.h"
+#include "verifier_percpu_addr.skel.h"
#include "verifier_precision.skel.h"
#include "verifier_prevent_map_lookup.skel.h"
#include "verifier_private_stack.skel.h"
@@ -102,6 +105,7 @@
#include "verifier_stack_arg_order.skel.h"
#include "verifier_stack_ptr.skel.h"
#include "verifier_store_release.skel.h"
+#include "verifier_subprog_insn_stats.skel.h"
#include "verifier_subprog_precision.skel.h"
#include "verifier_subprog_topo.skel.h"
#include "verifier_subreg.skel.h"
@@ -184,6 +188,7 @@ void test_verifier_bpf_get_stack(void) { RUN(verifier_bpf_get_stack); }
void test_verifier_bpf_trap(void) { RUN(verifier_bpf_trap); }
void test_verifier_bswap(void) { RUN(verifier_bswap); }
void test_verifier_btf_ctx_access(void) { RUN(verifier_btf_ctx_access); }
+void test_verifier_btf_flex_array(void) { RUN(verifier_btf_flex_array); }
void test_verifier_btf_unreliable_prog(void) { RUN(verifier_btf_unreliable_prog); }
void test_verifier_call_large_imm(void) { RUN(verifier_call_large_imm); }
void test_verifier_cfg(void) { RUN(verifier_cfg); }
@@ -214,6 +219,7 @@ void test_verifier_int_ptr(void) { RUN(verifier_int_ptr); }
void test_verifier_iterating_callbacks(void) { RUN(verifier_iterating_callbacks); }
void test_verifier_jeq_infer_not_null(void) { RUN(verifier_jeq_infer_not_null); }
void test_verifier_jit_convergence(void) { RUN(verifier_jit_convergence); }
+void test_verifier_kfunc_perfmon(void) { RUN(verifier_kfunc_perfmon); }
void test_verifier_load_acquire(void) { RUN(verifier_load_acquire); }
void test_verifier_ld_ind(void) { RUN(verifier_ld_ind); }
void test_verifier_ldsx(void) { RUN(verifier_ldsx); }
@@ -239,6 +245,7 @@ void test_verifier_netfilter_ctx(void) { RUN(verifier_netfilter_ctx); }
void test_verifier_netfilter_retcode(void) { RUN(verifier_netfilter_retcode); }
void test_verifier_bpf_fastcall(void) { RUN(verifier_bpf_fastcall); }
void test_verifier_or_jmp32_k(void) { RUN(verifier_or_jmp32_k); }
+void test_verifier_percpu_addr(void) { RUN(verifier_percpu_addr); }
void test_verifier_precision(void) { RUN(verifier_precision); }
void test_verifier_prevent_map_lookup(void) { RUN(verifier_prevent_map_lookup); }
void test_verifier_private_stack(void) { RUN(verifier_private_stack); }
@@ -262,6 +269,7 @@ void test_verifier_stack_arg(void) { RUN(verifier_stack_arg); }
void test_verifier_stack_arg_order(void) { RUN(verifier_stack_arg_order); }
void test_verifier_stack_ptr(void) { RUN(verifier_stack_ptr); }
void test_verifier_store_release(void) { RUN(verifier_store_release); }
+void test_verifier_subprog_insn_stats(void) { RUN(verifier_subprog_insn_stats); }
void test_verifier_subprog_precision(void) { RUN(verifier_subprog_precision); }
void test_verifier_subprog_topo(void) { RUN(verifier_subprog_topo); }
void test_verifier_subreg(void) { RUN(verifier_subreg); }
diff --git a/tools/testing/selftests/bpf/progs/arena_kfunc.c b/tools/testing/selftests/bpf/progs/arena_kfunc.c
index cdcea889da58..bf0d304e0e59 100644
--- a/tools/testing/selftests/bpf/progs/arena_kfunc.c
+++ b/tools/testing/selftests/bpf/progs/arena_kfunc.c
@@ -27,6 +27,7 @@ volatile u64 stash;
SEC("syscall")
__arch_x86_64
+__arch_arm64
__success __retval(0)
int arena_arg_forms(void *ctx)
{
@@ -70,6 +71,7 @@ int arena_arg_forms(void *ctx)
*/
SEC("syscall")
__arch_x86_64
+__arch_arm64
__success __retval(0)
int arena_arg_rebase(void *ctx)
{
@@ -111,6 +113,7 @@ int arena_arg_rebase(void *ctx)
SEC("syscall")
__arch_x86_64
+__arch_arm64
__success __retval(0)
int arena_args5(void *ctx)
{
@@ -142,6 +145,7 @@ int arena_args5(void *ctx)
SEC("syscall")
__arch_x86_64
+__arch_arm64
__success __retval(0)
int arena_arg_mixed(void *ctx)
{
@@ -169,6 +173,7 @@ int arena_arg_mixed(void *ctx)
/* kernel-side faults on unpopulated pages recover via the scratch page */
SEC("syscall")
__arch_x86_64
+__arch_arm64
__success __retval(0)
int arena_arg_unpopulated(void *ctx)
{
@@ -189,6 +194,7 @@ int arena_arg_unpopulated(void *ctx)
SEC("syscall")
__arch_x86_64
+__arch_arm64
__failure __msg("arena pointer requires a program with an associated arena")
int arena_arg_no_arena(void *ctx)
{
@@ -198,6 +204,7 @@ int arena_arg_no_arena(void *ctx)
SEC("syscall")
__arch_x86_64
+__arch_arm64
__failure __msg("is not a pointer to arena or scalar")
int arena_arg_bad_reg(void *ctx)
{
@@ -213,6 +220,7 @@ int arena_arg_bad_reg(void *ctx)
defined(__BPF_FEATURE_STACK_ARGUMENT)
SEC("syscall")
__arch_x86_64
+__arch_arm64
__failure __msg("arena pointer cannot be a stack argument")
int arena_arg_stack(void *ctx)
{
@@ -223,6 +231,7 @@ int arena_arg_stack(void *ctx)
#else
SEC("syscall")
__arch_x86_64
+__arch_arm64
__description("arena_arg_stack: not supported, dummy test")
__success
int arena_arg_stack(void *ctx)
diff --git a/tools/testing/selftests/bpf/progs/arena_kfunc_jit.c b/tools/testing/selftests/bpf/progs/arena_kfunc_jit.c
index c9b918662616..b5a01cbc33a7 100644
--- a/tools/testing/selftests/bpf/progs/arena_kfunc_jit.c
+++ b/tools/testing/selftests/bpf/progs/arena_kfunc_jit.c
@@ -32,6 +32,10 @@ __jited(" movl %edi, %edi")
__jited(" addq %r12, %rdi")
__jited("...")
__jited(" callq {{.*}}")
+__arch_arm64
+__jited("...")
+__jited(" add x0, x28, w0, uxtw")
+__jited(" {{(bl|mov) .*}}")
__success
int arena_arg_jit_rebase(void *ctx)
{
@@ -48,6 +52,12 @@ __jited(" testl %edi, %edi")
__jited(" je L0")
__jited(" addq %r12, %rdi")
__jited("L0: callq {{.*}}")
+__arch_arm64
+__jited("...")
+__jited(" mov w0, w0")
+__jited(" cbz w0, L0")
+__jited(" add x0, x28, w0, uxtw")
+__jited("L0: {{.*}}")
__success
int arena_arg_jit_nullable(void *ctx)
{
@@ -72,6 +82,16 @@ __jited(" testl %r8d, %r8d")
__jited(" je L0")
__jited(" addq %r12, %r8")
__jited("L0: callq {{.*}}")
+__arch_arm64
+__jited("...")
+__jited(" add x0, x28, w0, uxtw")
+__jited(" add x1, x28, w1, uxtw")
+__jited(" add x2, x28, w2, uxtw")
+__jited(" add x3, x28, w3, uxtw")
+__jited(" mov w4, w4")
+__jited(" cbz w4, L0")
+__jited(" add x4, x28, w4, uxtw")
+__jited("L0: {{.*}}")
__success
int arena_arg_jit_args5(void *ctx)
{
diff --git a/tools/testing/selftests/bpf/progs/arena_spin_lock.c b/tools/testing/selftests/bpf/progs/arena_spin_lock.c
index cf7cda79c16c..92e75ec3844c 100644
--- a/tools/testing/selftests/bpf/progs/arena_spin_lock.c
+++ b/tools/testing/selftests/bpf/progs/arena_spin_lock.c
@@ -23,6 +23,13 @@ int cs_count;
#if defined(ENABLE_ATOMICS_TESTS) && defined(__BPF_FEATURE_ADDR_SPACE_CAST)
arena_spinlock_t __arena lock;
int test_skip = 1;
+
+/*
+ * Storage for the queue nodes declared by bpf_arena_spin_lock.h. Each program
+ * linking the arena spinlock provides exactly one definition; libarena's lives
+ * in libarena/src/common.bpf.c.
+ */
+struct arena_qnode __arena __hidden qnodes[_Q_MAX_CPUS][_Q_MAX_NODES];
#else
int test_skip = 2;
#endif
diff --git a/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__untrusted_write.c b/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__untrusted_write.c
new file mode 100644
index 000000000000..688c2a049ae3
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__untrusted_write.c
@@ -0,0 +1,65 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include "bpf_experimental.h"
+#include "bpf_qdisc_common.h"
+#include "bpf_misc.h"
+
+char _license[] SEC("license") = "GPL";
+
+SEC("struct_ops")
+__failure __msg("only read is supported")
+int BPF_PROG(untrusted_write, struct sk_buff *skb, struct Qdisc *sch,
+ struct bpf_sk_buff_ptr *to_free)
+{
+ struct Qdisc *next = sch->next_sched;
+
+ /*
+ * sch is trusted, but the walk of next_sched yields a plain
+ * PTR_TO_BTF_ID which may fault on a dereference. A store through
+ * it does not get an exception table entry, there is no probed
+ * store to rewrite it into, hence it has to be rejected before
+ * bpf_qdisc_btf_struct_access() gets to allow the write to limit.
+ */
+ next->limit = 1000;
+
+ bpf_qdisc_skb_drop(skb, to_free);
+ return NET_XMIT_DROP;
+}
+
+SEC("struct_ops")
+__auxiliary
+struct sk_buff *BPF_PROG(bpf_qdisc_test_dequeue, struct Qdisc *sch)
+{
+ return NULL;
+}
+
+SEC("struct_ops")
+__auxiliary
+int BPF_PROG(bpf_qdisc_test_init, struct Qdisc *sch, struct nlattr *opt,
+ struct netlink_ext_ack *extack)
+{
+ return 0;
+}
+
+SEC("struct_ops")
+__auxiliary
+void BPF_PROG(bpf_qdisc_test_reset, struct Qdisc *sch)
+{
+}
+
+SEC("struct_ops")
+__auxiliary
+void BPF_PROG(bpf_qdisc_test_destroy, struct Qdisc *sch)
+{
+}
+
+SEC(".struct_ops")
+struct Qdisc_ops test = {
+ .enqueue = (void *)untrusted_write,
+ .dequeue = (void *)bpf_qdisc_test_dequeue,
+ .init = (void *)bpf_qdisc_test_init,
+ .reset = (void *)bpf_qdisc_test_reset,
+ .destroy = (void *)bpf_qdisc_test_destroy,
+ .id = "bpf_qdisc_test",
+};
diff --git a/tools/testing/selftests/bpf/progs/btf_data.c b/tools/testing/selftests/bpf/progs/btf_data.c
index ec34f7a6e038..8082c13490ab 100644
--- a/tools/testing/selftests/bpf/progs/btf_data.c
+++ b/tools/testing/selftests/bpf/progs/btf_data.c
@@ -68,3 +68,23 @@ int kfunc_d(struct root_struct *a, struct root_struct *b)
{
return 0;
}
+
+int kfunc_e(struct root_struct *a__arena,
+ struct root_struct *b__arena__nullable,
+ struct root_struct *c__arena,
+ struct root_struct *d__arena__nullable,
+ struct root_struct *e__arena)
+{
+ return 0;
+}
+
+int kfunc_f(struct root_struct *a, struct root_struct *b__arena, int flags)
+{
+ return 0;
+}
+
+struct root_struct *kfunc_g(struct root_struct *a__arena,
+ struct root_struct *b__arena__nullable)
+{
+ return a__arena;
+}
diff --git a/tools/testing/selftests/bpf/progs/dynptr_fail.c b/tools/testing/selftests/bpf/progs/dynptr_fail.c
index beaa73dc35f5..1cd61d72c166 100644
--- a/tools/testing/selftests/bpf/progs/dynptr_fail.c
+++ b/tools/testing/selftests/bpf/progs/dynptr_fail.c
@@ -1590,6 +1590,7 @@ __u32 hdr_size = sizeof(struct ethhdr);
/* Can't pass in variable-sized len to bpf_dynptr_slice */
SEC("?tc")
__failure __msg("must be a known constant")
+__msg("requires this memory size to be a verifier-known constant")
int dynptr_slice_var_len1(struct __sk_buff *skb)
{
struct bpf_dynptr ptr;
@@ -1609,6 +1610,7 @@ int dynptr_slice_var_len1(struct __sk_buff *skb)
/* Can't pass in variable-sized len to bpf_dynptr_slice */
SEC("?tc")
__failure __msg("must be a known constant")
+__msg("requires this memory size to be a verifier-known constant")
int dynptr_slice_var_len2(struct __sk_buff *skb)
{
char buffer[sizeof(struct ethhdr)] = {};
diff --git a/tools/testing/selftests/bpf/progs/dynptr_success.c b/tools/testing/selftests/bpf/progs/dynptr_success.c
index e0745b6e467e..b668ebd61fc7 100644
--- a/tools/testing/selftests/bpf/progs/dynptr_success.c
+++ b/tools/testing/selftests/bpf/progs/dynptr_success.c
@@ -10,6 +10,7 @@
#include "errno.h"
#define PAGE_SIZE_64K 65536
+#define TEST_SKB_LINEAR_SIZE (sizeof(struct ethhdr) + sizeof(struct iphdr))
char _license[] SEC("license") = "GPL";
@@ -212,6 +213,25 @@ int test_dynptr_skb_data(struct __sk_buff *skb)
}
SEC("?tc")
+int test_dynptr_skb_slice_non_linear(struct __sk_buff *skb)
+{
+ struct bpf_dynptr ptr;
+ void *data;
+
+ if (bpf_dynptr_from_skb(skb, 0, &ptr)) {
+ err = 1;
+ return 1;
+ }
+
+ /* Ensure we cannot read past the end of the buffer. */
+ data = bpf_dynptr_slice(&ptr, TEST_SKB_LINEAR_SIZE + 1, NULL, 1);
+ if (data)
+ err = 2;
+
+ return 1;
+}
+
+SEC("?tc")
int test_dynptr_skb_meta_data(struct __sk_buff *skb)
{
struct bpf_dynptr meta;
diff --git a/tools/testing/selftests/bpf/progs/exceptions.c b/tools/testing/selftests/bpf/progs/exceptions.c
index c8d716fbd419..91c81971e58c 100644
--- a/tools/testing/selftests/bpf/progs/exceptions.c
+++ b/tools/testing/selftests/bpf/progs/exceptions.c
@@ -212,6 +212,36 @@ int exception_throw_subprog(struct __sk_buff *ctx)
return 0;
}
+u64 exception_cb_stack_src = 0x1234;
+
+/*
+ * The address handed to the helper has to be this callback's own stack
+ * slot, not one from a frame that is already gone.
+ */
+__noinline int exception_cb_stack(u64 cookie)
+{
+ volatile u64 val = 0xdead;
+
+ bpf_probe_read_kernel((void *)&val, sizeof(val), &exception_cb_stack_src);
+ return val;
+}
+
+/* Throws from a subprogram that has a stack of its own. */
+__noinline static int throwing_subprog_stack(struct __sk_buff *ctx)
+{
+ volatile u64 pad[4] = {};
+
+ bpf_throw(pad[0]);
+ return 0;
+}
+
+SEC("tc")
+__exception_cb(exception_cb_stack)
+int exception_throw_subprog_stack_cb(struct __sk_buff *ctx)
+{
+ return throwing_subprog_stack(ctx);
+}
+
__noinline int assert_nz_gfunc(u64 c)
{
volatile u64 cookie = c;
diff --git a/tools/testing/selftests/bpf/progs/iters.c b/tools/testing/selftests/bpf/progs/iters.c
index 62d7df9e80be..c6699159dacd 100644
--- a/tools/testing/selftests/bpf/progs/iters.c
+++ b/tools/testing/selftests/bpf/progs/iters.c
@@ -2149,4 +2149,43 @@ __naked int stack_misc_vs_scalar_in_a_loop(void)
);
}
+__used
+static int loop_cb5(int i, __u64 *ctx)
+{
+ /* unsafe on a second iteration */
+ small_arr[*ctx] = i;
+ *ctx = 100500;
+ return 0;
+}
+
+SEC("raw_tp")
+__flag(BPF_F_TEST_STATE_FREQ)
+__failure __msg("memory access is {{.*}} and is outside of the object of size 64")
+__naked void loop_counter_precision_2nd_iter(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "*(u64 *)(r10 - 8) = 0;"
+ "r1 = 2;"
+ "if r0 == 42 goto +1;"
+ "r1 = 1;"
+ "r2 = loop_cb5 ll;"
+ "r3 = r10;"
+ "r3 += -8;"
+ "r4 = 0;"
+ /*
+ * Explore with nr_loops=1 on a first path and nr_loops=2 on a second path.
+ * Buggy verifier did not propagate r1 precision properly,
+ * and thus checkpoints created for nr_loops=1 case matched nr_loops=2 case.
+ */
+ "call %[bpf_loop];"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm(bpf_loop),
+ __imm(bpf_get_prandom_u32)
+ : __clobber_all
+ );
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/iters_state_safety.c b/tools/testing/selftests/bpf/progs/iters_state_safety.c
index 646026430e9b..e5bb9fe6d5e5 100644
--- a/tools/testing/selftests/bpf/progs/iters_state_safety.c
+++ b/tools/testing/selftests/bpf/progs/iters_state_safety.c
@@ -52,6 +52,28 @@ int create_and_destroy(void *ctx)
return 0;
}
+/* fp+0 is not a stack slot. bpf_get_spi(0) used to alias spi 0 (fp-8). */
+SEC("?raw_tp")
+__failure __msg("cannot pass in iter at an offset=0")
+int destroy_fp0_fail(void *ctx)
+{
+ struct bpf_iter_num iter;
+
+ asm volatile ("r1 = %[iter];"
+ "r2 = 0;"
+ "r3 = 1000;"
+ "call %[bpf_iter_num_new];"
+ /* r10 is fp+0, one byte above the top of the BPF stack */
+ "r1 = r10;"
+ "call %[bpf_iter_num_destroy];"
+ :
+ : __imm_ptr(iter), ITER_HELPERS
+ : __clobber_common
+ );
+
+ return 0;
+}
+
SEC("?raw_tp")
__failure __msg("Unreleased reference id=1")
int create_and_forget_to_destroy_fail(void *ctx)
diff --git a/tools/testing/selftests/bpf/progs/iters_task_failure.c b/tools/testing/selftests/bpf/progs/iters_task_failure.c
index fe3663dedbe1..566a1d3dffea 100644
--- a/tools/testing/selftests/bpf/progs/iters_task_failure.c
+++ b/tools/testing/selftests/bpf/progs/iters_task_failure.c
@@ -62,6 +62,52 @@ int BPF_PROG(iter_tasks_lock_and_unlock)
}
SEC("?fentry.s/" SYS_PREFIX "sys_getpgid")
+__failure __msg("expected an RCU CS when using bpf_iter_task_next")
+__flag(BPF_F_TEST_STATE_FREQ)
+int BPF_PROG(iter_tasks_rcu_state_pruning)
+{
+ struct bpf_iter_task it;
+
+ asm volatile (
+ "call %[bpf_rcu_read_lock];"
+ "r1 = %[it];"
+ "r2 = 0;"
+ "r3 = 0;" /* BPF_TASK_ITER_ALL_PROCS */
+ "call %[bpf_iter_task_new];"
+
+ "call %[bpf_get_prandom_u32];"
+ "if w0 == 0 goto unprotected_%=;"
+
+ /* Keep the outer RCU lock active on the straight-line path. */
+ "call %[bpf_rcu_read_lock];"
+ "call %[bpf_rcu_read_unlock];"
+ "goto merge_%=;"
+
+ "unprotected_%=:"
+ /* Create an unprotected gap on the taken path. */
+ "call %[bpf_rcu_read_unlock];"
+ "call %[bpf_rcu_read_lock];"
+
+ "merge_%=: r1 = %[it];"
+ "call %[bpf_iter_task_next];"
+ "r1 = %[it];"
+ "call %[bpf_iter_task_destroy];"
+ "call %[bpf_rcu_read_unlock];"
+ :
+ : __imm_ptr(it),
+ __imm(bpf_get_prandom_u32),
+ __imm(bpf_iter_task_new),
+ __imm(bpf_iter_task_next),
+ __imm(bpf_iter_task_destroy),
+ __imm(bpf_rcu_read_lock),
+ __imm(bpf_rcu_read_unlock)
+ : __clobber_common
+ );
+
+ return 0;
+}
+
+SEC("?fentry.s/" SYS_PREFIX "sys_getpgid")
__failure __msg("expected an RCU CS when using bpf_iter_css_next")
int BPF_PROG(iter_css_lock_and_unlock)
{
diff --git a/tools/testing/selftests/bpf/progs/ksock_common.h b/tools/testing/selftests/bpf/progs/ksock_common.h
new file mode 100644
index 000000000000..01edaeb9fdd4
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/ksock_common.h
@@ -0,0 +1,78 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/* Copyright (c) 2026 Isovalent */
+
+#ifndef _KSOCK_COMMON_H
+#define _KSOCK_COMMON_H
+
+#include "errno.h"
+
+#define SOCK_DGRAM 2
+#define IPPROTO_UDP 17
+
+struct bpf_ksock *bpf_ksock_create(const struct bpf_ksock_create_opts *opts,
+ u32 opts__sz, int *err__uninit) __ksym;
+int bpf_ksock_connect(struct bpf_ksock *ks, const union bpf_ksock_addr *addr,
+ u32 addr__sz) __ksym;
+struct bpf_ksock *bpf_ksock_acquire(struct bpf_ksock *ks) __ksym;
+void bpf_ksock_release(struct bpf_ksock *ks) __ksym;
+int bpf_ksock_send(struct bpf_ksock *ks, const void *data, u32 data__sz) __ksym;
+void bpf_rcu_read_lock(void) __ksym;
+void bpf_rcu_read_unlock(void) __ksym;
+
+struct __ksock_ctx_value {
+ struct bpf_ksock __kptr * ctx;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __type(key, int);
+ __type(value, struct __ksock_ctx_value);
+ __uint(max_entries, 1);
+} __ksock_ctx_map SEC(".maps");
+
+static inline struct __ksock_ctx_value *ksock_ctx_value_lookup(void)
+{
+ u32 key = 0;
+
+ return bpf_map_lookup_elem(&__ksock_ctx_map, &key);
+}
+
+static inline struct bpf_ksock *ksock_ctx_get(void)
+{
+ struct __ksock_ctx_value *v;
+ struct bpf_ksock *ks = NULL, *tmp;
+
+ v = ksock_ctx_value_lookup();
+ if (!v)
+ return NULL;
+
+ bpf_rcu_read_lock();
+ tmp = v->ctx;
+ if (tmp)
+ ks = bpf_ksock_acquire(tmp);
+ bpf_rcu_read_unlock();
+
+ return ks;
+}
+
+static inline int ksock_ctx_insert(struct bpf_ksock *ctx)
+{
+ struct __ksock_ctx_value *v;
+ struct bpf_ksock *old;
+
+ v = ksock_ctx_value_lookup();
+ if (!v) {
+ bpf_ksock_release(ctx);
+ return -ENOENT;
+ }
+
+ old = bpf_kptr_xchg(&v->ctx, ctx);
+ if (old) {
+ bpf_ksock_release(old);
+ return -EEXIST;
+ }
+
+ return 0;
+}
+
+#endif /* _KSOCK_COMMON_H */
diff --git a/tools/testing/selftests/bpf/progs/ksock_lsm.c b/tools/testing/selftests/bpf/progs/ksock_lsm.c
new file mode 100644
index 000000000000..9808451098ef
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/ksock_lsm.c
@@ -0,0 +1,72 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Isovalent */
+
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+#include <bpf/bpf_endian.h>
+#include "bpf_tracing_net.h"
+#include "ksock_common.h"
+
+char send_data[32] = "hello from bpf ksock";
+
+__be32 ipv4_remote;
+__u16 remote_port;
+int target_pid;
+int send_ret = -1;
+
+SEC("syscall")
+int ksock_setup(void *ctx)
+{
+ struct bpf_ksock_create_opts create_opts = {};
+ union bpf_ksock_addr addr = {};
+ struct bpf_ksock *ks;
+ int err = 0;
+
+ create_opts.family = AF_INET;
+ create_opts.type = SOCK_DGRAM;
+ create_opts.protocol = IPPROTO_UDP;
+
+ ks = bpf_ksock_create(&create_opts, sizeof(create_opts), &err);
+ if (!ks)
+ return err;
+
+ addr.sin.sin_family = AF_INET;
+ addr.sin.sin_port = bpf_htons(remote_port);
+ addr.sin.sin_addr.s_addr = ipv4_remote;
+
+ err = bpf_ksock_connect(ks, &addr, sizeof(addr));
+ if (err) {
+ bpf_ksock_release(ks);
+ return err;
+ }
+
+ err = ksock_ctx_insert(ks);
+ if (err && err != -EEXIST)
+ return err;
+ return 0;
+}
+
+SEC("lsm.s/socket_bind")
+int BPF_PROG(ksock_socket_bind, struct socket *sock, struct sockaddr *address,
+ int addrlen, int ret)
+{
+ struct bpf_ksock *ks;
+ u32 pid = bpf_get_current_pid_tgid() >> 32;
+
+ if (ret || pid != target_pid)
+ return ret;
+
+ ks = ksock_ctx_get();
+ if (!ks) {
+ send_ret = -ENOENT;
+ return ret;
+ }
+
+ send_ret = bpf_ksock_send(ks, send_data, sizeof(send_data));
+ bpf_ksock_release(ks);
+
+ return ret;
+}
+
+char __license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/ksock_lsm_verifier.c b/tools/testing/selftests/bpf/progs/ksock_lsm_verifier.c
new file mode 100644
index 000000000000..fd2ccfdb5802
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/ksock_lsm_verifier.c
@@ -0,0 +1,35 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Isovalent */
+
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+#include "bpf_misc.h"
+#include "ksock_common.h"
+
+char send_data[11] = "dummy data";
+
+SEC("lsm.s/socket_sendmsg")
+__description("bpf_ksock_send is rejected from socket_sendmsg LSM hook")
+__failure __msg("calling kernel function bpf_ksock_send is not allowed")
+int BPF_PROG(ksock_socket_sendmsg, struct socket *sock, struct msghdr *msg,
+ int size, int ret)
+{
+ struct __ksock_ctx_value *v;
+ struct bpf_ksock *ks;
+
+ v = ksock_ctx_value_lookup();
+ if (!v)
+ return ret;
+
+ ks = bpf_kptr_xchg(&v->ctx, NULL);
+ if (!ks)
+ return ret;
+
+ bpf_ksock_send(ks, send_data, sizeof(send_data));
+ bpf_ksock_release(ks);
+
+ return ret;
+}
+
+char __license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/ksock_wq.c b/tools/testing/selftests/bpf/progs/ksock_wq.c
new file mode 100644
index 000000000000..16a1873d132e
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/ksock_wq.c
@@ -0,0 +1,62 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Isovalent */
+
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include "bpf_experimental.h"
+#include "bpf_tracing_net.h"
+#include "errno.h"
+#include "ksock_common.h"
+
+struct ksock_wq_value {
+ struct bpf_wq work;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __uint(max_entries, 1);
+ __type(key, u32);
+ __type(value, struct ksock_wq_value);
+} work_map SEC(".maps");
+
+int create_err;
+u32 callback_done;
+
+static int ksock_wq_callback(void *map, int *key, void *value)
+{
+ struct bpf_ksock_create_opts opts = {
+ .family = AF_INET,
+ .type = SOCK_DGRAM,
+ .protocol = IPPROTO_UDP,
+ };
+ struct bpf_ksock *ks;
+ int err = 0;
+
+ ks = bpf_ksock_create(&opts, sizeof(opts), &err);
+ if (ks)
+ bpf_ksock_release(ks);
+ create_err = err;
+ __sync_fetch_and_add(&callback_done, 1);
+ return 0;
+}
+
+SEC("syscall")
+int ksock_wq_start(void *ctx)
+{
+ struct ksock_wq_value *value;
+ u32 key = 0;
+ int err;
+
+ value = bpf_map_lookup_elem(&work_map, &key);
+ if (!value)
+ return -ENOENT;
+ err = bpf_wq_init(&value->work, &work_map, 0);
+ if (err)
+ return err;
+ err = bpf_wq_set_callback(&value->work, ksock_wq_callback, 0);
+ if (err)
+ return err;
+ return bpf_wq_start(&value->work, 0);
+}
+
+char __license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/map_kptr_fail.c b/tools/testing/selftests/bpf/progs/map_kptr_fail.c
index f11848dfa78f..eee35d203b66 100644
--- a/tools/testing/selftests/bpf/progs/map_kptr_fail.c
+++ b/tools/testing/selftests/bpf/progs/map_kptr_fail.c
@@ -386,7 +386,16 @@ int kptr_xchg_possibly_null(struct __sk_buff *ctx)
}
SEC("?tc")
+/*
+ * A compiler with BPF_ST folds the constant into a store-immediate, which the
+ * verifier rejects on a different path (and with a different message) than the
+ * BPF_STX form.
+ */
+#ifdef __BPF_FEATURE_ST
+__failure __msg("BPF_ST imm must be 0 when storing to kptr at off=8")
+#else
__failure __msg("invalid kptr access, R")
+#endif
int reject_scalar_store_to_kptr(struct __sk_buff *ctx)
{
struct map_value *v;
@@ -400,4 +409,41 @@ int reject_scalar_store_to_kptr(struct __sk_buff *ctx)
return 0;
}
+SEC("?tc")
+__description("reject imprecise scalar store to kptr after state pruning")
+__failure __msg("invalid kptr access, R7 type=scalar")
+__naked void reject_imprecise_scalar_store_to_kptr(void)
+{
+ asm volatile (
+ "r0 = 0;"
+ "*(u32 *)(r10 - 4) = r0;"
+ "r2 = r10;"
+ "r2 += -4;"
+ "r1 = %[array_map] ll;"
+ "call %[bpf_map_lookup_elem];"
+ "if r0 == 0 goto l2_%=;"
+ "r6 = r0;"
+ "r9 = *(u64 *)(r6 + 0);"
+ "if r9 != 0 goto l0_%=;"
+ "r7 = 0;"
+ ".rept 10;"
+ "r5 = 1;"
+ ".endr;"
+ "goto l1_%=;"
+ "l0_%=:"
+ "r7 = 0x4141414141414141 ll;"
+ ".rept 10;"
+ "r5 = 1;"
+ ".endr;"
+ "l1_%=:"
+ "*(u64 *)(r6 + 8) = r7;"
+ "l2_%=:"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(array_map)
+ : __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/mem_rdonly_untrusted.c b/tools/testing/selftests/bpf/progs/mem_rdonly_untrusted.c
index b91271d4caa4..3e0d4f687aaa 100644
--- a/tools/testing/selftests/bpf/progs/mem_rdonly_untrusted.c
+++ b/tools/testing/selftests/bpf/progs/mem_rdonly_untrusted.c
@@ -3,6 +3,7 @@
#include <vmlinux.h>
#include <bpf/bpf_core_read.h>
#include "bpf_misc.h"
+#include "bpf_kfuncs.h"
#include "../test_kmods/bpf_testmod_kfunc.h"
SEC("tp_btf/sys_enter")
@@ -164,6 +165,239 @@ int mixed_mem_type(void *ctx)
return *p;
}
+struct {
+ __uint(type, BPF_MAP_TYPE_RINGBUF);
+ __uint(max_entries, 4096);
+} ringbuf SEC(".maps");
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __uint(max_entries, 1);
+ __type(key, u32);
+ __type(value, u64);
+} array SEC(".maps");
+
+char dynptr_data[8];
+
+int zero;
+
+SEC("socket")
+__success
+__log_level(2)
+__msg("r8 = *(u64 *)(r7 +0){{.*}}R7=untrusted_ptr_sock")
+__msg("r8 = *(u64 *)(r7 +0){{.*}}R7=ringbuf_mem")
+__retval(0)
+int mixed_mem_untrusted_btf_id_type(void *ctx)
+{
+ u64 *p, *q, v;
+
+ p = bpf_ringbuf_reserve(&ringbuf, sizeof(*p), 0);
+ if (!p)
+ return 1;
+ *p = 42;
+ q = bpf_rdonly_cast(0, bpf_core_type_id_kernel(struct sock));
+ /*
+ * The load below is reached with PTR_TO_MEM | MEM_RINGBUF on one
+ * path and with PTR_TO_BTF_ID | PTR_UNTRUSTED on the other. The
+ * merged type has to keep the BPF_PROBE_MEM rewrite, otherwise
+ * the NULL deref taken at runtime panics the kernel instead of
+ * returning 0.
+ */
+ asm volatile (
+ "r7 = %[p];"
+ "if %[zero] != 0 goto +1;"
+ "r7 = %[q];"
+ "r8 = *(u64 *)(r7 + 0);"
+ "%[v] = r8;"
+ : [v]"=r"(v)
+ : [p]"r"(p),
+ [q]"r"(q),
+ [zero]"r"(zero)
+ : "r7", "r8");
+ bpf_ringbuf_discard(p, 0);
+ return v;
+}
+
+SEC("socket")
+__success
+__log_level(2)
+__msg("r8 = *(u32 *)(r7 +0){{.*}}R7=ptr_nameidata")
+__msg("r8 = *(u32 *)(r7 +0){{.*}}R7=ringbuf_mem")
+__retval(0)
+int mixed_mem_btf_id_type(void *ctx)
+{
+ struct task_struct *task;
+ u32 *p, *q;
+ u64 v;
+
+ p = bpf_ringbuf_reserve(&ringbuf, sizeof(*p), 0);
+ if (!p)
+ return 1;
+ *p = 42;
+ task = bpf_get_current_task_btf();
+ /*
+ * A plain BTF pointer walk yields a bare PTR_TO_BTF_ID, and
+ * task->nameidata is NULL unless the task currently is in the
+ * middle of a path lookup.
+ */
+ q = (u32 *)&task->nameidata->flags;
+ /*
+ * Same as above, except that the other path yields a bare
+ * PTR_TO_BTF_ID. Merging it with PTR_TO_MEM used to drop the
+ * BPF_PROBE_MEM rewrite the bare PTR_TO_BTF_ID would have
+ * gotten on its own.
+ */
+ asm volatile (
+ "r7 = %[p];"
+ "if %[zero] != 0 goto +1;"
+ "r7 = %[q];"
+ "r8 = *(u32 *)(r7 + 0);"
+ "%[v] = r8;"
+ : [v]"=r"(v)
+ : [p]"r"(p),
+ [q]"r"(q),
+ [zero]"r"(zero)
+ : "r7", "r8");
+ bpf_ringbuf_discard(p, 0);
+ return v;
+}
+
+SEC("socket")
+__success
+__log_level(2)
+__msg("r8 = *(u32 *)(r7 +0){{.*}}R7=ptr_nameidata")
+__msg("r8 = *(u32 *)(r7 +0){{.*}}R7=rdonly_mem")
+__retval(0)
+int mixed_rdonly_mem_btf_id_type(void *ctx)
+{
+ struct task_struct *task;
+ struct bpf_dynptr dptr;
+ char buf[sizeof(u32)];
+ u32 *p, *q;
+ u64 v;
+
+ if (bpf_dynptr_from_mem(dynptr_data, sizeof(dynptr_data), 0, &dptr))
+ return 1;
+ p = bpf_dynptr_slice(&dptr, 0, buf, sizeof(buf));
+ if (!p)
+ return 1;
+ task = bpf_get_current_task_btf();
+ q = (u32 *)&task->nameidata->flags;
+ /*
+ * Same as above, except that the PTR_TO_MEM side already carries
+ * MEM_RDONLY. Merging it with a bare PTR_TO_BTF_ID used to yield
+ * PTR_TO_MEM | MEM_RDONLY, which is not rewritten either since
+ * only its PTR_UNTRUSTED variant is.
+ */
+ asm volatile (
+ "r7 = %[p];"
+ "if %[zero] != 0 goto +1;"
+ "r7 = %[q];"
+ "r8 = *(u32 *)(r7 + 0);"
+ "%[v] = r8;"
+ : [v]"=r"(v)
+ : [p]"r"(p),
+ [q]"r"(q),
+ [zero]"r"(zero)
+ : "r7", "r8");
+ return v;
+}
+
+SEC("socket")
+__success
+__log_level(2)
+__msg("r8 = *(u64 *)(r7 +0){{.*}}R7=ringbuf_mem")
+__msg("r8 = *(u64 *)(r7 +0){{.*}}R7=rdonly_untrusted_mem")
+__retval(0)
+int mixed_mem_mem_type(void *ctx)
+{
+ u64 *p, *q, v;
+
+ p = bpf_ringbuf_reserve(&ringbuf, sizeof(*p), 0);
+ if (!p)
+ return 1;
+ *p = 42;
+ q = bpf_rdonly_cast(0, 0);
+ /*
+ * Both paths are PTR_TO_MEM based, so they used to not trip the
+ * type mismatch check and skipped the merge altogether, leaving
+ * the insn with the PTR_TO_MEM | MEM_RINGBUF recorded first and
+ * hence without the BPF_PROBE_MEM rewrite the other path needs.
+ */
+ asm volatile (
+ "r7 = %[q];"
+ "if %[zero] == 0 goto +1;"
+ "r7 = %[p];"
+ "r8 = *(u64 *)(r7 + 0);"
+ "%[v] = r8;"
+ : [v]"=r"(v)
+ : [p]"r"(p),
+ [q]"r"(q),
+ [zero]"r"(zero)
+ : "r7", "r8");
+ bpf_ringbuf_discard(p, 0);
+ return v;
+}
+
+SEC("socket")
+__failure
+__msg("same insn cannot be used with different pointers")
+int mixed_map_value_mem_type(void *ctx)
+{
+ u64 *p, *q, v;
+ u32 key = 0;
+
+ p = bpf_map_lookup_elem(&array, &key);
+ if (!p)
+ return 1;
+ q = bpf_rdonly_cast(0, 0);
+ /*
+ * PTR_TO_MAP_VALUE is neither PTR_TO_MEM nor PTR_TO_BTF_ID based,
+ * so it cannot be merged into a type which keeps the BPF_PROBE_MEM
+ * rewrite the PTR_TO_MEM | MEM_RDONLY | PTR_UNTRUSTED of the other
+ * path needs. Both bases were mismatch ok, hence the load used to be
+ * accepted with the PTR_TO_MAP_VALUE recorded and the NULL deref on
+ * the second path panicked the kernel.
+ */
+ asm volatile (
+ "r7 = %[q];"
+ "if %[zero] == 0 goto +1;"
+ "r7 = %[p];"
+ "r8 = *(u64 *)(r7 + 0);"
+ "%[v] = r8;"
+ : [v]"=r"(v)
+ : [p]"r"(p),
+ [q]"r"(q),
+ [zero]"r"(zero)
+ : "r7", "r8");
+ return v;
+}
+
+SEC("socket")
+__failure
+__msg("same insn cannot be used with different pointers")
+int mixed_stack_mem_type(void *ctx)
+{
+ u64 *p = bpf_rdonly_cast(0, 0);
+ u64 s = 42, v;
+
+ /*
+ * Same as above, but for a PTR_TO_STACK on the other path.
+ */
+ asm volatile (
+ "r7 = %[p];"
+ "if %[zero] == 0 goto +1;"
+ "r7 = %[s];"
+ "r8 = *(u64 *)(r7 + 0);"
+ "%[v] = r8;"
+ : [v]"=r"(v)
+ : [p]"r"(p),
+ [s]"r"(&s),
+ [zero]"r"(zero)
+ : "r7", "r8");
+ return v;
+}
+
__attribute__((__aligned__(8)))
u8 global[] = {
0x11, 0x22, 0x33, 0x44,
diff --git a/tools/testing/selftests/bpf/progs/percpu_alloc_fail.c b/tools/testing/selftests/bpf/progs/percpu_alloc_fail.c
index 08379c3b6a03..3701f4ea58c7 100644
--- a/tools/testing/selftests/bpf/progs/percpu_alloc_fail.c
+++ b/tools/testing/selftests/bpf/progs/percpu_alloc_fail.c
@@ -33,6 +33,20 @@ struct {
__type(value, struct elem);
} array SEC(".maps");
+struct kernel_percpu_elem {
+ struct task_struct __percpu_kptr *task;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __uint(max_entries, 1);
+ __type(key, int);
+ __type(value, struct kernel_percpu_elem);
+} kernel_percpu_array SEC(".maps");
+
+struct task_struct *bpf_task_from_pid(s32 pid) __ksym;
+void bpf_task_release(struct task_struct *p) __ksym;
+
long ret;
SEC("?fentry/bpf_fentry_test1")
@@ -137,6 +151,51 @@ int BPF_PROG(test_array_map_5)
return 0;
}
+SEC("?syscall")
+__failure __msg("invalid kptr access, R2 type=trusted_ptr_ expected=ptr_task_struct")
+int reject_kernel_ptr_into_percpu_kptr(void *ctx)
+{
+ struct kernel_percpu_elem *e;
+ struct task_struct *p, *old;
+ int index = 0;
+
+ e = bpf_map_lookup_elem(&kernel_percpu_array, &index);
+ if (!e)
+ return 0;
+
+ p = bpf_task_from_pid(1);
+ if (!p)
+ return 0;
+
+ old = bpf_kptr_xchg(&e->task, p);
+ if (old)
+ bpf_task_release(old);
+ return 0;
+}
+
+SEC("?fentry.s/bpf_fentry_test1")
+__failure __msg("invalid kptr access, R2 type=ptr_ expected=ptr_val_t")
+int BPF_PROG(reject_plain_alloc_into_percpu_kptr)
+{
+ struct val_t __percpu_kptr *old;
+ struct val_t *p;
+ struct elem *e;
+ int index = 0;
+
+ e = bpf_map_lookup_elem(&array, &index);
+ if (!e)
+ return 0;
+
+ p = bpf_obj_new(struct val_t);
+ if (!p)
+ return 0;
+
+ old = bpf_kptr_xchg(&e->pc, p);
+ if (old)
+ bpf_percpu_obj_drop(old);
+ return 0;
+}
+
SEC("?fentry.s/bpf_fentry_test1")
__failure __msg("bpf_percpu_obj_new type ID argument must be of a struct of scalars")
int BPF_PROG(test_array_map_6)
diff --git a/tools/testing/selftests/bpf/progs/preempt_lock.c b/tools/testing/selftests/bpf/progs/preempt_lock.c
index 6d5fce7e6ffc..81c459435680 100644
--- a/tools/testing/selftests/bpf/progs/preempt_lock.c
+++ b/tools/testing/selftests/bpf/progs/preempt_lock.c
@@ -115,6 +115,58 @@ int preempt_sleepable_helper(void *ctx)
return 0;
}
+SEC("?uprobe.s")
+__failure __msg("sleepable helper bpf_get_stack#")
+int preempt_sleepable_get_stack(struct pt_regs *ctx)
+{
+ struct bpf_stack_build_id stack;
+
+ bpf_preempt_disable();
+ bpf_get_stack(ctx, &stack, sizeof(stack),
+ BPF_F_USER_STACK | BPF_F_USER_BUILD_ID);
+ bpf_preempt_enable();
+ return 0;
+}
+
+SEC("?uprobe.s")
+__failure __msg("sleepable helper bpf_get_task_stack#")
+int preempt_sleepable_get_task_stack(void *ctx)
+{
+ struct bpf_stack_build_id stack;
+ struct task_struct *task;
+
+ task = bpf_get_current_task_btf();
+ bpf_preempt_disable();
+ bpf_get_task_stack(task, &stack, sizeof(stack),
+ BPF_F_USER_STACK | BPF_F_USER_BUILD_ID);
+ bpf_preempt_enable();
+ return 0;
+}
+
+SEC("?uprobe.s")
+__success
+int sleepable_get_stack(struct pt_regs *ctx)
+{
+ struct bpf_stack_build_id stack;
+
+ bpf_get_stack(ctx, &stack, sizeof(stack),
+ BPF_F_USER_STACK | BPF_F_USER_BUILD_ID);
+ return 0;
+}
+
+SEC("?uprobe.s")
+__success
+int sleepable_get_task_stack(void *ctx)
+{
+ struct bpf_stack_build_id stack;
+ struct task_struct *task;
+
+ task = bpf_get_current_task_btf();
+ bpf_get_task_stack(task, &stack, sizeof(stack),
+ BPF_F_USER_STACK | BPF_F_USER_BUILD_ID);
+ return 0;
+}
+
SEC("?fentry.s/" SYS_PREFIX "sys_getpgid")
__failure __msg("kernel func bpf_copy_from_user_str is sleepable within non-preemptible region")
int preempt_sleepable_kfunc(void *ctx)
diff --git a/tools/testing/selftests/bpf/progs/pyperf.h b/tools/testing/selftests/bpf/progs/pyperf.h
index 86484f07e1d1..fd93a96e5901 100644
--- a/tools/testing/selftests/bpf/progs/pyperf.h
+++ b/tools/testing/selftests/bpf/progs/pyperf.h
@@ -85,9 +85,11 @@ static void *get_thread_state(void *tls_base, PidData *pidData)
return thread_state;
}
-static __always_inline bool get_frame_data(void *frame_ptr, PidData *pidData,
- FrameData *frame, Symbol *symbol)
+__weak bool __get_frame_data(long frame_ptr_, PidData *pidData __arg_nonnull,
+ FrameData *frame __arg_nonnull, Symbol *symbol __arg_nonnull)
{
+ void *frame_ptr = (void *)frame_ptr_;
+
// read data from PyFrameObject
bpf_probe_read_user(&frame->f_back,
sizeof(frame->f_back),
@@ -119,6 +121,12 @@ static __always_inline bool get_frame_data(void *frame_ptr, PidData *pidData,
return true;
}
+static __always_inline bool get_frame_data(void *frame_ptr, PidData *pidData,
+ FrameData *frame, Symbol *symbol)
+{
+ return __get_frame_data((long)frame_ptr, pidData, frame, symbol);
+}
+
struct {
__uint(type, BPF_MAP_TYPE_HASH);
__uint(max_entries, 1);
diff --git a/tools/testing/selftests/bpf/progs/raw_tp_null_fail.c b/tools/testing/selftests/bpf/progs/raw_tp_null_fail.c
index 0d58114a4955..725d73c9ffe1 100644
--- a/tools/testing/selftests/bpf/progs/raw_tp_null_fail.c
+++ b/tools/testing/selftests/bpf/progs/raw_tp_null_fail.c
@@ -22,3 +22,56 @@ int test_raw_tp_null_sched_pi_setprio_arg_2(void *ctx) {
asm volatile("r1 = *(u64 *)(r1 +8); r1 = *(u64 *)(r1 +0);" ::: __clobber_all);
return 0;
}
+
+/* Plain raw tracepoint arguments remain scalar values. */
+SEC("raw_tp/signal_generate")
+__success
+int test_raw_tp_signal_generate_info_scalar(void *ctx)
+{
+ asm volatile("r1 = *(u64 *)(r1 +8); if r1 != 1 goto +0;" ::: __clobber_all);
+ return 0;
+}
+
+/* tp_btf programs may inspect the sentinel as a scalar value. */
+SEC("tp_btf/signal_generate")
+__success
+int test_tp_btf_signal_generate_info_scalar(void *ctx)
+{
+ asm volatile("r1 = *(u64 *)(r1 +8); if r1 != 1 goto +0;" ::: __clobber_all);
+ return 0;
+}
+
+/* SEND_SIG_PRIV is non-NULL, so a NULL check cannot make info safe. */
+SEC("tp_btf/signal_generate")
+__failure __msg("R1 invalid mem access 'scalar'")
+int test_tp_btf_signal_generate_info_no_deref(void *ctx)
+{
+ asm volatile("r1 = *(u64 *)(r1 +8); if r1 == 0 goto +1; "
+ "r1 = *(u32 *)(r1 +0);" ::: __clobber_all);
+ return 0;
+}
+
+SEC("tp_btf/signal_deliver")
+__failure __msg("R1 invalid mem access 'scalar'")
+int test_tp_btf_signal_deliver_info_no_deref(void *ctx)
+{
+ asm volatile("r1 = *(u64 *)(r1 +8); r1 = *(u32 *)(r1 +0);" ::: __clobber_all);
+ return 0;
+}
+
+SEC("tp_btf/sched_process_wait")
+__failure __msg("R1 invalid mem access 'trusted_ptr_or_null_'")
+int test_raw_tp_null_sched_process_wait_arg_1(void *ctx)
+{
+ asm volatile("r1 = *(u64 *)(r1 +0); r1 = *(u32 *)(r1 +0);" ::: __clobber_all);
+ return 0;
+}
+
+SEC("tp_btf/sched_process_wait")
+__success
+int test_raw_tp_null_sched_process_wait_arg_1_checked(void *ctx)
+{
+ asm volatile("r1 = *(u64 *)(r1 +0); if r1 == 0 goto +1; "
+ "r1 = *(u32 *)(r1 +0);" ::: __clobber_all);
+ return 0;
+}
diff --git a/tools/testing/selftests/bpf/progs/rbtree_fail.c b/tools/testing/selftests/bpf/progs/rbtree_fail.c
index 555379952dcc..4504608196ab 100644
--- a/tools/testing/selftests/bpf/progs/rbtree_fail.c
+++ b/tools/testing/selftests/bpf/progs/rbtree_fail.c
@@ -16,6 +16,7 @@ struct node_data {
private(A) struct bpf_spin_lock glock;
private(A) struct bpf_rb_root groot __contains(node_data, node);
private(A) struct bpf_rb_root groot2 __contains(node_data, node);
+private(B) struct bpf_res_spin_lock res_glock;
static bool less(struct bpf_rb_node *a, const struct bpf_rb_node *b)
{
@@ -265,6 +266,53 @@ static bool less__bad_fn_call_first_unlock_after(struct bpf_rb_node *a, const st
return node_a->key < node_b->key;
}
+static bool less__bad_res_spin_unlock(struct bpf_rb_node *a, const struct bpf_rb_node *b)
+{
+ bpf_res_spin_unlock(&res_glock);
+ return false;
+}
+
+static __noinline void rbtree_cb_unlock_relock(void)
+{
+ bpf_spin_unlock(&glock);
+ bpf_spin_lock(&glock);
+}
+
+static __noinline void rbtree_cb_nested_unlock(void)
+{
+ rbtree_cb_unlock_relock();
+ asm volatile ("");
+}
+
+static bool less__bad_subprog_unlock(struct bpf_rb_node *a, const struct bpf_rb_node *b)
+{
+ struct node_data *node_a;
+ struct node_data *node_b;
+
+ node_a = container_of(a, struct node_data, node);
+ node_b = container_of(b, struct node_data, node);
+ rbtree_cb_nested_unlock();
+
+ return node_a->key < node_b->key;
+}
+
+static __noinline void rbtree_cb_noop(void)
+{
+ asm volatile ("");
+}
+
+static bool less__subprog_allowed(struct bpf_rb_node *a, const struct bpf_rb_node *b)
+{
+ struct node_data *node_a;
+ struct node_data *node_b;
+
+ node_a = container_of(a, struct node_data, node);
+ node_b = container_of(b, struct node_data, node);
+ rbtree_cb_noop();
+
+ return node_a->key < node_b->key;
+}
+
static __always_inline
long add_with_cb(bool (cb)(struct bpf_rb_node *a, const struct bpf_rb_node *b))
{
@@ -301,4 +349,40 @@ long rbtree_api_add_bad_cb_bad_fn_call_first_unlock_after(void *ctx)
return add_with_cb(less__bad_fn_call_first_unlock_after);
}
+SEC("?tc")
+__failure __msg("can't res_spin_{lock,unlock} in rbtree cb")
+long rbtree_api_add_bad_cb_res_spin_unlock(void *ctx)
+{
+ struct node_data *n;
+
+ n = bpf_obj_new(typeof(*n));
+ if (!n)
+ return 1;
+
+ bpf_spin_lock(&glock);
+ if (bpf_res_spin_lock(&res_glock)) {
+ bpf_spin_unlock(&glock);
+ bpf_obj_drop(n);
+ return 1;
+ }
+ bpf_rbtree_add(&groot, &n->node, less__bad_res_spin_unlock);
+ bpf_res_spin_unlock(&res_glock);
+ bpf_spin_unlock(&glock);
+ return 0;
+}
+
+SEC("?tc")
+__failure __msg("can't spin_{lock,unlock} in rbtree cb")
+long rbtree_api_add_bad_cb_subprog_unlock(void *ctx)
+{
+ return add_with_cb(less__bad_subprog_unlock);
+}
+
+SEC("?tc")
+__success
+long rbtree_api_add_cb_subprog_allowed(void *ctx)
+{
+ return add_with_cb(less__subprog_allowed);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/rcu_read_lock.c b/tools/testing/selftests/bpf/progs/rcu_read_lock.c
index b4e073168fb1..cdb255addbc3 100644
--- a/tools/testing/selftests/bpf/progs/rcu_read_lock.c
+++ b/tools/testing/selftests/bpf/progs/rcu_read_lock.c
@@ -549,3 +549,79 @@ int rcu_read_lock_sleepable_global_subprog_indirect(void *ctx)
bpf_rcu_read_unlock();
return 0;
}
+
+struct rcu_node_data {
+ long key;
+ struct bpf_rb_node node;
+};
+
+struct rcu_node_stash {
+ struct rcu_node_data __kptr *node;
+};
+
+/*
+ * Necessary so that LLVM emits BTF for rcu_node_data rather than just a
+ * fwd reference to it, same as in progs/local_kptr_stash.c.
+ */
+struct rcu_node_data *just_here_because_btf_bug;
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __uint(max_entries, 1);
+ __type(key, int);
+ __type(value, struct rcu_node_stash);
+} node_stash SEC(".maps");
+
+long non_own_ref_key;
+
+SEC("?fentry.s/" SYS_PREFIX "sys_getpgid")
+int non_own_ref_untrusted_ld(void *ctx)
+{
+ struct rcu_node_stash *stash;
+ struct rcu_node_data *node;
+ int key = 0;
+
+ stash = bpf_map_lookup_elem(&node_stash, &key);
+ if (!stash)
+ return 0;
+ bpf_rcu_read_lock();
+ node = stash->node;
+ if (!node) {
+ bpf_rcu_read_unlock();
+ return 0;
+ }
+ bpf_rcu_read_unlock();
+ /*
+ * The unlock leaves node as PTR_TO_BTF_ID | MEM_ALLOC | PTR_UNTRUSTED,
+ * and the load below has to get the BPF_PROBE_MEM rewrite for it,
+ * otherwise a bad address panics the kernel.
+ */
+ non_own_ref_key = node->key;
+ return 0;
+}
+
+long rcu_untrusted_wq_flags;
+
+SEC("?tp_btf/tcp_probe")
+int BPF_PROG(rcu_untrusted_union_ld, struct sock *sk)
+{
+ struct socket_wq *wq;
+
+ /*
+ * sk_wq sits in a two member union, so btf_struct_walk() marks the
+ * pointer PTR_UNTRUSTED, and the __rcu tag on the member adds MEM_RCU
+ * on top of it. struct sock is not on the __safe_rcu_or_null allow
+ * list, hence the two stay combined and the load below has to get the
+ * BPF_PROBE_MEM rewrite for PTR_TO_BTF_ID | PTR_UNTRUSTED | MEM_RCU,
+ * otherwise a bad address panics the kernel.
+ *
+ * The __rcu tag only reaches BTF on a clang built kernel, that is, one
+ * with CONFIG_PAHOLE_HAS_BTF_TAG. On a gcc built kernel the walk yields
+ * a plain untrusted pointer, which is rewritten either way.
+ */
+ wq = sk->sk_wq;
+ if (!wq)
+ return 0;
+ rcu_untrusted_wq_flags = wq->flags;
+ return 0;
+}
diff --git a/tools/testing/selftests/bpf/progs/refcounted_kptr.c b/tools/testing/selftests/bpf/progs/refcounted_kptr.c
index 61906f48025c..cae00f7b0a24 100644
--- a/tools/testing/selftests/bpf/progs/refcounted_kptr.c
+++ b/tools/testing/selftests/bpf/progs/refcounted_kptr.c
@@ -23,6 +23,15 @@ struct map_value {
struct node_data __kptr *node;
};
+struct node_refcount_only {
+ long key;
+ struct bpf_refcount refcount;
+};
+
+struct map_value_refcount_only {
+ struct node_refcount_only __kptr *node;
+};
+
struct {
__uint(type, BPF_MAP_TYPE_ARRAY);
__type(key, int);
@@ -30,6 +39,13 @@ struct {
__uint(max_entries, 2);
} stashed_nodes SEC(".maps");
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __type(key, int);
+ __type(value, struct map_value_refcount_only);
+ __uint(max_entries, 1);
+} stashed_refcount_only SEC(".maps");
+
struct node_acquire {
long key;
long data;
@@ -832,6 +848,51 @@ long rbtree_refcounted_node_ref_escapes_owning_input(void *ctx)
return 0;
}
+SEC("tc")
+__success
+long refcount_acquire_owning_input_no_null_check(void *ctx)
+{
+ struct node_refcount_only *n, *m;
+
+ n = bpf_obj_new(typeof(*n));
+ if (!n)
+ return 1;
+
+ m = bpf_refcount_acquire(n);
+ bpf_obj_drop(m);
+ bpf_obj_drop(n);
+
+ return 0;
+}
+
+SEC("?syscall")
+__success
+long refcount_acquire_rcu_map_kptr_null_checked(void *ctx)
+{
+ struct map_value_refcount_only *mapval;
+ struct node_refcount_only *n, *m;
+ int idx = 0;
+
+ mapval = bpf_map_lookup_elem(&stashed_refcount_only, &idx);
+ if (!mapval)
+ return 1;
+
+ bpf_rcu_read_lock();
+ n = mapval->node;
+ if (!n) {
+ bpf_rcu_read_unlock();
+ return 2;
+ }
+ m = bpf_refcount_acquire(n);
+ bpf_rcu_read_unlock();
+
+ if (!m)
+ return 3;
+ bpf_obj_drop(m);
+
+ return 0;
+}
+
static long __stash_map_empty_xchg(struct node_data *n, int idx)
{
struct map_value *mapval = bpf_map_lookup_elem(&stashed_nodes, &idx);
diff --git a/tools/testing/selftests/bpf/progs/refcounted_kptr_fail.c b/tools/testing/selftests/bpf/progs/refcounted_kptr_fail.c
index 024ef2aae200..338e43822ffe 100644
--- a/tools/testing/selftests/bpf/progs/refcounted_kptr_fail.c
+++ b/tools/testing/selftests/bpf/progs/refcounted_kptr_fail.c
@@ -19,6 +19,26 @@ struct node_refcounted {
struct bpf_refcount refcount;
};
+struct node_refcount_only {
+ long key;
+ struct bpf_refcount refcount;
+};
+
+struct map_value_refcount_only {
+ struct node_refcount_only __kptr *node;
+};
+
+struct rcu_graph_node {
+ struct bpf_rb_node node;
+ long data;
+};
+
+struct rcu_graph_node *just_here_because_btf_bug;
+
+struct map_value_rcu_graph {
+ struct rcu_graph_node __kptr *node;
+};
+
extern void bpf_rcu_read_lock(void) __ksym;
extern void bpf_rcu_read_unlock(void) __ksym;
@@ -27,6 +47,22 @@ private(A) struct bpf_spin_lock glock;
private(A) struct bpf_rb_root groot __contains(node_acquire, node);
private(B) struct bpf_spin_lock lock;
private(B) struct bpf_list_head head __contains(node_refcounted, list);
+private(C) struct bpf_spin_lock graph_lock;
+private(C) struct bpf_rb_root graph_root __contains(rcu_graph_node, node);
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __type(key, int);
+ __type(value, struct map_value_refcount_only);
+ __uint(max_entries, 1);
+} stashed_refcount_only SEC(".maps");
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __type(key, int);
+ __type(value, struct map_value_rcu_graph);
+ __uint(max_entries, 1);
+} stashed_rcu_graph SEC(".maps");
static bool less(struct bpf_rb_node *a, const struct bpf_rb_node *b)
{
@@ -63,6 +99,7 @@ long rbtree_refcounted_node_ref_escapes(void *ctx)
SEC("?tc")
__failure __msg("Possibly NULL pointer passed to trusted R1")
+__msg("requires a non-NULL value of type (void *)")
long refcount_acquire_maybe_null(void *ctx)
{
struct node_acquire *n, *m;
@@ -81,6 +118,128 @@ long refcount_acquire_maybe_null(void *ctx)
}
SEC("?tc")
+__failure __msg("R1 is neither owning or non-owning ref")
+__msg("expects a pointer to a BPF-managed refcounted object, but R1 is a context pointer")
+long refcount_acquire_non_object(void *ctx)
+{
+ return bpf_refcount_acquire(ctx) != NULL;
+}
+
+SEC("?syscall")
+__failure __msg("Possibly NULL pointer passed to trusted R1")
+long refcount_acquire_rcu_map_kptr_unchecked_drop(void *ctx)
+{
+ struct map_value_refcount_only *mapval;
+ struct node_refcount_only *tmp, *n, *m;
+ int idx = 0;
+
+ /* Force Clang to emit complete BTF for struct node_refcount_only. */
+ tmp = bpf_obj_new(typeof(*tmp));
+ if (!tmp)
+ return 3;
+ bpf_obj_drop(tmp);
+
+ mapval = bpf_map_lookup_elem(&stashed_refcount_only, &idx);
+ if (!mapval)
+ return 1;
+
+ bpf_rcu_read_lock();
+ n = mapval->node;
+ if (!n) {
+ bpf_rcu_read_unlock();
+ return 2;
+ }
+ m = bpf_refcount_acquire(n);
+ bpf_rcu_read_unlock();
+
+ bpf_obj_drop(m);
+
+ return 0;
+}
+
+SEC("?syscall")
+__failure
+__msg("bpf_rbtree_remove can only take non-owning or refcounted "
+ "bpf_rb_node pointer")
+long rbtree_remove_after_rcu_unlock(void *ctx)
+{
+ struct map_value_rcu_graph *mapval;
+ struct bpf_rb_node *rb_node;
+ struct rcu_graph_node *node;
+ int idx = 0;
+
+ mapval = bpf_map_lookup_elem(&stashed_rcu_graph, &idx);
+ if (!mapval)
+ return 0;
+
+ bpf_rcu_read_lock();
+ node = mapval->node;
+ if (!node) {
+ bpf_rcu_read_unlock();
+ return 0;
+ }
+ bpf_rcu_read_unlock();
+
+ bpf_spin_lock(&graph_lock);
+ rb_node = bpf_rbtree_remove(&graph_root, &node->node);
+ bpf_spin_unlock(&graph_lock);
+ if (rb_node)
+ bpf_obj_drop(container_of(rb_node, struct rcu_graph_node, node));
+
+ return 0;
+}
+
+SEC("?syscall")
+__failure __msg("R1 is neither owning or non-owning ref")
+long refcount_acquire_after_rcu_unlock(void *ctx)
+{
+ struct map_value_refcount_only *mapval;
+ struct node_refcount_only *node, *ref;
+ int idx = 0;
+
+ mapval = bpf_map_lookup_elem(&stashed_refcount_only, &idx);
+ if (!mapval)
+ return 0;
+
+ bpf_rcu_read_lock();
+ node = mapval->node;
+ if (!node) {
+ bpf_rcu_read_unlock();
+ return 0;
+ }
+ bpf_rcu_read_unlock();
+
+ ref = bpf_refcount_acquire(node);
+ if (ref)
+ bpf_obj_drop(ref);
+
+ return 0;
+}
+
+SEC("?syscall")
+__failure __msg("invalid mem access 'scalar'")
+long graph_kptr_after_spin_unlock(void *ctx)
+{
+ struct map_value_rcu_graph *mapval;
+ struct rcu_graph_node *node;
+ int idx = 0;
+
+ mapval = bpf_map_lookup_elem(&stashed_rcu_graph, &idx);
+ if (!mapval)
+ return 0;
+
+ bpf_spin_lock(&graph_lock);
+ node = mapval->node;
+ if (!node) {
+ bpf_spin_unlock(&graph_lock);
+ return 0;
+ }
+ bpf_spin_unlock(&graph_lock);
+
+ return node->data;
+}
+
+SEC("?tc")
__failure __msg("Unreleased reference id=3 alloc_insn={{[0-9]+}}")
long rbtree_refcounted_node_ref_escapes_owning_input(void *ctx)
{
diff --git a/tools/testing/selftests/bpf/progs/rhash.c b/tools/testing/selftests/bpf/progs/rhash.c
index fc2dac3a719e..aea4de8dc781 100644
--- a/tools/testing/selftests/bpf/progs/rhash.c
+++ b/tools/testing/selftests/bpf/progs/rhash.c
@@ -19,6 +19,11 @@ struct elem {
int val;
};
+struct special_elem {
+ struct task_struct __kptr *task;
+ int val;
+};
+
struct {
__uint(type, BPF_MAP_TYPE_RHASH);
__uint(map_flags, BPF_F_NO_PREALLOC);
@@ -27,6 +32,17 @@ struct {
__type(value, struct elem);
} rhmap SEC(".maps");
+struct {
+ __uint(type, BPF_MAP_TYPE_RHASH);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __uint(max_entries, 1);
+ __type(key, int);
+ __type(value, struct special_elem);
+} special_fields SEC(".maps");
+
+extern struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym;
+extern void bpf_task_release(struct task_struct *p) __ksym;
+
SEC("syscall")
int test_rhash_lookup_update(void *ctx)
{
@@ -246,3 +262,99 @@ int test_rhash_delete_nonexistent(void *ctx)
err = 0;
return 0;
}
+
+SEC("syscall")
+int test_rhash_kptr_update(void *ctx)
+{
+ struct special_elem val1 = { .val = 1 };
+ struct special_elem val2 = { .val = 2 };
+ struct task_struct *task, *old;
+ struct special_elem *elem;
+ int key = 0;
+
+ err = 1;
+ if (bpf_map_update_elem(&special_fields, &key, &val1, BPF_NOEXIST))
+ return 1;
+
+ err = 2;
+ elem = bpf_map_lookup_elem(&special_fields, &key);
+ if (!elem)
+ return 2;
+
+ err = 3;
+ task = bpf_task_acquire(bpf_get_current_task_btf());
+ if (!task)
+ return 3;
+
+ err = 4;
+ old = bpf_kptr_xchg(&elem->task, task);
+ if (old) {
+ bpf_task_release(old);
+ return 4;
+ }
+
+ err = 5;
+ if (bpf_map_update_elem(&special_fields, &key, &val2, BPF_EXIST))
+ return 5;
+
+ err = 6;
+ elem = bpf_map_lookup_elem(&special_fields, &key);
+ if (!elem || elem->val != 2)
+ return 6;
+
+ err = 7;
+ old = bpf_kptr_xchg(&elem->task, NULL);
+ if (!old)
+ return 7;
+ bpf_task_release(old);
+
+ err = 8;
+ if (bpf_map_delete_elem(&special_fields, &key))
+ return 8;
+
+ err = 0;
+ return 0;
+}
+
+SEC("syscall")
+int test_rhash_kptr_delete(void *ctx)
+{
+ struct special_elem val = {};
+ struct task_struct *task, *old;
+ struct special_elem *elem;
+ int key = 0;
+
+ err = 1;
+ if (bpf_map_update_elem(&special_fields, &key, &val, BPF_NOEXIST))
+ return 1;
+
+ err = 2;
+ elem = bpf_map_lookup_elem(&special_fields, &key);
+ if (!elem)
+ return 2;
+
+ err = 3;
+ task = bpf_task_acquire(bpf_get_current_task_btf());
+ if (!task)
+ return 3;
+
+ err = 4;
+ old = bpf_kptr_xchg(&elem->task, task);
+ if (old) {
+ bpf_task_release(old);
+ return 4;
+ }
+
+ err = 5;
+ if (bpf_map_delete_elem(&special_fields, &key))
+ return 5;
+
+ err = 6;
+ old = bpf_kptr_xchg(&elem->task, NULL);
+ if (!old)
+ return 6;
+ bpf_task_release(old);
+
+ err = 0;
+ return 0;
+}
diff --git a/tools/testing/selftests/bpf/progs/rhash_timer.c b/tools/testing/selftests/bpf/progs/rhash_timer.c
new file mode 100644
index 000000000000..2e06a463c605
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/rhash_timer.c
@@ -0,0 +1,98 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <errno.h>
+#include <bpf/bpf_helpers.h>
+
+#define CLOCK_MONOTONIC 1
+#define TIMER_NSEC (60ULL * 1000 * 1000 * 1000)
+
+struct timer_value {
+ struct bpf_timer timer;
+ u64 data;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_RHASH);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __uint(max_entries, 1);
+ __type(key, u64);
+ __type(value, struct timer_value);
+} timer_map SEC(".maps");
+
+u64 armed;
+u64 cancelled;
+long timer_init_err;
+long timer_set_callback_err;
+long timer_start_err;
+long timer_cancel_err;
+
+static int timer_cb(void *map, u64 *key, struct timer_value *value)
+{
+ return 0;
+}
+
+static long arm_timer_cb(struct bpf_map *map, u64 *key,
+ struct timer_value *value, void *ctx)
+{
+ u64 key_copy = *key;
+ long err;
+
+ err = bpf_map_delete_elem(map, &key_copy);
+ if (err)
+ return 1;
+
+ err = bpf_timer_init(&value->timer, map, CLOCK_MONOTONIC);
+ if (err) {
+ timer_init_err = err;
+ return 1;
+ }
+
+ err = bpf_timer_set_callback(&value->timer, timer_cb);
+ if (err) {
+ timer_set_callback_err = err;
+ return 1;
+ }
+
+ err = bpf_timer_start(&value->timer, TIMER_NSEC, BPF_F_TIMER_CPU_PIN);
+ if (err) {
+ timer_start_err = err;
+ return 1;
+ }
+
+ __sync_fetch_and_add(&armed, 1);
+ return 1;
+}
+
+static long cancel_timer_cb(struct bpf_map *map, u64 *key,
+ struct timer_value *value, void *ctx)
+{
+ long err;
+
+ err = bpf_timer_cancel(&value->timer);
+ if (err == -EINVAL)
+ return 1;
+ if (err < 0) {
+ timer_cancel_err = err;
+ return 1;
+ }
+
+ __sync_fetch_and_add(&cancelled, 1);
+ return 1;
+}
+
+SEC("syscall")
+int arm_deleted_timer(void *ctx)
+{
+ bpf_for_each_map_elem(&timer_map, arm_timer_cb, NULL, 0);
+ return 0;
+}
+
+SEC("syscall")
+int cancel_recycled_timer(void *ctx)
+{
+ bpf_for_each_map_elem(&timer_map, cancel_timer_cb, NULL, 0);
+ return 0;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/snprintf_btf_void.c b/tools/testing/selftests/bpf/progs/snprintf_btf_void.c
new file mode 100644
index 000000000000..44af80fbbb80
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/snprintf_btf_void.c
@@ -0,0 +1,24 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "btf_ptr.h"
+#include <bpf/bpf_helpers.h>
+
+__u32 type_id;
+/* A buffer we own to render the selected type from, kept in bounds. */
+char obj[256];
+char out[64];
+long ret;
+
+SEC("raw_tp/sys_enter")
+int dump_type(void *ctx)
+{
+ struct btf_ptr ptr = {
+ .ptr = obj,
+ .type_id = type_id,
+ .flags = 0,
+ };
+
+ ret = bpf_snprintf_btf(out, sizeof(out), &ptr, sizeof(ptr), 0);
+ return 0;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/sock_destroy_prog.c b/tools/testing/selftests/bpf/progs/sock_destroy_prog.c
index 9e0bf7a54cec..0a8887543218 100644
--- a/tools/testing/selftests/bpf/progs/sock_destroy_prog.c
+++ b/tools/testing/selftests/bpf/progs/sock_destroy_prog.c
@@ -7,6 +7,8 @@
#include "bpf_tracing_net.h"
__be16 serv_port = 0;
+int tw_found = 0;
+int tw_destroy_err = 0;
int bpf_sock_destroy(struct sock_common *sk) __ksym;
@@ -100,6 +102,34 @@ int iter_tcp6_server(struct bpf_iter__tcp *ctx)
return 0;
}
+SEC("iter/tcp")
+int iter_tcp6_timewait(struct bpf_iter__tcp *ctx)
+{
+ struct sock_common *sk_common = ctx->sk_common;
+ __u64 *val;
+ int key = 0;
+
+ if (!sk_common)
+ return 0;
+
+ if (sk_common->skc_family != AF_INET6)
+ return 0;
+
+ if (!bpf_skc_to_tcp_timewait_sock(sk_common))
+ return 0;
+
+ val = bpf_map_lookup_elem(&tcp_conn_sockets, &key);
+ if (!val)
+ return 0;
+ /* The timewait sock inherits the cookie of the closed client sock. */
+ if (bpf_get_socket_cookie(sk_common) != *val)
+ return 0;
+
+ tw_found++;
+ tw_destroy_err = bpf_sock_destroy(sk_common);
+
+ return 0;
+}
SEC("iter/udp")
int iter_udp6_client(struct bpf_iter__udp *ctx)
diff --git a/tools/testing/selftests/bpf/progs/sock_read_xattr.c b/tools/testing/selftests/bpf/progs/sock_read_xattr.c
new file mode 100644
index 000000000000..c4a8eae8cc3c
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/sock_read_xattr.c
@@ -0,0 +1,54 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Christian Brauner */
+
+#include <vmlinux.h>
+#include <bpf/bpf_tracing.h>
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_core_read.h>
+#include "bpf_experimental.h"
+#include "bpf_misc.h"
+
+char _license[] SEC("license") = "GPL";
+
+char value[16];
+int read_ret = -1;
+__u32 monitored_pid = 0;
+
+static __always_inline void read_xattr(struct socket *sock)
+{
+ struct bpf_dynptr value_ptr;
+
+ bpf_dynptr_from_mem(value, sizeof(value), 0, &value_ptr);
+ bpf_sock_read_xattr(sock, "user.bpf_test", &value_ptr);
+}
+
+SEC("lsm.s/socket_connect")
+__success
+int BPF_PROG(trusted_sock_ptr_sleepable, struct socket *sock)
+{
+ read_xattr(sock);
+ return 0;
+}
+
+SEC("lsm/socket_connect")
+__success
+int BPF_PROG(trusted_sock_ptr_non_sleepable, struct socket *sock)
+{
+ read_xattr(sock);
+ return 0;
+}
+
+SEC("lsm.s/socket_connect")
+__success
+int BPF_PROG(read_sock_xattr, struct socket *sock)
+{
+ struct bpf_dynptr value_ptr;
+ __u32 pid = bpf_get_current_pid_tgid() >> 32;
+
+ if (pid != monitored_pid)
+ return 0;
+
+ bpf_dynptr_from_mem(value, sizeof(value), 0, &value_ptr);
+ read_ret = bpf_sock_read_xattr(sock, "user.bpf_test", &value_ptr);
+ return 0;
+}
diff --git a/tools/testing/selftests/bpf/progs/stream.c b/tools/testing/selftests/bpf/progs/stream.c
index cf5533e11f39..8e8e1339dc74 100644
--- a/tools/testing/selftests/bpf/progs/stream.c
+++ b/tools/testing/selftests/bpf/progs/stream.c
@@ -65,6 +65,7 @@ __arch_x86_64
__arch_arm64
__arch_s390x
__arch_riscv64
+__arch_loongarch
__success __retval(0)
__stderr("ERROR: Timeout detected for may_goto instruction")
__stderr("CPU: {{[0-9]+}} UID: 0 PID: {{[0-9]+}} Comm: {{.*}}")
@@ -229,6 +230,107 @@ int stream_arena_load_acquire_fault(void *ctx)
return val;
}
+SEC("syscall")
+__arch_x86_64
+__arch_arm64
+__success __retval(0)
+__stderr("ERROR: Arena WRITE access at unmapped address 0x{{.*}}")
+__stderr("CPU: {{[0-9]+}} UID: 0 PID: {{[0-9]+}} Comm: {{.*}}")
+__stderr("Call trace:\n"
+"{{([a-zA-Z_][a-zA-Z0-9_]*\\+0x[0-9a-fA-F]+/0x[0-9a-fA-F]+\n"
+"|[ \t]+[^\n]+\n)*}}")
+int stream_arena_xchg_fault(void *ctx)
+{
+ static const struct bpf_insn xchg_insn = {
+ .code = 0xc3, /* BPF_STX | BPF_ATOMIC | BPF_W */
+ .dst_reg = 1, /* BPF_REG_1 */
+ .src_reg = 2, /* BPF_REG_2 */
+ .off = 0x7fff,
+ .imm = 0xe1, /* BPF_XCHG */
+ };
+ struct bpf_arena *ptr = (void *)&arena;
+ u64 user_vm_start, val;
+
+ /*
+ * Prevent GCC bounds warning: casting &arena to struct bpf_arena *
+ * triggers bounds checking since the map definition is smaller than
+ * struct bpf_arena. barrier_var() makes the pointer opaque to GCC,
+ * preventing the bounds analysis.
+ */
+ barrier_var(ptr);
+ user_vm_start = ptr->user_vm_start;
+ fault_addr = user_vm_start + 0x7fff;
+ bpf_addr_space_cast(user_vm_start, 0, 1);
+ /*
+ * A read-modify-write carrying BPF_FETCH writes to memory, so the fault
+ * has to be reported as a WRITE from the dst_reg address, but it also
+ * reads the old value into src_reg, so the exception handler has to
+ * clear src_reg. Poison it up front, the returned value must be 0.
+ */
+ asm volatile (
+ "r1 = %[user_vm_start];"
+ "r2 = 1;"
+ ".8byte %[xchg_insn];" /* r2 = xchg((u32 *)(r1 + 0x7fff), r2) */
+ "%[val] = r2;"
+ : [val] "=r" (val)
+ : [user_vm_start] "r" (user_vm_start),
+ __imm_insn(xchg_insn, xchg_insn)
+ : "r1", "r2"
+ );
+ return val;
+}
+
+SEC("syscall")
+__arch_x86_64
+__arch_arm64
+__success __retval(0)
+__stderr("ERROR: Arena WRITE access at unmapped address 0x{{.*}}")
+__stderr("CPU: {{[0-9]+}} UID: 0 PID: {{[0-9]+}} Comm: {{.*}}")
+__stderr("Call trace:\n"
+"{{([a-zA-Z_][a-zA-Z0-9_]*\\+0x[0-9a-fA-F]+/0x[0-9a-fA-F]+\n"
+"|[ \t]+[^\n]+\n)*}}")
+int stream_arena_cmpxchg_fault(void *ctx)
+{
+ static const struct bpf_insn cmpxchg_insn = {
+ .code = 0xc3, /* BPF_STX | BPF_ATOMIC | BPF_W */
+ .dst_reg = 1, /* BPF_REG_1 */
+ .src_reg = 2, /* BPF_REG_2 */
+ .off = 0x7fff,
+ .imm = 0xf1, /* BPF_CMPXCHG */
+ };
+ struct bpf_arena *ptr = (void *)&arena;
+ u64 user_vm_start, val;
+
+ /*
+ * Prevent GCC bounds warning: casting &arena to struct bpf_arena *
+ * triggers bounds checking since the map definition is smaller than
+ * struct bpf_arena. barrier_var() makes the pointer opaque to GCC,
+ * preventing the bounds analysis.
+ */
+ barrier_var(ptr);
+ user_vm_start = ptr->user_vm_start;
+ fault_addr = user_vm_start + 0x7fff;
+ bpf_addr_space_cast(user_vm_start, 0, 1);
+ /*
+ * Same as the exchange above, except that a BPF_CMPXCHG reads the old
+ * value into r0 rather than into src_reg, so r0 is the register the
+ * exception handler has to clear. It doubles as the compare value, but
+ * the comparison never happens since the access faults first.
+ */
+ asm volatile (
+ "r1 = %[user_vm_start];"
+ "r0 = 1;"
+ "r2 = 2;"
+ ".8byte %[cmpxchg_insn];" /* r0 = cmpxchg((u32 *)(r1 + 0x7fff), r0, r2) */
+ "%[val] = r0;"
+ : [val] "=r" (val)
+ : [user_vm_start] "r" (user_vm_start),
+ __imm_insn(cmpxchg_insn, cmpxchg_insn)
+ : "r0", "r1", "r2"
+ );
+ return val;
+}
+
static __noinline void subprog(void)
{
int __arena *addr = (int __arena *)0xdeadbeef;
diff --git a/tools/testing/selftests/bpf/progs/struct_ops_arena.c b/tools/testing/selftests/bpf/progs/struct_ops_arena.c
index ba04c73d8d96..8aa8639df91f 100644
--- a/tools/testing/selftests/bpf/progs/struct_ops_arena.c
+++ b/tools/testing/selftests/bpf/progs/struct_ops_arena.c
@@ -59,11 +59,28 @@ int test_arena_stack_cb(unsigned long long *ctx)
return 0;
}
+SEC("struct_ops/test_arena_multislot")
+int test_arena_multislot_cb(unsigned long long *ctx)
+{
+ u64 __arena *ptr = (u64 __arena *)ctx[2];
+
+ arena_touch++;
+ /*
+ * The 16-byte struct occupies ctx[0] and ctx[1], so @ptr is argument
+ * one but slot two. Getting that wrong hands the callback a scalar.
+ */
+ if (ctx[0] != 11 || ctx[1] != 22)
+ return 0xbad;
+ *ptr += 1;
+ return 0;
+}
+
SEC(".struct_ops.link")
struct bpf_testmod_ops3 testmod_arena = {
.test_arena = (void *)test_arena_cb,
.test_arena_nullable = (void *)test_arena_nullable_cb,
.test_arena_stack = (void *)test_arena_stack_cb,
+ .test_arena_multislot = (void *)test_arena_multislot_cb,
};
SEC("syscall")
@@ -109,6 +126,13 @@ int trigger(void *ctx)
if (*val != 44)
return 9;
+ /* a multi-slot arg precedes the arena pointer here */
+ ret = bpf_testmod_ops3_call_test_arena_multislot((u64 *)val);
+ if (ret)
+ return 10;
+ if (*val != 45)
+ return 11;
+
bpf_arena_free_pages(&arena, (void __arena *)val, 1);
#endif
return 0;
diff --git a/tools/testing/selftests/bpf/progs/tailcall_callback.c b/tools/testing/selftests/bpf/progs/tailcall_callback.c
index c41632cf423b..14fa7a87028e 100644
--- a/tools/testing/selftests/bpf/progs/tailcall_callback.c
+++ b/tools/testing/selftests/bpf/progs/tailcall_callback.c
@@ -45,6 +45,13 @@ int callback_loop(int index, void **cb_ctx)
}
static __noinline
+int callback_tail(int index, void **cb_ctx)
+{
+ bpf_tail_call_static(*cb_ctx, &jmp_table, 0);
+ return 0;
+}
+
+static __noinline
int callback_empty(int index, void *data)
{
return 0;
@@ -78,4 +85,13 @@ int tailcall_callback_2(struct __sk_buff *skb)
return 0;
}
+/* callback with a direct tail call is rejected without a verifier bug */
+SEC("tc")
+__failure __msg("cannot tail call within callback")
+int tailcall_callback_3(struct __sk_buff *skb)
+{
+ bpf_loop(1, callback_tail, &skb, 0);
+ return 0;
+}
+
char __license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/task_local_data.bpf.h b/tools/testing/selftests/bpf/progs/task_local_data.bpf.h
index 0df8a12fd61e..a31a399870be 100644
--- a/tools/testing/selftests/bpf/progs/task_local_data.bpf.h
+++ b/tools/testing/selftests/bpf/progs/task_local_data.bpf.h
@@ -61,6 +61,7 @@
#define TLD_ROUND_UP(x, y) ((((x) - 1) | TLD_ROUND_MASK(x, y)) + 1)
#define TLD_MAX_DATA_CNT (__PAGE_SIZE / sizeof(struct tld_metadata) - 1)
+#define TLD_DATA_SIZE (__PAGE_SIZE - sizeof(__u64))
#ifndef TLD_NAME_LEN
#define TLD_NAME_LEN 62
@@ -189,6 +190,8 @@ static int __tld_fetch_key(struct tld_object *tld_obj, const char *name, int i_s
return start + off;
off += TLD_ROUND_UP(metadata[i].size, 8);
+ if (off > TLD_DATA_SIZE)
+ break;
}
return -cnt;
diff --git a/tools/testing/selftests/bpf/progs/test_bpf_nf.c b/tools/testing/selftests/bpf/progs/test_bpf_nf.c
index df43649ecb78..eda9b7bbab75 100644
--- a/tools/testing/selftests/bpf/progs/test_bpf_nf.c
+++ b/tools/testing/selftests/bpf/progs/test_bpf_nf.c
@@ -190,8 +190,8 @@ nf_ct_test(struct nf_conn *(*lookup_fn)(void *, struct bpf_sock_tuple *, u32,
ct = alloc_fn(ctx, &bpf_tuple, sizeof(bpf_tuple.ipv4), &opts_def,
sizeof(opts_def));
if (ct) {
- __u16 sport = bpf_get_prandom_u32();
- __u16 dport = bpf_get_prandom_u32();
+ __u16 sport = bpf_get_prandom_u32() % 65535 + 1;
+ __u16 dport = bpf_get_prandom_u32() % 65535 + 1;
union nf_inet_addr saddr = {};
union nf_inet_addr daddr = {};
struct nf_conn *ct_ins;
@@ -293,8 +293,8 @@ nf_ct_opts_new_test(struct nf_conn *(*lookup_fn)(void *, struct bpf_sock_tuple *
ct = alloc_fn(ctx, &bpf_tuple, sizeof(bpf_tuple.ipv4), &opts_def,
sizeof(opts_def));
if (ct) {
- __u16 sport = bpf_get_prandom_u32();
- __u16 dport = bpf_get_prandom_u32();
+ __u16 sport = bpf_get_prandom_u32() % 65535 + 1;
+ __u16 dport = bpf_get_prandom_u32() % 65535 + 1;
union nf_inet_addr saddr = {};
union nf_inet_addr daddr = {};
struct nf_conn *ct_ins;
diff --git a/tools/testing/selftests/bpf/progs/test_global_percpu_data.c b/tools/testing/selftests/bpf/progs/test_global_percpu_data.c
new file mode 100644
index 000000000000..5dc21b3b4cb5
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/test_global_percpu_data.c
@@ -0,0 +1,89 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_misc.h"
+
+/* Used for testing map name. */
+int loong SEC(".percpu.looooooooong");
+int data3 SEC(".data.percpu");
+int data2 SEC(".percpu.data");
+
+int run;
+/* cpu_id as array to verify map value resizing. */
+int cpu_id[1] SEC(".percpu");
+int data SEC(".percpu") = -1;
+int nums[7] SEC(".percpu");
+bool set SEC(".percpu") = false;
+struct {
+ char set;
+ int i;
+ int nums[7];
+} struct_data SEC(".percpu") = {
+ .set = 0,
+ .i = -1,
+};
+
+SEC("raw_tp/task_rename")
+__auxiliary
+int update_percpu_data(void *ctx)
+{
+ struct_data.nums[6] = 0xc0de;
+ struct_data.set = 1;
+ struct_data.i = 1;
+ nums[6] = 0xc0de;
+ data = 1;
+ run++;
+ set = true;
+ cpu_id[0] = bpf_get_smp_processor_id();
+ return 0;
+}
+
+static const char fmt[] SEC(".percpu.fmt") = "data %d\n";
+
+SEC("?kprobe")
+__failure __msg("R{{[0-9]+}} points to percpu_array map which cannot be used as const string")
+int verifier_strncmp(void *ctx)
+{
+ return bpf_strncmp("test", 5, fmt);
+}
+
+SEC("?kprobe")
+__failure __msg("R{{[0-9]+}} points to percpu_array map which cannot be used as const string")
+int verifier_snprintf(void *ctx)
+{
+ u64 args[] = { data };
+ char buf[128];
+ int len;
+
+ len = bpf_snprintf(buf, sizeof(buf), fmt, args, sizeof(args));
+ if (len > 0)
+ bpf_printk("snprintf: %s\n", buf);
+ return 0;
+}
+
+volatile const __u32 num_cpus = 0;
+volatile const int num_off;
+volatile const int elem_sz;
+__u32 sum = 0;
+bool run_iter = false;
+
+SEC("iter/bpf_map_elem")
+__auxiliary
+int dump_percpu_data(struct bpf_iter__bpf_map_elem *ctx)
+{
+ void *pptr = ctx->value;
+ int i;
+
+ if (!pptr)
+ return 0;
+
+ run_iter = true;
+
+ for (i = 0; i < num_cpus; i++) {
+ sum += *(int *) (pptr + num_off);
+ pptr += elem_sz;
+ }
+ return 0;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c b/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c
index f678ee6bd7ea..55282f20fa32 100644
--- a/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c
+++ b/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c
@@ -14,17 +14,18 @@ struct array_map {
__type(key, int);
__type(value, struct foo);
__uint(max_entries, 1);
-} array_map SEC(".maps");
+} array_map SEC(".maps"), array_map_b SEC(".maps");
struct {
__uint(type, BPF_MAP_TYPE_ARRAY_OF_MAPS);
- __uint(max_entries, 1);
+ __uint(max_entries, 2);
__type(key, int);
__type(value, int);
__array(values, struct array_map);
} map_of_maps SEC(".maps") = {
.values = {
[0] = &array_map,
+ [1] = &array_map_b,
},
};
@@ -314,4 +315,66 @@ int lock_global_sleepable_subprog_indirect(struct __sk_buff *ctx)
return ret;
}
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __uint(max_entries, 2);
+ __type(key, int);
+ __type(value, struct foo);
+} callback_array_map SEC(".maps");
+
+struct callback_ctx {
+ struct foo *value;
+};
+
+static long lock_different_value(struct bpf_map *map, int *key,
+ struct foo *value, struct callback_ctx *ctx)
+{
+ bpf_spin_lock(&value->lock);
+ bpf_spin_unlock(&ctx->value->lock);
+ return 0;
+}
+
+static long nest_lock_different_value(struct bpf_map *map, int *key,
+ struct foo *value, void *data)
+{
+ struct callback_ctx ctx = { .value = value };
+
+ bpf_for_each_map_elem(&callback_array_map, lock_different_value, &ctx, 0);
+ return 0;
+}
+
+SEC("?tc")
+int callback_value_lock_identity(void *ctx)
+{
+ bpf_for_each_map_elem(&callback_array_map, nest_lock_different_value, NULL, 0);
+ return 0;
+}
+
+static long nest_lock_different_inner_value(struct bpf_map *map, int *key,
+ struct foo *value, void *data)
+{
+ struct callback_ctx ctx = { .value = value };
+ int inner_key = 1;
+ void *inner_map;
+
+ inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key);
+ if (!inner_map)
+ return 0;
+ bpf_for_each_map_elem(inner_map, lock_different_value, &ctx, 0);
+ return 0;
+}
+
+SEC("?tc")
+int callback_inner_map_value_lock_identity(void *ctx)
+{
+ int inner_key = 0;
+ void *inner_map;
+
+ inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key);
+ if (!inner_map)
+ return 0;
+ bpf_for_each_map_elem(inner_map, nest_lock_different_inner_value, NULL, 0);
+ return 0;
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c b/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c
new file mode 100644
index 000000000000..5c4c07545bc9
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c
@@ -0,0 +1,129 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <stdbool.h>
+#include <stddef.h>
+
+#include <linux/bpf.h>
+#include <linux/icmp.h>
+#include <linux/if_ether.h>
+#include <linux/in.h>
+#include <linux/ip.h>
+#include <linux/tcp.h>
+
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_endian.h>
+
+#define ICMP_SAMPLE_LEN (sizeof(struct iphdr) + 8)
+#define ICMP_HDRS_LEN (sizeof(struct iphdr) + sizeof(struct icmphdr))
+
+__be16 server_port = 0;
+__u16 pmtu = 0;
+
+long change_tail_ret = 1;
+long adjust_room_ret = 0;
+bool icmp_sent = false;
+bool icmp_err = false;
+
+static __always_inline __sum16 csum_fold(__wsum csum)
+{
+ csum = (csum & 0xffff) + (csum >> 16);
+ csum = (csum & 0xffff) + (csum >> 16);
+
+ return (__sum16)~csum;
+}
+
+SEC("tc/egress")
+int change_tail_icmp(struct __sk_buff *skb)
+{
+ __u8 smac[ETH_ALEN], dmac[ETH_ALEN];
+ void *data, *data_end;
+ struct icmphdr *icmp;
+ struct ethhdr *eth;
+ struct tcphdr *tcp;
+ __be32 saddr, daddr;
+ struct iphdr *ip;
+ __wsum csum;
+
+ if (icmp_sent || icmp_err)
+ return TCX_PASS;
+
+ data = (void *)(long)skb->data;
+ data_end = (void *)(long)skb->data_end;
+
+ eth = data;
+ if ((void *)(eth + 1) > data_end)
+ return TCX_PASS;
+ if (eth->h_proto != bpf_htons(ETH_P_IP))
+ return TCX_PASS;
+
+ ip = (void *)(eth + 1);
+ if ((void *)(ip + 1) > data_end)
+ return TCX_PASS;
+ if (ip->ihl != 5 || ip->protocol != IPPROTO_TCP)
+ return TCX_PASS;
+
+ tcp = (void *)(ip + 1);
+ if ((void *)(tcp + 1) > data_end)
+ return TCX_PASS;
+ if (tcp->dest != server_port)
+ return TCX_PASS;
+ if (bpf_ntohs(ip->tot_len) <= sizeof(*ip) + tcp->doff * 4)
+ return TCX_PASS;
+
+ __builtin_memcpy(smac, eth->h_source, ETH_ALEN);
+ __builtin_memcpy(dmac, eth->h_dest, ETH_ALEN);
+ saddr = ip->saddr;
+ daddr = ip->daddr;
+
+ change_tail_ret = bpf_skb_change_tail(skb, ETH_HLEN + ICMP_SAMPLE_LEN, 0);
+ if (change_tail_ret) {
+ icmp_err = true;
+ return TCX_PASS;
+ }
+
+ adjust_room_ret = bpf_skb_adjust_room(skb, ICMP_HDRS_LEN,
+ BPF_ADJ_ROOM_MAC,
+ BPF_F_ADJ_ROOM_NO_CSUM_RESET);
+ if (adjust_room_ret) {
+ icmp_err = true;
+ return TCX_DROP;
+ }
+
+ data = (void *)(long)skb->data;
+ data_end = (void *)(long)skb->data_end;
+
+ eth = data;
+ ip = (void *)(eth + 1);
+ icmp = (void *)(ip + 1);
+ if ((void *)icmp + sizeof(*icmp) + ICMP_SAMPLE_LEN > data_end) {
+ icmp_err = true;
+ return TCX_DROP;
+ }
+
+ __builtin_memcpy(eth->h_dest, smac, ETH_ALEN);
+ __builtin_memcpy(eth->h_source, dmac, ETH_ALEN);
+
+ __builtin_memset(icmp, 0, sizeof(*icmp));
+ icmp->type = ICMP_DEST_UNREACH;
+ icmp->code = ICMP_FRAG_NEEDED;
+ icmp->un.frag.mtu = bpf_htons(pmtu);
+
+ __builtin_memset(ip, 0, sizeof(*ip));
+ ip->version = 4;
+ ip->ihl = 5;
+ ip->ttl = 64;
+ ip->protocol = IPPROTO_ICMP;
+ ip->tot_len = bpf_htons(ICMP_HDRS_LEN + ICMP_SAMPLE_LEN);
+ ip->saddr = daddr;
+ ip->daddr = saddr;
+
+ csum = bpf_csum_diff(NULL, 0, (__be32 *)icmp,
+ sizeof(*icmp) + ICMP_SAMPLE_LEN, 0);
+ icmp->checksum = csum_fold(csum);
+ csum = bpf_csum_diff(NULL, 0, (__be32 *)ip, sizeof(*ip), 0);
+ ip->check = csum_fold(csum);
+ icmp_sent = true;
+ return bpf_redirect(skb->ifindex, BPF_F_INGRESS);
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/test_tc_tunnel.c b/tools/testing/selftests/bpf/progs/test_tc_tunnel.c
index 7376df405a6b..853bca962910 100644
--- a/tools/testing/selftests/bpf/progs/test_tc_tunnel.c
+++ b/tools/testing/selftests/bpf/progs/test_tc_tunnel.c
@@ -6,6 +6,7 @@
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_endian.h>
+#include <bpf/bpf_core_read.h>
#include "bpf_tracing_net.h"
#include "bpf_compiler.h"
@@ -37,6 +38,22 @@ struct vxlanhdr___local {
#define EXTPROTO_VXLAN 0x1
+#define SKB_GSO_UDP_TUNNEL_MASK (SKB_GSO_UDP_TUNNEL | \
+ SKB_GSO_UDP_TUNNEL_CSUM)
+
+#define SKB_GSO_TUNNEL_MASK (SKB_GSO_UDP_TUNNEL_MASK | \
+ SKB_GSO_GRE | \
+ SKB_GSO_GRE_CSUM | \
+ SKB_GSO_IPXIP4 | \
+ SKB_GSO_IPXIP6 | \
+ SKB_GSO_ESP)
+
+#define BPF_F_ADJ_ROOM_DECAP_L4_MASK (BPF_F_ADJ_ROOM_DECAP_L4_UDP | \
+ BPF_F_ADJ_ROOM_DECAP_L4_GRE)
+
+#define BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK (BPF_F_ADJ_ROOM_DECAP_IPXIP4 | \
+ BPF_F_ADJ_ROOM_DECAP_IPXIP6)
+
#define VXLAN_FLAGS bpf_htonl(1<<27)
#define VNI_ID 1
#define VXLAN_VNI bpf_htonl(VNI_ID << 8)
@@ -589,9 +606,12 @@ int __encap_ip6vxlan_eth(struct __sk_buff *skb)
return TC_ACT_OK;
}
-static int decap_internal(struct __sk_buff *skb, int off, int len, char proto)
+static int decap_internal(struct __sk_buff *skb, int off, int len, char proto,
+ __u64 ipxip_flag)
{
__u64 flags = BPF_F_ADJ_ROOM_FIXED_GSO;
+ struct sk_buff *kskb;
+ struct skb_shared_info *shinfo;
struct ipv6_opt_hdr ip6_opt_hdr;
struct gre_hdr greh;
struct udphdr udph;
@@ -599,10 +619,12 @@ static int decap_internal(struct __sk_buff *skb, int off, int len, char proto)
switch (proto) {
case IPPROTO_IPIP:
- flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV4;
+ flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV4 |
+ ipxip_flag;
break;
case IPPROTO_IPV6:
- flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV6;
+ flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV6 |
+ ipxip_flag;
break;
case NEXTHDR_DEST:
if (bpf_skb_load_bytes(skb, off + len, &ip6_opt_hdr,
@@ -610,10 +632,12 @@ static int decap_internal(struct __sk_buff *skb, int off, int len, char proto)
return TC_ACT_OK;
switch (ip6_opt_hdr.nexthdr) {
case IPPROTO_IPIP:
- flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV4;
+ flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV4 |
+ ipxip_flag;
break;
case IPPROTO_IPV6:
- flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV6;
+ flags |= BPF_F_ADJ_ROOM_DECAP_L3_IPV6 |
+ ipxip_flag;
break;
default:
return TC_ACT_OK;
@@ -621,6 +645,11 @@ static int decap_internal(struct __sk_buff *skb, int off, int len, char proto)
break;
case IPPROTO_GRE:
olen += sizeof(struct gre_hdr);
+ if (!bpf_core_enum_value_exists(enum bpf_adj_room_flags,
+ BPF_F_ADJ_ROOM_DECAP_L4_GRE))
+ return TC_ACT_SHOT;
+ flags |= BPF_F_ADJ_ROOM_DECAP_L4_GRE;
+
if (bpf_skb_load_bytes(skb, off + len, &greh, sizeof(greh)) < 0)
return TC_ACT_OK;
switch (bpf_ntohs(greh.protocol)) {
@@ -634,6 +663,10 @@ static int decap_internal(struct __sk_buff *skb, int off, int len, char proto)
break;
case IPPROTO_UDP:
olen += sizeof(struct udphdr);
+ if (!bpf_core_enum_value_exists(enum bpf_adj_room_flags,
+ BPF_F_ADJ_ROOM_DECAP_L4_UDP))
+ return TC_ACT_SHOT;
+ flags |= BPF_F_ADJ_ROOM_DECAP_L4_UDP;
if (bpf_skb_load_bytes(skb, off + len, &udph, sizeof(udph)) < 0)
return TC_ACT_OK;
switch (bpf_ntohs(udph.dest)) {
@@ -655,6 +688,40 @@ static int decap_internal(struct __sk_buff *skb, int off, int len, char proto)
if (bpf_skb_adjust_room(skb, -olen, BPF_ADJ_ROOM_MAC, flags))
return TC_ACT_SHOT;
+ kskb = bpf_cast_to_kern_ctx(skb);
+ shinfo = bpf_core_cast(kskb->head + kskb->end, struct skb_shared_info);
+ if (shinfo->gso_size) {
+ if ((flags & BPF_F_ADJ_ROOM_DECAP_L4_UDP) &&
+ (shinfo->gso_type & SKB_GSO_UDP_TUNNEL_MASK))
+ return TC_ACT_SHOT;
+
+ if ((flags & BPF_F_ADJ_ROOM_DECAP_L4_GRE) &&
+ (shinfo->gso_type & (SKB_GSO_GRE | SKB_GSO_GRE_CSUM)))
+ return TC_ACT_SHOT;
+
+ if ((flags & BPF_F_ADJ_ROOM_DECAP_IPXIP4) &&
+ (shinfo->gso_type & SKB_GSO_IPXIP4))
+ return TC_ACT_SHOT;
+
+ if ((flags & BPF_F_ADJ_ROOM_DECAP_IPXIP6) &&
+ (shinfo->gso_type & SKB_GSO_IPXIP6))
+ return TC_ACT_SHOT;
+
+ if (flags & (BPF_F_ADJ_ROOM_DECAP_L4_MASK |
+ BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK)) {
+ if ((shinfo->gso_type & SKB_GSO_TUNNEL_MASK) &&
+ !kskb->encapsulation)
+ return TC_ACT_SHOT;
+ if (!(shinfo->gso_type & SKB_GSO_TUNNEL_MASK) &&
+ kskb->encapsulation)
+ return TC_ACT_SHOT;
+ }
+ } else if ((flags & (BPF_F_ADJ_ROOM_DECAP_L4_MASK |
+ BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK)) &&
+ kskb->encapsulation) {
+ return TC_ACT_SHOT;
+ }
+
return TC_ACT_OK;
}
@@ -662,6 +729,10 @@ static int decap_ipv4(struct __sk_buff *skb)
{
struct iphdr iph_outer;
+ if (!bpf_core_enum_value_exists(enum bpf_adj_room_flags,
+ BPF_F_ADJ_ROOM_DECAP_IPXIP4))
+ return TC_ACT_SHOT;
+
if (bpf_skb_load_bytes(skb, ETH_HLEN, &iph_outer,
sizeof(iph_outer)) < 0)
return TC_ACT_OK;
@@ -670,19 +741,25 @@ static int decap_ipv4(struct __sk_buff *skb)
return TC_ACT_OK;
return decap_internal(skb, ETH_HLEN, sizeof(iph_outer),
- iph_outer.protocol);
+ iph_outer.protocol,
+ BPF_F_ADJ_ROOM_DECAP_IPXIP4);
}
static int decap_ipv6(struct __sk_buff *skb)
{
struct ipv6hdr iph_outer;
+ if (!bpf_core_enum_value_exists(enum bpf_adj_room_flags,
+ BPF_F_ADJ_ROOM_DECAP_IPXIP6))
+ return TC_ACT_SHOT;
+
if (bpf_skb_load_bytes(skb, ETH_HLEN, &iph_outer,
sizeof(iph_outer)) < 0)
return TC_ACT_OK;
return decap_internal(skb, ETH_HLEN, sizeof(iph_outer),
- iph_outer.nexthdr);
+ iph_outer.nexthdr,
+ BPF_F_ADJ_ROOM_DECAP_IPXIP6);
}
SEC("tc")
diff --git a/tools/testing/selftests/bpf/progs/test_usdt.c b/tools/testing/selftests/bpf/progs/test_usdt.c
index f00cb52874e0..0ee78fb050a1 100644
--- a/tools/testing/selftests/bpf/progs/test_usdt.c
+++ b/tools/testing/selftests/bpf/progs/test_usdt.c
@@ -149,5 +149,30 @@ int usdt_executed(struct pt_regs *ctx)
executed++;
return 0;
}
+
+int arg_total;
+int arg_bad;
+long arg_last[3];
+long expected_arg[3];
+int expected_pid;
+
+SEC("usdt")
+int BPF_USDT(usdt_check_arg, long arg1, long arg2, long arg3)
+{
+ if (expected_pid != (bpf_get_current_pid_tgid() >> 32))
+ return 0;
+
+ __sync_fetch_and_add(&arg_total, 1);
+ arg_last[0] = arg1;
+ arg_last[1] = arg2;
+ arg_last[2] = arg3;
+
+ if (arg1 != expected_arg[0] ||
+ arg2 != expected_arg[1] ||
+ arg3 != expected_arg[2])
+ __sync_fetch_and_add(&arg_bad, 1);
+
+ return 0;
+}
#endif
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/timer_mim_reject.c b/tools/testing/selftests/bpf/progs/timer_mim_reject.c
index dd3f1ed6d6e6..83f31138336b 100644
--- a/tools/testing/selftests/bpf/progs/timer_mim_reject.c
+++ b/tools/testing/selftests/bpf/progs/timer_mim_reject.c
@@ -43,7 +43,7 @@ static int timer_cb(void *map, int *key, struct hmap_elem *val)
return 0;
}
-SEC("fentry/bpf_fentry_test1")
+SEC("?fentry/bpf_fentry_test1")
int BPF_PROG(test1, int a)
{
struct hmap_elem init = {};
@@ -72,3 +72,85 @@ int BPF_PROG(test1, int a)
err |= 8;
return 0;
}
+
+struct callback_ctx {
+ void *map;
+};
+
+static int mismatch_iter_cb(void *map, int *key, struct hmap_elem *val, struct callback_ctx *ctx)
+{
+ bpf_timer_init(&val->timer, ctx->map, CLOCK_MONOTONIC);
+ return 0;
+}
+
+static int timer_mismatch_cb(void *map, int *key, struct hmap_elem *val)
+{
+ struct callback_ctx ctx = { .map = map };
+ struct bpf_map *inner_map2;
+ int array_key2 = ARRAY_KEY2;
+
+ inner_map2 = bpf_map_lookup_elem(&outer_arr, &array_key2);
+ if (!inner_map2)
+ return 0;
+ bpf_for_each_map_elem(inner_map2, mismatch_iter_cb, &ctx, 0);
+ return 0;
+}
+
+static int match_iter_cb(void *map, int *key, struct hmap_elem *val, struct callback_ctx *ctx)
+{
+ bpf_timer_init(&val->timer, map, CLOCK_MONOTONIC);
+ return 0;
+}
+
+static int timer_match_cb(void *map, int *key, struct hmap_elem *val)
+{
+ struct callback_ctx ctx = {};
+ struct bpf_map *inner_map2;
+ int array_key2 = ARRAY_KEY2;
+
+ inner_map2 = bpf_map_lookup_elem(&outer_arr, &array_key2);
+ if (!inner_map2)
+ return 0;
+ bpf_for_each_map_elem(inner_map2, match_iter_cb, &ctx, 0);
+ return 0;
+}
+
+SEC("?fentry/bpf_fentry_test1")
+int BPF_PROG(callback_map_uid_mismatch, int a)
+{
+ struct hmap_elem *val;
+ struct bpf_map *inner_map;
+ int array_key = ARRAY_KEY;
+ int hash_key = HASH_KEY;
+
+ inner_map = bpf_map_lookup_elem(&outer_arr, &array_key);
+ if (!inner_map)
+ return 0;
+ val = bpf_map_lookup_elem(inner_map, &hash_key);
+ if (!val)
+ return 0;
+
+ bpf_timer_init(&val->timer, inner_map, CLOCK_MONOTONIC);
+ bpf_timer_set_callback(&val->timer, timer_mismatch_cb);
+ return 0;
+}
+
+SEC("?fentry/bpf_fentry_test1")
+int BPF_PROG(callback_map_uid_match, int a)
+{
+ struct hmap_elem *val;
+ struct bpf_map *inner_map;
+ int array_key = ARRAY_KEY;
+ int hash_key = HASH_KEY;
+
+ inner_map = bpf_map_lookup_elem(&outer_arr, &array_key);
+ if (!inner_map)
+ return 0;
+ val = bpf_map_lookup_elem(inner_map, &hash_key);
+ if (!val)
+ return 0;
+
+ bpf_timer_init(&val->timer, inner_map, CLOCK_MONOTONIC);
+ bpf_timer_set_callback(&val->timer, timer_match_cb);
+ return 0;
+}
diff --git a/tools/testing/selftests/bpf/progs/verifier_arena.c b/tools/testing/selftests/bpf/progs/verifier_arena.c
index b241bbcf54a8..3e33766547c0 100644
--- a/tools/testing/selftests/bpf/progs/verifier_arena.c
+++ b/tools/testing/selftests/bpf/progs/verifier_arena.c
@@ -562,6 +562,55 @@ int arena_ptr_add_arena_ptr(void *ctx)
}
SEC("syscall")
+__failure __msg("same insn cannot be used with and without arena pointer")
+int mixed_arena_scalar_alu64_scalar_first(void *ctx)
+{
+ volatile register __u64 reg asm("r3");
+ __u32 pick_arena = bpf_get_prandom_u32();
+
+ reg = 1ULL << 32;
+
+ if (pick_arena) {
+ asm volatile (
+ "r9 = %[arena] ll;"
+ "%[reg] = 0;"
+ "%[reg] = addr_space_cast(%[reg], 0x0, 0x1);"
+ : [reg] "=r"(reg)
+ : __imm_addr(arena)
+ : "r9"
+ );
+ }
+
+ reg += 1;
+
+ return 0;
+}
+
+SEC("syscall")
+__failure __msg("same insn cannot be used with and without arena pointer")
+int mixed_arena_scalar_alu64_arena_first(void *ctx)
+{
+ volatile register __u64 reg asm("r3");
+ __u32 pick_scalar = bpf_get_prandom_u32();
+
+ asm volatile (
+ "r9 = %[arena] ll;"
+ "%[reg] = 0;"
+ "%[reg] = addr_space_cast(%[reg], 0x0, 0x1);"
+ : [reg] "=r"(reg)
+ : __imm_addr(arena)
+ : "r9"
+ );
+
+ if (pick_scalar)
+ reg = 1ULL << 32;
+
+ reg += 1;
+
+ return 0;
+}
+
+SEC("syscall")
__success __retval(0)
int scalar_xor_arena_ptr(void *ctx)
{
@@ -635,7 +684,37 @@ int non_arena_ptr_add_to_arena_ptr(void *ctx)
return 0;
}
-#endif
+SEC("socket")
+__description("arena and stack atomic at the same instruction")
+__failure __msg("same insn cannot be used with different pointers")
+__arch_x86_64
+__load_if_JITed()
+__naked void mixed_arena_stack_atomic(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r6 = r10; \
+ r6 += -8; \
+ r9 = 0; \
+ *(u64 *)(r6 + 0) = r9; \
+ r7 = 8192; \
+ r7 = addr_space_cast(r7, 0, 1); \
+ call %[bpf_get_prandom_u32]; \
+ if w0 != 0 goto 1f; \
+ r8 = r6; \
+ goto 2f; \
+1: r8 = r7; \
+2: r9 = 1; \
+ lock *(u64 *)(r8 + 0) += r9; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+#endif /* defined(__BPF_FEATURE_ADDR_SPACE_CAST) */
static __noinline
u32 __arena *check_arena_arg_nonglobal(u32 __arena *arg)
diff --git a/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c b/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c
index 6bf95550a024..1b653bfb63eb 100644
--- a/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c
+++ b/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c
@@ -9,6 +9,11 @@
char _license[] SEC("license") = "GPL";
+struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym;
+void bpf_task_release(struct task_struct *p) __ksym;
+void bpf_rcu_read_lock(void) __ksym;
+void bpf_rcu_read_unlock(void) __ksym;
+
/* Timer tests */
struct timer_elem {
@@ -62,10 +67,109 @@ int timer_sleepable_prog(void *ctx)
return 0;
}
+static int timer_sys_bpf_cb(void *map, int *key, struct bpf_timer *timer)
+{
+ __u64 attr = 0;
+
+ bpf_sys_bpf(BPF_MAP_FREEZE, &attr, sizeof(attr));
+ return 0;
+}
+
+SEC("syscall")
+__failure __msg("sleepable helper bpf_sys_bpf#{{[0-9]+}} in non-sleepable prog")
+int timer_sys_bpf_prog(void *ctx)
+{
+ struct timer_elem *val;
+ int key = 0;
+
+ val = bpf_map_lookup_elem(&timer_map, &key);
+ if (!val)
+ return 0;
+
+ bpf_timer_init(&val->t, &timer_map, 0);
+ bpf_timer_set_callback(&val->t, timer_sys_bpf_cb);
+ return 0;
+}
+
+static int timer_sys_close_cb(void *map, int *key, struct bpf_timer *timer)
+{
+ bpf_sys_close(0);
+ return 0;
+}
+
+SEC("syscall")
+__failure __msg("sleepable helper bpf_sys_close#{{[0-9]+}} in non-sleepable prog")
+int timer_sys_close_prog(void *ctx)
+{
+ struct timer_elem *val;
+ int key = 0;
+
+ val = bpf_map_lookup_elem(&timer_map, &key);
+ if (!val)
+ return 0;
+
+ bpf_timer_init(&val->t, &timer_map, 0);
+ bpf_timer_set_callback(&val->t, timer_sys_close_cb);
+ return 0;
+}
+
+static int timer_btf_find_cb(void *map, int *key, struct bpf_timer *timer)
+{
+ char name[] = "task_struct";
+
+ bpf_btf_find_by_name_kind(name, sizeof(name), BTF_KIND_STRUCT, 0);
+ return 0;
+}
+
+SEC("syscall")
+__failure __msg("sleepable helper bpf_btf_find_by_name_kind#{{[0-9]+}} in non-sleepable prog")
+int timer_btf_find_prog(void *ctx)
+{
+ struct timer_elem *val;
+ int key = 0;
+
+ val = bpf_map_lookup_elem(&timer_map, &key);
+ if (!val)
+ return 0;
+
+ bpf_timer_init(&val->t, &timer_map, 0);
+ bpf_timer_set_callback(&val->t, timer_btf_find_cb);
+ return 0;
+}
+
+SEC("syscall")
+__success
+int syscall_sys_bpf_prog(void *ctx)
+{
+ __u64 attr = 0;
+
+ bpf_sys_bpf(BPF_MAP_FREEZE, &attr, sizeof(attr));
+ return 0;
+}
+
+SEC("syscall")
+__success
+int syscall_sys_close_prog(void *ctx)
+{
+ bpf_sys_close(0);
+ return 0;
+}
+
+SEC("syscall")
+__success
+int syscall_btf_find_prog(void *ctx)
+{
+ char name[] = "task_struct";
+
+ bpf_btf_find_by_name_kind(name, sizeof(name), BTF_KIND_STRUCT, 0);
+ return 0;
+}
+
/* Workqueue tests */
struct wq_elem {
struct bpf_wq w;
+ struct task_struct __kptr *task;
};
struct {
@@ -119,6 +223,106 @@ int wq_sleepable_prog(void *ctx)
return 0;
}
+__noinline int wq_global_acquire(void)
+{
+ struct task_struct *task, *acquired;
+ struct wq_elem *val;
+ int key = 0;
+
+ val = bpf_map_lookup_elem(&wq_map, &key);
+ if (!val)
+ return 0;
+
+ task = val->task;
+ if (!task)
+ return 0;
+
+ acquired = bpf_task_acquire(task);
+ if (acquired)
+ bpf_task_release(acquired);
+ return 0;
+}
+
+static int wq_global_rcu_cb(void *map, int *key, void *value)
+{
+ wq_global_acquire();
+ return 0;
+}
+
+SEC("fentry/bpf_fentry_test1")
+__failure __msg("R1 must be a rcu pointer")
+int wq_global_rcu_prog(void *ctx)
+{
+ struct wq_elem *val;
+ int key = 0;
+
+ val = bpf_map_lookup_elem(&wq_map, &key);
+ if (!val)
+ return 0;
+
+ bpf_wq_init(&val->w, &wq_map, 0);
+ bpf_wq_set_callback(&val->w, wq_global_rcu_cb, 0);
+ return 0;
+}
+
+static int wq_global_rcu_lock_cb(void *map, int *key, void *value)
+{
+ bpf_rcu_read_lock();
+ wq_global_acquire();
+ bpf_rcu_read_unlock();
+ return 0;
+}
+
+SEC("fentry/bpf_fentry_test1")
+__success
+int wq_global_rcu_lock_prog(void *ctx)
+{
+ struct wq_elem *val;
+ int key = 0;
+
+ /* Verify the same global subprog in non-sleepable and protected contexts. */
+ wq_global_acquire();
+
+ val = bpf_map_lookup_elem(&wq_map, &key);
+ if (!val)
+ return 0;
+
+ bpf_wq_init(&val->w, &wq_map, 0);
+ bpf_wq_set_callback(&val->w, wq_global_rcu_lock_cb, 0);
+ return 0;
+}
+
+__weak __noinline int wq_global_no_rcu(void)
+{
+ return 0;
+}
+
+static int wq_global_no_rcu_cb(void *map, int *key, void *value)
+{
+ wq_global_no_rcu();
+ return 0;
+}
+
+SEC("fentry/bpf_fentry_test1")
+__success __log_level(4)
+__msg("subprog {{[0-9]+}} (wq_global_no_rcu) global insns_self 4 insns_total 4 stack 0")
+int wq_global_no_rcu_prog(void *ctx)
+{
+ struct wq_elem *val;
+ int key = 0;
+
+ /* Verify the same global in non-sleepable and unprotected contexts. */
+ wq_global_no_rcu();
+
+ val = bpf_map_lookup_elem(&wq_map, &key);
+ if (!val)
+ return 0;
+
+ bpf_wq_init(&val->w, &wq_map, 0);
+ bpf_wq_set_callback(&val->w, wq_global_no_rcu_cb, 0);
+ return 0;
+}
+
/* Task work tests */
struct task_work_elem {
diff --git a/tools/testing/selftests/bpf/progs/verifier_basic_stack.c b/tools/testing/selftests/bpf/progs/verifier_basic_stack.c
index d3df7a9f1d8c..0eb495ce85c1 100644
--- a/tools/testing/selftests/bpf/progs/verifier_basic_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_basic_stack.c
@@ -27,7 +27,8 @@ __naked void stack_out_of_bounds(void)
SEC("socket")
__description("uninitialized stack1")
-__success __log_level(4) __msg("stack depth 8")
+__success __log_level(4)
+__msg("subprog 0 (uninitialized_stack1) main {{.*}} stack 8")
__failure_unpriv __msg_unpriv("invalid read from stack")
__naked void uninitialized_stack1(void)
{
@@ -45,7 +46,8 @@ __naked void uninitialized_stack1(void)
SEC("socket")
__description("uninitialized stack2")
-__success __log_level(4) __msg("stack depth 8")
+__success __log_level(4)
+__msg("subprog 0 (uninitialized_stack2) main insns_self {{[0-9]+}} insns_total {{[0-9]+}} stack 8")
__failure_unpriv __msg_unpriv("invalid read from stack")
__naked void uninitialized_stack2(void)
{
diff --git a/tools/testing/selftests/bpf/progs/verifier_bounds.c b/tools/testing/selftests/bpf/progs/verifier_bounds.c
index 1a273e416fed..df8d5309657e 100644
--- a/tools/testing/selftests/bpf/progs/verifier_bounds.c
+++ b/tools/testing/selftests/bpf/progs/verifier_bounds.c
@@ -2267,6 +2267,47 @@ __naked void deduce64_from_32_wrapping_32bit(void)
: __clobber_all);
}
+/*
+ * Unprivileged variable pointer arithmetic on a PTR_TO_MAP_VALUE whose
+ * offset collapses to a constant. The Spectre-v1 speculative path snapshots
+ * the pointer while its r32 has just been blanked but its offset not yet
+ * synced; the following register move used to trip reg_bounds_sanity_check()
+ * ("const subreg tnum out of sync with range bounds"). With
+ * BPF_F_TEST_REG_INVARIANTS that violation turns into a load failure, so the
+ * unprivileged program must still load.
+ */
+SEC("socket")
+__success __success_unpriv
+__flag(BPF_F_TEST_REG_INVARIANTS)
+__naked void spec_ptr_alu_const_offset(void)
+{
+ asm volatile (" \
+ call %[bpf_ktime_get_ns]; \
+ *(u64*)(r10 - 16) = r0; \
+ r1 = 0; \
+ *(u64*)(r10 - 8) = r1; \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = %[map_hash_8b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto l0_%=; \
+ r1 = *(u64*)(r10 - 16); \
+ r2 = 0x40000000; \
+ if r1 > r2 goto l0_%=; \
+ if r1 s> 1 goto l0_%=; /* r1 in [0, 1] */ \
+ r0 += r1; /* ptr += bounded scalar */ \
+ r9 = r0; /* used to trip the warning */ \
+ *(u8*)(r0 + 0) = r1; \
+l0_%=: r0 = 0; \
+ exit; \
+ "
+ :
+ : __imm(bpf_ktime_get_ns),
+ __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_8b)
+ : __clobber_all);
+}
+
/* Check that range_within() compares cnum ranges, not min/max projections. */
SEC("socket")
__failure __msg("div by zero")
diff --git a/tools/testing/selftests/bpf/progs/verifier_bpf_fastcall.c b/tools/testing/selftests/bpf/progs/verifier_bpf_fastcall.c
index 83707faea049..a73b837553fb 100644
--- a/tools/testing/selftests/bpf/progs/verifier_bpf_fastcall.c
+++ b/tools/testing/selftests/bpf/progs/verifier_bpf_fastcall.c
@@ -10,7 +10,8 @@
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 8")
+__log_level(4)
+__msg("subprog 0 (simple) main insns_self {{[0-9]+}} insns_total {{[0-9]+}} stack 8")
__xlated("4: r5 = 5")
__xlated("5: r0 = ")
__xlated("6: r0 = &(void __percpu *)(r0)")
@@ -96,7 +97,8 @@ __naked void canary_zero_spills(void)
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 16")
+__log_level(4)
+__msg("subprog 0 (wrong_reg_in_pattern1) main {{.*}} stack 16")
__xlated("1: *(u64 *)(r10 -16) = r1")
__xlated("...")
__xlated("3: r0 = &(void __percpu *)(r0)")
@@ -598,7 +600,8 @@ __naked static void subprogs_use_independent_offsets_aux(void)
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 8")
+__log_level(4)
+__msg("subprog 0 (helper_call_does_not_prevent_bpf_fastcall) main {{.*}} stack 8")
__xlated("2: r0 = &(void __percpu *)(r0)")
__success
__naked void helper_call_does_not_prevent_bpf_fastcall(void)
@@ -618,9 +621,120 @@ __naked void helper_call_does_not_prevent_bpf_fastcall(void)
: __clobber_all);
}
+/* A jump to the first spill executes the whole pattern, rewrite is safe. */
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 24")
+__log_level(4)
+__msg("subprog 0 (jump_to_first_spill) main {{.*}} stack 0")
+__xlated("2: if r0 == 0x2a goto pc+0")
+__xlated("3: r0 = ")
+__xlated("4: r0 = &(void __percpu *)(r0)")
+__success
+__naked void jump_to_first_spill(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 1;"
+ "if r0 == 42 goto l0_%=;"
+"l0_%=:"
+ "*(u64 *)(r10 - 8) = r1;"
+ "call %[bpf_get_smp_processor_id];"
+ "r1 = *(u64 *)(r10 - 8);"
+ "exit;"
+ :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_get_smp_processor_id)
+ : __clobber_all);
+}
+
+/* A jump to the call skips the spill, the pattern must be kept. */
+SEC("raw_tp")
+__arch_x86_64
+__log_level(4)
+__msg("subprog 0 (jump_to_call) main {{.*}} stack 8")
+__xlated("2: if r0 == 0x2a goto pc+1")
+__xlated("3: *(u64 *)(r10 -8) = r1")
+__xlated("...")
+__xlated("7: r1 = *(u64 *)(r10 -8)")
+__success
+__naked void jump_to_call(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 1;"
+ "if r0 == 42 goto l0_%=;"
+ "*(u64 *)(r10 - 8) = r1;"
+"l0_%=:"
+ "call %[bpf_get_smp_processor_id];"
+ "r1 = *(u64 *)(r10 - 8);"
+ "exit;"
+ :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_get_smp_processor_id)
+ : __clobber_all);
+}
+
+/* A jump to the fill skips the spill, the pattern must be kept. */
+SEC("raw_tp")
+__arch_x86_64
+__log_level(4)
+__msg("subprog 0 (jump_to_fill) main {{.*}} stack 8")
+__xlated("2: if r0 == 0x2a goto pc+4")
+__xlated("3: *(u64 *)(r10 -8) = r1")
+__xlated("...")
+__xlated("7: r1 = *(u64 *)(r10 -8)")
+__success
+__naked void jump_to_fill(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 1;"
+ "if r0 == 42 goto l0_%=;"
+ "*(u64 *)(r10 - 8) = r1;"
+ "call %[bpf_get_smp_processor_id];"
+"l0_%=:"
+ "r1 = *(u64 *)(r10 - 8);"
+ "exit;"
+ :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_get_smp_processor_id)
+ : __clobber_all);
+}
+
+/* Same as above, but the fill is entered by an unconditional jump. */
+SEC("raw_tp")
+__arch_x86_64
+__log_level(4)
+__msg("subprog 0 (unconditional_jump_to_fill) main {{.*}} stack 8")
+__xlated("3: *(u64 *)(r10 -8) = r1")
+__xlated("...")
+__xlated("7: r1 = *(u64 *)(r10 -8)")
+__xlated("8: exit")
+__xlated("9: goto pc-3")
+__success
+__naked void unconditional_jump_to_fill(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 1;"
+ "if r0 == 42 goto l1_%=;"
+ "*(u64 *)(r10 - 8) = r1;"
+ "call %[bpf_get_smp_processor_id];"
+"l0_%=:"
+ "r1 = *(u64 *)(r10 - 8);"
+ "exit;"
+"l1_%=:"
+ "goto l0_%=;"
+ :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_get_smp_processor_id)
+ : __clobber_all);
+}
+
+SEC("raw_tp")
+__arch_x86_64
+__log_level(4)
+__msg("subprog 0 (may_goto_interaction_x86_64) main {{.*}} stack 24")
/* may_goto counter at -24 */
__xlated("0: *(u64 *)(r10 -24) =")
/* may_goto timestamp at -16 */
@@ -661,7 +775,9 @@ __naked void may_goto_interaction_x86_64(void)
SEC("raw_tp")
__arch_arm64
__arch_riscv64
-__log_level(4) __msg("stack depth 24")
+__arch_loongarch
+__log_level(4)
+__msg("subprog 0 (may_goto_interaction) main {{.*}} stack 24")
/* may_goto counter at -24 */
__xlated("0: *(u64 *)(r10 -24) =")
/* may_goto timestamp at -16 */
@@ -708,7 +824,9 @@ __naked static void dummy_loop_callback(void)
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 32+0")
+__log_level(4)
+__msg("subprog 0 (bpf_loop_interaction1) main {{.*}} stack 32")
+__msg("subprog 1 (dummy_loop_callback) static {{.*}} stack 0")
__xlated("2: r1 = 1")
__xlated("3: r0 =")
__xlated("4: r0 = &(void __percpu *)(r0)")
@@ -756,7 +874,9 @@ __naked int bpf_loop_interaction1(void)
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 40+0")
+__log_level(4)
+__msg("subprog 0 (bpf_loop_interaction2) main {{.*}} stack 40")
+__msg("subprog 1 (dummy_loop_callback) static {{.*}} stack 0")
/* call bpf_get_smp_processor_id */
__xlated("2: r1 = 42")
__xlated("3: r0 =")
@@ -800,7 +920,10 @@ __naked int bpf_loop_interaction2(void)
SEC("raw_tp")
__arch_x86_64
-__log_level(4) __msg("stack depth 512+0 max 512")
+__log_level(4)
+__msg("stack depth max 512")
+__msg("subprog 0 (cumulative_stack_depth) main {{.*}} stack 512")
+__msg("subprog 1 (cumulative_stack_depth_subprog) static {{.*}} stack 0")
/* just to print xlated version when debugging */
__xlated("r0 = &(void __percpu *)(r0)")
__success
diff --git a/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c b/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c
new file mode 100644
index 000000000000..59b84261f622
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c
@@ -0,0 +1,56 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+
+#include "bpf_experimental.h"
+#include "bpf_misc.h"
+
+struct test_empty_event {};
+
+struct test_flex_batch {
+ int nr;
+ struct test_empty_event events[];
+};
+
+struct map_value {
+ struct test_flex_batch __kptr *batch;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __type(key, int);
+ __type(value, struct map_value);
+ __uint(max_entries, 1);
+} batches SEC(".maps");
+
+SEC("syscall")
+__description("btf walk into flexible array of zero-sized elements")
+__failure __msg("access beyond struct test_flex_batch at off 4 size 1")
+int stash_and_peek(void *ctx)
+{
+ struct test_flex_batch *b, *old;
+ struct map_value *v;
+ int key = 0;
+
+ v = bpf_map_lookup_elem(&batches, &key);
+ if (!v)
+ return 0;
+
+ b = bpf_obj_new(struct test_flex_batch);
+ if (!b)
+ return 0;
+ b->nr = 1;
+
+ old = bpf_kptr_xchg(&v->batch, b);
+ if (old)
+ bpf_obj_drop(old);
+
+ b = v->batch;
+ if (!b)
+ return 0;
+
+ return b->nr + *(char *)&b->events[0];
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_cfg.c b/tools/testing/selftests/bpf/progs/verifier_cfg.c
index c1f55e1d80a4..3c3bb03e8217 100644
--- a/tools/testing/selftests/bpf/progs/verifier_cfg.c
+++ b/tools/testing/selftests/bpf/progs/verifier_cfg.c
@@ -3,6 +3,7 @@
#include <linux/bpf.h>
#include <bpf/bpf_helpers.h>
+#include "../../../include/linux/filter.h"
#include "bpf_misc.h"
SEC("socket")
@@ -56,6 +57,19 @@ __naked void out_of_range_jump2(void)
}
SEC("socket")
+__description("invalid DW LDSX instruction in diagnostics")
+__failure __msg("BUG_ldx_99")
+__log_level(2)
+__naked void invalid_dw_ldsx(void)
+{
+ asm volatile (" \
+ .8byte %[ldsx_dw]; \
+" :
+ : __imm_insn(ldsx_dw, BPF_RAW_INSN(BPF_LDX | BPF_MEMSX | BPF_DW, BPF_REG_0, BPF_REG_0, 0, 0))
+ : __clobber_all);
+}
+
+SEC("socket")
__description("loop (back-edge)")
__failure __msg("unreachable insn 1")
__msg_unpriv("back-edge")
diff --git a/tools/testing/selftests/bpf/progs/verifier_cgroup_storage.c b/tools/testing/selftests/bpf/progs/verifier_cgroup_storage.c
index 9a13f5c11ac7..884080a5bffc 100644
--- a/tools/testing/selftests/bpf/progs/verifier_cgroup_storage.c
+++ b/tools/testing/selftests/bpf/progs/verifier_cgroup_storage.c
@@ -305,4 +305,33 @@ __naked void cpu_cgroup_storage_access_6(void)
: __clobber_all);
}
+/*
+ * Verification takes two paths: with r2 being scalar zero on path (1)
+ * and with r2 being some other scalar on path (2).
+ * Check that the verifier does not use checkpoints created
+ * on path (1) to prune path (2).
+ */
+SEC("cgroup/skb")
+__failure
+__flag(BPF_F_TEST_STATE_FREQ)
+__msg("get_local_storage() doesn't support non-zero flags")
+__naked void non_zero_flags_on_a_pruned_path(void)
+{
+ asm volatile (" \
+ call %[bpf_get_prandom_u32]; \
+ /* r2 is 0 on the path explored first, 1 on the other */\
+ r2 = 1; \
+ if r0 == 0 goto 1f; \
+ r2 = 0; \
+1: r1 = %[cgroup_storage] ll; \
+ call %[bpf_get_local_storage]; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_get_local_storage),
+ __imm_addr(cgroup_storage)
+ : __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c b/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c
index 0bdeb7bc4687..dcc2dd46751a 100644
--- a/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c
+++ b/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c
@@ -56,6 +56,30 @@ int trusted_task_arg_nullable(void *ctx)
return res;
}
+/*
+ * Check that the verifier does not use checkpoints created
+ * on path with r1 == 0 to prune path with r1 != 0.
+ */
+SEC("?tp_btf/task_newtask")
+__failure
+__flag(BPF_F_TEST_STATE_FREQ)
+__msg("R1 type=scalar expected=ptr_, trusted_ptr_, rcu_ptr_")
+__naked int null_btf_id_arg_global_subprog(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 42;"
+ "if r0 > 42 goto 1f;"
+ "r1 = 0;"
+ "1:"
+ "call subprog_trusted_task_nullable;"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
__weak int subprog_trusted_task_nonnull(struct task_struct *task __arg_trusted)
{
return task->pid + task->tgid;
@@ -326,4 +350,57 @@ int anything_to_untrusted_mem(void *ctx)
return 0;
}
+struct pkt_arg {
+ __u64 x;
+ __u8 pad[56];
+};
+
+__weak int subprog_pkt_ptr_no_change(struct pkt_arg *p)
+{
+ if (!p)
+ return 0;
+
+ return p->x;
+}
+
+SEC("?tc")
+__success
+int pkt_ptr_to_global_mem_arg_no_change(struct __sk_buff *skb)
+{
+ void *data = (void *)(long)skb->data;
+ void *data_end = (void *)(long)skb->data_end;
+ struct pkt_arg *p = data;
+
+ if ((void *)(p + 1) > data_end)
+ return 0;
+
+ return subprog_pkt_ptr_no_change(p);
+}
+
+__weak int subprog_pkt_ptr_changes_data(struct __sk_buff *skb __arg_ctx,
+ struct pkt_arg *p)
+{
+ if (!p)
+ return 0;
+
+ bpf_skb_pull_data(skb, 0);
+ return p->x;
+}
+
+SEC("?tc")
+__failure __log_level(2)
+__msg("R2 is a packet pointer, but func#{{[0-9]+}} may change packet data")
+__msg("Caller passes invalid args into func#{{[0-9]+}} ('subprog_pkt_ptr_changes_data')")
+int pkt_ptr_to_global_mem_arg_changes_data(struct __sk_buff *skb)
+{
+ void *data = (void *)(long)skb->data;
+ void *data_end = (void *)(long)skb->data_end;
+ struct pkt_arg *p = data;
+
+ if ((void *)(p + 1) > data_end)
+ return 0;
+
+ return subprog_pkt_ptr_changes_data(skb, p);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_global_subprogs.c b/tools/testing/selftests/bpf/progs/verifier_global_subprogs.c
index 67dc352addfd..966f49348787 100644
--- a/tools/testing/selftests/bpf/progs/verifier_global_subprogs.c
+++ b/tools/testing/selftests/bpf/progs/verifier_global_subprogs.c
@@ -52,7 +52,15 @@ __msg("('global_calls_good_only') is global and assumed valid.")
/* eventually global_good() is transitively validated as well */
__msg("Validating global_good() func")
__msg("('global_good') is safe for any args that match its prototype")
-__msg("insns processed {{[0-9]+\\+[0-9]+\\+[0-9]+$}}")
+__msg("subprog 0 (chained_global_func_calls_success) main insns_self 7 insns_total 7 stack")
+__msg("subprog {{[0-9]+}} (global_calls_good_only) global insns_self 2 insns_total 2 stack")
+#if defined(__BPF_CPU_VERSION__) && __BPF_CPU_VERSION__ >= 4
+__msg("subprog {{[0-9]+}} (global_good) global insns_self 3 insns_total 3 stack")
+__msg("processed 12 insns")
+#else
+__msg("subprog {{[0-9]+}} (global_good) global insns_self 5 insns_total 5 stack")
+__msg("processed 14 insns")
+#endif
int chained_global_func_calls_success(void)
{
int sum = 0;
diff --git a/tools/testing/selftests/bpf/progs/verifier_gotox.c b/tools/testing/selftests/bpf/progs/verifier_gotox.c
index 5b18c9a27717..0e27c2c79c57 100644
--- a/tools/testing/selftests/bpf/progs/verifier_gotox.c
+++ b/tools/testing/selftests/bpf/progs/verifier_gotox.c
@@ -47,6 +47,54 @@ DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_src_reg, BPF_REG_1, 0, 0, __fa
DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_off, BPF_REG_0, 1, 0, __failure __msg("BPF_JA|BPF_X uses reserved fields"))
DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_imm, BPF_REG_0, 0, 1, __failure __msg("BPF_JA|BPF_X uses reserved fields"))
+#define DEFINE_TERMINAL_GOTOX_PROG(NAME, BASE) \
+ __naked void NAME(void) \
+ { \
+ asm volatile (" \
+ .pushsection .jumptables,\"\",@progbits; \
+jt0_%=: \
+ .quad ret0_%= - " BASE "; \
+ .size jt0_%=, 8; \
+ .global jt0_%=; \
+ .popsection; \
+ \
+ r0 = jt0_%= ll; \
+ r0 = *(u64 *)(r0 + 0); \
+ goto end_%=; \
+ret0_%=: \
+ r0 = 0; \
+ exit; \
+end_%=: \
+ .8byte %[gotox_r0]; \
+" : \
+ : __imm_insn(gotox_r0, BPF_RAW_INSN(BPF_JMP | BPF_JA | BPF_X, \
+ BPF_REG_0, 0, 0, 0)) \
+ : __clobber_all); \
+ }
+
+SEC("socket")
+__success __retval(0)
+DEFINE_TERMINAL_GOTOX_PROG(jump_table_terminal_gotox, "socket")
+
+static __noinline __used
+DEFINE_TERMINAL_GOTOX_PROG(terminal_gotox_subprog1, ".text")
+
+static __noinline __used int terminal_gotox_subprog2(void)
+{
+ return 0;
+}
+
+SEC("socket")
+__success __retval(0)
+__naked void jump_table_terminal_gotox_subprog(void)
+{
+ asm volatile (" \
+ call terminal_gotox_subprog1; \
+ call terminal_gotox_subprog2; \
+ exit; \
+" ::: __clobber_all);
+}
+
/*
* Gotox is forbidden when there is no jump table loaded
* which points to the sub-function where the gotox is used
diff --git a/tools/testing/selftests/bpf/progs/verifier_iterating_callbacks.c b/tools/testing/selftests/bpf/progs/verifier_iterating_callbacks.c
index 75dd922e4e9f..1fbcc5228306 100644
--- a/tools/testing/selftests/bpf/progs/verifier_iterating_callbacks.c
+++ b/tools/testing/selftests/bpf/progs/verifier_iterating_callbacks.c
@@ -169,6 +169,23 @@ static int iter_limit_cb(__u32 idx, struct num_context *ctx)
}
SEC("?raw_tp")
+__failure __msg("R1 type=ctx expected=scalar")
+__naked void bpf_loop_reject_pointer(void)
+{
+ asm volatile (
+ "r2 = %[iter_limit_cb];"
+ "r3 = 0;"
+ "r4 = 0;"
+ "call %[bpf_loop];"
+ "exit;"
+ :
+ : __imm_ptr(iter_limit_cb),
+ __imm(bpf_loop)
+ : __clobber_common
+ );
+}
+
+SEC("?raw_tp")
__success
int bpf_loop_iter_limit_ok(void *unused)
{
diff --git a/tools/testing/selftests/bpf/progs/verifier_jeq_infer_not_null.c b/tools/testing/selftests/bpf/progs/verifier_jeq_infer_not_null.c
index 3d1e8de4390c..3c789c565b18 100644
--- a/tools/testing/selftests/bpf/progs/verifier_jeq_infer_not_null.c
+++ b/tools/testing/selftests/bpf/progs/verifier_jeq_infer_not_null.c
@@ -3,7 +3,9 @@
#include <linux/bpf.h>
#include <bpf/bpf_helpers.h>
+#include <stdbool.h>
#include "bpf_misc.h"
+#include "bpf_kfuncs.h"
struct {
__uint(type, BPF_MAP_TYPE_XSKMAP);
@@ -12,6 +14,13 @@ struct {
__type(value, int);
} map_xskmap SEC(".maps");
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(max_entries, 1);
+ __type(key, int);
+ __type(value, int);
+} map_hash SEC(".maps");
+
/* This is equivalent to the following program:
*
* r6 = skb->sk;
@@ -264,4 +273,188 @@ __naked void jne_reg_reg_null_check(void)
: __clobber_all);
}
+/*
+ * A comparison between PTR_TO_MEM | MEM_RDONLY | PTR_UNTRUSTED and
+ * PTR_TO_MAP_VALUE_OR_NULL should not infer that map pointer is not null.
+ * A bug in check_cond_jmp_op() made such inference possible.
+ */
+SEC("raw_tp")
+__failure
+__msg("error: invalid dereference of R0 (a nullable map value pointer)")
+__msg(">>> 11 | (61) r0 = *(u32 *)(r0 +0)")
+__naked void untrusted_mem_does_not_infer_map_value_non_null(void)
+{
+ asm volatile (" \
+ /* r6 = bpf_rdonly_cast(0, 0); */ \
+ r1 = 0; \
+ r2 = 0; \
+ call %[bpf_rdonly_cast]; \
+ r6 = r0; \
+ /* r0 = bpf_map_lookup_elem(map_hash, &key); */ \
+ *(u64 *)(r10 - 8) = 0; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ /* \
+ * buggy verifier assumed that r6 can't be null \
+ * and marked r0 non-null as well. \
+ */ \
+ if r6 != r0 goto 1f; \
+ r0 = *(u32 *)(r0 + 0); \
+1: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_rdonly_cast),
+ __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash)
+ : __clobber_all);
+}
+
+/*
+ * A pointer with an offset that is not bounded from above may be null at
+ * runtime, hence it is not a witness for the pointer it is compared with.
+ */
+SEC("socket")
+__failure
+__msg("error: invalid dereference of R7 (a nullable map value pointer)")
+__naked void unbounded_offset_does_not_infer_map_value_non_null(void)
+{
+ asm volatile (" \
+ /* r6 = bpf_map_lookup_elem(map_hash, &0); */ \
+ *(u64 *)(r10 - 8) = 0; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto 1f; \
+ r6 = r0; \
+ /* r7 = bpf_map_lookup_elem(map_hash, &1); */ \
+ *(u64 *)(r10 - 8) = 1; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ r7 = r0; \
+ /* pointer - pointer is an unknown scalar */ \
+ r8 = r7; \
+ r8 -= r6; \
+ /* r8 is in [0, S64_MAX] */ \
+ r8 <<= 1; \
+ r8 >>= 1; \
+ /* r6 may wrap to zero at runtime */ \
+ r6 += r8; \
+ if r7 != r6 goto 1f; \
+ r0 = *(u8 *)(r7 + 0); \
+1: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash)
+ : __clobber_all);
+}
+
+/* Same, but the offset is bounded, so the inference is still done. */
+SEC("socket")
+__success
+__naked void bounded_offset_infers_map_value_non_null(void)
+{
+ asm volatile (" \
+ /* r6 = bpf_map_lookup_elem(map_hash, &0); */ \
+ *(u64 *)(r10 - 8) = 0; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto 1f; \
+ r6 = r0; \
+ /* r7 = bpf_map_lookup_elem(map_hash, &1); */ \
+ *(u64 *)(r10 - 8) = 1; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ r7 = r0; \
+ /* pointer - pointer is an unknown scalar */ \
+ r8 = r7; \
+ r8 -= r6; \
+ /* r8 is in [0, 3] */ \
+ r8 &= 3; \
+ r6 += r8; \
+ if r7 != r6 goto 1f; \
+ r0 = *(u8 *)(r7 + 0); \
+1: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash)
+ : __clobber_all);
+}
+
+/*
+ * The low 32 bits of a map value pointer may be zero, hence a 32-bit
+ * compare with zero cannot be predicted from the pointer being non-NULL
+ * and both successors of such a jump have to be verified.
+ */
+SEC("socket")
+__failure __msg("invalid access to map value, value_size=4 off=32 size=4")
+__naked void jmp32_ptr_vs_zero_jne(void)
+{
+ asm volatile (" \
+ /* r0 = bpf_map_lookup_elem(map_hash, &key); */ \
+ *(u64 *)(r10 - 8) = 0; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto 1f; \
+ if w0 != 0 goto 1f; \
+ r0 = *(u32 *)(r0 + 32); \
+1: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash)
+ : __clobber_all);
+}
+
+/*
+ * The below program is explored in two paths: r6 == 0 and r6 == 1.
+ * On the first path comparison "if r0 == r6 goto 2f" should mark r6 as precise,
+ * otherwise unsafe path with r6 == 1 would be incorrectly pruned.
+ */
+SEC("socket")
+__failure
+__flag(BPF_F_TEST_STATE_FREQ)
+__msg("error: invalid dereference of R0 (a nullable map value pointer)")
+__naked void imprecise_zero_does_not_infer_map_value_non_null(void)
+{
+ asm volatile (" \
+ call %[bpf_get_prandom_u32]; \
+ /* r6 is 0 on the path explored first, 1 on the other */\
+ r6 = 1; \
+ if r0 == 0 goto 1f; \
+ r6 = 0; \
+ /* r0 = bpf_map_lookup_elem(map_hash, &0); */ \
+1: *(u64 *)(r10 - 8) = 0; \
+ r1 = %[map_hash] ll; \
+ r2 = r10; \
+ r2 += -8; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == r6 goto 2f; \
+ r0 = *(u8 *)(r0 + 0); \
+2: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash)
+ : __clobber_all);
+}
+
+void kfunc_root(void)
+{
+ bpf_rdonly_cast(0, 0);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c b/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c
new file mode 100644
index 000000000000..76c39ef30e96
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c
@@ -0,0 +1,75 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_misc.h"
+
+void *user_ptr;
+char dynptr_buf[8];
+u64 kaddr;
+
+extern struct kmem_cache *bpf_get_kmem_cache(u64 addr) __ksym;
+
+SEC("socket")
+__success
+__caps_unpriv(CAP_BPF)
+__failure_unpriv
+__msg_unpriv("bpf_rdonly_cast is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
+int rdonly_cast_noperfmon(void *ctx)
+{
+ char *p = bpf_rdonly_cast(0, 0);
+
+ return p[0x7fff];
+}
+
+SEC("socket")
+__success
+__caps_unpriv(CAP_BPF)
+__failure_unpriv
+__msg_unpriv("bpf_probe_read_kernel_dynptr is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
+int probe_read_kernel_dynptr_noperfmon(void *ctx)
+{
+ struct bpf_dynptr dptr;
+
+ bpf_dynptr_from_mem(dynptr_buf, sizeof(dynptr_buf), 0, &dptr);
+ bpf_probe_read_kernel_dynptr(&dptr, 0, sizeof(dynptr_buf), user_ptr);
+ return 0;
+}
+
+SEC("socket")
+__success
+__caps_unpriv(CAP_BPF)
+__failure_unpriv
+__msg_unpriv("bpf_stream_vprintk is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
+int stream_vprintk_noperfmon(void *ctx)
+{
+ bpf_stream_printk(BPF_STDOUT, "%pB", (void *)kaddr);
+ return 0;
+}
+
+SEC("socket")
+__success
+__caps_unpriv(CAP_BPF)
+__failure_unpriv
+__msg_unpriv("bpf_get_kmem_cache is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
+int get_kmem_cache_noperfmon(void *ctx)
+{
+ return !!bpf_get_kmem_cache(kaddr);
+}
+
+__weak int subprog_untrusted_read(void *p __arg_untrusted)
+{
+ return *(char *)p;
+}
+
+SEC("socket")
+__success
+__caps_unpriv(CAP_BPF)
+__failure_unpriv
+__msg_unpriv("rdonly_untrusted_mem access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
+int arg_untrusted_read_noperfmon(void *ctx)
+{
+ return subprog_untrusted_read(0);
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_ld_ind.c b/tools/testing/selftests/bpf/progs/verifier_ld_ind.c
index 09e81b99eecb..32989f981fb6 100644
--- a/tools/testing/selftests/bpf/progs/verifier_ld_ind.c
+++ b/tools/testing/selftests/bpf/progs/verifier_ld_ind.c
@@ -194,6 +194,102 @@ __naked void ld_ind_subprog_both_paths_safe(void)
::: __clobber_all);
}
+__naked __noinline __used
+static int ld_abs_callback(void)
+{
+ asm volatile (
+ "r6 = *(u64 *)(r2 + 0);"
+ ".8byte %[ld_abs];"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm_insn(ld_abs, BPF_LD_ABS(BPF_W, 0))
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("ld_abs: reject in callback")
+__failure __msg("cannot use BPF_LD_[ABS|IND] within callback")
+int ld_abs_callback_reject(struct __sk_buff *skb)
+{
+ bpf_loop(1, ld_abs_callback, &skb, 0);
+ return 0;
+}
+
+__naked __noinline __used
+static int ld_ind_callback_subprog(void)
+{
+ asm volatile (
+ "r6 = r1;"
+ "r7 = 0;"
+ ".8byte %[ld_ind];"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm_insn(ld_ind, BPF_LD_IND(BPF_W, BPF_REG_7, 0))
+ : __clobber_all);
+}
+
+__naked __noinline __used
+static int ld_ind_callback(void)
+{
+ asm volatile (
+ "r1 = *(u64 *)(r2 + 0);"
+ "call ld_ind_callback_subprog;"
+ "exit;"
+ ::: __clobber_all);
+}
+
+SEC("socket")
+__description("ld_ind: reject in callback subprog")
+__failure __msg("cannot use BPF_LD_[ABS|IND] within callback")
+int ld_ind_callback_subprog_reject(struct __sk_buff *skb)
+{
+ bpf_loop(1, ld_ind_callback, &skb, 0);
+ return 0;
+}
+
+static __noinline int ld_ind_global_static(struct __sk_buff *skb)
+{
+ asm volatile (
+ "r6 = %[skb];"
+ "r7 = 0;"
+ ".8byte %[ld_ind];"
+ :
+ : [skb] "r"(skb),
+ __imm_insn(ld_ind, BPF_LD_IND(BPF_W, BPF_REG_7, 0))
+ : __clobber_common, "r6", "r7");
+ return skb->mark;
+}
+
+__noinline int ld_ind_global(struct __sk_buff *skb)
+{
+ return ld_ind_global_static(skb);
+}
+
+static int ld_ind_global_callback(__u32 index, struct __sk_buff **ctx)
+{
+ ld_ind_global(*ctx);
+ return 0;
+}
+
+SEC("socket")
+__description("ld_ind: reject in callback global subprog")
+__failure __msg("cannot use BPF_LD_[ABS|IND] within callback")
+int ld_ind_global_callback_reject(struct __sk_buff *skb)
+{
+ bpf_loop(1, ld_ind_global_callback, &skb, 0);
+ return 0;
+}
+
+SEC("socket")
+__description("ld_ind: allow in non-callback global subprog")
+__success
+int ld_ind_global_subprog_ok(struct __sk_buff *skb)
+{
+ return ld_ind_global(skb);
+}
+
/*
* ld_{abs,ind} in subprogs require scalar (int) return type in BTF.
* A test with void return must be rejected.
diff --git a/tools/testing/selftests/bpf/progs/verifier_loops1.c b/tools/testing/selftests/bpf/progs/verifier_loops1.c
index d248ce877f14..48a966cda199 100644
--- a/tools/testing/selftests/bpf/progs/verifier_loops1.c
+++ b/tools/testing/selftests/bpf/progs/verifier_loops1.c
@@ -303,4 +303,40 @@ __naked void maybe_exit_scc_bug1(void)
::: __clobber_all);
}
+/*
+ * The loop reads zero from the caller's stack on its first iteration and
+ * one from the callee's stack on its second iteration. At the loop header,
+ * only the frame number of the pointer in r1 changes.
+ */
+static __naked __noinline __used
+void loop_stack_frames_reg(void)
+{
+ asm volatile (
+ "*(u64 *)(r10 - 8) = 1;"
+"1:"
+ "r0 = *(u64 *)(r1 + 0);"
+ "if r0 != 0 goto 2f;"
+ "r1 = r10;"
+ "r1 += -8;"
+ "goto 1b;"
+"2:"
+ "exit;"
+ ::: __clobber_all);
+}
+
+SEC("xdp")
+__description("bounded loop changing stack frame in a register")
+__success __retval(1)
+__flag(BPF_F_TEST_STATE_FREQ)
+__naked void bounded_loop_stack_frames_reg(void)
+{
+ asm volatile (
+ "*(u64 *)(r10 - 8) = 0;"
+ "r1 = r10;"
+ "r1 += -8;"
+ "call loop_stack_frames_reg;"
+ "exit;"
+ ::: __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_map_in_map.c b/tools/testing/selftests/bpf/progs/verifier_map_in_map.c
index 7918646e5bfc..d3be69a9a755 100644
--- a/tools/testing/selftests/bpf/progs/verifier_map_in_map.c
+++ b/tools/testing/selftests/bpf/progs/verifier_map_in_map.c
@@ -155,6 +155,7 @@ l0_%=: r0 = 0; \
SEC("socket")
__description("forgot null checking on the inner map pointer")
__failure __msg("R1 type=map_ptr_or_null expected=map_ptr")
+__msg("map_ptr_or_null, but this argument accepts map_ptr")
__failure_unpriv
__naked void on_the_inner_map_pointer(void)
{
diff --git a/tools/testing/selftests/bpf/progs/verifier_may_goto_1.c b/tools/testing/selftests/bpf/progs/verifier_may_goto_1.c
index 0e211f030d0d..db7e30da234f 100644
--- a/tools/testing/selftests/bpf/progs/verifier_may_goto_1.c
+++ b/tools/testing/selftests/bpf/progs/verifier_may_goto_1.c
@@ -12,6 +12,7 @@ __arch_x86_64
__arch_s390x
__arch_arm64
__arch_riscv64
+__arch_loongarch
__xlated("0: r0 = 1")
__xlated("1: exit")
__success
@@ -33,6 +34,7 @@ __arch_x86_64
__arch_s390x
__arch_arm64
__arch_riscv64
+__arch_loongarch
__xlated("0: r0 = 1")
__xlated("1: exit")
__success
@@ -56,6 +58,7 @@ __arch_x86_64
__arch_s390x
__arch_arm64
__arch_riscv64
+__arch_loongarch
__xlated("0: r0 = 1")
__xlated("1: exit")
__success
@@ -83,6 +86,7 @@ __arch_x86_64
__arch_s390x
__arch_arm64
__arch_riscv64
+__arch_loongarch
__xlated("0: *(u64 *)(r10 -16) = 65535")
__xlated("1: *(u64 *)(r10 -8) = 0")
__xlated("2: r12 = *(u64 *)(r10 -16)")
@@ -120,6 +124,7 @@ __arch_x86_64
__arch_s390x
__arch_arm64
__arch_riscv64
+__arch_loongarch
__success
__retval(0)
__naked void timed_may_goto_preserves_regs(void)
diff --git a/tools/testing/selftests/bpf/progs/verifier_netfilter_ctx.c b/tools/testing/selftests/bpf/progs/verifier_netfilter_ctx.c
index e2cbc5bda65e..b5d7f567d0d4 100644
--- a/tools/testing/selftests/bpf/progs/verifier_netfilter_ctx.c
+++ b/tools/testing/selftests/bpf/progs/verifier_netfilter_ctx.c
@@ -113,4 +113,82 @@ int with_valid_ctx_access_test6(struct bpf_nf_ctx *ctx)
return th->dest == bpf_htons(22) ? NF_ACCEPT : NF_DROP;
}
+SEC("netfilter")
+__description("netfilter test prog with skb write access")
+__failure __msg("only read is supported")
+int skb_len_write(struct bpf_nf_ctx *ctx)
+{
+ ctx->skb->len = 1;
+ return 1;
+}
+
+SEC("netfilter")
+__description("netfilter test prog with skb data write access")
+__failure __msg("cannot write into rdonly_untrusted_mem")
+int skb_data_write(struct bpf_nf_ctx *ctx)
+{
+ ctx->skb->data[0] = 0;
+ return 1;
+}
+
+SEC("netfilter")
+__description("netfilter test prog with bpf_dynptr_write")
+__success __failure_unpriv
+__retval(0)
+int with_dynptr_write(struct bpf_nf_ctx *ctx)
+{
+ struct __sk_buff *skb = (struct __sk_buff *)ctx->skb;
+ struct bpf_dynptr ptr;
+ u8 buffer[1] = {};
+
+ if (bpf_dynptr_from_skb(skb, 0, &ptr))
+ return 1;
+
+ if (bpf_dynptr_write(&ptr, 0, buffer, sizeof(buffer), 0))
+ return 0; /* must always fail */
+
+ return 1;
+}
+
+SEC("netfilter")
+__description("netfilter test prog with bpf_dynptr_slice_rdwr")
+__failure __msg("the prog does not allow writes to packet data")
+int with_dynptr_rdwr(struct bpf_nf_ctx *ctx)
+{
+ struct __sk_buff *skb = (struct __sk_buff *)ctx->skb;
+ u8 buffer_iph[20] = {};
+ struct bpf_dynptr ptr;
+ struct iphdr *iph;
+
+ if (bpf_dynptr_from_skb(skb, 0, &ptr))
+ return 1;
+
+ iph = bpf_dynptr_slice_rdwr(&ptr, 0, buffer_iph, sizeof(buffer_iph));
+ if (!iph)
+ return 0;
+
+ return 1;
+}
+
+SEC("netfilter")
+__description("netfilter test prog with bpf_dynptr_slice + write")
+__failure __msg("cannot write into rdonly_mem")
+int with_dynptr_store(struct bpf_nf_ctx *ctx)
+{
+ struct __sk_buff *skb = (struct __sk_buff *)ctx->skb;
+ u8 buffer_iph[20] = {};
+ struct bpf_dynptr ptr;
+ struct iphdr *iph;
+
+ if (bpf_dynptr_from_skb(skb, 0, &ptr))
+ return 1;
+
+ iph = bpf_dynptr_slice(&ptr, 0, buffer_iph, sizeof(buffer_iph));
+ if (!iph)
+ return 0;
+ iph->protocol = 42;
+
+ return 1;
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_percpu_addr.c b/tools/testing/selftests/bpf/progs/verifier_percpu_addr.c
new file mode 100644
index 000000000000..967f4e6e3a49
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/verifier_percpu_addr.c
@@ -0,0 +1,72 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_misc.h"
+
+#if defined(__TARGET_ARCH_x86)
+
+int percpu_data SEC(".percpu");
+
+/*
+ * An ld_imm64 of a per-CPU map value is followed by a mov_percpu_addr that
+ * reuses the same register, so check that the add resolves into the register
+ * the address was loaded into, for every register.
+ */
+SEC("raw_tp")
+__description("per-CPU address resolution")
+__success
+__arch_x86_64
+__jited(" movabsq $0x{{.*}}, %rax")
+__jited(" addq %gs:{{.*}}, %rax")
+__jited(" movabsq $0x{{.*}}, %rdi")
+__jited(" addq %gs:{{.*}}, %rdi")
+__jited(" movabsq $0x{{.*}}, %rsi")
+__jited(" addq %gs:{{.*}}, %rsi")
+__jited(" movabsq $0x{{.*}}, %rdx")
+__jited(" addq %gs:{{.*}}, %rdx")
+__jited(" movabsq $0x{{.*}}, %rcx")
+__jited(" addq %gs:{{.*}}, %rcx")
+__jited(" movabsq $0x{{.*}}, %r8")
+__jited(" addq %gs:{{.*}}, %r8")
+__jited(" movabsq $0x{{.*}}, %rbx")
+__jited(" addq %gs:{{.*}}, %rbx")
+__jited(" movabsq $0x{{.*}}, %r13")
+__jited(" addq %gs:{{.*}}, %r13")
+__jited(" movabsq $0x{{.*}}, %r14")
+__jited(" addq %gs:{{.*}}, %r14")
+__jited(" movabsq $0x{{.*}}, %r15")
+__jited(" addq %gs:{{.*}}, %r15")
+__naked void percpu_addr(void)
+{
+ asm volatile (" \
+ r0 = %[percpu_data] ll; \
+ r1 = %[percpu_data] ll; \
+ r2 = %[percpu_data] ll; \
+ r3 = %[percpu_data] ll; \
+ r4 = %[percpu_data] ll; \
+ r5 = %[percpu_data] ll; \
+ r6 = %[percpu_data] ll; \
+ r7 = %[percpu_data] ll; \
+ r8 = %[percpu_data] ll; \
+ r9 = %[percpu_data] ll; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_addr(percpu_data)
+ : __clobber_all);
+}
+
+#else
+
+SEC("raw_tp")
+__description("percpu addr dummy")
+__success
+int dummy_test(void)
+{
+ return 0;
+}
+
+#endif
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_precision.c b/tools/testing/selftests/bpf/progs/verifier_precision.c
index 6f325876efdd..f4459561bf39 100644
--- a/tools/testing/selftests/bpf/progs/verifier_precision.c
+++ b/tools/testing/selftests/bpf/progs/verifier_precision.c
@@ -2,8 +2,10 @@
/* Copyright (C) 2023 SUSE LLC */
#include <linux/bpf.h>
#include <bpf/bpf_helpers.h>
+#include <stdbool.h>
#include "../../../include/linux/filter.h"
#include "bpf_misc.h"
+#include "bpf_kfuncs.h"
struct {
__uint(type, BPF_MAP_TYPE_ARRAY);
@@ -642,4 +644,102 @@ __naked int bpf_atomic_cmpxchg_32bit_precision(void)
: __clobber_all);
}
+/*
+ * Verification takes two paths: with r1 being scalar zero on path (1)
+ * and with r1 being some other scalar on path (2).
+ * Check that the verifier does not use checkpoints created
+ * on path (1) to prune path (2).
+ */
+SEC("?tc")
+__flag(BPF_F_TEST_STATE_FREQ)
+__failure __msg("R1 type=scalar expected=fp")
+__naked int null_mem_arg_zero_size(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 42;"
+ "if r0 > 42 goto 1f;"
+ "r1 = 0;"
+ "1:"
+ "r2 = 0;"
+ "r3 = 0;"
+ "r4 = 0;"
+ "r5 = 0;"
+ /*
+ * ARG_PTR_TO_MEM | PTR_MAYBE_NULL parameter can be NULL,
+ * but can't be some other scalar value.
+ */
+ "call %[bpf_csum_diff];"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm(bpf_get_prandom_u32),
+ __imm(bpf_csum_diff)
+ : __clobber_all);
+}
+
+__weak int subprog_mem_arg(int *p)
+{
+ if (p)
+ return *p;
+ return 0;
+}
+
+/*
+ * Verification takes two paths: with r1 being scalar zero on path (1)
+ * and with r1 being some other scalar on path (2).
+ * Check that the verifier does not use checkpoints created
+ * on path (1) to prune path (2).
+ */
+SEC("?raw_tp")
+__flag(BPF_F_TEST_STATE_FREQ)
+__failure __msg("R1 type=scalar expected=fp")
+__naked int null_mem_arg_global_subprog(void)
+{
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r1 = 42;"
+ "if r0 > 42 goto 1f;"
+ "r1 = 0;"
+ "1:"
+ "call subprog_mem_arg;"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+/* Same as above, check that path with r3 == 0 does not prune the path with r3 != 0 */
+SEC("?tc")
+__flag(BPF_F_TEST_STATE_FREQ)
+__failure __msg("R3 type=scalar expected=fp")
+int null_kfunc_arg_dynptr_slice(struct __sk_buff *skb)
+{
+ struct bpf_dynptr ptr;
+
+ bpf_dynptr_from_skb(skb, 0, &ptr);
+ asm volatile (
+ "call %[bpf_get_prandom_u32];"
+ "r3 = 42;"
+ "if r0 > 42 goto 1f;"
+ "r3 = 0;"
+ "1:"
+ "r1 = %[ptr];"
+ "r2 = 0;"
+ "r4 = 8;"
+ "call %[bpf_dynptr_slice];"
+ :
+ : __imm_ptr(ptr),
+ __imm(bpf_get_prandom_u32),
+ __imm(bpf_dynptr_slice)
+ : __clobber_common);
+ return 0;
+}
+
+void __kfunc_btf_root(void)
+{
+ bpf_dynptr_slice(0, 0, 0, 0);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_private_stack.c b/tools/testing/selftests/bpf/progs/verifier_private_stack.c
index bb8206e10880..ea0a7e73331d 100644
--- a/tools/testing/selftests/bpf/progs/verifier_private_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_private_stack.c
@@ -86,7 +86,9 @@ __naked static void cumulative_stack_depth_subprog(void)
SEC("kprobe")
__description("Private stack, subtree > MAX_BPF_STACK")
__success
-__log_level(4) __msg("stack depth 512+32 max 512")
+__log_level(4) __msg("stack depth max 512")
+__msg("subprog 0 (private_stack_nested_1) main {{.*}} stack 512")
+__msg("subprog 1 (cumulative_stack_depth_subprog) static {{.*}} stack 32")
__arch_x86_64
/* private stack fp for the main prog */
__jited(" movabsq $0x{{.*}}, %r9")
@@ -331,7 +333,11 @@ SEC("fentry/bpf_fentry_test9")
__description("Private stack, async callback, potential nesting")
__success __retval(0)
__load_if_JITed()
-__log_level(4) __msg("stack depth 8+0+256+0 max 272")
+__log_level(4) __msg("stack depth max 272")
+__msg("subprog 0 (private_stack_async_callback_2) main {{.*}} stack 8")
+__msg("subprog 1 (timer_cb1) static {{.*}} stack 0")
+__msg("subprog 2 (subprog1) static {{.*}} stack 256")
+__msg("subprog 3 (subprog2) static {{.*}} stack 0")
__arch_x86_64
__jited(" subq $0x100, %rsp")
__arch_arm64
@@ -355,7 +361,10 @@ int private_stack_async_callback_2(void)
SEC("fentry/bpf_fentry_test9")
__description("private stack, max stack depth is private stack")
__success
-__log_level(4) __msg("stack depth 8+256+0 max 256")
+__log_level(4) __msg("stack depth max 256")
+__msg("subprog 0 (private_stack_max_depth) main {{.*}} stack 8")
+__msg("subprog 1 (subprog1) static insns_self {{[0-9]+}} insns_total {{[0-9]+}} stack 256")
+__msg("subprog 2 (subprog2) static insns_self {{[0-9]+}} insns_total {{[0-9]+}} stack 0")
int private_stack_max_depth(void)
{
int x = 0;
diff --git a/tools/testing/selftests/bpf/progs/verifier_scalar_ids.c b/tools/testing/selftests/bpf/progs/verifier_scalar_ids.c
index 663d15fc5fd2..256547048cc4 100644
--- a/tools/testing/selftests/bpf/progs/verifier_scalar_ids.c
+++ b/tools/testing/selftests/bpf/progs/verifier_scalar_ids.c
@@ -380,13 +380,14 @@ SEC("socket")
__success __log_level(2)
__flag(BPF_F_TEST_STATE_FREQ)
/*
- * check that r0 and r5 have different IDs after 'if',
- * collect_linked_regs() can't tie more than 5 registers for a single insn.
+ * check that r5 is unlinked after 'if', collect_linked_regs() can't tie
+ * more than 5 registers for a single insn and the register compared by
+ * the jump is not exempt from that.
*/
-__msg("7: (25) if r0 > 0x7 goto pc+0 ; R0=scalar(id=1")
+__msg("7: (25) if r5 > 0x7 goto pc+0 ; R5=scalar(smin=")
__msg("12: (bf) r5 = r5 ; R5=scalar(id=2")
/* check that r{0-4} are marked precise after 'if' */
-__msg("frame0: regs=r0 stack= before 7: (25) if r0 > 0x7 goto pc+0")
+__msg("frame0: regs=r0 stack= before 7: (25) if r5 > 0x7 goto pc+0")
__msg("frame0: parent state regs=r0,r1,r2,r3,r4 stack=:")
__naked void linked_regs_too_many_regs(void)
{
@@ -400,8 +401,8 @@ __naked void linked_regs_too_many_regs(void)
"r3 = r0;"
"r4 = r0;"
"r5 = r0;"
- /* propagate range for r{0-5} */
- "if r0 > 7 goto +0;"
+ /* r{0-4} fill the record, r5 does not fit and is unlinked */
+ "if r5 > 7 goto +0;"
/* keep r{1-4} live */
"r1 = r1;"
"r2 = r2;"
diff --git a/tools/testing/selftests/bpf/progs/verifier_sock.c b/tools/testing/selftests/bpf/progs/verifier_sock.c
index 4f2f3209eec8..bf9f6fb6582c 100644
--- a/tools/testing/selftests/bpf/progs/verifier_sock.c
+++ b/tools/testing/selftests/bpf/progs/verifier_sock.c
@@ -88,6 +88,44 @@ l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_family]); \
: __clobber_all);
}
+SEC("socket")
+__description("skb->sk: sk->rx_queue_mapping [no sign extension]")
+__success __success_unpriv __retval(0)
+__naked void sk_rx_queue_mapping_no_sign_ext(void)
+{
+ asm volatile (" \
+ r1 = *(u64*)(r1 + %[__sk_buff_sk]); \
+ if r1 != 0 goto l0_%=; \
+ r0 = 0xdead; \
+ exit; \
+l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_rx_queue_mapping]); \
+ r0 >>= 32; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)),
+ __imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping))
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("skb->sk: sk->rx_queue_mapping [narrow load mask]")
+__success __success_unpriv __retval(0)
+__naked void sk_rx_queue_mapping_narrow_load_mask(void)
+{
+ asm volatile (" \
+ r1 = *(u64*)(r1 + %[__sk_buff_sk]); \
+ if r1 != 0 goto l0_%=; \
+ r0 = 0xdead; \
+ exit; \
+l0_%=: r0 = *(u16*)(r1 + %[bpf_sock_rx_queue_mapping]); \
+ r0 >>= 16; \
+ exit; \
+" :
+ : __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)),
+ __imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping))
+ : __clobber_all);
+}
+
SEC("cgroup/skb")
__description("skb->sk: sk->type [fullsock field]")
__failure __msg("invalid sock_common access")
diff --git a/tools/testing/selftests/bpf/progs/verifier_spill_fill.c b/tools/testing/selftests/bpf/progs/verifier_spill_fill.c
index 8b166c42c4e0..39a1766dae3f 100644
--- a/tools/testing/selftests/bpf/progs/verifier_spill_fill.c
+++ b/tools/testing/selftests/bpf/progs/verifier_spill_fill.c
@@ -1403,6 +1403,46 @@ __naked void partial_fill_from_cleaned_pointer_spill(void)
::: __clobber_all);
}
+SEC("raw_tp")
+__failure
+__msg("access may be outside object bounds")
+__flag(BPF_F_TEST_STATE_FREQ)
+__naked void imprecise_scalar_spill_half_dead(void)
+{
+ asm volatile (
+ /*
+ * Fork two paths: the one explored first spills an imprecise zero,
+ * the one explored second, an imprecise non-zero scalar.
+ */
+ "call %[bpf_get_prandom_u32];"
+ "if r0 > 42 goto 1f;"
+ "r6 = 0;"
+ "goto 2f;"
+"1:"
+ /* causes out of bounds access on a second path. */
+ "r6 = 100500;"
+"2:"
+ /* Force a checkpoint before the spill. */
+ "goto +0;"
+ "*(u64 *)(r10 - 8) = r6;"
+ /*
+ * Force stack cleanup, only the low half of the spill is alive,
+ * so the dead high half is degraded to raw stack bytes.
+ * Buggy verifier converted it to STACK_ZERO w/o proper precision propagation.
+ */
+ "goto +0;"
+ "r7 = *(u32 *)(r10 - 4);"
+ /* Use r7 as an offset into a one-byte buffer. */
+ "r1 = %[single_byte_buf] ll;"
+ "r1 += r7;"
+ "r0 = *(u8 *)(r1 + 0);"
+ "exit;"
+:
+: __imm(bpf_get_prandom_u32),
+ __imm_addr(single_byte_buf)
+: __clobber_all);
+}
+
/* check valid spill/fill, ptr to tp buffer */
SEC("raw_tracepoint.w")
__success
diff --git a/tools/testing/selftests/bpf/progs/verifier_subprog_insn_stats.c b/tools/testing/selftests/bpf/progs/verifier_subprog_insn_stats.c
new file mode 100644
index 000000000000..8f6082fdb5c8
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/verifier_subprog_insn_stats.c
@@ -0,0 +1,223 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_misc.h"
+
+struct timer_value {
+ struct bpf_timer timer;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARRAY);
+ __uint(max_entries, 1);
+ __type(key, __u32);
+ __type(value, struct timer_value);
+} timer_map SEC(".maps");
+
+SEC("?raw_tp")
+__success __log_level(4)
+__msg("subprog 0 (stats_main_only) main insns_self 2 insns_total 2 stack 0")
+__msg("processed 2 insns")
+__naked int stats_main_only(void)
+{
+ asm volatile (
+ "r0 = 0;"
+ "exit;"
+ );
+}
+
+__naked __noinline __used
+static int stats_chain_leaf(void)
+{
+ asm volatile (
+ "r0 = 0;"
+ "exit;"
+ );
+}
+
+__naked __noinline __used
+static int stats_chain_parent(void)
+{
+ asm volatile (
+ "call stats_chain_leaf;"
+ "exit;"
+ );
+}
+
+SEC("?raw_tp")
+__success __log_level(4)
+/*
+ * self: 2 + 2 + 2 = 6
+ * totals: leaf 2, parent 2 + 2 = 4, main 2 + 4 = 6
+ */
+__msg("subprog 0 (stats_static_chain) main insns_self 2 insns_total 6 stack 0")
+__msg("subprog {{[0-9]+}} (stats_chain_parent) static insns_self 2 insns_total 4 stack 0")
+__msg("subprog {{[0-9]+}} (stats_chain_leaf) static insns_self 2 insns_total 2 stack 0")
+__msg("processed 6 insns")
+__naked int stats_static_chain(void)
+{
+ asm volatile (
+ "call stats_chain_parent;"
+ "exit;"
+ );
+}
+
+__naked __noinline __used
+static int stats_shared_leaf(void)
+{
+ asm volatile (
+ "r0 = 0;"
+ "exit;"
+ );
+}
+
+__naked __noinline __used
+int stats_global_root(void)
+{
+ asm volatile (
+ "call stats_shared_leaf;"
+ "exit;"
+ );
+}
+
+SEC("?raw_tp")
+__success __log_level(4)
+/*
+ * stats_shared_leaf is explored once under each independent root.
+ * self: main 3 + leaf 4 + global 2 = 9
+ * root totals: main 5 + global 4 = 9
+ */
+__msg("subprog 0 (stats_shared_roots) main insns_self 3 insns_total 5 stack 0")
+__msg("subprog {{[0-9]+}} (stats_shared_leaf) static insns_self 4 insns_total 4 stack 0")
+__msg("subprog {{[0-9]+}} (stats_global_root) global insns_self 2 insns_total 4 stack 0")
+__msg("processed 9 insns")
+__naked int stats_shared_roots(void)
+{
+ asm volatile (
+ "call stats_shared_leaf;"
+ "call stats_global_root;"
+ "exit;"
+ );
+}
+
+__noinline __used
+static int stats_async_leaf(void *map, __u32 *key, struct bpf_timer *timer)
+{
+ return 0;
+}
+
+__noinline __used
+static __u64 stats_async_schedule(struct bpf_map *map, __u32 *key,
+ struct timer_value *value, void *ctx)
+{
+ asm volatile (
+ "r1 = %[timer];"
+ "r2 = %[stats_async_leaf];"
+ "call %[bpf_timer_set_callback];"
+ :
+ : [timer] "r" (value),
+ __imm_ptr(stats_async_leaf),
+ __imm(bpf_timer_set_callback)
+ : __clobber_common
+ );
+ return 0;
+}
+
+SEC("?raw_tp")
+__success __log_level(4)
+/*
+ * self: 9 + 7 + 2 = 18
+ * totals: leaf 2, scheduler 7, main root 18
+ */
+__msg("subprog 0 (stats_async_direct) main insns_self 9 insns_total 18 stack 0")
+__msg("subprog {{[0-9]+}} (stats_async_schedule) static insns_self 7 insns_total 7 stack 0")
+__msg("subprog {{[0-9]+}} (stats_async_leaf) static insns_self 2 insns_total 2 stack 0")
+__msg("processed 18 insns")
+__naked int stats_async_direct(void)
+{
+ asm volatile (
+ "r1 = %[timer_map] ll;"
+ "r2 = %[stats_async_schedule];"
+ "r3 = 0;"
+ "r4 = 0;"
+ "call %[bpf_for_each_map_elem];"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm_addr(timer_map),
+ __imm_ptr(stats_async_schedule),
+ __imm(bpf_for_each_map_elem)
+ : __clobber_common
+ );
+}
+
+__noinline __used
+static int stats_async_nested_leaf(void *map, __u32 *key, struct bpf_timer *timer)
+{
+ return 0;
+}
+
+__noinline __used
+static int stats_async_outer(void *map, __u32 *key, struct bpf_timer *timer)
+{
+ asm volatile (
+ "r1 = %[timer];"
+ "r2 = %[stats_async_nested_leaf];"
+ "call %[bpf_timer_set_callback];"
+ :
+ : [timer] "r" (timer),
+ __imm_ptr(stats_async_nested_leaf),
+ __imm(bpf_timer_set_callback)
+ : __clobber_common
+ );
+ return 0;
+}
+
+__noinline __used
+static __u64 stats_async_nested_schedule(struct bpf_map *map, __u32 *key,
+ struct timer_value *value, void *ctx)
+{
+ asm volatile (
+ "r1 = %[timer];"
+ "r2 = %[stats_async_outer];"
+ "call %[bpf_timer_set_callback];"
+ :
+ : [timer] "r" (value),
+ __imm_ptr(stats_async_outer),
+ __imm(bpf_timer_set_callback)
+ : __clobber_common
+ );
+ return 0;
+}
+
+SEC("?raw_tp")
+__success __log_level(4)
+/*
+ * self: 9 + 7 + 7 + 2 = 25
+ * totals: leaf 2, outer 7, scheduler 7, main root 25
+ */
+__msg("subprog 0 (stats_async_nested) main insns_self 9 insns_total 25 stack 0")
+__msg("subprog {{[0-9]+}} (stats_async_nested_schedule) static insns_self 7 insns_total 7 stack 0")
+__msg("subprog {{[0-9]+}} (stats_async_outer) static insns_self 7 insns_total 7 stack 0")
+__msg("subprog {{[0-9]+}} (stats_async_nested_leaf) static insns_self 2 insns_total 2 stack 0")
+__msg("processed 25 insns")
+__naked int stats_async_nested(void)
+{
+ asm volatile (
+ "r1 = %[timer_map] ll;"
+ "r2 = %[stats_async_nested_schedule];"
+ "r3 = 0;"
+ "r4 = 0;"
+ "call %[bpf_for_each_map_elem];"
+ "r0 = 0;"
+ "exit;"
+ :
+ : __imm_addr(timer_map),
+ __imm_ptr(stats_async_nested_schedule),
+ __imm(bpf_for_each_map_elem)
+ : __clobber_common
+ );
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_subprog_precision.c b/tools/testing/selftests/bpf/progs/verifier_subprog_precision.c
index d21d32f6a676..dc0c7034c04f 100644
--- a/tools/testing/selftests/bpf/progs/verifier_subprog_precision.c
+++ b/tools/testing/selftests/bpf/progs/verifier_subprog_precision.c
@@ -287,9 +287,9 @@ __msg("17: (b7) r0 = 0")
__msg("18: (95) exit")
__msg("returning from callee:")
__msg("to caller at 9:")
-__msg("frame 0: propagating r1,r4")
+__msg("frame 0: propagating r1,r3,r4")
__msg("mark_precise: frame0: last_idx 9 first_idx 9 subseq_idx -1")
-__msg("mark_precise: frame0: regs=r1,r4 stack= before 18: (95) exit")
+__msg("mark_precise: frame0: regs=r1,r3,r4 stack= before 18: (95) exit")
__msg("from 18 to 9: safe")
__naked int callback_result_precise(void)
{
@@ -419,9 +419,9 @@ __msg("to caller at 9:")
/* r1, r4 are always precise for bpf_loop(),
* r6 was marked before backtracking to callback body.
*/
-__msg("frame 0: propagating r1,r4,r6")
+__msg("frame 0: propagating r1,r3,r4,r6")
__msg("mark_precise: frame0: last_idx 9 first_idx 9 subseq_idx -1")
-__msg("mark_precise: frame0: regs=r1,r4,r6 stack= before 16: (95) exit")
+__msg("mark_precise: frame0: regs=r1,r3,r4,r6 stack= before 16: (95) exit")
__msg("mark_precise: frame1: regs= stack= before 15: (b7) r0 = 0")
__msg("mark_precise: frame1: regs= stack= before 9: (85) call bpf_loop")
__msg("mark_precise: frame0: parent state regs= stack=:")
@@ -575,9 +575,9 @@ __msg("to caller at 10:")
/* r1, r4 are always precise for bpf_loop(),
* fp-8 was marked before backtracking to callback body.
*/
-__msg("frame 0: propagating r1,r4,fp-8")
+__msg("frame 0: propagating r1,r3,r4,fp-8")
__msg("mark_precise: frame0: last_idx 10 first_idx 10 subseq_idx -1")
-__msg("mark_precise: frame0: regs=r1,r4 stack=-8 before 18: (95) exit")
+__msg("mark_precise: frame0: regs=r1,r3,r4 stack=-8 before 18: (95) exit")
__msg("mark_precise: frame1: regs= stack= before 17: (b7) r0 = 0")
__msg("mark_precise: frame1: regs= stack= before 10: (85) call bpf_loop#181")
__msg("mark_precise: frame0: parent state regs= stack=:")
@@ -846,4 +846,55 @@ __naked int subprog_result_tail_call(void)
);
}
+__naked __noinline __used
+static int ld_abs_subprog(void)
+{
+ asm volatile (
+ "r6 = r1;"
+ "r7 = r1;"
+ ".8byte %[ld_abs];"
+ "exit;"
+ :
+ : __imm_insn(ld_abs, BPF_LD_ABS(BPF_W, 0))
+ : __clobber_all);
+}
+
+/*
+ * Buggy verifier did not properly backtrack early subprogram exit
+ * modelled for BPF_LD | BPF_ABS instruction, causing a segfault.
+ */
+SEC("socket")
+__success
+__log_level(2)
+/* early exit path */
+__msg("3: (0f) r1 += r7")
+__msg("mark_precise: frame0: regs=r7 stack= before 2: (bf) r1 = r10")
+__msg("mark_precise: frame0: regs=r7 stack= before 9: (20) r0 = *(u32 *)skb[0]")
+__msg("mark_precise: frame1: regs= stack= before 8: (bf) r7 = r1")
+__msg("mark_precise: frame1: regs= stack= before 7: (bf) r6 = r1")
+__msg("mark_precise: frame1: regs= stack= before 1: (85) call pc+5")
+__msg("mark_precise: frame0: regs=r7 stack= before 0: (b7) r7 = -8")
+/* fallthrough path */
+__msg("3: (0f) r1 += r7")
+__msg("mark_precise: frame0: regs=r7 stack= before 2: (bf) r1 = r10")
+__msg("mark_precise: frame0: regs=r7 stack= before 10: (95) exit")
+__msg("mark_precise: frame1: regs= stack= before 9: (20) r0 = *(u32 *)skb[0]")
+__msg("mark_precise: frame1: regs= stack= before 8: (bf) r7 = r1")
+__msg("mark_precise: frame1: regs= stack= before 7: (bf) r6 = r1")
+__msg("mark_precise: frame1: regs= stack= before 1: (85) call pc+5")
+__msg("mark_precise: frame0: regs=r7 stack= before 0: (b7) r7 = -8")
+__naked int ld_abs_backtrack_both_paths(void)
+{
+ asm volatile (
+ "r7 = -8;"
+ "call ld_abs_subprog;"
+ "r1 = r10;"
+ "r1 += r7;" /* mark r7 as precise */
+ "*(u64 *)(r1 + 0) = 0;"
+ "r0 = 0;"
+ "exit;"
+ ::: __clobber_all
+ );
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_uninit.c b/tools/testing/selftests/bpf/progs/verifier_uninit.c
index 7718cd7d19ce..691018a46049 100644
--- a/tools/testing/selftests/bpf/progs/verifier_uninit.c
+++ b/tools/testing/selftests/bpf/progs/verifier_uninit.c
@@ -9,6 +9,7 @@
SEC("socket")
__description("read uninitialized register")
__failure __msg("R2 !read_ok")
+__msg("R2 has never been initialized on this path")
__failure_unpriv
__naked void read_uninitialized_register(void)
{
diff --git a/tools/testing/selftests/bpf/progs/verifier_unpriv.c b/tools/testing/selftests/bpf/progs/verifier_unpriv.c
index 42de5cff7e52..3069e70fbcbd 100644
--- a/tools/testing/selftests/bpf/progs/verifier_unpriv.c
+++ b/tools/testing/selftests/bpf/progs/verifier_unpriv.c
@@ -96,6 +96,24 @@ __naked void pseudo_btf_id_log_masks_address(void)
: __clobber_all);
}
+static int pseudo_func_callback(__u32 index, void *ctx)
+{
+ return 0;
+}
+
+SEC("socket")
+__description("unpriv: pseudo function policy diagnostic")
+__success __failure_unpriv
+__msg_unpriv("loading/calling other bpf or kernel functions")
+__not_msg_unpriv("BPF-to-BPF function call")
+__msg_unpriv("policy check failed for BPF function reference")
+__msg_unpriv("avoid BPF function references in unprivileged")
+int unpriv_pseudo_func_policy(void *ctx)
+{
+ bpf_loop(1, pseudo_func_callback, NULL, 0);
+ return 0;
+}
+
SEC("socket")
__description("unpriv: return pointer")
__success __failure_unpriv __msg_unpriv("R0 leaks addr")
diff --git a/tools/testing/selftests/bpf/progs/verifier_var_off.c b/tools/testing/selftests/bpf/progs/verifier_var_off.c
index 24cd0a763673..a63e33675091 100644
--- a/tools/testing/selftests/bpf/progs/verifier_var_off.c
+++ b/tools/testing/selftests/bpf/progs/verifier_var_off.c
@@ -198,7 +198,8 @@ __success
/* Check that the maximum stack depth is correctly maintained according to the
* maximum possible variable offset.
*/
-__log_level(4) __msg("stack depth 16")
+__log_level(4)
+__msg("subprog 0 (stack_write_priv_vs_unpriv) main {{.*}} stack 16")
__failure_unpriv
/* Variable stack access is rejected for unprivileged.
*/
@@ -238,7 +239,8 @@ __success
/* Check that the maximum stack depth is correctly maintained according to the
* maximum possible variable offset.
*/
-__log_level(4) __msg("stack depth 16")
+__log_level(4)
+__msg("subprog 0 (stack_write_followed_by_read) main {{.*}} stack 16")
__failure_unpriv
__msg_unpriv("R2 variable stack access prohibited for !root")
__retval(0)
diff --git a/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c b/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c
index 0b86d95a4133..9866bc154194 100644
--- a/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c
+++ b/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c
@@ -1719,4 +1719,39 @@ l0_%=: r0 = 0; \
: __clobber_all);
}
+SEC("xdp")
+__description("XDP pkt regsafe preserves packet pointer class displacement")
+__failure __msg("R2 min value is outside of the allowed memory range")
+__flag(BPF_F_ANY_ALIGNMENT) __flag(BPF_F_TEST_STATE_FREQ)
+__naked void pkt_regsafe_class_displacement(void)
+{
+ asm volatile (" \
+ r8 = *(u32 *)(r1 + %[xdp_md_data_end]); \
+ r9 = *(u32 *)(r1 + %[xdp_md_data]); \
+ r4 = *(u32 *)(r1 + %[xdp_md_rx_queue_index]); \
+ r4 &= 15; \
+ r0 = *(u32 *)(r1 + %[xdp_md_ingress_ifindex]); \
+ if r0 != 0 goto l0_%=; \
+ r2 = r9; \
+ r2 += r4; \
+ r3 = r2; \
+ r3 += 8; \
+ goto l1_%=; \
+l0_%=: r4 &= 3; \
+ r4 += 8; \
+ r2 = r9; \
+ r2 += r4; \
+ r3 = r2; \
+l1_%=: if r3 > r8 goto l2_%=; \
+ r0 = *(u64 *)(r2 + 0); \
+l2_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm_const(xdp_md_data, offsetof(struct xdp_md, data)),
+ __imm_const(xdp_md_data_end, offsetof(struct xdp_md, data_end)),
+ __imm_const(xdp_md_rx_queue_index, offsetof(struct xdp_md, rx_queue_index)),
+ __imm_const(xdp_md_ingress_ifindex, offsetof(struct xdp_md, ingress_ifindex))
+ : __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_zext.c b/tools/testing/selftests/bpf/progs/verifier_zext.c
index 8f2362da91d6..572017fe28fb 100644
--- a/tools/testing/selftests/bpf/progs/verifier_zext.c
+++ b/tools/testing/selftests/bpf/progs/verifier_zext.c
@@ -356,6 +356,32 @@ __naked void arena_ptr(void)
: __clobber_all);
}
+/*
+ * Result of a 32-bit cmpxchg is always explicitly zero extended.
+ * Check that this holds for arenas (BPF_PROBE_ATOMIC instruction flavor).
+ */
+SEC("socket")
+__success
+__xlated("probe r0 = atomic_cmpxchg((u32 *)(r1 +0), r0, r2)")
+__xlated("w0 = w0")
+__naked void zext_arena_cmpxchg32(void)
+{
+ asm volatile (" \
+ r9 = %[arena] ll; /* associate the arena with the program */ \
+ r1 = 0; \
+ r1 = addr_space_cast(r1, 0, 1); \
+ r0 = 0; \
+ r2 = 0; \
+ .8byte %[cmpxchg32]; \
+ r0 >>= 32; /* make the upper half live */ \
+ exit; \
+" :
+ : __imm_addr(arena),
+ __imm_insn(cmpxchg32,
+ BPF_ATOMIC_OP(BPF_W, BPF_CMPXCHG, BPF_REG_1, BPF_REG_2, 0))
+ : __clobber_all);
+}
+
#endif
/* Check if probe mem loads keep their zero extension. */
diff --git a/tools/testing/selftests/bpf/progs/veristat_bar.c b/tools/testing/selftests/bpf/progs/veristat_bar.c
new file mode 100644
index 000000000000..83d2a2a1dfc9
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/veristat_bar.c
@@ -0,0 +1,3 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */
+#include "veristat_foo.c"
diff --git a/tools/testing/selftests/bpf/progs/veristat_foo.c b/tools/testing/selftests/bpf/progs/veristat_foo.c
new file mode 100644
index 000000000000..bd24b97664b4
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/veristat_foo.c
@@ -0,0 +1,31 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+
+/*
+ * Programs below exist only to exercise veristat's -f name filters,
+ * their bodies are irrelevant, only the names matter.
+ * This file is also included by veristat_bar.c, so that the same set of
+ * program names is available in two differently named object files.
+ */
+
+SEC("socket")
+int foo(void *ctx)
+{
+ return 0;
+}
+
+SEC("socket")
+int bar(void *ctx)
+{
+ return 0;
+}
+
+SEC("socket")
+int buz(void *ctx)
+{
+ return 0;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/test_kmods/bpf_testmod.c b/tools/testing/selftests/bpf/test_kmods/bpf_testmod.c
index a6133f7521f3..9366a3c578f1 100644
--- a/tools/testing/selftests/bpf/test_kmods/bpf_testmod.c
+++ b/tools/testing/selftests/bpf/test_kmods/bpf_testmod.c
@@ -402,12 +402,19 @@ static int bpf_testmod_ops3__test_arena_stack(u64 a, u64 b, u64 c, u64 d,
return 0;
}
+static int bpf_testmod_ops3__test_arena_multislot(struct bpf_testmod_arena_pair p,
+ u64 *ptr__arena)
+{
+ return 0;
+}
+
static struct bpf_testmod_ops3 __bpf_testmod_ops3 = {
.test_1 = bpf_testmod_test_3,
.test_2 = bpf_testmod_test_4,
.test_arena = bpf_testmod_ops3__test_arena,
.test_arena_nullable = bpf_testmod_ops3__test_arena_nullable,
.test_arena_stack = bpf_testmod_ops3__test_arena_stack,
+ .test_arena_multislot = bpf_testmod_ops3__test_arena_multislot,
};
static void bpf_testmod_test_struct_ops3(void)
@@ -441,6 +448,13 @@ __bpf_kfunc int bpf_testmod_ops3_call_test_arena_stack(u64 *ptr__arena)
return st_ops3->test_arena_stack(1, 2, 3, 4, 5, 6, 7, 8, ptr__arena);
}
+__bpf_kfunc int bpf_testmod_ops3_call_test_arena_multislot(u64 *ptr__arena)
+{
+ struct bpf_testmod_arena_pair p = { .a = 11, .b = 22 };
+
+ return st_ops3->test_arena_multislot(p, ptr__arena);
+}
+
struct bpf_testmod_btf_type_tag_1 {
int a;
};
@@ -852,6 +866,7 @@ BTF_ID_FLAGS(func, bpf_testmod_ops3_call_test_2)
BTF_ID_FLAGS(func, bpf_testmod_ops3_call_test_arena)
BTF_ID_FLAGS(func, bpf_testmod_ops3_call_test_arena_nullable)
BTF_ID_FLAGS(func, bpf_testmod_ops3_call_test_arena_stack)
+BTF_ID_FLAGS(func, bpf_testmod_ops3_call_test_arena_multislot)
BTF_ID_FLAGS(func, bpf_kfunc_get_default_trusted_ptr_test);
BTF_ID_FLAGS(func, bpf_kfunc_put_default_trusted_ptr_test);
BTF_KFUNCS_END(bpf_testmod_common_kfunc_ids)
diff --git a/tools/testing/selftests/bpf/test_kmods/bpf_testmod.h b/tools/testing/selftests/bpf/test_kmods/bpf_testmod.h
index 33f2af5b7085..210b919290cc 100644
--- a/tools/testing/selftests/bpf/test_kmods/bpf_testmod.h
+++ b/tools/testing/selftests/bpf/test_kmods/bpf_testmod.h
@@ -103,6 +103,12 @@ struct bpf_testmod_ops2 {
int (*test_1)(void);
};
+/* 16 bytes, so it takes two argument slots when passed by value */
+struct bpf_testmod_arena_pair {
+ u64 a;
+ u64 b;
+};
+
struct bpf_testmod_ops3 {
int (*test_1)(void);
int (*test_2)(void);
@@ -112,6 +118,8 @@ struct bpf_testmod_ops3 {
/* enough leading args to force @ptr onto the stack on x86 and arm64 */
int (*test_arena_stack)(u64 a, u64 b, u64 c, u64 d, u64 e, u64 f,
u64 g, u64 h, u64 *ptr);
+ /* a multi-slot leading arg, so @ptr is not at the slot its arg index suggests */
+ int (*test_arena_multislot)(struct bpf_testmod_arena_pair p, u64 *ptr);
};
struct st_ops_args {
diff --git a/tools/testing/selftests/bpf/test_kmods/bpf_testmod_kfunc.h b/tools/testing/selftests/bpf/test_kmods/bpf_testmod_kfunc.h
index c4383acb53c1..7d81070eefe7 100644
--- a/tools/testing/selftests/bpf/test_kmods/bpf_testmod_kfunc.h
+++ b/tools/testing/selftests/bpf/test_kmods/bpf_testmod_kfunc.h
@@ -123,6 +123,7 @@ void bpf_testmod_test_mod_kfunc(int i) __ksym;
int bpf_testmod_ops3_call_test_arena(__u64 *ptr__arena) __ksym;
int bpf_testmod_ops3_call_test_arena_nullable(__u64 *ptr__arena__nullable) __ksym;
int bpf_testmod_ops3_call_test_arena_stack(__u64 *ptr__arena) __ksym;
+int bpf_testmod_ops3_call_test_arena_multislot(__u64 *ptr__arena) __ksym;
__u64 bpf_kfunc_call_test1(struct sock *sk, __u32 a, __u64 b,
__u32 c, __u64 d) __ksym;
diff --git a/tools/testing/selftests/bpf/test_progs.c b/tools/testing/selftests/bpf/test_progs.c
index aa06bab30966..46eb201b96a3 100644
--- a/tools/testing/selftests/bpf/test_progs.c
+++ b/tools/testing/selftests/bpf/test_progs.c
@@ -424,10 +424,12 @@ static void jsonw_write_log_message(json_writer_t *w, char *log_buf, size_t log_
}
}
+/* @quiet elides the human readable output, the JSON report is unaffected */
static void dump_test_log(const struct prog_test_def *test,
const struct test_state *test_state,
bool skip_ok_subtests,
bool par_exec_result,
+ bool quiet,
json_writer_t *w)
{
bool test_failed = test_state->error_cnt > 0;
@@ -449,7 +451,7 @@ static void dump_test_log(const struct prog_test_def *test,
if (verbose() && !par_exec_result)
return;
- if (test_state->log_cnt && print_test)
+ if (test_state->log_cnt && print_test && !quiet)
print_test_log(test_state->log_buf, test_state->log_cnt);
if (w && print_test) {
@@ -471,15 +473,16 @@ static void dump_test_log(const struct prog_test_def *test,
if ((skip_ok_subtests && !subtest_failed) || subtest_filtered)
continue;
- if (subtest_state->log_cnt && print_subtest) {
+ if (subtest_state->log_cnt && print_subtest && !quiet) {
print_test_log(subtest_state->log_buf,
subtest_state->log_cnt);
}
- print_subtest_name(test->test_num, i + 1,
- test->test_name, subtest_state->name,
- test_result(subtest_state->error_cnt,
- subtest_state->skipped));
+ if (!quiet)
+ print_subtest_name(test->test_num, i + 1,
+ test->test_name, subtest_state->name,
+ test_result(subtest_state->error_cnt,
+ subtest_state->skipped));
if (w && print_subtest) {
jsonw_start_object(w);
@@ -496,7 +499,8 @@ static void dump_test_log(const struct prog_test_def *test,
jsonw_end_object(w);
}
- print_test_result(test, test_state);
+ if (!quiet)
+ print_test_result(test, test_state);
}
/* A bunch of tests set custom affinity per-thread and/or per-process. Reset
@@ -899,6 +903,7 @@ enum ARG_KEYS {
ARG_JSON_SUMMARY = 'J',
ARG_TRAFFIC_MONITOR = 'm',
ARG_WATCHDOG_TIMEOUT = 'w',
+ ARG_NO_ERROR_SUMMARY = -2,
};
static const struct argp_option opts[] = {
@@ -931,6 +936,8 @@ static const struct argp_option opts[] = {
#endif
{ "watchdog-timeout", ARG_WATCHDOG_TIMEOUT, "SECONDS", 0,
"Kill the process if tests are not making progress for specified number of seconds." },
+ { "no-error-summary", ARG_NO_ERROR_SUMMARY, NULL, 0,
+ "Do not re-print the aggregated error logs of failed tests at the end of the run." },
{},
};
@@ -1132,6 +1139,9 @@ static error_t parse_arg(int key, char *arg, struct argp_state *state)
case ARG_DEBUG:
env->debug = true;
break;
+ case ARG_NO_ERROR_SUMMARY:
+ env->error_summary = false;
+ break;
case ARG_JSON_SUMMARY:
env->json = fopen(arg, "w");
if (env->json == NULL) {
@@ -1304,7 +1314,7 @@ static void dump_crash_log(void)
if (env.test) {
env.test_state->error_cnt++;
- dump_test_log(env.test, env.test_state, true, false, NULL);
+ dump_test_log(env.test, env.test_state, true, false, false, NULL);
}
}
@@ -1462,7 +1472,7 @@ static void run_one_test(int test_num)
free(stop_libbpf_log_capture());
- dump_test_log(test, state, false, false, NULL);
+ dump_test_log(test, state, false, false, false, NULL);
}
struct dispatch_data {
@@ -1623,7 +1633,7 @@ static void *dispatch_thread(void *ctx)
} while (false);
pthread_mutex_lock(&stdout_output_lock);
- dump_test_log(test, state, false, true, NULL);
+ dump_test_log(test, state, false, true, false, NULL);
pthread_mutex_unlock(&stdout_output_lock);
} /* while (true) */
error:
@@ -1646,8 +1656,8 @@ done:
static void calculate_summary_and_print_errors(struct test_env *env)
{
- int i;
- int succ_cnt = 0, fail_cnt = 0, sub_succ_cnt = 0, skip_cnt = 0;
+ int i, j;
+ int succ_cnt = 0, fail_cnt = 0, sub_succ_cnt = 0, sub_fail_cnt = 0, skip_cnt = 0;
json_writer_t *w = NULL;
for (i = 0; i < prog_test_cnt; i++) {
@@ -1660,10 +1670,14 @@ static void calculate_summary_and_print_errors(struct test_env *env)
sub_succ_cnt += state->sub_succ_cnt;
skip_cnt += state->skip_cnt;
- if (state->error_cnt)
+ if (state->error_cnt) {
fail_cnt++;
- else if (!test->not_built)
+ for (j = 0; j < state->subtest_num; j++)
+ if (state->subtest_states[j].error_cnt)
+ sub_fail_cnt++;
+ } else if (!test->not_built) {
succ_cnt++;
+ }
}
if (env->json) {
@@ -1678,6 +1692,7 @@ static void calculate_summary_and_print_errors(struct test_env *env)
jsonw_uint_field(w, "success_subtest", sub_succ_cnt);
jsonw_uint_field(w, "skipped", skip_cnt);
jsonw_uint_field(w, "failed", fail_cnt);
+ jsonw_uint_field(w, "failed_subtest", sub_fail_cnt);
jsonw_name(w, "results");
jsonw_start_array(w);
}
@@ -1686,9 +1701,14 @@ static void calculate_summary_and_print_errors(struct test_env *env)
* We only print error logs summary when there are failed tests and
* verbose mode is not enabled. Otherwise, results may be inconsistent.
*
+ * --no-error-summary elides the human readable dump. The walk still
+ * happens when a JSON report was requested, so the JSON output keeps
+ * its per-test results; with no JSON report there is nothing left to
+ * do and the whole loop is skipped.
*/
- if (!verbose() && fail_cnt) {
- printf("\nAll error logs:\n");
+ if (!verbose() && fail_cnt && (env->error_summary || w)) {
+ if (env->error_summary)
+ printf("\nAll error logs:\n");
/* print error logs again */
for (i = 0; i < prog_test_cnt; i++) {
@@ -1698,7 +1718,8 @@ static void calculate_summary_and_print_errors(struct test_env *env)
if (!state->tested || !state->error_cnt)
continue;
- dump_test_log(test, state, true, true, w);
+ dump_test_log(test, state, true, true,
+ !env->error_summary, w);
}
}
@@ -1712,12 +1733,12 @@ static void calculate_summary_and_print_errors(struct test_env *env)
fclose(env->json);
if (env->not_built_cnt)
- printf("Summary: %d/%d PASSED, %d SKIPPED (%d not built), %d FAILED\n",
+ printf("Summary: %d/%d PASSED, %d SKIPPED (%d not built), %d/%d FAILED\n",
succ_cnt, sub_succ_cnt, skip_cnt, env->not_built_cnt,
- fail_cnt);
+ fail_cnt, sub_fail_cnt);
else
- printf("Summary: %d/%d PASSED, %d SKIPPED, %d FAILED\n",
- succ_cnt, sub_succ_cnt, skip_cnt, fail_cnt);
+ printf("Summary: %d/%d PASSED, %d SKIPPED, %d/%d FAILED\n",
+ succ_cnt, sub_succ_cnt, skip_cnt, fail_cnt, sub_fail_cnt);
env->succ_cnt = succ_cnt;
env->sub_succ_cnt = sub_succ_cnt;
@@ -2028,6 +2049,7 @@ int main(int argc, char **argv)
env.secs_till_notify = 10;
env.secs_till_kill = 120;
+ env.error_summary = true;
err = argp_parse(&argp, argc, argv, 0, NULL, &env);
if (err)
return err;
diff --git a/tools/testing/selftests/bpf/test_progs.h b/tools/testing/selftests/bpf/test_progs.h
index 2cf950afcd85..ea493c477fbd 100644
--- a/tools/testing/selftests/bpf/test_progs.h
+++ b/tools/testing/selftests/bpf/test_progs.h
@@ -105,6 +105,7 @@ struct test_env {
struct test_selector tmon_selector;
bool verifier_stats;
bool debug;
+ bool error_summary;
enum verbosity verbosity;
bool jit_enabled;
@@ -123,7 +124,7 @@ struct test_env {
int succ_cnt; /* successful tests */
int sub_succ_cnt; /* successful sub-tests */
- int fail_cnt; /* total failed tests + sub-tests */
+ int fail_cnt; /* failed tests */
int skip_cnt; /* skipped tests */
int not_built_cnt; /* tests not built */
diff --git a/tools/testing/selftests/bpf/test_verifier.c b/tools/testing/selftests/bpf/test_verifier.c
index a8ae03c57bba..bffb7360434c 100644
--- a/tools/testing/selftests/bpf/test_verifier.c
+++ b/tools/testing/selftests/bpf/test_verifier.c
@@ -1560,7 +1560,7 @@ static void do_test_single(struct bpf_test *test, bool unpriv,
opts.expected_attach_type = test->expected_attach_type;
if (expected_ret == VERBOSE_ACCEPT)
- opts.log_level = 2;
+ opts.log_level = 2 | 4;
else if (verbose)
opts.log_level = verif_log_level | 4; /* force stats */
else
diff --git a/tools/testing/selftests/bpf/usdt.h b/tools/testing/selftests/bpf/usdt.h
index c71e21df38b3..75687f50f4e2 100644
--- a/tools/testing/selftests/bpf/usdt.h
+++ b/tools/testing/selftests/bpf/usdt.h
@@ -313,7 +313,7 @@ struct usdt_sema { volatile unsigned short active; };
#if defined(__ia64__) || defined(__s390__) || defined(__s390x__)
#define USDT_NOP nop 0
#elif defined(__x86_64__)
-#define USDT_NOP .byte 0x90, 0x0f, 0x1f, 0x44, 0x00, 0x0 /* nop, nop5 */
+#define USDT_NOP .byte 0x90, 0x66, 0x2e, 0x0f, 0x1f, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00 /* nop, nop10 */
#else
#define USDT_NOP nop
#endif
diff --git a/tools/testing/selftests/bpf/usdt_2.c b/tools/testing/selftests/bpf/usdt_2.c
index 789883aaca4c..5e38f8605b02 100644
--- a/tools/testing/selftests/bpf/usdt_2.c
+++ b/tools/testing/selftests/bpf/usdt_2.c
@@ -3,7 +3,7 @@
#if defined(__x86_64__)
/*
- * Include usdt.h with default nop,nop5 instructions combo.
+ * Include usdt.h with default nop,nop10 instructions combo.
*/
#include "usdt.h"
@@ -13,4 +13,17 @@ void usdt_2(void)
USDT(optimized_attach, usdt_2);
}
+static volatile unsigned long usdt_red_zone_arg1 = 0xDEADBEEF;
+static volatile unsigned long usdt_red_zone_arg2 = 0xCAFEBABE;
+static volatile unsigned long usdt_red_zone_arg3 = 0xFEEDFACE;
+
+void __attribute__((noinline)) usdt_red_zone_trigger(void)
+{
+ unsigned long a1 = usdt_red_zone_arg1;
+ unsigned long a2 = usdt_red_zone_arg2;
+ unsigned long a3 = usdt_red_zone_arg3;
+
+ USDT(optimized_attach, usdt_red_zone, a1, a2, a3);
+}
+
#endif
diff --git a/tools/testing/selftests/bpf/verifier/calls.c b/tools/testing/selftests/bpf/verifier/calls.c
index 8cd626e04551..eb6e3baef412 100644
--- a/tools/testing/selftests/bpf/verifier/calls.c
+++ b/tools/testing/selftests/bpf/verifier/calls.c
@@ -1091,7 +1091,17 @@
/* stack_main=32, stack_A=256, stack_B=64
* and max(main+A, main+A+B) < 512
*/
- .result = ACCEPT,
+ .result = VERBOSE_ACCEPT,
+ .errstr = "stack depth max 352\t"
+ "subprog 0 (<unknown>) main insns_self \t"
+ " insns_total \t"
+ " stack 32\t"
+ "subprog 1 (<unknown>) static insns_self \t"
+ " insns_total \t"
+ " stack 256\t"
+ "subprog 2 (<unknown>) static insns_self \t"
+ " insns_total \t"
+ " stack 64",
},
{
"calls: stack depth check using three frames. test2",
diff --git a/tools/testing/selftests/bpf/verifier/pseudo_func.c b/tools/testing/selftests/bpf/verifier/pseudo_func.c
new file mode 100644
index 000000000000..63c5c67d51de
--- /dev/null
+++ b/tools/testing/selftests/bpf/verifier/pseudo_func.c
@@ -0,0 +1,45 @@
+/*
+ * Buggy verifier accepted the program below while not patching BPF_PSEUDO_FUNC
+ * load instruction to contain a real address. Which resulted in a function call
+ * to a bogus address.
+ */
+{
+ "BPF_PSEUDO_FUNC reference to the main program",
+ .insns = {
+ /* r6 = bpf_map_lookup_elem(&timer_map, &(int){0}); */
+ BPF_ST_MEM(BPF_W, BPF_REG_10, -4, 0),
+ BPF_MOV64_REG(BPF_REG_2, BPF_REG_10),
+ BPF_ALU64_IMM(BPF_ADD, BPF_REG_2, -4),
+ BPF_LD_MAP_FD(BPF_REG_1, 0),
+ BPF_EMIT_CALL(BPF_FUNC_map_lookup_elem),
+ BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 10),
+ BPF_MOV64_REG(BPF_REG_6, BPF_REG_0),
+ /* bpf_timer_init(r6, &timer_map, 0); */
+ BPF_MOV64_REG(BPF_REG_1, BPF_REG_6),
+ BPF_LD_MAP_FD(BPF_REG_2, 0),
+ BPF_MOV64_IMM(BPF_REG_3, 0),
+ BPF_EMIT_CALL(BPF_FUNC_timer_init),
+ /* bpf_timer_set_callback(r6, <insn #0>); */
+ BPF_MOV64_REG(BPF_REG_1, BPF_REG_6),
+ BPF_RAW_INSN(BPF_LD | BPF_IMM | BPF_DW, BPF_REG_2, BPF_PSEUDO_FUNC, 0, -15),
+ BPF_RAW_INSN(0, 0, 0, 0, 0),
+ BPF_EMIT_CALL(BPF_FUNC_timer_set_callback),
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_EXIT_INSN(),
+ },
+ .prog_type = BPF_PROG_TYPE_TRACEPOINT,
+ .fixup_map_timer = { 3, 9 },
+ .result = REJECT,
+ .errstr = "callback function cannot be the main program",
+ .func_info = { { 0, 4 /* main_prog */ } },
+ .func_info_cnt = 1,
+ .btf_strings = "\0int\0ctx\0main_prog",
+ .btf_types = {
+ /* 1: int */ BTF_TYPE_INT_ENC(1, BTF_INT_SIGNED, 0, 32, 4),
+ /* 2: void* */ BTF_PTR_ENC(0),
+ /* 3: int __(void *) */ BTF_FUNC_PROTO_ENC(1, 1),
+ BTF_FUNC_PROTO_ARG_ENC(5, 2),
+ /* 4: main_prog */ BTF_FUNC_ENC(9, 3),
+ BTF_END_RAW
+ }
+},
diff --git a/tools/testing/selftests/bpf/veristat.c b/tools/testing/selftests/bpf/veristat.c
index c9c257784ee3..e70741c6b9b7 100644
--- a/tools/testing/selftests/bpf/veristat.c
+++ b/tools/testing/selftests/bpf/veristat.c
@@ -514,6 +514,40 @@ cleanup:
return err == 0;
}
+/* Exact filter match */
+static bool name_filter_matches(struct filter *f, const char *filename, const char *prog_name)
+{
+ if (f->any_glob)
+ return glob_matches(filename, f->any_glob) ||
+ (prog_name && glob_matches(prog_name, f->any_glob));
+ if (f->file_glob && f->prog_glob)
+ return prog_name &&
+ glob_matches(filename, f->file_glob) &&
+ glob_matches(prog_name, f->prog_glob);
+ if (f->file_glob)
+ return glob_matches(filename, f->file_glob);
+ if (f->prog_glob)
+ return prog_name && glob_matches(prog_name, f->prog_glob);
+ return false;
+}
+
+/* Check if the filter does not outright reject the file name */
+static bool name_filter_may_match(struct filter *f, const char *filename)
+{
+ if (f->file_glob)
+ return glob_matches(filename, f->file_glob);
+ /*
+ * If we don't know program name yet, any_glob filter
+ * has to assume that current BPF object file might be
+ * relevant; we'll check again later on after opening
+ * BPF object file, at which point program name will
+ * be known finally.
+ */
+ if (f->any_glob || f->prog_glob)
+ return true;
+ return false;
+}
+
static bool should_process_file_prog(const char *filename, const char *prog_name)
{
struct filter *f;
@@ -521,16 +555,7 @@ static bool should_process_file_prog(const char *filename, const char *prog_name
for (i = 0; i < env.deny_filter_cnt; i++) {
f = &env.deny_filters[i];
- if (f->kind != FILTER_NAME)
- continue;
-
- if (f->any_glob && glob_matches(filename, f->any_glob))
- return false;
- if (f->any_glob && prog_name && glob_matches(prog_name, f->any_glob))
- return false;
- if (f->file_glob && glob_matches(filename, f->file_glob))
- return false;
- if (f->prog_glob && prog_name && glob_matches(prog_name, f->prog_glob))
+ if (f->kind == FILTER_NAME && name_filter_matches(f, filename, prog_name))
return false;
}
@@ -540,24 +565,15 @@ static bool should_process_file_prog(const char *filename, const char *prog_name
continue;
allow_cnt++;
- if (f->any_glob) {
- if (glob_matches(filename, f->any_glob))
- return true;
- /* If we don't know program name yet, any_glob filter
- * has to assume that current BPF object file might be
- * relevant; we'll check again later on after opening
- * BPF object file, at which point program name will
- * be known finally.
- */
- if (!prog_name || glob_matches(prog_name, f->any_glob))
- return true;
- } else {
- if (f->file_glob && !glob_matches(filename, f->file_glob))
- continue;
- if (f->prog_glob && prog_name && !glob_matches(prog_name, f->prog_glob))
- continue;
+ if (prog_name && name_filter_matches(f, filename, prog_name))
+ return true;
+ /*
+ * If there is no prog_name and the file name is not blocked by
+ * the filter, allow to open the file. Afterwards there would be
+ * a second refining query with prog_name set.
+ */
+ if (!prog_name && name_filter_may_match(f, filename))
return true;
- }
}
/* if there are no file/prog name allow filters, allow all progs,
@@ -703,6 +719,12 @@ static int append_filter(struct filter **filters, int *cnt, const char *str)
}
}
+ if ((!f->any_glob && !f->file_glob && !f->prog_glob) ||
+ (f->any_glob && strcmp(f->any_glob, "") == 0)) {
+ fprintf(stderr, "Invalid filter: '%s'\n", str);
+ return -EINVAL;
+ }
+
*cnt += 1;
return 0;
}
@@ -993,13 +1015,15 @@ static void free_verif_stats(struct verif_stats *stats, size_t stat_cnt)
static char verif_log_buf[64 * 1024];
-#define MAX_PARSED_LOG_LINES 100
+/* Keep room for all 256 subprogram records and trailing statistics. */
+#define MAX_PARSED_LOG_LINES 300
static int parse_verif_log(char * const buf, size_t buf_sz, struct verif_stats *s)
{
const char *cur;
- int pos, lines, sub_stack, cnt = 0;
- char *state = NULL, *token, stack[512];
+ long sub_stack;
+ int pos, lines, cnt = 0;
+ char *state = NULL, *token, stack[512] = {};
buf[buf_sz - 1] = '\0';
@@ -1025,11 +1049,24 @@ static int parse_verif_log(char * const buf, size_t buf_sz, struct verif_stats *
&s->stats[MARK_READ_MAX_LEN]))
continue;
+ /*
+ * New kernels emit one "subprog <id> (<name>) <kind>" record
+ * per subprogram with the stack depth at the end, while old
+ * kernels emit a single "stack depth <a+...+n> max <max>"
+ * line. Match both formats so veristat works against either
+ * kernel.
+ */
+ if (sscanf(cur, "stack depth max %ld", &s->stats[MAX_STACK]) == 1)
+ continue;
+ if (sscanf(cur, "subprog %*d %*s %*s insns_self %*d insns_total %*d stack %ld", &sub_stack) == 1) {
+ s->stats[STACK] += sub_stack;
+ continue;
+ }
if (2 == sscanf(cur, "stack depth %511s max %ld", stack, &s->stats[MAX_STACK]))
continue;
}
while ((token = strtok_r(cnt++ ? NULL : stack, "+", &state))) {
- if (sscanf(token, "%d", &sub_stack) == 0)
+ if (sscanf(token, "%ld", &sub_stack) == 0)
break;
s->stats[STACK] += sub_stack;
}
diff --git a/tools/testing/selftests/bpf/vmtest.sh b/tools/testing/selftests/bpf/vmtest.sh
index 9ca802285393..6a3d026d76bd 100755
--- a/tools/testing/selftests/bpf/vmtest.sh
+++ b/tools/testing/selftests/bpf/vmtest.sh
@@ -428,8 +428,17 @@ main()
if [[ $# -eq 0 && "${debug_shell}" == "no" ]]; then
echo "No command specified, will run ${DEFAULT_COMMAND} in the vm"
- else
- command="$@"
+ elif [[ $# -gt 0 ]]; then
+ # Quote each argument so the command survives into the guest: the
+ # host expands ${command} into the generated init script, which
+ # the guest bash then parses as shell source. Without the %q
+ # escapes an argument with a space or a glob (e.g. -a 'verifier_*')
+ # is re-split and expanded against /root/bpf there.
+ #
+ # Skip this when there is no command: printf '%q ' would still
+ # apply the format once and emit '', which is not the empty
+ # command that -s (debug shell) expects.
+ command=$(printf '%q ' "$@")
fi
local kconfig_file="${OUTPUT_DIR}/latest.config"
diff --git a/tools/testing/selftests/cgroup/lib/include/cgroup_util.h b/tools/testing/selftests/cgroup/lib/include/cgroup_util.h
index febc1723d090..5d39c709ac7a 100644
--- a/tools/testing/selftests/cgroup/lib/include/cgroup_util.h
+++ b/tools/testing/selftests/cgroup/lib/include/cgroup_util.h
@@ -7,7 +7,9 @@
#endif
#define MB(x) (x << 20)
+#define GB(x) ((unsigned long long)(x) << 30)
+#define NSEC_PER_USEC 1000L
#define USEC_PER_SEC 1000000L
#define NSEC_PER_SEC 1000000000L
diff --git a/tools/testing/selftests/cgroup/test_core.c b/tools/testing/selftests/cgroup/test_core.c
index 88ca832d4fc1..20d2b63774c3 100644
--- a/tools/testing/selftests/cgroup/test_core.c
+++ b/tools/testing/selftests/cgroup/test_core.c
@@ -427,7 +427,6 @@ static int test_cgcore_no_internal_process_constraint_on_threads(const char *roo
cleanup:
cg_enter_current(root);
- cg_enter_current(root);
if (child)
cg_destroy(child);
if (parent)
@@ -795,10 +794,9 @@ static int lesser_ns_open_thread_fn(void *arg)
static int test_cgcore_lesser_ns_open(const char *root)
{
static char stack[65536];
- const uid_t test_euid = 65534; /* usually nobody, any !root is fine */
int ret = KSFT_FAIL;
char *cg_test_a = NULL, *cg_test_b = NULL;
- char *cg_test_a_procs = NULL, *cg_test_b_procs = NULL;
+ char *cg_test_b_procs = NULL;
int cg_test_b_procs_fd = -1;
struct lesser_ns_open_thread_arg targ = { .fd = -1 };
pid_t pid;
@@ -813,10 +811,9 @@ static int test_cgcore_lesser_ns_open(const char *root)
if (!cg_test_a || !cg_test_b)
goto cleanup;
- cg_test_a_procs = cg_name(cg_test_a, "cgroup.procs");
cg_test_b_procs = cg_name(cg_test_b, "cgroup.procs");
- if (!cg_test_a_procs || !cg_test_b_procs)
+ if (!cg_test_b_procs)
goto cleanup;
if (cg_create(cg_test_a) || cg_create(cg_test_b))
@@ -825,10 +822,6 @@ static int test_cgcore_lesser_ns_open(const char *root)
if (cg_enter_current(cg_test_b))
goto cleanup;
- if (chown(cg_test_a_procs, test_euid, -1) ||
- chown(cg_test_b_procs, test_euid, -1))
- goto cleanup;
-
targ.path = cg_test_b_procs;
pid = clone(lesser_ns_open_thread_fn, stack + sizeof(stack),
CLONE_NEWCGROUP | CLONE_FILES | CLONE_VM | SIGCHLD,
@@ -863,7 +856,6 @@ cleanup:
if (cg_test_a)
cg_destroy(cg_test_a);
free(cg_test_b_procs);
- free(cg_test_a_procs);
free(cg_test_b);
free(cg_test_a);
return ret;
@@ -927,7 +919,6 @@ int main(int argc, char *argv[])
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), &nsdelegate)) {
if (setup_named_v1_root(root, sizeof(root), CG_NAMED_NAME))
ksft_exit_skip("cgroup v2 isn't mounted and could not setup named v1 hierarchy\n");
@@ -940,6 +931,7 @@ int main(int argc, char *argv[])
ksft_exit_skip("Failed to set memory controller\n");
post_v2_setup:
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_cpu.c b/tools/testing/selftests/cgroup/test_cpu.c
index 7a40d76b9548..735a53bb222b 100644
--- a/tools/testing/selftests/cgroup/test_cpu.c
+++ b/tools/testing/selftests/cgroup/test_cpu.c
@@ -291,6 +291,8 @@ static int test_cpucg_nice(const char *root)
user_usec = cg_read_key_long(cpucg, "cpu.stat", "user_usec");
nice_usec = cg_read_key_long(cpucg, "cpu.stat", "nice_usec");
+ if (user_usec <= 0)
+ goto cleanup;
if (!values_close_report(nice_usec, expected_nice_usec, 1))
goto cleanup;
@@ -640,21 +642,49 @@ test_cpucg_nested_weight_underprovisioned(const char *root)
}
/*
+ * Best effort attempt to get the kernel's HZ value from the config.
+ * Return the HZ value if found otherwise return 1000 (the default) to
+ * indicate failure.
+ */
+static long
+get_config_hz(void)
+{
+ long hz = 1000;
+ FILE *f;
+ char cmd[256] = "zcat /proc/config.gz 2>/dev/null | grep '^CONFIG_HZ='";
+
+ f = popen(cmd, "r");
+
+ if (!f)
+ return hz;
+
+ if (fscanf(f, "CONFIG_HZ=%ld", &hz) == EOF)
+ goto out;
+
+out:
+ pclose(f);
+ return hz;
+}
+
+/*
* This test creates a cgroup with some maximum value within a period, and
* verifies that a process in the cgroup is not overscheduled.
*/
static int test_cpucg_max(const char *root)
{
int ret = KSFT_FAIL;
+ long hz = get_config_hz();
long quota_usec = 1000;
long default_period_usec = 100000; /* cpu.max's default period */
long duration_seconds = 1;
- long duration_usec = duration_seconds * USEC_PER_SEC;
+ long duration_usec;
long usage_usec, n_periods, remainder_usec, expected_usage_usec;
char *cpucg;
char quota_buf[32];
+ duration_usec = duration_seconds * USEC_PER_SEC * 1000 / hz;
+
snprintf(quota_buf, sizeof(quota_buf), "%ld", quota_usec);
cpucg = cg_name(root, "cpucg_test");
@@ -670,8 +700,8 @@ static int test_cpucg_max(const char *root)
struct cpu_hog_func_param param = {
.nprocs = 1,
.ts = {
- .tv_sec = duration_seconds,
- .tv_nsec = 0,
+ .tv_sec = duration_usec / USEC_PER_SEC,
+ .tv_nsec = duration_usec % USEC_PER_SEC * NSEC_PER_USEC,
},
.clock_type = CPU_HOG_CLOCK_WALL,
};
@@ -710,15 +740,18 @@ cleanup:
static int test_cpucg_max_nested(const char *root)
{
int ret = KSFT_FAIL;
+ long hz = get_config_hz();
long quota_usec = 1000;
long default_period_usec = 100000; /* cpu.max's default period */
long duration_seconds = 1;
- long duration_usec = duration_seconds * USEC_PER_SEC;
+ long duration_usec;
long usage_usec, n_periods, remainder_usec, expected_usage_usec;
char *parent, *child;
char quota_buf[32];
+ duration_usec = duration_seconds * USEC_PER_SEC * 1000 / hz;
+
snprintf(quota_buf, sizeof(quota_buf), "%ld", quota_usec);
parent = cg_name(root, "cpucg_parent");
@@ -741,8 +774,8 @@ static int test_cpucg_max_nested(const char *root)
struct cpu_hog_func_param param = {
.nprocs = 1,
.ts = {
- .tv_sec = duration_seconds,
- .tv_nsec = 0,
+ .tv_sec = duration_usec / USEC_PER_SEC,
+ .tv_nsec = duration_usec % USEC_PER_SEC * NSEC_PER_USEC,
},
.clock_type = CPU_HOG_CLOCK_WALL,
};
@@ -799,7 +832,6 @@ int main(int argc, char *argv[])
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
@@ -807,6 +839,7 @@ int main(int argc, char *argv[])
if (cg_write(root, "cgroup.subtree_control", "+cpu"))
ksft_exit_skip("Failed to set cpu controller\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_cpuset.c b/tools/testing/selftests/cgroup/test_cpuset.c
index c5cf8b56ceb8..3dfadd280c1c 100644
--- a/tools/testing/selftests/cgroup/test_cpuset.c
+++ b/tools/testing/selftests/cgroup/test_cpuset.c
@@ -1,7 +1,13 @@
// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <assert.h>
#include <linux/limits.h>
+#include <pthread.h>
+#include <sched.h>
#include <signal.h>
+#include <sys/syscall.h>
+#include <unistd.h>
#include "kselftest.h"
#include "cgroup_util.h"
@@ -232,6 +238,246 @@ cleanup:
return ret;
}
+static int get_cpu_affinity(cpu_set_t *mask)
+{
+ CPU_ZERO(mask);
+ return sched_getaffinity(0, sizeof(*mask), mask);
+}
+
+static int cpu_set_equal(cpu_set_t *dst, unsigned long mask)
+{
+ cpu_set_t expected;
+
+ CPU_ZERO(&expected);
+ assert(sizeof(mask) < CPU_SETSIZE);
+
+ for (int cpu = 0; cpu < sizeof(mask) * 8; ++cpu)
+ if ((1UL << cpu) & mask)
+ CPU_SET(cpu, &expected);
+
+ return CPU_EQUAL(&expected, dst);
+}
+
+enum test_phase {
+ AFFINITY_SETUP,
+ AFFINITY_CONTROLLER_DISABLED,
+ AFFINITY_COMPLETE,
+ AFFINITY_ERROR
+};
+
+struct thread_args {
+ const char *cgroup;
+ cpu_set_t *affinity_before;
+ cpu_set_t *affinity_after;
+ int affinity_before_ready;
+};
+
+static pthread_mutex_t test_mutex = PTHREAD_MUTEX_INITIALIZER;
+static pthread_cond_t test_cond = PTHREAD_COND_INITIALIZER;
+static enum test_phase test_phase;
+
+static void *affinity_thread_fn(void *arg)
+{
+ struct thread_args *args = (struct thread_args *)arg;
+
+ if (cg_enter_current_thread(args->cgroup))
+ goto fail;
+
+ if (get_cpu_affinity(args->affinity_before) != 0)
+ goto fail;
+
+ pthread_mutex_lock(&test_mutex);
+ args->affinity_before_ready = 1;
+ pthread_cond_broadcast(&test_cond);
+
+ while (test_phase < AFFINITY_CONTROLLER_DISABLED)
+ pthread_cond_wait(&test_cond, &test_mutex);
+ pthread_mutex_unlock(&test_mutex);
+
+ if (get_cpu_affinity(args->affinity_after) != 0)
+ goto fail;
+
+
+ return NULL;
+
+fail:
+ pthread_mutex_lock(&test_mutex);
+ test_phase = AFFINITY_ERROR;
+ pthread_cond_broadcast(&test_cond);
+ pthread_mutex_unlock(&test_mutex);
+ return NULL;
+}
+
+/*
+ * Test that disabling cpuset controller properly updates thread affinity.
+ *
+ * This test exposes a bug in cpuset_attach() where threads in child cgroups
+ * don't get their affinity updated when the cpuset controller is disabled.
+ *
+ * Setup:
+ * - Create parent cgroup with cpuset.cpus=0-1
+ * - Create child A with cpuset.cpus=0-1
+ * - Create child B with cpuset.cpus=1
+ * - Place multithreaded process: group leader + thread_a in A, thread_b in B
+ * - Disable cpuset controller on parent
+ *
+ * Expected: thread_b's affinity should expand from {1} to {0-1}
+ * Buggy: thread_b's affinity remains {1}
+ */
+static int test_cpuset_affinity_on_controller_disable(const char *root)
+{
+ char *parent = NULL, *child_a = NULL, *child_b = NULL;
+ pthread_t thread_a, thread_b;
+ int thread_a_created = 0, thread_b_created = 0;
+ cpu_set_t affinity_a_before, affinity_a_after;
+ cpu_set_t affinity_b_before, affinity_b_after;
+ int ret = KSFT_FAIL;
+
+ parent = cg_name(root, "cpuset_affinity_test");
+ if (!parent)
+ goto cleanup;
+ if (cg_create(parent))
+ goto cleanup;
+ if (cg_write(parent, "cgroup.type", "threaded"))
+ goto cleanup;
+
+ child_a = cg_name(parent, "A");
+ if (!child_a)
+ goto cleanup;
+ if (cg_create(child_a))
+ goto cleanup;
+ if (cg_write(child_a, "cgroup.type", "threaded"))
+ goto cleanup;
+
+ child_b = cg_name(parent, "B");
+ if (!child_b)
+ goto cleanup;
+ if (cg_create(child_b))
+ goto cleanup;
+ if (cg_write(child_b, "cgroup.type", "threaded"))
+ goto cleanup;
+
+ /* Now enable cpuset controller in parent */
+ if (cg_write(parent, "cgroup.subtree_control", "+cpuset"))
+ goto skip;
+
+ /*
+ * Set CPU affinity constraints
+ * Skip the test if the setting of "cpuset.cpus" fails as the test
+ * system may not have CPU 1.
+ */
+ if (cg_write(parent, "cpuset.cpus", "0-1"))
+ goto skip;
+ if (cg_write(child_a, "cpuset.cpus", "0-1"))
+ goto skip;
+ if (cg_write(child_b, "cpuset.cpus", "1"))
+ goto skip;
+
+ /* Move group leader (main thread) to child A */
+ if (cg_enter_current(child_a))
+ goto cleanup;
+
+ /* Create threads - they will move themselves to their respective cgroups */
+ test_phase = AFFINITY_SETUP;
+
+ struct thread_args args_a = {
+ .cgroup = child_a,
+ .affinity_before = &affinity_a_before,
+ .affinity_after = &affinity_a_after,
+ .affinity_before_ready = 0,
+ };
+ if (pthread_create(&thread_a, NULL, affinity_thread_fn, &args_a))
+ goto cleanup;
+ thread_a_created = 1;
+
+ struct thread_args args_b = {
+ .cgroup = child_b,
+ .affinity_before = &affinity_b_before,
+ .affinity_after = &affinity_b_after,
+ .affinity_before_ready = 0,
+ };
+ if (pthread_create(&thread_b, NULL, affinity_thread_fn, &args_b))
+ goto cleanup_threads;
+ thread_b_created = 1;
+
+ pthread_mutex_lock(&test_mutex);
+ while ((test_phase < AFFINITY_ERROR) &&
+ (args_a.affinity_before_ready + args_b.affinity_before_ready < 2))
+ pthread_cond_wait(&test_cond, &test_mutex);
+
+ /* If a thread failed during setup, bail out */
+ if (test_phase == AFFINITY_ERROR) {
+ pthread_mutex_unlock(&test_mutex);
+ goto cleanup_threads;
+ }
+ pthread_mutex_unlock(&test_mutex);
+
+ if (!cpu_set_equal(&affinity_a_before, 0x3)) {
+ ksft_print_msg("FAIL: thread_a initial affinity incorrect\n");
+ goto cleanup_threads;
+ }
+
+ if (!cpu_set_equal(&affinity_b_before, 0x2)) {
+ ksft_print_msg("FAIL: thread_b initial affinity incorrect\n");
+ goto cleanup_threads;
+ }
+
+ /* Disable cpuset controller - this should trigger affinity update */
+ if (cg_write(parent, "cgroup.subtree_control", "-cpuset"))
+ goto cleanup_threads;
+
+ /* Signal threads to save their final affinity and exit */
+ pthread_mutex_lock(&test_mutex);
+ test_phase = AFFINITY_CONTROLLER_DISABLED;
+ pthread_cond_broadcast(&test_cond);
+ pthread_mutex_unlock(&test_mutex);
+
+ pthread_join(thread_a, NULL);
+ pthread_join(thread_b, NULL);
+
+ /* Verify thread affinities AFTER disabling controller */
+ if (!cpu_set_equal(&affinity_a_after, 0x3)) {
+ ksft_print_msg("FAIL: thread_a final affinity incorrect\n");
+ goto cleanup;
+ }
+
+ if (!cpu_set_equal(&affinity_b_after, 0x3)) {
+ ksft_print_msg("FAIL: thread_b affinity did not expand to {0-1}\n");
+ goto cleanup;
+ }
+
+ ret = KSFT_PASS;
+ goto cleanup;
+
+skip:
+ ret = KSFT_SKIP;
+ goto cleanup;
+
+cleanup_threads:
+ pthread_mutex_lock(&test_mutex);
+ test_phase = AFFINITY_COMPLETE;
+ pthread_cond_broadcast(&test_cond);
+ pthread_mutex_unlock(&test_mutex);
+
+ if (thread_a_created)
+ pthread_join(thread_a, NULL);
+ if (thread_b_created)
+ pthread_join(thread_b, NULL);
+
+cleanup:
+ /* Move back to root before cleanup */
+ cg_enter_current(root);
+
+ cg_destroy(child_b);
+ free(child_b);
+ cg_destroy(child_a);
+ free(child_a);
+ cg_destroy(parent);
+ free(parent);
+
+ return ret;
+}
+
#define T(x) { x, #x }
struct cpuset_test {
@@ -241,6 +487,7 @@ struct cpuset_test {
T(test_cpuset_perms_object_allow),
T(test_cpuset_perms_object_deny),
T(test_cpuset_perms_subtree),
+ T(test_cpuset_affinity_on_controller_disable),
};
#undef T
@@ -250,7 +497,6 @@ int main(int argc, char *argv[])
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
@@ -258,6 +504,7 @@ int main(int argc, char *argv[])
if (cg_write(root, "cgroup.subtree_control", "+cpuset"))
ksft_exit_skip("Failed to set cpuset controller\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_cpuset_prs.sh b/tools/testing/selftests/cgroup/test_cpuset_prs.sh
index 0d41aa0d343d..131d8b4551ef 100755
--- a/tools/testing/selftests/cgroup/test_cpuset_prs.sh
+++ b/tools/testing/selftests/cgroup/test_cpuset_prs.sh
@@ -20,7 +20,7 @@ skip_test() {
WAIT_INOTIFY=$(cd $(dirname $0); pwd)/wait_inotify
# Find cgroup v2 mount point
-CGROUP2=$(mount -t cgroup2 | head -1 | awk -e '{print $3}')
+CGROUP2=$(mount -t cgroup2 | head -1 | awk '{print $3}')
[[ -n "$CGROUP2" ]] || skip_test "Cgroup v2 mount point not found!"
SUBPARTS_CPUS=$CGROUP2/.__DEBUG__.cpuset.cpus.subpartitions
CPULIST=$(cat $CGROUP2/cpuset.cpus.effective)
@@ -495,13 +495,26 @@ REMOTE_TEST_MATRIX=(
# Narrowing cpuset.cpus to previously sibling-excluded CPUs should
# not return CPUs that were never actually owned.
" C1-4:P1 . C1-2:P1 C1-3:P2 . . \
- . . . C3 . . p1:4|c11:1-2|c12:3 \
+ . . . C3 . . p1:4|c11:1-2|c12:3 \
p1:P1|c11:P1|c12:P2 3"
# Expanding cpuset.cpus to include a previously sibling-excluded CPU
# after the sibling has become a member should correctly request it.
" C1-4:P1 . C1-2:P1 C1-3:P2 . . \
- . . P0 C2-3 . . p1:1,4|c11:1|c12:2-3 \
+ . . P0 C2-3 . . p1:1,4|c11:1|c12:2-3 \
p1:P1|c11:P0|c12:P2 2-3"
+ # Changing a sibling partition's cpuset.cpus to overlap with another
+ # sibling partition should invalidate itself and return only actually
+ # allocated CPUs (effective_xcpus) to the parent.
+ " C1-4:P1 . C1-2:P1 C2-4:P2 . . \
+ . . . C1-2 . . p1:3-4|c11:1-2|c12:3-4 \
+ p1:P1|c11:P1|c12:P-2"
+ # Cpusets with empty cpuset.cpus should inherit parent's effective_cpus
+ " C1-4:P1 C5-6 C1-2 . C5 . \
+ . P1 P1 . . . p1:3-4|p2:5-6|c11:1-2|c12:3-4|c21:5|c22:5-6 \
+ p1:P1|p2:P1|c11:P1"
+ " C1-4:P1 C5-6 C1-2 . C5 . \
+ . P1 P1 . O5=0 . p1:3-4|p2:6|c11:1-2|c12:3-4|c21:6|c22:6 \
+ p1:P1|p2:P1|c11:P1"
)
#
@@ -513,6 +526,7 @@ write_cpu_online()
CPU=${1%=*}
VAL=${1#*=}
CPUFILE=//sys/devices/system/cpu/cpu${CPU}/online
+ echo $VAL > $CPUFILE || return 1
if [[ $VAL -eq 0 ]]
then
OFFLINE_CPUS="$OFFLINE_CPUS $CPU"
@@ -522,7 +536,6 @@ write_cpu_online()
sort | uniq -u)
}
fi
- echo $VAL > $CPUFILE
pause 0.05
}
@@ -590,7 +603,8 @@ set_ctrl_state()
eval $COMM $REDIRECT
;;
O*) VAL=${CMD#?}
- write_cpu_online $VAL
+ COMM="write_cpu_online $VAL"
+ eval $COMM $REDIRECT
;;
T*) COMM="echo 0 > $TFILE"
eval $COMM $REDIRECT
@@ -783,7 +797,6 @@ check_isolcpus()
EXPECTED_ISOLCPUS=$1
ISCPUS=${CGROUP2}/cpuset.cpus.isolated
ISOLCPUS=$(cat $ISCPUS)
- HKICPUS=$(cat /sys/devices/system/cpu/isolated)
LASTISOLCPU=
SCHED_DOMAINS=/sys/kernel/debug/sched/domains
if [[ $EXPECTED_ISOLCPUS = . ]]
@@ -822,11 +835,6 @@ check_isolcpus()
EXPECTED_ISOLCPUS=$EXPECTED_SDOMAIN
#
- # The inverse of HK_TYPE_DOMAIN cpumask in $HKICPUS should match $ISOLCPUS
- #
- [[ "$ISOLCPUS" != "$HKICPUS" ]] && return 1
-
- #
# Use the sched domain in debugfs to check isolated CPUs, if available
#
[[ -d $SCHED_DOMAINS ]] || return 0
@@ -1148,6 +1156,63 @@ test_isolated()
}
#
+# Select an online CPU isolated from scheduler domains at boot.
+# $1: test name used in the skip message
+#
+get_boot_isolated_cpu()
+{
+ TEST_NAME=$1
+ BOOT_ISOLATED_FILE=/sys/devices/system/cpu/isolated
+
+ [[ -r $BOOT_ISOLATED_FILE ]] || {
+ echo "$TEST_NAME test SKIPPED: boot isolation state unavailable"
+ return 1
+ }
+ BOOT_CPUS=$(cat $BOOT_ISOLATED_FILE)
+ [[ -n "$BOOT_CPUS" ]] || {
+ echo "$TEST_NAME test SKIPPED: no boot-isolated CPU"
+ return 1
+ }
+
+ BOOT_CPU=$(echo "$BOOT_CPUS" | sed -e 's/[,-].*//')
+ CPU_ONLINE=/sys/devices/system/cpu/cpu${BOOT_CPU}/online
+ [[ ! -e $CPU_ONLINE || $(cat $CPU_ONLINE) -eq 1 ]] || {
+ echo "$TEST_NAME test SKIPPED: CPU $BOOT_CPU is offline"
+ return 1
+ }
+}
+
+#
+# A CPU isolated at boot must stay isolated after it is released by a dynamic
+# isolated partition.
+#
+test_boot_isolated()
+{
+ TEST_NAME="Boot-isolated CPU partition release"
+ get_boot_isolated_cpu "$TEST_NAME" || return 0
+ echo "Running $TEST_NAME test ..."
+
+ cd $CGROUP2/test
+ echo member > cpuset.cpus.partition
+ echo $BOOT_CPU > cpuset.cpus
+ [[ $(cat cpuset.cpus.effective) = "$BOOT_CPU" ]] || {
+ echo "$TEST_NAME test SKIPPED: CPU $BOOT_CPU is unavailable"
+ echo "" > cpuset.cpus
+ cd $CGROUP2
+ return 0
+ }
+ test_partition isolated
+ test_partition member
+ check_isolcpus "." || {
+ echo "Boot-isolated CPU $BOOT_CPU was lost after partition release"
+ exit 1
+ }
+ echo "" > cpuset.cpus
+ cd $CGROUP2
+ echo "$TEST_NAME test PASSED."
+}
+
+#
# Wait for inotify event for the given file and read it
# $1: cgroup file to wait for
# $2: file to store the read result
@@ -1218,5 +1283,6 @@ trap cleanup 0 2 3 6
run_state_test TEST_MATRIX
run_remote_state_test REMOTE_TEST_MATRIX
test_isolated
+test_boot_isolated
test_inotify
echo "All tests PASSED."
diff --git a/tools/testing/selftests/cgroup/test_cpuset_v1_hp.sh b/tools/testing/selftests/cgroup/test_cpuset_v1_hp.sh
index 7406c24be1ac..da97f1643f9a 100755
--- a/tools/testing/selftests/cgroup/test_cpuset_v1_hp.sh
+++ b/tools/testing/selftests/cgroup/test_cpuset_v1_hp.sh
@@ -14,7 +14,7 @@ skip_test() {
[[ $(id -u) -eq 0 ]] || skip_test "Test must be run as root!"
# Find cpuset v1 mount point
-CPUSET=$(mount -t cgroup | grep cpuset | head -1 | awk -e '{print $3}')
+CPUSET=$(mount -t cgroup | grep cpuset | head -1 | awk '{print $3}')
[[ -n "$CPUSET" ]] || skip_test "cpuset v1 mount point not found!"
#
diff --git a/tools/testing/selftests/cgroup/test_freezer.c b/tools/testing/selftests/cgroup/test_freezer.c
index 0569e93fa6b0..f28bb02e9783 100644
--- a/tools/testing/selftests/cgroup/test_freezer.c
+++ b/tools/testing/selftests/cgroup/test_freezer.c
@@ -1491,9 +1491,9 @@ int main(int argc, char *argv[])
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_hugetlb_memcg.c b/tools/testing/selftests/cgroup/test_hugetlb_memcg.c
index b627d84358b1..8c5aced813b6 100644
--- a/tools/testing/selftests/cgroup/test_hugetlb_memcg.c
+++ b/tools/testing/selftests/cgroup/test_hugetlb_memcg.c
@@ -199,7 +199,10 @@ out:
int main(int argc, char **argv)
{
char root[PATH_MAX];
- int ret = EXIT_SUCCESS, has_memory_hugetlb_acc;
+ int has_memory_hugetlb_acc;
+
+ ksft_print_header();
+ ksft_set_plan(1);
has_memory_hugetlb_acc = proc_mount_contains("memory_hugetlb_accounting");
if (has_memory_hugetlb_acc < 0)
@@ -211,7 +214,7 @@ int main(int argc, char **argv)
if (get_hugepage_size() != 2048) {
ksft_print_msg("test_hugetlb_memcg requires 2MB hugepages\n");
ksft_test_result_skip("test_hugetlb_memcg\n");
- return ret;
+ ksft_finished();
}
if (cg_find_unified_root(root, sizeof(root), NULL))
@@ -233,10 +236,9 @@ int main(int argc, char **argv)
ksft_test_result_skip("test_hugetlb_memcg\n");
break;
default:
- ret = EXIT_FAILURE;
ksft_test_result_fail("test_hugetlb_memcg\n");
break;
}
- return ret;
+ ksft_finished();
}
diff --git a/tools/testing/selftests/cgroup/test_kill.c b/tools/testing/selftests/cgroup/test_kill.c
index f6cd23a8ecc7..bac1ddd8cb94 100644
--- a/tools/testing/selftests/cgroup/test_kill.c
+++ b/tools/testing/selftests/cgroup/test_kill.c
@@ -7,6 +7,7 @@
#include <stdlib.h>
#include <string.h>
#include <sys/types.h>
+#include <sys/wait.h>
#include <unistd.h>
#include "kselftest.h"
@@ -261,6 +262,59 @@ cleanup:
return ret;
}
+/*
+ * Test that a cgroup that was killed in the past can still be the target
+ * of clone3(CLONE_INTO_CGROUP): writing cgroup.kill must only kill the
+ * tasks in the cgroup at the time of the write, not tasks cloned into
+ * it afterwards.
+ */
+static int test_cgkill_clone_into_killed(const char *root)
+{
+ pid_t pid;
+ int cgroup_fd = -EBADF;
+ int ret = KSFT_FAIL;
+ char *cgroup = NULL;
+
+ cgroup = cg_name(root, "cg_test_clone_into_killed");
+ if (!cgroup)
+ goto cleanup;
+
+ if (cg_create(cgroup))
+ goto cleanup;
+
+ /* Kill the cgroup while it is still empty. */
+ if (cg_write(cgroup, "cgroup.kill", "1"))
+ goto cleanup;
+
+ cgroup_fd = dirfd_open_opath(cgroup);
+ if (cgroup_fd < 0)
+ goto cleanup;
+
+ pid = clone_into_cgroup(cgroup_fd);
+ if (pid < 0) {
+ if (errno == ENOSYS)
+ ret = KSFT_SKIP;
+ goto cleanup;
+ }
+
+ if (pid == 0)
+ exit(EXIT_SUCCESS);
+
+ /* The child must not be SIGKILLed; it has to exit cleanly. */
+ if (clone_reap(pid, WEXITED) != EXIT_SUCCESS)
+ goto cleanup;
+
+ ret = KSFT_PASS;
+
+cleanup:
+ if (cgroup_fd >= 0)
+ close(cgroup_fd);
+ if (cgroup)
+ cg_destroy(cgroup);
+ free(cgroup);
+ return ret;
+}
+
#define T(x) { x, #x }
struct cgkill_test {
int (*fn)(const char *root);
@@ -269,6 +323,7 @@ struct cgkill_test {
T(test_cgkill_simple),
T(test_cgkill_tree),
T(test_cgkill_forkbomb),
+ T(test_cgkill_clone_into_killed),
};
#undef T
@@ -278,9 +333,9 @@ int main(int argc, char *argv[])
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_kmem.c b/tools/testing/selftests/cgroup/test_kmem.c
index 1db0ba1226b9..437f2d35f205 100644
--- a/tools/testing/selftests/cgroup/test_kmem.c
+++ b/tools/testing/selftests/cgroup/test_kmem.c
@@ -145,7 +145,7 @@ static int cg_run_in_subcgroups(const char *parent,
return -1;
}
- if (cg_run(child, fn, NULL)) {
+ if (cg_run(child, fn, arg)) {
cg_destroy(child);
free(child);
return -1;
@@ -426,7 +426,6 @@ int main(int argc, char **argv)
int i;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
@@ -441,6 +440,7 @@ int main(int argc, char **argv)
if (cg_write(root, "cgroup.subtree_control", "+memory"))
ksft_exit_skip("Failed to set memory controller\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_memcontrol.c b/tools/testing/selftests/cgroup/test_memcontrol.c
index 0ebf796f3cff..0ed82347044e 100644
--- a/tools/testing/selftests/cgroup/test_memcontrol.c
+++ b/tools/testing/selftests/cgroup/test_memcontrol.c
@@ -30,7 +30,7 @@ static int page_size;
int get_temp_fd(void)
{
- return open(".", O_TMPFILE | O_RDWR | O_EXCL);
+ return open(".", O_TMPFILE | O_RDWR | O_EXCL, 0600);
}
int alloc_pagecache(int fd, size_t size)
@@ -1798,7 +1798,6 @@ int main(int argc, char **argv)
page_size = BUF_SIZE;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
@@ -1823,6 +1822,7 @@ int main(int argc, char **argv)
ksft_exit_skip("Failed to query cgroup mount option\n");
has_localevents = proc_status;
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_pids.c b/tools/testing/selftests/cgroup/test_pids.c
index 9a387c815d2c..710109b53dfe 100644
--- a/tools/testing/selftests/cgroup/test_pids.c
+++ b/tools/testing/selftests/cgroup/test_pids.c
@@ -148,7 +148,6 @@ int main(int argc, char **argv)
char root[PATH_MAX];
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
@@ -163,6 +162,7 @@ int main(int argc, char **argv)
if (cg_write(root, "cgroup.subtree_control", "+pids"))
ksft_exit_skip("Failed to set pids controller\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (int i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c
index 49b36ee79160..8df54b59513a 100644
--- a/tools/testing/selftests/cgroup/test_zswap.c
+++ b/tools/testing/selftests/cgroup/test_zswap.c
@@ -20,6 +20,8 @@ static int page_size;
#define PATH_ZSWAP "/sys/module/zswap"
#define PATH_ZSWAP_ENABLED "/sys/module/zswap/parameters/enabled"
+#define PATH_ZSWAP_SHRINKER_ENABLED "/sys/module/zswap/parameters/shrinker_enabled"
+#define PATH_ZSWAP_STORED_PAGES "/sys/kernel/debug/zswap/stored_pages"
static int read_int(const char *path, size_t *value)
{
@@ -55,7 +57,7 @@ static int read_min_free_kb(size_t *value)
static int get_zswap_stored_pages(size_t *value)
{
- return read_int("/sys/kernel/debug/zswap/stored_pages", value);
+ return read_int(PATH_ZSWAP_STORED_PAGES, value);
}
static long get_cg_wb_count(const char *cg)
@@ -443,6 +445,16 @@ static int test_zswap_writeback_disabled(const char *root)
return test_zswap_writeback(root, false);
}
+static bool zswap_shrinker_enabled(void)
+{
+ char value[2];
+
+ if (read_text(PATH_ZSWAP_SHRINKER_ENABLED, value, sizeof(value)) <= 0)
+ return 0;
+
+ return value[0] == 'Y';
+}
+
/*
* When trying to store a memcg page in zswap, if the memcg hits its memory
* limit in zswap, writeback should affect only the zswapped pages of that
@@ -452,6 +464,7 @@ static int test_no_invasive_cgroup_shrink(const char *root)
{
int ret = KSFT_FAIL;
unsigned int off;
+ long zswpwb_before, zswpwb_after, zswpwb_target;
size_t allocation_size = page_size * 1024;
unsigned int nr_pages = allocation_size / page_size;
char zswap_max_buf[32], mem_max_buf[32];
@@ -487,6 +500,14 @@ static int test_no_invasive_cgroup_shrink(const char *root)
if (cg_read_key_long(zw_group, "memory.stat", "zswapped") < 1)
goto out;
+ /* If the shrinker is enabled, try to let the writebacks finish first */
+ if (zswap_shrinker_enabled())
+ sleep(5);
+
+ zswpwb_before = get_cg_wb_count(zw_group);
+ if (zswpwb_before < 0)
+ goto out;
+
/* Push wb_group memory into zswap with hard-to-compress data to trigger wb */
if (cg_enter_current(wb_group))
goto out;
@@ -499,9 +520,13 @@ static int test_no_invasive_cgroup_shrink(const char *root)
getrandom(&wb_allocation[off], page_size/4, 0);
}
- /* Verify that only zswapped memory from gwb_group has been written back */
- if (wait_for_writeback(wb_group, 5000) > 0 && get_cg_wb_count(zw_group) == 0)
+ /* Verify that only zswapped memory from wb_group has been written back */
+ zswpwb_target = wait_for_writeback(wb_group, 5000);
+ zswpwb_after = get_cg_wb_count(zw_group);
+
+ if (zswpwb_target > 0 && zswpwb_before == zswpwb_after)
ret = KSFT_PASS;
+
out:
cg_enter_current(root);
if (zw_group) {
@@ -570,8 +595,16 @@ static int test_no_kmem_bypass(const char *root)
/* Read sys info and compute test values accordingly */
if (sysinfo(&sys_info) != 0)
return KSFT_FAIL;
- if (sys_info.totalram > 5000000000)
+ if (sys_info.totalram > GB(4)) {
+ ksft_print_msg(
+ "requires less than 4GB total ram, sys_info.totalram: %.1fGB\n",
+ (double)sys_info.totalram / GB(1));
+ return KSFT_SKIP;
+ }
+ if (access(PATH_ZSWAP_STORED_PAGES, R_OK)) {
+ ksft_print_msg("debugfs not mounted at /sys/kernel/debug\n");
return KSFT_SKIP;
+ }
values = mmap(0, sizeof(struct no_kmem_bypass_child_args), PROT_READ |
PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0);
if (values == MAP_FAILED)
@@ -810,7 +843,6 @@ int main(int argc, char **argv)
page_size = BUF_SIZE;
ksft_print_header();
- ksft_set_plan(ARRAY_SIZE(tests));
if (cg_find_unified_root(root, sizeof(root), NULL))
ksft_exit_skip("cgroup v2 isn't mounted\n");
@@ -827,6 +859,7 @@ int main(int argc, char **argv)
if (cg_write(root, "cgroup.subtree_control", "+memory"))
ksft_exit_skip("Failed to set memory controller\n");
+ ksft_set_plan(ARRAY_SIZE(tests));
for (i = 0; i < ARRAY_SIZE(tests); i++) {
switch (tests[i].fn(root)) {
case KSFT_PASS:
diff --git a/tools/testing/selftests/clock-helpers.h b/tools/testing/selftests/clock-helpers.h
new file mode 100644
index 000000000000..01451f538e71
--- /dev/null
+++ b/tools/testing/selftests/clock-helpers.h
@@ -0,0 +1,76 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+
+#ifndef __CLOCK_HELPERS_H
+#define __CLOCK_HELPERS_H
+
+#include <sys/types.h>
+#include <time.h>
+
+#define MSEC_PER_SEC 1000LL
+#define USEC_PER_MSEC 1000LL
+#define NSEC_PER_USEC 1000LL
+#define NSEC_PER_MSEC 1000000LL
+#define USEC_PER_SEC 1000000LL
+#define NSEC_PER_SEC 1000000000LL
+#define PSEC_PER_SEC 1000000000000LL
+#define FSEC_PER_SEC 1000000000000000LL
+
+#ifndef CLOCK_AUX
+#define CLOCK_AUX 16
+#endif
+
+#ifndef MAX_AUX_CLOCKS
+#define MAX_AUX_CLOCKS 8
+#endif
+
+#ifndef CLOCK_AUX_LAST
+#define CLOCK_AUX_LAST (CLOCK_AUX + MAX_AUX_CLOCKS - 1)
+#endif
+
+__attribute__((unused))
+static inline const char *clock_name(clockid_t clockid)
+{
+ switch (clockid) {
+ case CLOCK_REALTIME:
+ return "CLOCK_REALTIME";
+ case CLOCK_MONOTONIC:
+ return "CLOCK_MONOTONIC";
+ case CLOCK_PROCESS_CPUTIME_ID:
+ return "CLOCK_PROCESS_CPUTIME_ID";
+ case CLOCK_THREAD_CPUTIME_ID:
+ return "CLOCK_THREAD_CPUTIME_ID";
+ case CLOCK_MONOTONIC_RAW:
+ return "CLOCK_MONOTONIC_RAW";
+ case CLOCK_REALTIME_COARSE:
+ return "CLOCK_REALTIME_COARSE";
+ case CLOCK_MONOTONIC_COARSE:
+ return "CLOCK_MONOTONIC_COARSE";
+ case CLOCK_BOOTTIME:
+ return "CLOCK_BOOTTIME";
+ case CLOCK_REALTIME_ALARM:
+ return "CLOCK_REALTIME_ALARM";
+ case CLOCK_BOOTTIME_ALARM:
+ return "CLOCK_BOOTTIME_ALARM";
+ case CLOCK_TAI:
+ return "CLOCK_TAI";
+ case CLOCK_AUX + 0:
+ return "CLOCK_AUX0";
+ case CLOCK_AUX + 1:
+ return "CLOCK_AUX1";
+ case CLOCK_AUX + 2:
+ return "CLOCK_AUX2";
+ case CLOCK_AUX + 3:
+ return "CLOCK_AUX3";
+ case CLOCK_AUX + 4:
+ return "CLOCK_AUX4";
+ case CLOCK_AUX + 5:
+ return "CLOCK_AUX5";
+ case CLOCK_AUX + 6:
+ return "CLOCK_AUX6";
+ case CLOCK_AUX + 7:
+ return "CLOCK_AUX7";
+ };
+ return "UNKNOWN_CLOCKID";
+}
+
+#endif /* __CLOCK_HELPERS_H */
diff --git a/tools/testing/selftests/cpufreq/governor.sh b/tools/testing/selftests/cpufreq/governor.sh
index fe37df79c087..cf59e63f8e14 100755
--- a/tools/testing/selftests/cpufreq/governor.sh
+++ b/tools/testing/selftests/cpufreq/governor.sh
@@ -16,6 +16,12 @@ source cpufreq.sh
CUR_GOV=
CUR_FREQ=
+# Per-policy backup, keyed by policy so multiple policies can be saved at once
+# (backup_governor/restore_governor also keep CUR_GOV/CUR_FREQ for callers that
+# read them directly).
+declare -A SAVED_GOVERNORS
+declare -A SAVED_FREQS
+
# Find governor's directory path
# $1: policy, $2: governor
find_gov_directory()
@@ -39,11 +45,13 @@ find_current_governor()
backup_governor()
{
CUR_GOV=$(find_current_governor $1)
+ SAVED_GOVERNORS[$1]=$CUR_GOV
printf "Governor backup done for $1: $CUR_GOV\n"
if [ $CUR_GOV == "userspace" ]; then
CUR_FREQ=$(find_current_freq $1)
+ SAVED_FREQS[$1]=$CUR_FREQ
printf "Governor frequency backup done for $1: $CUR_FREQ\n"
fi
@@ -53,11 +61,13 @@ backup_governor()
# $1: policy
restore_governor()
{
+ CUR_GOV=${SAVED_GOVERNORS[$1]}
__switch_governor $1 $CUR_GOV
printf "Governor restored for $1 to $CUR_GOV\n"
if [ $CUR_GOV == "userspace" ]; then
+ CUR_FREQ=${SAVED_FREQS[$1]}
set_cpu_frequency $1 $CUR_FREQ
printf "Governor frequency restored for $1: $CUR_FREQ\n"
fi
@@ -65,6 +75,17 @@ restore_governor()
printf "\n"
}
+# Save/restore governors for every policy at once
+save_all_governors()
+{
+ for_each_policy backup_governor
+}
+
+restore_all_governors()
+{
+ for_each_policy restore_governor
+}
+
# param:
# $1: policy, $2: governor
__switch_governor()
@@ -100,11 +121,6 @@ switch_governor()
# $1: policy, $2: governor
switch_show_governor()
{
- cur_gov=find_current_governor
- if [ $cur_gov == "userspace" ]; then
- cur_freq=find_current_freq
- fi
-
# switch governor
__switch_governor $1 $2
diff --git a/tools/testing/selftests/cpufreq/special-tests.sh b/tools/testing/selftests/cpufreq/special-tests.sh
index 8d40505dc468..e87ed7c8e5e5 100755
--- a/tools/testing/selftests/cpufreq/special-tests.sh
+++ b/tools/testing/selftests/cpufreq/special-tests.sh
@@ -40,7 +40,9 @@ simple_lockdep()
{
printf "** Test: Running ${FUNCNAME[0]} **\n"
+ save_all_governors
for_each_policy __simple_lockdep
+ restore_all_governors
}
# Test 2
@@ -56,7 +58,10 @@ concurrent_lockdep()
{
printf "** Test: Running ${FUNCNAME[0]} **\n"
+ save_all_governors
for_each_policy_concurrent __concurrent_lockdep
+ wait
+ restore_all_governors
}
# Test 3
@@ -65,20 +70,26 @@ quick_shuffle()
# this is called concurrently from governor_race
for I in `seq 1000`
do
- echo ondemand | sudo tee $CPUFREQROOT/policy*/scaling_governor &
- echo userspace | sudo tee $CPUFREQROOT/policy*/scaling_governor &
+ echo ondemand | tee $CPUFREQROOT/policy*/scaling_governor &
+ echo userspace | tee $CPUFREQROOT/policy*/scaling_governor &
done
+ wait
}
governor_race()
{
printf "** Test: Running ${FUNCNAME[0]} **\n"
+ save_all_governors
+
# run 8 concurrent instances
for I in `seq 8`
do
quick_shuffle &
done
+ wait
+
+ restore_all_governors
}
# Test 4
@@ -112,5 +123,8 @@ hotplug_with_updates_cpu()
hotplug_with_updates()
{
+ save_all_governors
for_each_non_boot_cpu hotplug_with_updates_cpu
+ wait
+ restore_all_governors
}
diff --git a/tools/testing/selftests/damon/Makefile b/tools/testing/selftests/damon/Makefile
index 2180c328a825..ece244e5c5b9 100644
--- a/tools/testing/selftests/damon/Makefile
+++ b/tools/testing/selftests/damon/Makefile
@@ -13,6 +13,7 @@ TEST_PROGS += sysfs.py
TEST_PROGS += sysfs_update_schemes_tried_regions_wss_estimation.py
TEST_PROGS += damos_quota.py damos_quota_goal.py damos_apply_interval.py
TEST_PROGS += damos_tried_regions.py damon_nr_regions.py
+TEST_PROGS += sysfs_refresh.py
TEST_PROGS += reclaim.sh lru_sort.sh
# regression tests (reproducers of previously found bugs)
diff --git a/tools/testing/selftests/damon/_damon_sysfs.py b/tools/testing/selftests/damon/_damon_sysfs.py
index 8b12cc048440..e6a2265d721e 100644
--- a/tools/testing/selftests/damon/_damon_sysfs.py
+++ b/tools/testing/selftests/damon/_damon_sysfs.py
@@ -271,7 +271,7 @@ class DamosFilter:
self.type_ = type_
self.matching = matching
self.allow = allow
- self.memcg_path = memcg_path,
+ self.memcg_path = memcg_path
self.addr_start = addr_start
self.addr_end = addr_end
self.target_idx = target_idx
@@ -624,17 +624,23 @@ class DamonCtx:
pause = None
idx = None
- def __init__(self, ops='paddr', monitoring_attrs=DamonAttrs(), targets=[],
- schemes=[], pause=False):
+ def __init__(self, ops='paddr', monitoring_attrs=None, targets=None,
+ schemes=None, pause=False):
self.ops = ops
+ if monitoring_attrs is None:
+ monitoring_attrs = DamonAttrs()
self.monitoring_attrs = monitoring_attrs
self.monitoring_attrs.context = self
+ if targets is None:
+ targets = []
self.targets = targets
for idx, target in enumerate(self.targets):
target.idx = idx
target.context = self
+ if schemes is None:
+ schemes = []
self.schemes = schemes
for idx, scheme in enumerate(self.schemes):
scheme.idx = idx
@@ -692,12 +698,14 @@ class DamonCtx:
class Kdamond:
state = None
pid = None
+ refresh_ms = None
contexts = None
idx = None # index of this kdamond between siblings
kdamonds = None # parent
- def __init__(self, contexts=[]):
+ def __init__(self, contexts=[], refresh_ms=None):
self.contexts = contexts
+ self.refresh_ms = refresh_ms
for idx, context in enumerate(self.contexts):
context.idx = idx
context.kdamond = self
@@ -720,6 +728,11 @@ class Kdamond:
err = context.stage()
if err is not None:
return err
+ if self.refresh_ms is not None:
+ err = write_file(os.path.join(self.sysfs_dir(), 'refresh_ms'),
+ '%d' % self.refresh_ms)
+ if err is not None:
+ return err
err = write_file(os.path.join(self.sysfs_dir(), 'state'), 'on')
if err is not None:
return err
@@ -831,7 +844,7 @@ class Kdamond:
for goal in scheme.quota.goals:
err = goal.stage()
if err is not None:
- print('commit_schemes_quota_goals failed stagign: %s'%
+ print('commit_schemes_quota_goals failed staging: %s'%
err)
exit(1)
return write_file(os.path.join(self.sysfs_dir(), 'state'),
diff --git a/tools/testing/selftests/damon/damos_apply_interval.py b/tools/testing/selftests/damon/damos_apply_interval.py
index f04d43702481..0f2f36584e48 100755
--- a/tools/testing/selftests/damon/damos_apply_interval.py
+++ b/tools/testing/selftests/damon/damos_apply_interval.py
@@ -56,7 +56,7 @@ def main():
# Because the second scheme was having the apply interval that is ten times
# lower than that of the first scheme, the second scheme should be tried
# about ten times more frequently than the first scheme. For possible
- # timing errors, check if it was at least nine times more freuqnetly tried.
+ # timing errors, check if it was at least nine times more frequently tried.
ratio = nr_tried_stats[1] / nr_tried_stats[0]
if ratio < 9:
print('%d / %d = %f (< 9)' %
diff --git a/tools/testing/selftests/damon/damos_quota_goal.py b/tools/testing/selftests/damon/damos_quota_goal.py
index f76e0412b564..661e4ba4765a 100755
--- a/tools/testing/selftests/damon/damos_quota_goal.py
+++ b/tools/testing/selftests/damon/damos_quota_goal.py
@@ -66,7 +66,7 @@ def main():
# effective quota was already minimum that cannot be more reduced
if expect_increase is False and last_effective_bytes == 1:
continue
- print('efective bytes not changed: %d' % goal.effective_bytes)
+ print('effective bytes not changed: %d' % goal.effective_bytes)
exit(1)
increased = last_effective_bytes < goal.effective_bytes
diff --git a/tools/testing/selftests/damon/damos_tried_regions.py b/tools/testing/selftests/damon/damos_tried_regions.py
index 3b347eb28bd2..d6472e6a6e08 100755
--- a/tools/testing/selftests/damon/damos_tried_regions.py
+++ b/tools/testing/selftests/damon/damos_tried_regions.py
@@ -55,10 +55,10 @@ def main():
collected_nr_regions.sort()
sample = collected_nr_regions[4]
print('50-th percentile nr_regions: %d' % sample)
- print('expectation (>= 14) is %s' % 'met' if sample >= 14 else 'not met')
+ print('expectation (>= 14) is %s' % ('met' if sample >= 14 else 'not met'))
if collected_nr_regions[4] < 14:
print('full nr_regions:')
- print('\n'.join(collected_nr_regions))
+ print('\n'.join(['%d' % x for x in collected_nr_regions]))
exit(1)
if __name__ == '__main__':
diff --git a/tools/testing/selftests/damon/drgn_dump_damon_status.py b/tools/testing/selftests/damon/drgn_dump_damon_status.py
index 972948e6215f..09552e91bc78 100755
--- a/tools/testing/selftests/damon/drgn_dump_damon_status.py
+++ b/tools/testing/selftests/damon/drgn_dump_damon_status.py
@@ -59,7 +59,6 @@ def region_to_dict(region):
['ar', addr_range_to_dict],
['sampling_addr', int],
['nr_accesses', int],
- ['nr_accesses_bp', int],
['age', int],
])
@@ -163,7 +162,7 @@ def damos_filter_to_dict(damos_filter):
int(damos_filter.addr_range.end)]
elif type_ == 'target':
dict_['target_idx'] = int(damos_filter.target_idx)
- elif type_ == 'hugeapge_size':
+ elif type_ == 'hugepage_size':
dict_['sz_range'] = [int(damos_filter.sz_range.min),
int(damos_filter.sz_range.max)]
return dict_
diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py
index aa03a1187489..3ffa054b6386 100755
--- a/tools/testing/selftests/damon/sysfs.py
+++ b/tools/testing/selftests/damon/sysfs.py
@@ -119,7 +119,7 @@ def assert_access_pattern_committed(pattern, dump):
'max_nr_accesses', dump)
assert_true(dump['min_age_region'] == pattern.age[0], 'min_age_region',
dump)
- assert_true(dump['max_age_region'] == pattern.age[1], 'miaxage_region',
+ assert_true(dump['max_age_region'] == pattern.age[1], 'max_age_region',
dump)
def assert_scheme_committed(scheme, dump):
@@ -129,7 +129,7 @@ def assert_scheme_committed(scheme, dump):
'cold': 1,
'pageout': 2,
'hugepage': 3,
- 'nohugeapge': 4,
+ 'nohugepage': 4,
'collapse': 5,
'lru_prio': 6,
'lru_deprio': 7,
@@ -250,6 +250,35 @@ def assert_ctxs_committed(kdamonds):
if ctx in ctxs_paused_for_dump:
ctx.pause = False
+def test_memcg_filter_memcg_path_staging():
+ global kdamonds
+ memcg_filter = _damon_sysfs.DamosFilter(
+ type_='memcg', matching=True, allow=True, memcg_path='/')
+ kdamonds = _damon_sysfs.Kdamonds(
+ [_damon_sysfs.Kdamond(
+ contexts=[_damon_sysfs.DamonCtx(
+ targets=[_damon_sysfs.DamonTarget(pid=-1)],
+ schemes=[_damon_sysfs.Damos(
+ ops_filters=[memcg_filter])],
+ )])])
+ kdamonds.start()
+
+ shown, rd_err = _damon_sysfs.read_file(
+ os.path.join(memcg_filter.sysfs_dir(), 'memcg_path'))
+ if rd_err is not None:
+ print('memcg_path staging: sysfs read (%s)' % rd_err)
+ kdamonds.stop()
+ exit(1)
+ if shown.rstrip('\n') != memcg_filter.memcg_path:
+ print('memcg_path staging: memcg_path readback '
+ '(shown=%s, expected=%s)' %
+ (shown.rstrip('\n'), memcg_filter.memcg_path))
+ kdamonds.stop()
+ exit(1)
+
+ kdamonds.stop()
+ kdamonds = None
+
def main():
global kdamonds
kdamonds = _damon_sysfs.Kdamonds(
@@ -356,5 +385,7 @@ def main():
assert_ctxs_committed(kdamonds)
kdamonds.stop()
+ test_memcg_filter_memcg_path_staging()
+
if __name__ == '__main__':
main()
diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh
index 78f4badb5beb..f7fb94b84e71 100755
--- a/tools/testing/selftests/damon/sysfs.sh
+++ b/tools/testing/selftests/damon/sysfs.sh
@@ -3,7 +3,7 @@
source _common.sh
-# Kselftest frmework requirement - SKIP code is 4.
+# Kselftest framework requirement - SKIP code is 4.
ksft_skip=4
ensure_write_succ()
@@ -28,7 +28,7 @@ ensure_write_fail()
if (echo "$content" > "$file") 2> /dev/null
then
- echo "writing $content to $file succeed ($fail_reason)"
+ echo "writing $content to $file succeeded ($reason)"
echo "expected failure because $reason"
exit 1
fi
@@ -67,7 +67,7 @@ ensure_file()
echo "$file permission: expected $permission but $perm"
exit 1
fi
- elif [ "$to_ensure" = "not_exist" ] && [ -f "$dir" ]
+ elif [ "$to_ensure" = "not_exist" ] && [ -f "$file" ]
then
echo "$file is not expected but found"
exit 1
@@ -99,14 +99,55 @@ test_stats()
done
}
+test_dest()
+{
+ dest_dir=$1
+ ensure_file "$dest_dir/id" "exist" "600"
+ ensure_file "$dest_dir/weight" "exist" "600"
+}
+
+test_dests()
+{
+ dests_dir=$1
+ ensure_file "$dests_dir/nr_dests" "exist" "600"
+ ensure_write_succ "$dests_dir/nr_dests" "1" "valid input"
+ test_dest "$dests_dir/0"
+
+ ensure_write_succ "$dests_dir/nr_dests" "2" "valid input"
+ test_dest "$dests_dir/0"
+ test_dest "$dests_dir/1"
+
+ ensure_write_succ "$dests_dir/nr_dests" "0" "valid input"
+ ensure_dir "$dests_dir/0" "not_exist"
+ ensure_dir "$dests_dir/1" "not_exist"
+}
+
test_filter()
{
filter_dir=$1
ensure_file "$filter_dir/type" "exist" "600"
- ensure_write_succ "$filter_dir/type" "anon" "valid input"
- ensure_write_succ "$filter_dir/type" "memcg" "valid input"
- ensure_write_succ "$filter_dir/type" "addr" "valid input"
- ensure_write_succ "$filter_dir/type" "target" "valid input"
+
+ local dir_name=$(basename "$(dirname "$filter_dir")")
+ if [ "$dir_name" = "filters" ] || [ "$dir_name" = "ops_filters" ]
+ then
+ ensure_write_succ "$filter_dir/type" "anon" "valid input"
+ ensure_write_succ "$filter_dir/type" "memcg" "valid input"
+ fi
+ if [ "$dir_name" = "filters" ] || [ "$dir_name" = "core_filters" ]
+ then
+ ensure_write_succ "$filter_dir/type" "addr" "valid input"
+ ensure_write_succ "$filter_dir/type" "target" "valid input"
+ fi
+ if [ "$dir_name" = "core_filters" ]
+ then
+ ensure_write_fail "$filter_dir/type" "anon" "ops type"
+ ensure_write_fail "$filter_dir/type" "memcg" "ops type"
+ fi
+ if [ "$dir_name" = "ops_filters" ]
+ then
+ ensure_write_fail "$filter_dir/type" "addr" "core type"
+ ensure_write_fail "$filter_dir/type" "target" "core type"
+ fi
ensure_write_fail "$filter_dir/type" "foo" "invalid input"
ensure_file "$filter_dir/matching" "exist" "600"
ensure_file "$filter_dir/memcg_path" "exist" "600"
@@ -158,6 +199,20 @@ test_goal()
ensure_dir "$goal_dir" "exist"
ensure_file "$goal_dir/target_value" "exist" "600"
ensure_file "$goal_dir/current_value" "exist" "600"
+ ensure_file "$goal_dir/target_metric" "exist" "600"
+ local fpath="$goal_dir/target_metric"
+ ensure_write_succ "$fpath" "user_input" "valid input"
+ ensure_write_succ "$fpath" "some_mem_psi_us" "valid input"
+ ensure_write_succ "$fpath" "node_mem_used_bp" "valid input"
+ ensure_write_succ "$fpath" "node_mem_free_bp" "valid input"
+ ensure_write_succ "$fpath" "node_memcg_used_bp" "valid input"
+ ensure_write_succ "$fpath" "node_memcg_free_bp" "valid input"
+ ensure_write_succ "$fpath" "active_mem_bp" "valid input"
+ ensure_write_succ "$fpath" "inactive_mem_bp" "valid input"
+ ensure_write_succ "$fpath" "node_eligible_mem_bp" "valid input"
+ ensure_write_fail "$fpath" "foo" "invalid input"
+ ensure_file "$goal_dir/nid" "exist" "600"
+ ensure_file "$goal_dir/path" "exist" "600"
}
test_goals()
@@ -207,7 +262,10 @@ test_scheme()
ensure_file "$scheme_dir/apply_interval_us" "exist" "600"
test_quotas "$scheme_dir/quotas"
test_watermarks "$scheme_dir/watermarks"
+ test_dests "$scheme_dir/dests"
test_filters "$scheme_dir/filters"
+ test_filters "$scheme_dir/core_filters"
+ test_filters "$scheme_dir/ops_filters"
test_stats "$scheme_dir/stats"
test_tried_regions "$scheme_dir/tried_regions"
}
@@ -346,8 +404,13 @@ test_probes()
ensure_write_succ "$probes_dir/nr_probes" "1" "valid input"
test_probe "$probes_dir/0"
+ ensure_write_succ "$probes_dir/nr_probes" "2" "valid input"
+ test_probe "$probes_dir/0"
+ test_probe "$probes_dir/1"
+
ensure_write_succ "$probes_dir/nr_probes" "0" "valid input"
ensure_dir "$probes_dir/0" "not_exist"
+ ensure_dir "$probes_dir/1" "not_exist"
}
test_monitoring_attrs()
@@ -363,7 +426,7 @@ test_context()
{
context_dir=$1
ensure_dir "$context_dir" "exist"
- ensure_file "$context_dir/avail_operations" "exit" 400
+ ensure_file "$context_dir/avail_operations" "exist" 400
ensure_file "$context_dir/operations" "exist" 600
ensure_file "$context_dir/addr_unit" "exist" 600
ensure_file "$context_dir/pause" "exist" 600
diff --git a/tools/testing/selftests/damon/sysfs_refresh.py b/tools/testing/selftests/damon/sysfs_refresh.py
new file mode 100755
index 000000000000..012b7e8f509f
--- /dev/null
+++ b/tools/testing/selftests/damon/sysfs_refresh.py
@@ -0,0 +1,75 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: GPL-2.0
+
+import os
+import subprocess
+import time
+
+import _damon_sysfs
+
+def main():
+ # Continuously access a memory region for far longer than the test needs,
+ # so the kdamond always has a live target to monitor while we poll.
+ sz_region = 10 * 1024 * 1024
+ proc = subprocess.Popen(
+ ['./access_memory', '1', '%d' % sz_region, '60000', 'repeat'])
+
+ # A 'stat' scheme with the default (maximally wide) access pattern matches
+ # every monitored region, so its 'nr_tried' stat increases as the kdamond
+ # runs. refresh_ms should make DAMON update the schemes' stats files under
+ # sysfs on its own, without a manual 'update_schemes_stats' request.
+ kdamond = _damon_sysfs.Kdamond(
+ refresh_ms=100,
+ contexts=[_damon_sysfs.DamonCtx(
+ ops='vaddr',
+ targets=[_damon_sysfs.DamonTarget(pid=proc.pid)],
+ schemes=[_damon_sysfs.Damos(action='stat')],
+ )])
+ kdamonds = _damon_sysfs.Kdamonds([kdamond])
+
+ err = kdamonds.start()
+ if err is not None:
+ # Kernels older than the refresh_ms feature have no such file; treat
+ # that as unsupported rather than a failure.
+ if not os.path.exists(os.path.join(kdamond.sysfs_dir(), 'refresh_ms')):
+ proc.terminate()
+ proc.wait()
+ print('kdamond has no refresh_ms file; skipping')
+ exit(_damon_sysfs.ksft_skip)
+ proc.terminate()
+ proc.wait()
+ print('kdamond start failed: %s' % err)
+ exit(1)
+
+ scheme = kdamond.contexts[0].schemes[0]
+ nr_tried_path = os.path.join(scheme.sysfs_dir(), 'stats', 'nr_tried')
+
+ try:
+ # Poll the stat file directly. We never request an update (e.g.
+ # 'update_schemes_stats'), so 'nr_tried' can become non-zero only
+ # through the periodic refresh that refresh_ms enables.
+ nr_tried = 0
+ deadline = time.monotonic() + 10
+ while time.monotonic() < deadline:
+ if proc.poll() is not None:
+ print('the access_memory target exited unexpectedly')
+ exit(1)
+ content, err = _damon_sysfs.read_file(nr_tried_path)
+ if err is not None:
+ print('reading %s failed: %s' % (nr_tried_path, err))
+ exit(1)
+ nr_tried = int(content)
+ if nr_tried > 0:
+ break
+ time.sleep(0.1)
+ finally:
+ kdamonds.stop()
+ proc.terminate()
+ proc.wait()
+
+ if nr_tried == 0:
+ print('refresh_ms did not auto-update the schemes stats')
+ exit(1)
+
+if __name__ == '__main__':
+ main()
diff --git a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py
index 35c724a63f6c..16fdc6e7fc56 100755
--- a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py
+++ b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py
@@ -7,7 +7,7 @@ import time
import _damon_sysfs
def pass_wss_estimation(sz_region):
- # access two regions of given size, 2 seocnds per each region
+ # access two regions of given size, 2 seconds per each region
proc = subprocess.Popen(
['./access_memory', '2', '%d' % sz_region, '2000', 'repeat'])
kdamonds = _damon_sysfs.Kdamonds([_damon_sysfs.Kdamond(
diff --git a/tools/testing/selftests/dax/Makefile b/tools/testing/selftests/dax/Makefile
new file mode 100644
index 000000000000..25a4f3d73a5b
--- /dev/null
+++ b/tools/testing/selftests/dax/Makefile
@@ -0,0 +1,6 @@
+# SPDX-License-Identifier: GPL-2.0
+all:
+
+TEST_PROGS := dax-kmem-hotplug.sh
+
+include ../lib.mk
diff --git a/tools/testing/selftests/dax/config b/tools/testing/selftests/dax/config
new file mode 100644
index 000000000000..4c9aaeb6ceb4
--- /dev/null
+++ b/tools/testing/selftests/dax/config
@@ -0,0 +1,4 @@
+CONFIG_DEV_DAX=m
+CONFIG_DEV_DAX_KMEM=m
+CONFIG_MEMORY_HOTPLUG=y
+CONFIG_MEMORY_HOTREMOVE=y
diff --git a/tools/testing/selftests/dax/dax-kmem-hotplug.sh b/tools/testing/selftests/dax/dax-kmem-hotplug.sh
new file mode 100755
index 000000000000..9299120ee509
--- /dev/null
+++ b/tools/testing/selftests/dax/dax-kmem-hotplug.sh
@@ -0,0 +1,317 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Exercise the dax/kmem "state" sysfs attribute:
+# /sys/bus/dax/devices/daxX.Y/state -> unplugged | online | online_kernel | online_movable
+#
+# The test needs a dax device already bound to the kmem driver.
+#
+# This test mutates a device's memory: online/offline cycles migrate any
+# in-use pages, and the optional unbind subtest wedges the device until
+# reboot. The tester must identify the target device and opt into the
+# destructive unbind tests.
+#
+# DAX_KMEM_TEST_DEV=daxX.Y test this specific device
+# DAX_KMEM_TEST_DEV=auto auto-discover the first kmem-bound dax device
+# (best-effort: it may be a device in use!)
+# DAX_KMEM_TEST_UNBIND=1 also run the destructive unbind-while-online test
+#
+# If DAX_KMEM_TEST_DEV is unset the whole test SKIPs.
+#
+# A dax device can be provisioned with the memmap= boot param, e.g.:
+# memmap=2G!4G
+#
+# then, in the booted system:
+#
+# ndctl create-namespace -m devdax -e namespace0.0 -f
+# daxctl reconfigure-device -N -m system-ram dax0.0 # bind kmem
+# DAX_KMEM_TEST_DEV=auto ./dax-kmem-hotplug.sh
+
+# shellcheck disable=SC1091
+DIR="$(dirname "$(readlink -f "$0")")"
+. "$DIR"/../kselftest/ktap_helpers.sh
+
+DAX_BASE=/sys/bus/dax/devices
+MEM_BASE=/sys/devices/system/memory
+
+memtotal_kb() { awk '/^MemTotal:/ {print $2}' /proc/meminfo; }
+get_state() { cat "$HP" 2>/dev/null; }
+# set_state STATE -- write a state to the state attribute; returns the
+# write's exit status (0 = accepted by the kernel)
+set_state() { echo "$1" > "$HP" 2>/dev/null; }
+
+is_kmem_dax() {
+ local drv
+ [ -e "$DAX_BASE/$1/state" ] || return 1
+ drv=$(readlink "$DAX_BASE/$1/driver" 2>/dev/null)
+ [ "$(basename "${drv:-}")" = kmem ]
+}
+
+find_kmem_dax() {
+ local d
+ for d in "$DAX_BASE"/dax*; do
+ is_kmem_dax "$(basename "$d")" || continue
+ basename "$d"
+ return 0
+ done
+ return 1
+}
+
+# find_device_blocks -- print every memoryN block backing this dax device.
+# The blocks are derived from the device's own range(s) in /proc/iomem (the
+# reserved resource is named after the device), so we act on *its* blocks
+# rather than guessing by NUMA node - the target node may also hold unrelated
+# (and non-offlineable) memory.
+find_device_blocks() {
+ local bs
+ bs=$(cat "$MEM_BASE/block_size_bytes" 2>/dev/null) # hex, no leading 0x
+ [ -n "$bs" ] || return 1
+ grep -E " : ${DAX}\$" /proc/iomem | while read -r line; do
+ local range s e i
+ range=${line%% :*}; range=${range// /}
+ s=${range%-*}; e=${range#*-}
+ for (( i = 0x$s / 0x$bs; i <= 0x$e / 0x$bs; i++ )); do
+ echo "memory$i"
+ done
+ done
+}
+
+# find_device_block -- print the first online block backing this dax device.
+find_device_block() {
+ local b
+ for b in $(find_device_blocks); do
+ [ -f "$MEM_BASE/$b/state" ] || continue
+ [ "$(cat "$MEM_BASE/$b/state")" = online ] || continue
+ echo "$b"
+ return 0
+ done
+ return 1
+}
+
+ktap_print_header
+
+if [ "$UID" != 0 ]; then
+ ktap_skip_all "must be run as root"
+ exit "$KSFT_SKIP"
+fi
+
+# Device selection is opt-in - see the header for why.
+DEV_SEL=${DAX_KMEM_TEST_DEV:-}
+if [ -z "$DEV_SEL" ]; then
+ ktap_skip_all "set DAX_KMEM_TEST_DEV=<daxX.Y|auto> to opt in (mutates device memory)"
+ exit "$KSFT_SKIP"
+fi
+if [ "$DEV_SEL" = auto ]; then
+ DAX=$(find_kmem_dax)
+else
+ DAX=$DEV_SEL
+fi
+if [ -z "$DAX" ] || ! is_kmem_dax "$DAX"; then
+ ktap_skip_all "no kmem-bound dax device with a state attribute (${DEV_SEL})"
+ exit "$KSFT_SKIP"
+fi
+HP=$DAX_BASE/$DAX/state
+ORIG=$(get_state)
+
+# A failure to reach the baseline is environmental (memory in use), not an
+# interface failure, so skip rather than fail.
+set_state unplugged; rc=$?
+if [ "$rc" != 0 ] || [ "$(get_state)" != unplugged ]; then
+ ktap_skip_all "$DAX: cannot reach 'unplugged' baseline (memory in use?)"
+ [ -n "$ORIG" ] && set_state "$ORIG"
+ exit "$KSFT_SKIP"
+fi
+mt_unplugged=$(memtotal_kb)
+
+DRV=/sys/bus/dax/drivers/kmem
+AOB=$MEM_BASE/auto_online_blocks
+
+ktap_print_msg "using $DAX (initial state was: $ORIG)"
+ktap_set_plan 10
+
+# A public (N_MEMORY) kmem node onlined into a kernel zone (online/online_kernel)
+# collects unmovable allocations and can then never be offlined, which would
+# wedge the device for the rest of this test. So this test only ever
+# successfully onlines online_movable, the one mode that is reliably unpluggable.
+
+set_state online_movable; rc=$?
+mt_online=$(memtotal_kb)
+if [ "$rc" = 0 ] && [ "$(get_state)" = online_movable ] && [ "$mt_online" -gt "$mt_unplugged" ]; then
+ ktap_test_pass "online_movable: state=online_movable, MemTotal $mt_unplugged -> $mt_online kB"
+else
+ ktap_test_fail "online_movable: rc=$rc state=$(get_state) MemTotal $mt_unplugged -> $mt_online"
+fi
+
+set_state online_movable; rc=$?
+if [ "$rc" = 0 ] && [ "$(get_state)" = online_movable ]; then
+ ktap_test_pass "online_movable idempotent"
+else
+ ktap_test_fail "online_movable idempotent: rc=$rc state=$(get_state)"
+fi
+
+# A different online type is rejected without an intervening unplug. The write
+# is refused before any hotplug, so this never actually onlines a kernel zone.
+set_state online_kernel; rc=$?
+if [ "$rc" != 0 ] && [ "$(get_state)" = online_movable ]; then
+ ktap_test_pass "reject online_kernel without intervening unplug (no kernel-zone online)"
+else
+ ktap_test_fail "online_movable->online_kernel not rejected: rc=$rc state=$(get_state)"
+fi
+
+set_state unplugged; rc=$?
+mt=$(memtotal_kb)
+if [ "$rc" = 0 ] && [ "$(get_state)" = unplugged ] && [ "$mt" -lt "$mt_online" ]; then
+ ktap_test_pass "unplug from online_movable: MemTotal $mt_online -> $mt kB"
+else
+ ktap_test_fail "unplug from online_movable: rc=$rc state=$(get_state) MemTotal $mt_online -> $mt"
+fi
+
+before=$(get_state)
+set_state bogus_state; rc=$?
+if [ "$rc" != 0 ] && [ "$(get_state)" = "$before" ]; then
+ ktap_test_pass "reject invalid state string"
+else
+ ktap_test_fail "invalid state not rejected: rc=$rc state=$(get_state)"
+fi
+
+# An online_movable -> unplug cycle must re-acquire the per-range resources on
+# each online and release them on each unplug. Assert every iteration grows
+# MemTotal past the baseline and returns exactly to it; memory left online after
+# unplug (off > baseline) is a partial-free failure.
+set_state unplugged
+cycle_ok=1; fail_i=0; on=0; off=0
+for i in 1 2 3; do
+ if ! set_state online_movable; then cycle_ok=0; fail_i=$i; break; fi
+ on=$(memtotal_kb)
+ if ! set_state unplugged; then cycle_ok=0; fail_i=$i; break; fi
+ off=$(memtotal_kb)
+ # online must grow past baseline, and unplug must return to it - a
+ # partial free (memory left online) is a failure, not just off == on.
+ if [ "$on" -le "$mt_unplugged" ] || [ "$off" -gt "$mt_unplugged" ]; then
+ cycle_ok=0; fail_i=$i; break
+ fi
+done
+if [ "$cycle_ok" = 1 ]; then
+ ktap_test_pass "online_movable/unplug cycle re-acquires resources (3x: added and freed each time)"
+else
+ ktap_test_fail "online_movable/unplug cycle regressed at iteration $fail_i (on=$on off=$off baseline=$mt_unplugged)"
+fi
+
+# Desync: toggle a block through the legacy per-block memoryN/state interface
+# behind the driver's back, then unplug the whole device via daxX.Y/state.
+#
+# The driver only updates daxX.Y/state on its own writes, so it still reports
+# online_movable while a block underneath is already offline.
+#
+# Whole-device unplug must still succeed (within reason, an actor changing a
+# device from online_movable to online_kernel can no longer guarantee unplug).
+# At the very least, an already-offline block should not produce an error.
+set_state unplugged
+set_state online_movable
+blk=$(find_device_block)
+if [ -n "$blk" ] && echo offline > "$MEM_BASE/$blk/state" 2>/dev/null; then
+ # daxX.Y/state is now stale (still online_movable); unplug the device.
+ set_state unplugged; rc=$?
+ mt=$(memtotal_kb)
+ if [ "$rc" = 0 ] && [ "$(get_state)" = unplugged ] && [ "$mt" -le "$mt_unplugged" ]; then
+ ktap_test_pass "unplug tolerates a block pre-offlined via memoryN/state ($blk)"
+ else
+ ktap_test_fail "desync unplug: rc=$rc state=$(get_state) MemTotal=$mt baseline=$mt_unplugged"
+ fi
+else
+ set_state unplugged 2>/dev/null
+ ktap_test_skip "could not locate a device block to offline for desync test"
+fi
+
+# change system default online policy while the device is unbound, and show
+# the new system default policy is utilized across bindings.
+set_state unplugged
+if [ -w "$AOB" ] && [ -w "$DRV/unbind" ] && [ -w "$DRV/bind" ]; then
+ orig_aob=$(cat "$AOB")
+ echo "$DAX" > "$DRV/unbind" 2>/dev/null
+ echo offline > "$AOB" 2>/dev/null
+ echo "$DAX" > "$DRV/bind" 2>/dev/null
+ sleep 1
+ st=$(get_state)
+ echo "$orig_aob" > "$AOB" 2>/dev/null # restore system policy
+ if [ "$st" = offline ]; then
+ ktap_test_pass "online policy resolved at bind: auto_online_blocks=offline -> state=offline"
+ else
+ ktap_test_fail "bind-time policy not honored: state=$st (expected offline)"
+ fi
+ set_state unplugged 2>/dev/null
+else
+ ktap_test_skip "auto_online_blocks or driver bind/unbind not writable"
+fi
+
+# Blocks offlined out-of-band (via memoryN/state) leave daxX.Y/state stale
+# (still online_movable) while every block is actually offline. A driver unbind
+# must still hot-remove the offline memory and free its resources rather than
+# trust the stale state and leak until reboot. Unbind uses remove_memory(),
+# which never offlines, so removing already-offline blocks is non-destructive and
+# the device rebinds cleanly afterwards.
+if [ -w "$DRV/unbind" ] && [ -w "$DRV/bind" ]; then
+ set_state unplugged
+ set_state online_movable
+ offl_ok=1
+ for b in $(find_device_blocks); do
+ [ -f "$MEM_BASE/$b/state" ] || continue
+ [ "$(cat "$MEM_BASE/$b/state")" = online ] || continue
+ echo offline > "$MEM_BASE/$b/state" 2>/dev/null || offl_ok=0
+ done
+ # daxX.Y/state is now stale (still online_movable) while all blocks are
+ # offline; the unbind must hot-remove them anyway.
+ if [ "$offl_ok" = 1 ] && [ "$(get_state)" = online_movable ]; then
+ echo "$DAX" > "$DRV/unbind" 2>/dev/null
+ mt_after=$(memtotal_kb)
+ leaked=$(grep -cE " : ${DAX}\$" /proc/iomem) # before rebind
+ echo "$DAX" > "$DRV/bind" 2>/dev/null # restore for later steps
+ sleep 1
+ if [ "$mt_after" -le "$mt_unplugged" ] && [ "$leaked" = 0 ]; then
+ ktap_test_pass "unbind with stale online state hot-removes offlined blocks (no leak)"
+ else
+ ktap_test_fail "desync unbind leaked: MemTotal=$mt_after baseline=$mt_unplugged iomem_left=$leaked"
+ fi
+ set_state unplugged 2>/dev/null
+ else
+ ktap_test_skip "could not offline all device blocks for desync-unbind test"
+ fi
+else
+ ktap_test_skip "driver bind/unbind not writable for desync-unbind test"
+fi
+
+[ -n "$ORIG" ] && set_state "$ORIG"
+
+# DESTRUCTIVE and opt-in only (DAX_KMEM_TEST_UNBIND=1):
+#
+# unbinding the driver while memory is online causes the resources to leak - but
+# the unbind should not deadlock. Instead the driver leaks it with a warning.
+
+# This leaves the memory online and the device unbound until reboot, so it runs
+# last and only when explicitly requested. online_movable only: this test
+# never onlines a public node into a kernel zone.
+
+if [ "${DAX_KMEM_TEST_UNBIND:-}" = 1 ] && [ -w "$DRV/unbind" ]; then
+ set_state unplugged; set_state online_movable
+fi
+if [ "${DAX_KMEM_TEST_UNBIND:-}" = 1 ] && [ "$(get_state)" = online_movable ] &&
+ [ -w "$DRV/unbind" ]; then
+ mt_on=$(memtotal_kb)
+ dmesg -C 2>/dev/null
+ echo "$DAX" > "$DRV/unbind" 2>/dev/null
+ mt_after=$(memtotal_kb)
+ # The leaked "System RAM (kmem)" regions stay in the iomem tree; reading
+ # their names dereferences res_name, which a buggy unbind already freed.
+ # Walk /proc/iomem to provoke that use-after-free (caught by KASAN).
+ cat /proc/iomem > /dev/null 2>&1
+ splat=$(dmesg 2>/dev/null | grep -ciE "KASAN|BUG:|use-after-free|general protection|Oops|refcount_t")
+ if [ "$splat" = 0 ] && [ "$mt_after" -ge "$mt_on" ]; then
+ ktap_test_pass "unbind while online: memory left online, no UAF/oops (MemTotal $mt_on -> $mt_after kB)"
+ else
+ ktap_test_fail "unbind while online regressed: splat=$splat MemTotal $mt_on -> $mt_after kB"
+ fi
+else
+ ktap_test_skip "destructive unbind-while-online test (set DAX_KMEM_TEST_UNBIND=1)"
+fi
+
+ktap_finished
diff --git a/tools/testing/selftests/dax/settings b/tools/testing/selftests/dax/settings
new file mode 100644
index 000000000000..ba4d85f74cd6
--- /dev/null
+++ b/tools/testing/selftests/dax/settings
@@ -0,0 +1 @@
+timeout=90
diff --git a/tools/testing/selftests/dmabuf-heaps/dmabuf-heap.c b/tools/testing/selftests/dmabuf-heaps/dmabuf-heap.c
index fc9694fc4e89..1d49df671919 100644
--- a/tools/testing/selftests/dmabuf-heaps/dmabuf-heap.c
+++ b/tools/testing/selftests/dmabuf-heaps/dmabuf-heap.c
@@ -390,6 +390,116 @@ static void test_alloc_errors(char *heap_name)
close(heap_fd);
}
+/*
+ * count_open_fds - return the number of open file descriptors.
+ *
+ * The fd opened by opendir() itself is counted, but since it is opened
+ * and closed within each call, it cancels out when comparing two counts.
+ * Returns -1 on error.
+ */
+static int count_open_fds(void)
+{
+ DIR *d = opendir("/proc/self/fd");
+ struct dirent *de;
+ int count = 0;
+
+ if (!d)
+ return -1;
+
+ while ((de = readdir(d)))
+ if (de->d_name[0] != '.')
+ count++;
+ closedir(d);
+ return count;
+}
+
+/*
+ * test_alloc_no_fd_leak_on_efault - verify no fd is leaked when
+ * copy_to_user() fails during DMA_HEAP_IOCTL_ALLOC.
+ *
+ * The bug: dma_buf_fd() called fd_install() before copy_to_user().
+ * If copy_to_user() then failed (e.g. via mprotect), the fd was
+ * silently installed in the fd table but never returned to userspace.
+ *
+ * The fix: reserve the fd with get_unused_fd_flags() first, attempt
+ * copy_to_user(), and only call fd_install() on success.
+ *
+ * We trigger the failure by placing the ioctl argument in a private
+ * anonymous page and flipping it to PROT_READ before the ioctl.
+ * Inside the kernel, copy_from_user() reads from the page (reads are
+ * allowed under PROT_READ, so it succeeds), but copy_to_user() that
+ * writes the fd number back faults, returning -EFAULT. We then
+ * count open file descriptors before and after; with the bug an extra
+ * fd is left in the table.
+ */
+static void test_alloc_no_fd_leak_on_efault(char *heap_name)
+{
+ int heap_fd = -1;
+ int fd_before, fd_after;
+ int ret;
+ long page_size;
+ struct dma_heap_allocation_data *req;
+
+ ksft_print_msg("Testing fd leak when copy_to_user() fails:\n");
+
+ heap_fd = dmabuf_heap_open(heap_name);
+
+ page_size = sysconf(_SC_PAGESIZE);
+
+ /*
+ * Place the ioctl argument in its own private anonymous page so
+ * we can flip its protection independently.
+ */
+ req = mmap(NULL, page_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (req == MAP_FAILED) {
+ ksft_test_result_fail("mmap failed: %s\n", strerror(errno));
+ goto out;
+ }
+
+ memset(req, 0, sizeof(*req));
+ req->len = page_size;
+ req->fd_flags = O_RDWR | O_CLOEXEC;
+
+ fd_before = count_open_fds();
+ if (fd_before < 0) {
+ ksft_test_result_fail("count_open_fds: %s\n", strerror(errno));
+ munmap(req, page_size);
+ goto out;
+ }
+
+ /*
+ * Make the page read-only so copy_to_user() will fault. The
+ * ioctl must fail with -1; if it returns success the test setup
+ * is broken (mprotect is synchronous, so there is no race).
+ */
+ mprotect(req, page_size, PROT_READ);
+
+ ret = ioctl(heap_fd, DMA_HEAP_IOCTL_ALLOC, req);
+
+ /* Re-allow writes so munmap can clean up */
+ mprotect(req, page_size, PROT_READ | PROT_WRITE);
+ munmap(req, page_size);
+
+ if (ret != -1) {
+ ksft_test_result_fail("ioctl returned %d, expected -1 EFAULT\n",
+ ret);
+ goto out;
+ }
+
+ fd_after = count_open_fds();
+ if (fd_after < 0) {
+ ksft_test_result_fail("count_open_fds: %s\n", strerror(errno));
+ goto out;
+ }
+
+ ksft_test_result(fd_before == fd_after,
+ "fd leak on EFAULT: before=%d after=%d\n",
+ fd_before, fd_after);
+out:
+ close(heap_fd);
+}
+
static int numer_of_heaps(void)
{
DIR *d = opendir(DEVPATH);
@@ -420,7 +530,7 @@ int main(void)
return KSFT_SKIP;
}
- ksft_set_plan(11 * numer_of_heaps());
+ ksft_set_plan(12 * numer_of_heaps());
while ((dir = readdir(d))) {
if (!strncmp(dir->d_name, ".", 2))
@@ -435,6 +545,7 @@ int main(void)
test_alloc_zeroed(dir->d_name, ONE_MEG);
test_alloc_compat(dir->d_name);
test_alloc_errors(dir->d_name);
+ test_alloc_no_fd_leak_on_efault(dir->d_name);
}
closedir(d);
diff --git a/tools/testing/selftests/drivers/net/bonding/bond_topo_2d1c.sh b/tools/testing/selftests/drivers/net/bonding/bond_topo_2d1c.sh
index 167aa4a4a12a..903c7a6c7287 100644
--- a/tools/testing/selftests/drivers/net/bonding/bond_topo_2d1c.sh
+++ b/tools/testing/selftests/drivers/net/bonding/bond_topo_2d1c.sh
@@ -48,7 +48,7 @@ gateway_create()
ip -n ${g_ns} link add br0 type bridge
ip -n ${g_ns} link set br0 up
ip -n ${g_ns} addr add ${g_ip4}/24 dev br0
- ip -n ${g_ns} addr add ${g_ip6}/24 dev br0
+ ip -n ${g_ns} addr add ${g_ip6}/24 dev br0 nodad
}
gateway_destroy()
@@ -75,7 +75,7 @@ server_create()
ip -n ${s_ns} link set bond0 up
ip -n ${s_ns} addr add ${s_ip4}/24 dev bond0
- ip -n ${s_ns} addr add ${s_ip6}/24 dev bond0
+ ip -n ${s_ns} addr add ${s_ip6}/24 dev bond0 nodad
}
# Reset bond with new mode and options
@@ -97,9 +97,7 @@ bond_reset()
ip -n ${s_ns} link set bond0 up
ip -n ${s_ns} addr add ${s_ip4}/24 dev bond0
- ip -n ${s_ns} addr add ${s_ip6}/24 dev bond0
- # Wait for IPv6 address ready as it needs DAD
- slowwait 2 ip netns exec ${s_ns} ping6 ${c_ip6} -c 1 -W 0.1 &> /dev/null
+ ip -n ${s_ns} addr add ${s_ip6}/24 dev bond0 nodad
}
server_destroy()
@@ -124,7 +122,7 @@ client_create()
ip -n ${c_ns} link set eth0 up
ip -n ${c_ns} addr add ${c_ip4}/24 dev eth0
- ip -n ${c_ns} addr add ${c_ip6}/24 dev eth0
+ ip -n ${c_ns} addr add ${c_ip6}/24 dev eth0 nodad
}
client_destroy()
diff --git a/tools/testing/selftests/drivers/net/config b/tools/testing/selftests/drivers/net/config
index 2070e890e064..4838adf27fa1 100644
--- a/tools/testing/selftests/drivers/net/config
+++ b/tools/testing/selftests/drivers/net/config
@@ -4,8 +4,12 @@ CONFIG_DEBUG_INFO_BTF_MODULES=n
CONFIG_INET_PSP=y
CONFIG_IPV6=y
CONFIG_MACSEC=m
+CONFIG_NET_ACT_SKBEDIT=m
CONFIG_NET_CLS_ACT=y
CONFIG_NET_CLS_BPF=y
+CONFIG_NET_CLS_FLOWER=m
+CONFIG_NET_CLS_FW=m
+CONFIG_NET_CLS_MATCHALL=m
CONFIG_NETCONSOLE=m
CONFIG_NETCONSOLE_DYNAMIC=y
CONFIG_NETCONSOLE_EXTENDED_LOG=y
@@ -14,7 +18,9 @@ CONFIG_NETKIT=y
CONFIG_NET_SCH_ETF=m
CONFIG_NET_SCH_FQ=m
CONFIG_NET_SCH_INGRESS=y
+CONFIG_NET_SCH_PRIO=m
CONFIG_PPP=y
CONFIG_PPPOE=y
+CONFIG_TLS=y
CONFIG_VLAN_8021Q=m
CONFIG_XDP_SOCKETS=y
diff --git a/tools/testing/selftests/drivers/net/hw/Makefile b/tools/testing/selftests/drivers/net/hw/Makefile
index fd0535a96d84..78bb0169350b 100644
--- a/tools/testing/selftests/drivers/net/hw/Makefile
+++ b/tools/testing/selftests/drivers/net/hw/Makefile
@@ -19,7 +19,7 @@ TEST_GEN_FILES := \
TEST_PROGS = \
csum.py \
- devlink_port_split.py \
+ devlink_rate_cross_esw.py \
devlink_rate_tc_bw.py \
devmem.py \
ethtool.sh \
@@ -53,6 +53,10 @@ TEST_PROGS = \
xsk_reconfig.py \
#
+TEST_PROGS_EXTENDED := \
+ devlink_port_split.py \
+# end of TEST_PROGS_EXTENDED
+
TEST_FILES := \
devmem_lib.py \
ethtool_lib.sh \
diff --git a/tools/testing/selftests/drivers/net/hw/config b/tools/testing/selftests/drivers/net/hw/config
index ed8642b68094..d89a9ba17655 100644
--- a/tools/testing/selftests/drivers/net/hw/config
+++ b/tools/testing/selftests/drivers/net/hw/config
@@ -15,11 +15,13 @@ CONFIG_IPV6_SIT=y
CONFIG_IPV6_TUNNEL=y
CONFIG_NET_CLS_ACT=y
CONFIG_NET_CLS_BPF=y
+CONFIG_NET_DEVMEM=y
CONFIG_NET_IPGRE=y
CONFIG_NET_IPGRE_DEMUX=y
CONFIG_NET_IPIP=y
CONFIG_NETKIT=y
CONFIG_NET_SCH_INGRESS=y
+CONFIG_SYNC_FILE=y
CONFIG_UDMABUF=y
CONFIG_USER_NS=y
CONFIG_VXLAN=y
diff --git a/tools/testing/selftests/drivers/net/hw/csum.py b/tools/testing/selftests/drivers/net/hw/csum.py
index 3e3a89a34afe..0e99198f8d39 100755
--- a/tools/testing/selftests/drivers/net/hw/csum.py
+++ b/tools/testing/selftests/drivers/net/hw/csum.py
@@ -1,7 +1,7 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: GPL-2.0
-"""Run the tools/testing/selftests/net/csum testsuite."""
+"""Run the tools/testing/selftests/net/lib/csum testsuite."""
from os import path
diff --git a/tools/testing/selftests/drivers/net/hw/devlink_rate_cross_esw.py b/tools/testing/selftests/drivers/net/hw/devlink_rate_cross_esw.py
new file mode 100755
index 000000000000..4416f024cb76
--- /dev/null
+++ b/tools/testing/selftests/drivers/net/hw/devlink_rate_cross_esw.py
@@ -0,0 +1,296 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: GPL-2.0
+
+"""
+Devlink Rate Cross-eswitch Scheduling Test Suite
+==================================================
+
+Control-plane tests for cross-eswitch TX scheduling via devlink-rate.
+Validates that VFs from different PFs on the same chip can share
+rate groups using the cross-device parent-dev attribute.
+
+Preconditions:
+- NETIF points to a bond device with exactly two interfaces.
+- the interfaces must be two PFs from different devices sharing the same chip.
+- (for mlx5): the two interfaces are in switchdev mode and configured in a LAG:
+ - devlink dev eswitch set $DEV1 mode switchdev
+ - devlink dev eswitch set $DEV2 mode switchdev
+ - devlink dev param set $DEV1 name esw_multiport value 1 cmode runtime
+ - devlink dev param set $DEV2 name esw_multiport value 1 cmode runtime
+- test cases will be skipped if:
+ - the number of interfaces in the bond device is != 2.
+ - the kernel doesn't support devlink rates.
+ - the devlink API doesn't support cross-device parents (ENODEV).
+ - cross-esw rate scheduling returns EOPNOTSUPP.
+"""
+
+import errno
+import glob
+import os
+import time
+
+from lib.py import ksft_pr, ksft_eq, ksft_run, ksft_exit
+from lib.py import KsftSkipEx, KsftFailEx
+from lib.py import NetDrvEnv, DevlinkFamily
+from lib.py import NlError
+from lib.py import cmd, defer, ip, tool
+
+
+# --- Discovery and setup ---
+
+
+def get_bond_slaves(bond_ifname):
+ """Returns sorted list of slave netdev names for a bond."""
+ pattern = f"/sys/class/net/{bond_ifname}/lower_*"
+ lowers = glob.glob(pattern)
+ if not lowers:
+ raise KsftSkipEx(f"No bond slaves for {bond_ifname}")
+ slaves = []
+ for path in sorted(lowers):
+ name = os.path.basename(path)
+ if name.startswith("lower_"):
+ name = name[len("lower_"):]
+ slaves.append(name)
+ return slaves
+
+
+def discover_pfs(cfg):
+ """Discovers both PFs from bond slaves."""
+ slaves = get_bond_slaves(cfg.ifname)
+ if len(slaves) != 2:
+ raise KsftSkipEx(f"Need 2 bond slaves, found {len(slaves)}")
+
+ pf0, pf1 = slaves[0], slaves[1]
+ ksft_pr(f"PF0: {pf0} PF1: {pf1}")
+ return pf0, pf1
+
+
+def get_pci_addr(ifname):
+ """Resolves PCI address for a network interface."""
+ return os.path.basename(os.path.realpath(f"/sys/class/net/{ifname}/device"))
+
+
+def get_vf_port_index(pf_pci):
+ """Finds devlink port-index for vf0 under pf_pci."""
+ ports = tool("devlink", "port show", json=True)["port"]
+ for port_name, props in ports.items():
+ if port_name.startswith(f"pci/{pf_pci}/") and props.get("vfnum") == 0:
+ return int(port_name.split("/")[-1])
+ raise KsftSkipEx(f"VF port not found for {pf_pci}")
+
+
+def cleanup_esw(pf):
+ """Removes VFs if created by tests."""
+ cmd(f"echo 0 > /sys/class/net/{pf}/device/sriov_numvfs", shell=True, fail=False)
+
+
+def setup_esw(pf):
+ """Creates 1 VF on 'pf'."""
+ path = f"/sys/class/net/{pf}/device/sriov_numvfs"
+ cmd(f"echo 0 > {path}", shell=True)
+ cmd(f"echo 1 > {path}", shell=True)
+ defer(cleanup_esw, pf)
+ time.sleep(2)
+
+ vf_dir = f"/sys/class/net/{pf}/device/virtfn0/net"
+ entries = os.listdir(vf_dir) if os.path.isdir(vf_dir) else []
+ if not entries:
+ raise KsftSkipEx(f"VF not found for {pf}")
+ ip(f"link set dev {entries[0]} up")
+
+ pf_pci = get_pci_addr(pf)
+ vf_idx = get_vf_port_index(pf_pci)
+ ksft_pr(f"Created VF {vf_idx} on PF {pf} ({pf_pci})")
+ return pf_pci, vf_idx
+
+
+# --- Rate operation helpers ---
+
+
+def rate_new(devnl, dev_pci, node_name, **kwargs):
+ """Creates rate node."""
+ params = {
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "rate-node-name": node_name,
+ }
+ params.update(kwargs)
+ try:
+ devnl.rate_new(params)
+ except NlError as e:
+ if e.error == errno.EOPNOTSUPP:
+ raise KsftSkipEx("rate_new not supported") from e
+ raise KsftFailEx("rate_new failed") from e
+
+
+def rate_get(devnl, dev_pci, node_name):
+ """Gets rate node."""
+ params = {
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "rate-node-name": node_name,
+ }
+ return devnl.rate_get(params)
+
+
+def rate_get_leaf(devnl, dev_pci, port_index):
+ """Gets rate leaf (VF)."""
+ params = {
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "port-index": port_index,
+ }
+ return devnl.rate_get(params)
+
+
+def rate_del(devnl, dev_pci, node_name):
+ """Deletes rate node."""
+ devnl.rate_del({
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "rate-node-name": node_name,
+ })
+
+
+def rate_set_leaf(devnl, dev_pci, port_index, **kwargs):
+ """Sets rate attributes on a leaf (VF)."""
+ params = {
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "port-index": port_index,
+ }
+ params.update(kwargs)
+ try:
+ devnl.rate_set(params)
+ except NlError as e:
+ if e.error == errno.EOPNOTSUPP:
+ raise KsftSkipEx("rate_set not supported") from e
+ raise KsftFailEx("rate_set failed") from e
+
+
+def rate_set_leaf_parent(devnl, dev_pci, port_index,
+ parent_name, parent_dev_pci=None):
+ """Sets a leaf's parent, optionally cross-esw."""
+ params = {
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "port-index": port_index,
+ "rate-parent-node-name": parent_name,
+ }
+ if parent_dev_pci:
+ params["parent-dev"] = {
+ "bus-name": "pci",
+ "dev-name": parent_dev_pci,
+ }
+ try:
+ devnl.rate_set(params)
+ except NlError as e:
+ if e.error == errno.EOPNOTSUPP:
+ raise KsftSkipEx("rate_set not supported") from e
+ if parent_dev_pci and e.error == errno.ENODEV:
+ raise KsftSkipEx("Cross-esw scheduling not supported") from e
+ raise KsftFailEx("rate_set failed") from e
+
+
+def rate_clear_leaf_parent(devnl, dev_pci, port_index):
+ """Clears a leaf's parent."""
+ rate_set_leaf_parent(devnl, dev_pci, port_index, "")
+
+
+def rate_set_node(devnl, dev_pci, node_name, **kwargs):
+ """Sets rate attributes on a node."""
+ params = {
+ "bus-name": "pci",
+ "dev-name": dev_pci,
+ "rate-node-name": node_name,
+ }
+ params.update(kwargs)
+ devnl.rate_set(params)
+
+
+# --- Test cases ---
+
+
+def test_same_esw_parent(cfg):
+ """Assigns PF0's VF to PF0's group (same esw baseline)."""
+ pf0, _ = discover_pfs(cfg)
+ pf0_pci, vf0_idx = setup_esw(pf0)
+
+ rate_new(cfg.devnl, pf0_pci, "group0")
+ defer(rate_del, cfg.devnl, pf0_pci, "group0")
+ ksft_pr("rate-new succeeded")
+
+ rate_set_leaf_parent(cfg.devnl, pf0_pci, vf0_idx, "group0")
+ defer(rate_clear_leaf_parent, cfg.devnl, pf0_pci, vf0_idx)
+
+ ksft_pr("Same-esw parent assignment succeeded")
+
+
+def test_cross_esw_parent(cfg):
+ """Sets cross-esw parent, then clear it."""
+ pf0, pf1 = discover_pfs(cfg)
+ pf0_pci, _ = setup_esw(pf0)
+ pf1_pci, vf1_idx = setup_esw(pf1)
+
+ rate_new(cfg.devnl, pf0_pci, "group1")
+ defer(rate_del, cfg.devnl, pf0_pci, "group1")
+ ksft_pr("rate-new succeeded")
+
+ rate_set_leaf_parent(cfg.devnl, pf1_pci, vf1_idx,
+ "group1", parent_dev_pci=pf0_pci)
+ defer(rate_clear_leaf_parent, cfg.devnl, pf1_pci, vf1_idx)
+
+ ksft_pr("Cross-esw parent set and clear succeeded")
+
+
+def test_tx_rates_on_cross_esw(cfg):
+ """Sets tx_max on group and tx_share on leaves in a cross-esw setup."""
+ pf0, pf1 = discover_pfs(cfg)
+ pf0_pci, vf0_idx = setup_esw(pf0)
+ pf1_pci, vf1_idx = setup_esw(pf1)
+
+ rate_new(cfg.devnl, pf0_pci, "group2", **{"rate-tx-max": 10000000})
+ defer(rate_del, cfg.devnl, pf0_pci, "group2")
+ ksft_pr("rate-new succeeded")
+
+ rate_set_leaf_parent(cfg.devnl, pf1_pci, vf1_idx,
+ "group2", parent_dev_pci=pf0_pci)
+ defer(rate_clear_leaf_parent, cfg.devnl, pf1_pci, vf1_idx)
+ ksft_pr("set parent cross-esw succeeded")
+
+ rate_set_leaf_parent(cfg.devnl, pf0_pci, vf0_idx, "group2")
+ defer(rate_clear_leaf_parent, cfg.devnl, pf0_pci, vf0_idx)
+ ksft_pr("set parent same esw succeeded")
+
+ rate_set_leaf(cfg.devnl, pf0_pci, vf0_idx, **{"rate-tx-share": 1000000})
+ rate = rate_get_leaf(cfg.devnl, pf0_pci, vf0_idx)
+ ksft_eq(rate["rate-tx-share"], 1000000)
+ rate_set_leaf(cfg.devnl, pf1_pci, vf1_idx, **{"rate-tx-share": 2000000})
+ rate = rate_get_leaf(cfg.devnl, pf1_pci, vf1_idx)
+ ksft_eq(rate["rate-tx-share"], 2000000)
+ rate_set_node(cfg.devnl, pf0_pci, "group2", **{"rate-tx-max": 250000000})
+ rate = rate_get(cfg.devnl, pf0_pci, "group2")
+ ksft_eq(rate["rate-tx-max"], 250000000)
+
+ ksft_pr("tx_max and tx_share set on cross-esw group")
+
+
+def main() -> None:
+ """Main function."""
+
+ with NetDrvEnv(__file__, nsim_test=False) as cfg:
+ cfg.devnl = DevlinkFamily()
+
+ ksft_run(
+ cases=[
+ test_same_esw_parent,
+ test_cross_esw_parent,
+ test_tx_rates_on_cross_esw,
+ ],
+ args=(cfg,),
+ )
+ ksft_exit()
+
+
+if __name__ == "__main__":
+ main()
diff --git a/tools/testing/selftests/drivers/net/hw/devmem.py b/tools/testing/selftests/drivers/net/hw/devmem.py
index 031cf9905f65..82c11ffc4add 100755
--- a/tools/testing/selftests/drivers/net/hw/devmem.py
+++ b/tools/testing/selftests/drivers/net/hw/devmem.py
@@ -2,7 +2,8 @@
# SPDX-License-Identifier: GPL-2.0
from os import path
-from devmem_lib import setup_test, run_rx, run_tx, run_tx_chunks, run_rx_hds
+from devmem_lib import (setup_test, run_rx, run_tx, run_tx_chunks, run_rx_hds,
+ run_rx_large_niov)
from lib.py import ksft_run, ksft_exit, ksft_disruptive
from lib.py import NetDrvEpEnv
@@ -30,11 +31,17 @@ def check_rx_hds(cfg) -> None:
run_rx_hds(cfg)
+def check_rx_large_niov(cfg) -> None:
+ """Run the devmem RX test with rx-page-size = 16 KiB."""
+ run_rx_large_niov(cfg)
+
+
def main() -> None:
"""Run the devmem test cases."""
with NetDrvEpEnv(__file__) as cfg:
setup_test(cfg, path.abspath(path.dirname(__file__) + "/ncdevmem"))
- ksft_run([check_rx, check_tx, check_tx_chunks, check_rx_hds],
+ ksft_run([check_rx, check_tx, check_tx_chunks, check_rx_hds,
+ check_rx_large_niov],
args=(cfg,))
ksft_exit()
diff --git a/tools/testing/selftests/drivers/net/hw/devmem_lib.py b/tools/testing/selftests/drivers/net/hw/devmem_lib.py
index 0921ff03eb81..3554954a6691 100644
--- a/tools/testing/selftests/drivers/net/hw/devmem_lib.py
+++ b/tools/testing/selftests/drivers/net/hw/devmem_lib.py
@@ -1,6 +1,8 @@
# SPDX-License-Identifier: GPL-2.0
+# pylint: disable=invalid-name,too-many-arguments
"""Shared helpers for devmem TCP selftests."""
+import os
import re
from lib.py import (bkg, cmd, defer, ethtool, rand_port, wait_port_listen,
@@ -8,19 +10,82 @@ from lib.py import (bkg, cmd, defer, ethtool, rand_port, wait_port_listen,
NetdevFamily)
-def require_devmem(cfg):
- """Probe ncdevmem on cfg.ifname and SKIP the test if devmem isn't supported."""
- if not hasattr(cfg, "devmem_probed"):
- probe_command = f"{cfg.bin_local} -f {cfg.ifname}"
- cfg.devmem_supported = cmd(probe_command, fail=False, shell=True).ret == 0
- cfg.devmem_probed = True
+RX_PAGE_SIZE_DEFAULT = 0
+RX_PAGE_SIZE_16K = 16384
+
+PROBE_RX_PAGE_SIZES = (RX_PAGE_SIZE_DEFAULT, RX_PAGE_SIZE_16K)
+
+NR_HUGEPAGES_FILE = "/proc/sys/vm/nr_hugepages"
+
+
+def _is_aligned(value, alignment):
+ """Equivalent of the kernel IS_ALIGNED(value, alignment).
+
+ alignment must be a power of two.
+ """
+ return (value & (alignment - 1)) == 0
+
+
+def _restore_nr_hugepages(nr_hugepages):
+ with open(NR_HUGEPAGES_FILE, 'w', encoding='utf-8') as f:
+ f.write(str(nr_hugepages))
+
+
+def _reserve_hugepages(want=64):
+ """Raise nr_hugepages to @want and arrange for it to be restored."""
+ with open(NR_HUGEPAGES_FILE, 'r+', encoding='utf-8') as f:
+ nr_hugepages = int(f.read().strip())
+ if nr_hugepages >= want:
+ return
+ f.seek(0)
+ f.write(str(want))
+ defer(_restore_nr_hugepages, nr_hugepages)
+
+
+def _probe_devmem(cfg, rx_page_size):
+ """Return True if ncdevmem can bind cfg.ifname at @rx_page_size."""
+ probe_command = f"{cfg.bin_local} -f {cfg.ifname}"
+ if rx_page_size != RX_PAGE_SIZE_DEFAULT:
+ probe_command += f" -b {rx_page_size}"
+ return cmd(probe_command, fail=False, shell=True).ret == 0
+
- if not cfg.devmem_supported:
+def require_devmem(cfg, rx_page_size=RX_PAGE_SIZE_DEFAULT):
+ """Probe ncdevmem on cfg.ifname and SKIP the test if devmem isn't supported."""
+ if rx_page_size not in PROBE_RX_PAGE_SIZES:
+ raise RuntimeError(
+ f"rx-page-size={rx_page_size} is missing from "
+ f"PROBE_RX_PAGE_SIZES, so it was never probed.")
+
+ if not hasattr(cfg, "devmem_supported"):
+ _reserve_hugepages()
+ # Probe every size upfront: in nk tests a leased queue may land in
+ # ncdevmem's queue range and cause the probe to fail.
+ cfg.devmem_supported = {size: _probe_devmem(cfg, size)
+ for size in PROBE_RX_PAGE_SIZES}
+
+ if not cfg.devmem_supported[RX_PAGE_SIZE_DEFAULT]:
raise KsftSkipEx("Test requires devmem support")
+ if rx_page_size != RX_PAGE_SIZE_DEFAULT:
+ page_size = os.sysconf("SC_PAGE_SIZE")
+ if not _is_aligned(rx_page_size, page_size):
+ raise KsftSkipEx(
+ f"rx-page-size={rx_page_size} is invalid for this platform "
+ f"(must be a multiple of PAGE_SIZE={page_size})")
+
+ if not cfg.devmem_supported[rx_page_size]:
+ raise KsftSkipEx(
+ f"Test requires devmem rx-page-size={rx_page_size} support")
+
def configure_nic(cfg):
"""Channels, rings, RSS, queue lease for netkit devmem."""
+ if not hasattr(cfg, "devmem_supported"):
+ raise RuntimeError(
+ "require_devmem() must be called before configure_nic(), which "
+ "may lease a queue away and make later probes fail.")
+
if not hasattr(cfg, 'netns'):
return
@@ -37,14 +102,13 @@ def configure_nic(cfg):
rings = ethnl.rings_get({'header': {'dev-index': cfg.ifindex}})
orig_rx_rings = rings['rx']
orig_hds_thresh = rings.get('hds-thresh', 0)
- orig_data_split = rings.get('tcp-data-split', 'unknown')
ethnl.rings_set({'header': {'dev-index': cfg.ifindex},
'tcp-data-split': 'enabled',
'hds-thresh': 0,
'rx': min(64, orig_rx_rings)})
defer(ethnl.rings_set, {'header': {'dev-index': cfg.ifindex},
- 'tcp-data-split': orig_data_split,
+ 'tcp-data-split': 'unknown',
'hds-thresh': orig_hds_thresh,
'rx': orig_rx_rings})
@@ -76,7 +140,8 @@ def set_flow_rule(cfg, port):
return int(re.search(r'ID (\d+)', output).group(1))
-def ncdevmem_rx(cfg, port, verify=True, fail_on_linear=False, flow_steer=False):
+def ncdevmem_rx(cfg, port, verify=True, fail_on_linear=False, flow_steer=False,
+ rx_page_size=RX_PAGE_SIZE_DEFAULT):
"""Build the ncdevmem RX listener command."""
if hasattr(cfg, 'netns'):
flow_rule_id = set_flow_rule(cfg, port)
@@ -96,6 +161,8 @@ def ncdevmem_rx(cfg, port, verify=True, fail_on_linear=False, flow_steer=False):
extras.append("-v 7")
if fail_on_linear:
extras.append("-L")
+ if rx_page_size != RX_PAGE_SIZE_DEFAULT:
+ extras.append(f"-b {rx_page_size}")
parts = [cfg.bin_local, "-l", f"-f {ifname}", f"-s {addr}",
f"-p {port}", *extras]
@@ -202,6 +269,33 @@ def run_tx_chunks(cfg):
ksft_eq(socat.stdout.strip(), "hello\nworld")
+def run_rx_large_niov(cfg):
+ """Run the devmem RX test with a large niov (rx-page-size > PAGE_SIZE).
+
+ Sweep payload sizes that straddle the niov boundary: below, equal to,
+ and above rx_page_size, to exercise sub-niov, exact-niov, and multi-niov
+ RX paths.
+ """
+ require_devmem(cfg, rx_page_size=RX_PAGE_SIZE_16K)
+ _reserve_hugepages()
+ configure_nic(cfg)
+ netns = getattr(cfg, "netns", None)
+
+ for size in [1024, 4096, 8192, 16384, 32768, 65536]:
+ port = rand_port()
+ socat = socat_send(cfg, port)
+ listen_cmd = ncdevmem_rx(cfg, port,
+ flow_steer=not netns,
+ rx_page_size=RX_PAGE_SIZE_16K)
+ data_pipe = (f"yes $(echo -e \x01\x02\x03\x04\x05\x06) | "
+ f"head -c {size} | {socat}")
+ with bkg(listen_cmd, exit_wait=True, ns=netns) as ncdevmem:
+ wait_port_listen(port, proto="tcp", ns=netns)
+ cmd(data_pipe, host=cfg.remote, shell=True)
+ ksft_eq(ncdevmem.ret, 0,
+ f"large-niov failed for payload size {size}")
+
+
def run_rx_hds(cfg):
"""Run the HDS test by running devmem RX across a segment size sweep."""
require_devmem(cfg)
diff --git a/tools/testing/selftests/drivers/net/hw/ethtool_rmon.sh b/tools/testing/selftests/drivers/net/hw/ethtool_rmon.sh
index 2ec19edddfaa..a074834cbe59 100755
--- a/tools/testing/selftests/drivers/net/hw/ethtool_rmon.sh
+++ b/tools/testing/selftests/drivers/net/hw/ethtool_rmon.sh
@@ -65,6 +65,8 @@ bucket_test()
run_on "$iface" \
"$MZ" "$iface" -q -c "$num_tx" -p "$len" -a own -b bcast -d 10us
+ hw_stats_settle "$iface"
+
after=$(run_on "$iface" ethtool --json -S "$iface" --groups rmon | \
jq -r ".[0].rmon[\"${set}-pktsNtoM\"][$bucket].val")
diff --git a/tools/testing/selftests/drivers/net/hw/ethtool_std_stats.sh b/tools/testing/selftests/drivers/net/hw/ethtool_std_stats.sh
index c085d2a4c989..09f8128c51f3 100755
--- a/tools/testing/selftests/drivers/net/hw/ethtool_std_stats.sh
+++ b/tools/testing/selftests/drivers/net/hw/ethtool_std_stats.sh
@@ -43,10 +43,12 @@ traffic_test()
done
# shellcheck disable=SC2086 # needs split options
- run_on "$iface" "$MZ" "$iface" -q -c "$num_tx" $pkt_format
+ run_on "$iface" "$MZ" "$iface" -q -d 10usec -c "$num_tx" $pkt_format
# shellcheck disable=SC2086 # needs split options
- run_on "$neigh" "$MZ" "$neigh" -q -c "$num_rx" $pkt_format
+ run_on "$neigh" "$MZ" "$neigh" -q -d 10usec -c "$num_rx" $pkt_format
+
+ hw_stats_settle "$int"
for i in "${!counters[@]}"; do
read -r int grp cnt target exact_check xfail_message \
diff --git a/tools/testing/selftests/drivers/net/hw/ncdevmem.c b/tools/testing/selftests/drivers/net/hw/ncdevmem.c
index d96e8a3b5a65..918e3b51f3b8 100644
--- a/tools/testing/selftests/drivers/net/hw/ncdevmem.c
+++ b/tools/testing/selftests/drivers/net/hw/ncdevmem.c
@@ -40,6 +40,7 @@
#include <linux/uio.h>
#include <stdarg.h>
+#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
@@ -61,6 +62,7 @@
#include <sys/time.h>
#include <linux/memfd.h>
+#include <sys/param.h>
#include <linux/dma-buf.h>
#include <linux/errqueue.h>
#include <linux/udmabuf.h>
@@ -79,6 +81,7 @@
#define PAGE_SHIFT 12
#define TEST_PREFIX "ncdevmem"
#define NUM_PAGES 16000
+#define MB(x) ((x) << 20)
#ifndef MSG_SOCK_DEVMEM
#define MSG_SOCK_DEVMEM 0x2000000
@@ -100,6 +103,7 @@ static unsigned int dmabuf_id;
static uint32_t tx_dmabuf_id;
static int waittime_ms = 500;
static bool fail_on_linear;
+static uint32_t rx_page_size;
/* System state loaded by current_config_load() */
#define MAX_FLOWS 8
@@ -142,6 +146,7 @@ static struct memory_buffer *udmabuf_alloc(size_t size)
{
struct udmabuf_create create;
struct memory_buffer *ctx;
+ unsigned int memfd_flags;
int ret;
ctx = malloc(sizeof(*ctx));
@@ -150,15 +155,20 @@ static struct memory_buffer *udmabuf_alloc(size_t size)
ctx->size = size;
- ctx->devfd = open("/dev/udmabuf", O_RDWR);
+ ctx->devfd = open("/dev/udmabuf", O_RDONLY);
if (ctx->devfd < 0) {
pr_err("[skip,no-udmabuf: Unable to access DMA buffer device file]");
goto err_free_ctx;
}
- ctx->memfd = memfd_create("udmabuf-test", MFD_ALLOW_SEALING);
+ memfd_flags = MFD_ALLOW_SEALING;
+ if (rx_page_size > getpagesize())
+ memfd_flags |= MFD_HUGETLB | MFD_HUGE_2MB;
+
+ ctx->memfd = memfd_create("udmabuf-test", memfd_flags);
if (ctx->memfd < 0) {
- pr_err("[skip,no-memfd]");
+ pr_err("[skip,no-memfd%s]",
+ (memfd_flags & MFD_HUGETLB) ? " (need hugepages)" : "");
goto err_close_dev;
}
@@ -168,6 +178,11 @@ static struct memory_buffer *udmabuf_alloc(size_t size)
goto err_close_memfd;
}
+ if (memfd_flags & MFD_HUGETLB) {
+ size = roundup(size, MB(2));
+ ctx->size = size;
+ }
+
ret = ftruncate(ctx->memfd, size);
if (ret == -1) {
pr_err("[FAIL,memfd-truncate]");
@@ -699,6 +714,8 @@ static int bind_rx_queue(unsigned int ifindex, unsigned int dmabuf_fd,
netdev_bind_rx_req_set_ifindex(req, ifindex);
netdev_bind_rx_req_set_fd(req, dmabuf_fd);
__netdev_bind_rx_req_set_queues(req, queues, n_queue_index);
+ if (rx_page_size)
+ netdev_bind_rx_req_set_rx_page_size(req, rx_page_size);
rsp = netdev_bind_rx(*ys, req);
if (!rsp) {
@@ -1411,7 +1428,7 @@ int main(int argc, char *argv[])
int is_server = 0, opt;
int ret, err = 1;
- while ((opt = getopt(argc, argv, "Lls:c:p:v:q:t:f:z:n")) != -1) {
+ while ((opt = getopt(argc, argv, "Lls:c:p:v:q:t:f:z:nb:")) != -1) {
switch (opt) {
case 'L':
fail_on_linear = true;
@@ -1446,6 +1463,19 @@ int main(int argc, char *argv[])
case 'n':
skip_config = 1;
break;
+ case 'b': {
+ unsigned long val;
+
+ errno = 0;
+ val = strtoul(optarg, NULL, 0);
+ if ((val == ULONG_MAX && errno == ERANGE) ||
+ val > UINT32_MAX) {
+ pr_err("invalid rx_page_size: %s", optarg);
+ return 1;
+ }
+ rx_page_size = val;
+ break;
+ }
case '?':
fprintf(stderr, "unknown option: %c\n", optopt);
break;
diff --git a/tools/testing/selftests/drivers/net/hw/nk_devmem.py b/tools/testing/selftests/drivers/net/hw/nk_devmem.py
index 300ed2a70ab4..61c6f31f01e5 100755
--- a/tools/testing/selftests/drivers/net/hw/nk_devmem.py
+++ b/tools/testing/selftests/drivers/net/hw/nk_devmem.py
@@ -3,7 +3,8 @@
"""Test devmem TCP with netkit."""
import os
-from devmem_lib import setup_test, run_rx, run_tx, run_tx_chunks, run_rx_hds
+from devmem_lib import (setup_test, run_rx, run_tx, run_tx_chunks, run_rx_hds,
+ run_rx_large_niov)
from lib.py import ksft_run, ksft_exit, ksft_disruptive
from lib.py import NetDrvContEnv
@@ -31,6 +32,11 @@ def check_nk_rx_hds(cfg) -> None:
run_rx_hds(cfg)
+def check_nk_rx_large_niov(cfg) -> None:
+ """Run the devmem RX large-niov test through netkit."""
+ run_rx_large_niov(cfg)
+
+
def main() -> None:
"""Run the netkit devmem test cases."""
with NetDrvContEnv(__file__, rxqueues=2, primary_rx_redirect=True) as cfg:
@@ -38,7 +44,7 @@ def main() -> None:
os.path.join(os.path.dirname(os.path.abspath(__file__)),
"ncdevmem"))
ksft_run([check_nk_rx, check_nk_tx, check_nk_tx_chunks,
- check_nk_rx_hds], args=(cfg,))
+ check_nk_rx_hds, check_nk_rx_large_niov], args=(cfg,))
ksft_exit()
diff --git a/tools/testing/selftests/drivers/net/hw/rss_ctx.py b/tools/testing/selftests/drivers/net/hw/rss_ctx.py
index f36f76d6ca59..5b25fa89c629 100755
--- a/tools/testing/selftests/drivers/net/hw/rss_ctx.py
+++ b/tools/testing/selftests/drivers/net/hw/rss_ctx.py
@@ -651,9 +651,14 @@ def test_rss_context_overlap(cfg, other_ctx=0):
ntuple = defer(ethtool, f"-N {cfg.ifname} delete {ntuple_id}")
# Test the main context
- cnts = _get_rx_cnts(cfg)
- GenerateTraffic(cfg, port=port).wait_pkts_and_stop(20000)
- cnts = _get_rx_cnts(cfg, prev=cnts)
+ attempts = 3
+ for attempt in range(attempts):
+ cnts = _get_rx_cnts(cfg)
+ GenerateTraffic(cfg, port=port).wait_pkts_and_stop(20000)
+ cnts = _get_rx_cnts(cfg, prev=cnts)
+ if sum(cnts[:2]) >= 7000 and sum(cnts[2:4]) >= 7000:
+ break
+ ksft_pr(f"Skewed queue distribution, attempt {attempt + 1}/{attempts}: " + str(cnts))
ksft_ge(sum(cnts[ :4]), 20000, "traffic on main context: " + str(cnts))
ksft_ge(sum(cnts[ :2]), 7000, "traffic on main context (1/2): " + str(cnts))
diff --git a/tools/testing/selftests/drivers/net/hw/toeplitz.py b/tools/testing/selftests/drivers/net/hw/toeplitz.py
index cd7e080e6f84..571732198b93 100755
--- a/tools/testing/selftests/drivers/net/hw/toeplitz.py
+++ b/tools/testing/selftests/drivers/net/hw/toeplitz.py
@@ -21,6 +21,8 @@ from lib.py import ksft_variants, KsftNamedVariant, KsftSkipEx, KsftFailEx
ETH_RSS_HASH_TOP = 1
# Must match RPS_MAX_CPUS in toeplitz.c
RPS_MAX_CPUS = 16
+# Cap Rx queues so IRQ pinning leaves free CPUs in the RPS_MAX_CPUS range
+QUEUE_CAP = 8
def _check_rps_and_rfs_not_configured(cfg):
@@ -48,6 +50,25 @@ def _get_cpu_for_irq(irq):
return int(data)
+def _cap_queue_count(cfg):
+ ehdr = {"header": {"dev-index": cfg.ifindex}}
+ chans = cfg.ethnl.channels_get(ehdr)
+
+ config = {}
+ restore = {}
+ for key in ("combined-count", "rx-count"):
+ cur = chans.get(key, 0)
+ if cur > QUEUE_CAP:
+ config[key] = QUEUE_CAP
+ restore[key] = cur
+
+ if not config:
+ return
+
+ cfg.ethnl.channels_set(ehdr | config)
+ defer(cfg.ethnl.channels_set, ehdr | restore)
+
+
def _get_irq_cpus(cfg):
"""
Read the list of IRQs for the device Rx queues.
@@ -177,6 +198,7 @@ def test(cfg, proto_flag, ipver, grp):
]
if grp:
+ _cap_queue_count(cfg)
_check_rps_and_rfs_not_configured(cfg)
if grp == "rss":
irq_cpus = ",".join([str(x) for x in _get_irq_cpus(cfg)])
diff --git a/tools/testing/selftests/drivers/net/hw/tso.py b/tools/testing/selftests/drivers/net/hw/tso.py
index 802bb4868046..67f6c9ca9a64 100755
--- a/tools/testing/selftests/drivers/net/hw/tso.py
+++ b/tools/testing/selftests/drivers/net/hw/tso.py
@@ -1,7 +1,7 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: GPL-2.0
-"""Run the tools/testing/selftests/net/csum testsuite."""
+"""A simple test for TSO."""
import fcntl
import socket
diff --git a/tools/testing/selftests/drivers/net/lib/py/env.py b/tools/testing/selftests/drivers/net/lib/py/env.py
index e4ab99b905b1..25903f580b40 100644
--- a/tools/testing/selftests/drivers/net/lib/py/env.py
+++ b/tools/testing/selftests/drivers/net/lib/py/env.py
@@ -7,7 +7,7 @@ import time
import json
from pathlib import Path
from lib.py import KsftSkipEx, KsftXfailEx
-from lib.py import ksft_setup, wait_file
+from lib.py import ksft_pr, ksft_setup, wait_file
from lib.py import cmd, ethtool, ip, CmdExitFailure
from lib.py import NetNS, NetdevSimDev, UserNetNS
from .remote import Remote
@@ -31,6 +31,7 @@ class NetDrvEnvBase:
# Following attrs must be set be inheriting classes
self.dev = None
+ self.ifname = None
def _load_env_file(self):
env = os.environ.copy()
@@ -58,6 +59,22 @@ class NetDrvEnvBase:
def __del__(self):
pass
+ def _print_dev_info(self):
+ """
+ Show whether the test ran on real hardware or netdevsim.
+ Useful to confirm when results are shared on the mailing list.
+ """
+ driver = "unknown"
+ try:
+ info = ethtool(f"-i {self.ifname}").stdout
+ for line in info.splitlines():
+ if line.startswith("driver:"):
+ driver = line.split(':', 1)[1].strip() or driver
+ break
+ except (CmdExitFailure, FileNotFoundError):
+ pass
+ ksft_pr(f"Interface: {self.ifname}, driver: {driver}")
+
def __enter__(self):
ip(f"link set dev {self.dev['ifname']} up")
wait_file(f"/sys/class/net/{self.dev['ifname']}/carrier",
@@ -94,6 +111,7 @@ class NetDrvEnv(NetDrvEnvBase):
self.dev = self._ns.nsims[0].dev
self.ifname = self.dev['ifname']
self.ifindex = self.dev['ifindex']
+ self._print_dev_info()
def __del__(self):
if self._ns:
@@ -114,10 +132,11 @@ class NetDrvEpEnv(NetDrvEnvBase):
nsim_v4_pfx = "192.0.2."
nsim_v6_pfx = "2001:db8::"
- def __init__(self, src_path, nsim_test=None):
+ def __init__(self, src_path, nsim_test=None, queue_count=None):
super().__init__(src_path)
self._stats_settle_time = None
+ self._queue_count = queue_count
# Things we try to destroy
self.remote = None
@@ -159,16 +178,11 @@ class NetDrvEpEnv(NetDrvEnvBase):
self.remote = Remote(kind, args, src_path)
- self.addr_ipver = "6" if self.addr_v["6"] else "4"
- self.addr = self.addr_v[self.addr_ipver]
- self.remote_addr = self.remote_addr_v[self.addr_ipver]
-
- # Bracketed addresses, some commands need IPv6 to be inside []
- self.baddr = f"[{self.addr_v['6']}]" if self.addr_v["6"] else self.addr_v["4"]
- self.remote_baddr = f"[{self.remote_addr_v['6']}]" if self.remote_addr_v["6"] else self.remote_addr_v["4"]
+ self.set_ipver("6" if self.addr_v["6"] else "4")
self.ifname = self.dev['ifname']
self.ifindex = self.dev['ifindex']
+ self._print_dev_info()
# resolve remote interface name
self.remote_ifname = self.resolve_remote_ifc()
@@ -179,9 +193,13 @@ class NetDrvEpEnv(NetDrvEnvBase):
self._required_cmd = {}
def create_local(self):
+ nsim_kwargs = {}
+ if self._queue_count:
+ nsim_kwargs["queue_count"] = self._queue_count
+
self._netns = NetNS()
- self._ns = NetdevSimDev()
- self._ns_peer = NetdevSimDev(ns=self._netns)
+ self._ns = NetdevSimDev(**nsim_kwargs)
+ self._ns_peer = NetdevSimDev(ns=self._netns, **nsim_kwargs)
with open("/proc/self/ns/net") as nsfd0, \
open("/var/run/netns/" + self._netns.name) as nsfd1:
@@ -252,6 +270,25 @@ class NetDrvEpEnv(NetDrvEnvBase):
if not self.addr_v[ipver] or not self.remote_addr_v[ipver]:
raise KsftSkipEx(f"Test requires IPv{ipver} connectivity")
+ def set_ipver(self, ipver):
+ """
+ Modify the IP version used by the generic address fields.
+ """
+ if ipver == getattr(self, "addr_ipver", None):
+ return
+
+ self.require_ipver(ipver)
+
+ self.addr_ipver = ipver
+ self.addr = self.addr_v[ipver]
+ self.remote_addr = self.remote_addr_v[ipver]
+
+ # Bracketed addresses, some commands need IPv6 to be inside []
+ self.baddr = (f"[{self.addr_v['6']}]" if ipver == "6"
+ else self.addr_v["4"])
+ self.remote_baddr = (f"[{self.remote_addr_v['6']}]" if ipver == "6"
+ else self.remote_addr_v["4"])
+
def require_nsim(self, nsim_test=True):
"""Require or exclude netdevsim for this test"""
if nsim_test and self._ns is None:
diff --git a/tools/testing/selftests/drivers/net/netconsole/Makefile b/tools/testing/selftests/drivers/net/netconsole/Makefile
index b56c70b7e274..f0674c0017fc 100644
--- a/tools/testing/selftests/drivers/net/netconsole/Makefile
+++ b/tools/testing/selftests/drivers/net/netconsole/Makefile
@@ -13,6 +13,7 @@ TEST_PROGS := \
netcons_resume.sh \
netcons_sysdata.sh \
netcons_torture.sh \
+ netcons_userdata.sh \
# end of TEST_PROGS
include ../../../lib.mk
diff --git a/tools/testing/selftests/drivers/net/netconsole/netcons_userdata.sh b/tools/testing/selftests/drivers/net/netconsole/netcons_userdata.sh
new file mode 100755
index 000000000000..113903f4ce1c
--- /dev/null
+++ b/tools/testing/selftests/drivers/net/netconsole/netcons_userdata.sh
@@ -0,0 +1,229 @@
+#!/usr/bin/env bash
+# SPDX-License-Identifier: GPL-2.0
+
+# Exercise the netconsole userdata payload.
+#
+# The first part checks that the payload the target transmits follows what
+# configfs says: a value shows up in the next message, an update replaces the
+# previous one, clearing the value drops the entry, and so does removing the
+# key.
+#
+# The second part rewrites values, creates and deletes keys, and clears the
+# payload entirely while messages are being sent, so the transmit path keeps
+# picking up payloads that are being replaced underneath it. It runs twice,
+# once with a payload small enough to fit in a single packet and once large
+# enough to be fragmented.
+#
+# Author: Breno Leitao <leitao@debian.org>
+
+set -euo pipefail
+
+SCRIPTDIR=$(dirname "$(readlink -e "${BASH_SOURCE[0]}")")
+
+source "${SCRIPTDIR}"/../lib/sh/lib_netcons.sh
+
+# Number of times each torture worker loops
+ITERATIONS=${1:-200}
+
+# Keys owned by each torture worker. Workers do not share keys, so a failing
+# configfs operation means a real problem and not a lost race.
+CHURN_KEY="churnkey"
+TRANSIENT_KEY="transientkey"
+# Number of keys used to push a message past MAX_PRINT_CHUNK
+BULK_KEYS=8
+
+USERDATA_DIR="${NETCONS_PATH}/userdata"
+# Values are capped at MAX_EXTRADATA_VALUE_LEN(200) bytes, so ${BULK_KEYS}
+# entries of this size are enough to force fragmentation
+LONG_VALUE=$(printf -- 'v%.0s' {1..190})
+
+function write_key() {
+ local KEY="${1}"
+ local VALUE="${2}"
+
+ mkdir -p "${USERDATA_DIR}/${KEY}"
+ echo "${VALUE}" > "${USERDATA_DIR}/${KEY}/value"
+}
+
+# Send a single message and capture it on the destination interface
+function send_and_capture() {
+ rm -f "${OUTPUT_FILE}"
+
+ listen_port_and_save_to "${OUTPUT_FILE}" &
+ wait_for_port "${NAMESPACE}" "${PORT}" "${IP_VERSION}"
+ echo "${MSG}: ${TARGET}" > /dev/kmsg
+ busywait "${BUSYWAIT_TIMEOUT}" test -s "${OUTPUT_FILE}" || true
+ pkill_socat
+ validate_msg "${OUTPUT_FILE}"
+}
+
+function expect_in_msg() {
+ local WANTED="${1}"
+
+ if ! grep -q -- "${WANTED}" "${OUTPUT_FILE}"; then
+ echo "FAIL: '${WANTED}' not found in ${OUTPUT_FILE}" >&2
+ cat "${OUTPUT_FILE}" >&2
+ exit "${ksft_fail}"
+ fi
+}
+
+function expect_not_in_msg() {
+ local UNWANTED="${1}"
+
+ if grep -q -- "${UNWANTED}" "${OUTPUT_FILE}"; then
+ echo "FAIL: '${UNWANTED}' found in ${OUTPUT_FILE}" >&2
+ cat "${OUTPUT_FILE}" >&2
+ exit "${ksft_fail}"
+ fi
+}
+
+# Every write publishes a new payload and frees the previous one. An empty
+# value is skipped when the payload is formatted, so this also drives the
+# target through having no payload at all.
+function churn_value() {
+ local i
+
+ for i in $(seq "${ITERATIONS}")
+ do
+ echo "value${i}" > "${USERDATA_DIR}/${CHURN_KEY}/value"
+ echo > "${USERDATA_DIR}/${CHURN_KEY}/value"
+ done
+}
+
+# Create and delete a key underneath the sender
+function churn_key() {
+ local i
+
+ for i in $(seq "${ITERATIONS}")
+ do
+ mkdir "${USERDATA_DIR}/${TRANSIENT_KEY}"
+ echo "transient${i}" > "${USERDATA_DIR}/${TRANSIENT_KEY}/value"
+ rmdir "${USERDATA_DIR}/${TRANSIENT_KEY}"
+ done
+}
+
+# Keep the transmit path busy while the payload is being replaced
+function send_messages() {
+ local i
+
+ for i in $(seq "${ITERATIONS}")
+ do
+ echo "${MSG}: ${TARGET} ${i}" > /dev/kmsg
+ done
+}
+
+# Run the workers concurrently and fail if any of them hits an error
+function run_workers() {
+ local PIDS=()
+ local WORKER
+ local RET=0
+ local PID
+
+ for WORKER in "$@"
+ do
+ "${WORKER}" &
+ PIDS+=("$!")
+ done
+
+ # Reap every worker before reporting a failure, otherwise a surviving
+ # worker keeps writing to configfs while the exit trap cleans it up.
+ for PID in "${PIDS[@]}"
+ do
+ wait "${PID}" || RET=1
+ done
+
+ if [[ "${RET}" -ne 0 ]]
+ then
+ echo "FAIL: userdata torture worker failed" >&2
+ exit "${ksft_fail}"
+ fi
+}
+
+function create_bulk_keys() {
+ local i
+
+ for i in $(seq "${BULK_KEYS}")
+ do
+ write_key "bulk${i}" "${LONG_VALUE}"
+ done
+}
+
+function delete_bulk_keys() {
+ local i
+
+ for i in $(seq "${BULK_KEYS}")
+ do
+ rmdir "${USERDATA_DIR}/bulk${i}"
+ done
+}
+
+# ========== #
+# Start here #
+# ========== #
+
+modprobe netdevsim 2> /dev/null || true
+modprobe netconsole 2> /dev/null || true
+
+IP_VERSION="ipv4"
+# The content of kmsg will be saved to the following file
+OUTPUT_FILE="/tmp/${TARGET}"
+
+# Check for basic system dependency and exit if not found
+check_for_dependencies
+# Set current loglevel to KERN_INFO(6), and default to KERN_NOTICE(5)
+echo "6 5" > /proc/sys/kernel/printk
+# Remove the namespace, interfaces and netconsole target on exit
+trap cleanup EXIT
+# Create one namespace and two interfaces
+set_network "${IP_VERSION}"
+# Create a dynamic target for netconsole
+create_dynamic_target
+
+# ===================================================
+# TEST #1
+# A value written to configfs reaches the destination
+# ===================================================
+write_key "${USERDATA_KEY}" "first"
+send_and_capture
+expect_in_msg "${USERDATA_KEY}=first"
+
+# ===================================================
+# TEST #2
+# Updating the value replaces the previous payload
+# ===================================================
+write_key "${USERDATA_KEY}" "second"
+send_and_capture
+expect_in_msg "${USERDATA_KEY}=second"
+expect_not_in_msg "${USERDATA_KEY}=first"
+
+# ===================================================
+# TEST #3
+# Clearing the value drops the entry
+# ===================================================
+echo > "${USERDATA_DIR}/${USERDATA_KEY}/value"
+send_and_capture
+expect_not_in_msg "${USERDATA_KEY}="
+
+# ===================================================
+# TEST #4
+# Removing the key drops the entry
+# ===================================================
+write_key "${USERDATA_KEY}" "third"
+rmdir "${USERDATA_DIR}/${USERDATA_KEY}"
+send_and_capture
+expect_not_in_msg "${USERDATA_KEY}="
+rm "${OUTPUT_FILE}"
+
+# ===================================================
+# TEST #5
+# Torture the payload while messages are being sent,
+# first unfragmented and then fragmented
+# ===================================================
+write_key "${CHURN_KEY}" "${USERDATA_VALUE}"
+run_workers churn_value churn_key send_messages
+
+create_bulk_keys
+run_workers churn_value churn_key send_messages
+delete_bulk_keys
+
+exit "${ksft_pass}"
diff --git a/tools/testing/selftests/drivers/net/netdevsim/ethtool-coalesce.sh b/tools/testing/selftests/drivers/net/netdevsim/ethtool-coalesce.sh
index 9adfba8f87e6..b9fcafad4258 100755
--- a/tools/testing/selftests/drivers/net/netdevsim/ethtool-coalesce.sh
+++ b/tools/testing/selftests/drivers/net/netdevsim/ethtool-coalesce.sh
@@ -116,12 +116,14 @@ done
# bool settings which ethtool displays on the same line
ethtool -C $NSIM_NETDEV adaptive-rx on
-s=$(ethtool -c $NSIM_NETDEV | grep -q "Adaptive RX: on TX: off")
-check $? "$s" ""
+s=$(ethtool -c $NSIM_NETDEV)
+echo "$s" | grep -q "Adaptive RX: on TX: off"
+check $? "" ""
ethtool -C $NSIM_NETDEV adaptive-tx on
-s=$(ethtool -c $NSIM_NETDEV | grep -q "Adaptive RX: on TX: on")
-check $? "$s" ""
+s=$(ethtool -c $NSIM_NETDEV)
+echo "$s" | grep -q "Adaptive RX: on TX: on"
+check $? "" ""
if [ $num_errors -eq 0 ]; then
echo "PASSED all $((num_passes)) checks"
diff --git a/tools/testing/selftests/drivers/net/psp.py b/tools/testing/selftests/drivers/net/psp.py
index 315648a770d0..a5b1e14f120f 100755
--- a/tools/testing/selftests/drivers/net/psp.py
+++ b/tools/testing/selftests/drivers/net/psp.py
@@ -23,6 +23,8 @@ from lib.py import NetNSEnter
from lib.py import bkg, rand_port, wait_port_listen
from lib.py import ip
+TCP_ULP = 31
+
def _get_outq(s):
one = b'\0' * 4
@@ -333,6 +335,50 @@ def assoc_version_mismatch(cfg):
ksft_eq(the_exception.nl_msg.error, -errno.EINVAL)
+def _require_tls_ulp():
+ with socket.create_server(("localhost", 0)) as srv, \
+ socket.create_connection(srv.getsockname()) as s:
+ try:
+ s.setsockopt(socket.SOL_TCP, TCP_ULP, b"tls")
+ except OSError as exc:
+ raise KsftSkipEx("kTLS not available") from exc
+
+
+def assoc_psp_ulp_exclusive(cfg):
+ """ Test that a TCP ULP cannot be attached to a PSP socket """
+ _init_psp_dev(cfg)
+ _require_tls_ulp()
+
+ with _make_clr_conn(cfg) as s:
+ try:
+ cfg.pspnl.rx_assoc({"version": 0,
+ "dev-id": cfg.psp_dev_id,
+ "sock-fd": s.fileno()})
+ with ksft_raises(OSError) as cm:
+ s.setsockopt(socket.SOL_TCP, TCP_ULP, b"tls")
+ ksft_eq(cm.exception.errno, errno.EINVAL)
+ finally:
+ _close_conn(cfg, s)
+
+
+def assoc_ulp_psp_exclusive(cfg):
+ """ Test that a PSP assoc cannot be added to a socket with a TCP ULP """
+ _init_psp_dev(cfg)
+ _require_tls_ulp()
+
+ with _make_clr_conn(cfg) as s:
+ try:
+ s.setsockopt(socket.SOL_TCP, TCP_ULP, b"tls")
+ with ksft_raises(NlError) as cm:
+ cfg.pspnl.rx_assoc({"version": 0,
+ "dev-id": cfg.psp_dev_id,
+ "sock-fd": s.fileno()})
+ ksft_eq(cm.exception.nl_msg.error, -errno.EINVAL)
+ ksft_eq(cm.exception.nl_msg.extack['bad-attr'], ".sock-fd")
+ finally:
+ _close_conn(cfg, s)
+
+
def assoc_twice(cfg):
""" Test reusing Tx assoc for two sockets """
_init_psp_dev(cfg)
diff --git a/tools/testing/selftests/drivers/net/ring_reconfig.py b/tools/testing/selftests/drivers/net/ring_reconfig.py
index f9530a8b0856..2bc329b77134 100755
--- a/tools/testing/selftests/drivers/net/ring_reconfig.py
+++ b/tools/testing/selftests/drivers/net/ring_reconfig.py
@@ -5,10 +5,25 @@
Test channel and ring size configuration via ethtool (-L / -G).
"""
+import socket
+import struct
+import time
+
from lib.py import ksft_run, ksft_exit, ksft_pr
from lib.py import ksft_eq
+from lib.py import KsftSkipEx, KsftXfailEx
from lib.py import NetDrvEpEnv, EthtoolFamily, GenerateTraffic
-from lib.py import defer, NlError
+from lib.py import cmd, defer, rand_port, tc, NlError
+
+# Added in Python 3.13; fallback to 61 for x86/ARM/MIPS
+SO_TXTIME = getattr(socket, "SO_TXTIME", 61)
+
+# Not always exported by the socket module; asm-generic value (x86/ARM/MIPS).
+SO_SNDBUFFORCE = getattr(socket, "SO_SNDBUFFORCE", 32)
+
+# TX ring size the test shrinks to so the ring fills quickly.
+MIN_TX_RING = 32
+MAX_TX_RING = 1024
def channels(cfg) -> None:
@@ -151,14 +166,248 @@ def ringparam(cfg) -> None:
GenerateTraffic(cfg).wait_pkts_and_stop(10000)
+def _write_file(path, val):
+ """Write val to a file."""
+ with open(path, "w", encoding="utf-8") as fp:
+ fp.write(str(val))
+
+
+def _write_sysfs(path, val):
+ """Write val to a sysfs file, restoring the original value on exit."""
+ with open(path, "r", encoding="utf-8") as fp:
+ orig_val = fp.read().strip()
+ if str(val) == orig_val:
+ return
+ _write_file(path, val)
+ defer(_write_file, path, orig_val)
+
+
+def _get_qdisc_backlog(cfg, mq_handle, queue):
+ """Return the qdisc backlog (bytes) for the given TX queue's leaf."""
+ target_parent = f"{mq_handle}{queue + 1:x}"
+ for q in tc(f"-s qdisc show dev {cfg.ifname}", json=True):
+ if q.get("parent", "") == target_parent:
+ return q.get("backlog") or 0
+ return 0
+
+
+def _setup_fq_qdisc(cfg, port, target_queue, other_queue, flow_limit):
+ """Put an fq qdisc on target_queue's leaf and return the mq handle in use.
+
+ We must not disturb the device's existing TX/RX qdisc policy. On a real
+ NIC the root mq already has an addressable handle, so we leave the root
+ and every other queue alone and only swap this one leaf, restoring its
+ original qdisc afterwards.
+
+ @flow_limit raises fq's per-flow packet limit (default 100) so a single
+ flow can back up more packets than the Tx ring holds and thus overflow it.
+ """
+ qdiscs = tc(f"qdisc show dev {cfg.ifname}", json=True)
+ root = next((q for q in qdiscs if q.get("root")), None)
+
+ if root and root["kind"] == "mq" and root["handle"] != "0:":
+ # Addressable mq (previously-configured): touch only the target queue's
+ # leaf and restore its original qdisc afterwards.
+ mq_handle = root["handle"]
+ parent = f"{mq_handle}{target_queue + 1:x}"
+ orig = next((q for q in qdiscs if q.get("parent") == parent), None)
+ orig_kind = orig["kind"] if orig else \
+ cmd("sysctl -n net.core.default_qdisc").stdout.strip()
+ defer(tc, f"qdisc replace dev {cfg.ifname} parent {parent} {orig_kind}")
+ elif root is None or root["kind"] in ("mq", "noqueue"):
+ # The auto-attached root mq has handle 0: on any device (real or sim),
+ # which the kernel rejects as a qdisc parent. A 0: handle means the mq
+ # is the untouched kernel default - no custom child qdiscs can hang off
+ # an unaddressable parent - so installing a real handle and restoring
+ # the default mq on exit preserves the device's effective policy.
+ mq_handle = "1:"
+ tc(f"qdisc replace dev {cfg.ifname} root handle {mq_handle} mq")
+ defer(tc, f"qdisc replace dev {cfg.ifname} root mq")
+ parent = f"{mq_handle}{target_queue + 1:x}"
+ else:
+ raise KsftSkipEx(f"root qdisc '{root['kind']}' is not mq; "
+ "refusing to disturb existing qdisc policy")
+
+ try:
+ tc(f"qdisc replace dev {cfg.ifname} parent {parent} fq "
+ f"flow_limit {flow_limit} limit {flow_limit * 2}")
+ except Exception as exc:
+ raise KsftSkipEx(
+ f"fq not available (CONFIG_NET_SCH_FQ): {exc}") from exc
+
+ qdisc_j = tc(f"qdisc show dev {cfg.ifname}", json=True)
+ has_clsact = any(q['kind'] == 'clsact' for q in qdisc_j)
+ if not has_clsact:
+ tc(f"qdisc add dev {cfg.ifname} clsact")
+ defer(tc, f"qdisc del dev {cfg.ifname} clsact")
+
+ proto = "ipv6" if int(cfg.addr_ipver) == 6 else "ip"
+ try:
+ tc(f"filter add dev {cfg.ifname} egress protocol {proto} "
+ f"pref 1 flower ip_proto udp dst_port {port} "
+ f"action skbedit queue_mapping {target_queue}")
+ except Exception as exc:
+ raise KsftSkipEx("tc flower/act_skbedit not available") from exc
+ defer(tc, f"filter del dev {cfg.ifname} egress pref 1")
+
+ tc(f"filter add dev {cfg.ifname} egress pref 101 "
+ f"matchall action skbedit queue_mapping {other_queue}")
+ defer(tc, f"filter del dev {cfg.ifname} egress pref 101")
+
+ return mq_handle
+
+
+def _create_sotxtime_socket(cfg, sndbuf):
+ """Create a UDP socket with SO_TXTIME enabled, bound to the test device."""
+ sock = socket.socket(socket.AF_INET6 if cfg.addr_ipver == "6"
+ else socket.AF_INET, socket.SOCK_DGRAM)
+ try:
+ sock.setsockopt(socket.SOL_SOCKET, SO_TXTIME, struct.pack("Ii", 1, 0))
+ except OSError as exc:
+ sock.close()
+ raise KsftSkipEx("SO_TXTIME not supported") from exc
+ sock.setsockopt(socket.SOL_SOCKET, socket.SO_BINDTODEVICE,
+ cfg.ifname.encode())
+ # Deferred completions keep every in-flight skb charged to the socket, so
+ # size the send buffer to hold the whole burst. SO_SNDBUFFORCE bypasses
+ # net.core.wmem_max (the test runs as root).
+ try:
+ sock.setsockopt(socket.SOL_SOCKET, SO_SNDBUFFORCE, sndbuf)
+ except OSError:
+ sock.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, sndbuf)
+ return sock
+
+
+def _send_sotxtime_burst(cfg, sock, port, count, delay_ns, pkt_size):
+ """Send count UDP packets scheduled delay_ns ahead using SO_TXTIME."""
+ payload = b'\x00' * pkt_size
+ txtime_ns = time.clock_gettime_ns(time.CLOCK_MONOTONIC) + delay_ns
+
+ ancdata = [(socket.SOL_SOCKET, SO_TXTIME, struct.pack("Q", txtime_ns))]
+ if int(cfg.addr_ipver) == 6:
+ dest = (cfg.remote_addr, port, 0, 0)
+ else:
+ dest = (cfg.remote_addr, port)
+ for _ in range(count):
+ sock.sendmsg([payload], ancdata, 0, dest)
+
+
+def _set_small_tx_ring(cfg, ehdr):
+ """Set the Tx ring to the smallest size the driver accepts.
+
+ Start at 32 so the ring fills quickly, then grow exponentially (64,
+ 128, 256, ...) up to 1024. Some drivers enforce a minimum well above 32
+ (e.g. bnxt needs a large ring for software UDP segmentation), so raise
+ the lower bound until the driver accepts it, giving up past 1024.
+ """
+ size = MIN_TX_RING
+ while size <= MAX_TX_RING:
+ try:
+ cfg.eth.rings_set(ehdr | {'tx': size})
+ return size
+ except NlError:
+ size = size * 2
+ continue
+ raise KsftSkipEx("driver rejects all tx ring sizes up to 1024")
+
+
+def reconfig_tx_stall(cfg) -> None:
+ """Test that qdisc backlog drains after ring reconfiguration."""
+ target_queue = 1
+ other_queue = 0
+
+ ehdr = {'header': {'dev-index': cfg.ifindex}}
+ chans = cfg.eth.channels_get(ehdr)
+
+ if "combined-max" not in chans:
+ raise KsftSkipEx("device does not support combined channels")
+ if chans.get("combined-max", 0) < 2:
+ raise KsftSkipEx("device does not support 2+ combined channels")
+ if chans["combined-count"] < 2:
+ defer(cfg.eth.channels_set,
+ ehdr | {"combined-count": chans["combined-count"]})
+ cfg.eth.channels_set(ehdr | {"combined-count": 2})
+
+ rings = cfg.eth.rings_get(ehdr)
+ if 'rx' not in rings or 'tx' not in rings:
+ raise KsftSkipEx("device does not expose rx/tx ring params")
+ tx_cur = rings['tx']
+ if tx_cur <= MIN_TX_RING:
+ raise KsftSkipEx("tx ring size already at minimum")
+ defer(cfg.eth.rings_set, ehdr | {'tx': tx_cur})
+
+ # Use the smallest Tx ring the driver accepts (32, growing to 1024).
+ tx_ring = _set_small_tx_ring(cfg, ehdr)
+
+ # Slow completions so the ring stays full after FQ releases packets
+ napi_defer = f"/sys/class/net/{cfg.ifname}/napi_defer_hard_irqs"
+ gro_timeout = f"/sys/class/net/{cfg.ifname}/gro_flush_timeout"
+ _write_sysfs(napi_defer, 100)
+ _write_sysfs(gro_timeout, 1000000000)
+
+ port = rand_port()
+ # A single flow must overflow the ring, so send twice the ring depth and
+ # let fq hold that many packets for the flow.
+ pkt_count = tx_ring * 2
+ mq_handle = _setup_fq_qdisc(cfg, port, target_queue, other_queue,
+ tx_ring * 2)
+
+ # Size each packet to one MTU (less L3/L4 headers to avoid fragmentation).
+ pkt_size = cfg.dev['mtu'] - (48 if int(cfg.addr_ipver) == 6 else 28)
+
+ # Each queued skb charges the socket its truesize (~2x the payload), so
+ # budget the send buffer for the whole in-flight burst.
+ sock = _create_sotxtime_socket(cfg, pkt_count * pkt_size * 2)
+ defer(sock.close)
+
+ for delay_ms in [100, 200, 500]:
+ _send_sotxtime_burst(cfg, sock, port, pkt_count,
+ delay_ms * 1_000_000, pkt_size)
+ ksft_pr(f"Sent {pkt_count} SO_TXTIME packets (+{delay_ms}ms)")
+ time.sleep(delay_ms / 1000 + 0.3)
+
+ backlog = _get_qdisc_backlog(cfg, mq_handle, target_queue)
+ if backlog:
+ break
+ else:
+ # A device that completes Tx synchronously (e.g. a software/virtual
+ # driver like netdevsim) never keeps the ring full long enough for a
+ # backlog to form, so the wake-vs-start behavior can't be exercised.
+ # Treat that as an expected failure rather than a hard failure.
+ raise KsftXfailEx("could not build qdisc backlog")
+
+ ksft_pr(f"Backlog before reconfig: {backlog} bytes")
+
+ # Trigger ring reconfig — driver should call wake, not just start.
+ # Grow back to the original size so the driver actually switches channels
+ # (setting the current size is a no-op the driver short-circuits).
+ cfg.eth.rings_set(ehdr | {'tx': tx_cur})
+
+ # Let completions proceed normally
+ _write_sysfs(napi_defer, 0)
+ _write_sysfs(gro_timeout, 0)
+
+ # Poll for backlog to drain
+ for _ in range(100):
+ backlog = _get_qdisc_backlog(cfg, mq_handle, target_queue)
+ if not backlog:
+ break
+ time.sleep(0.1)
+
+ ksft_eq(0, backlog,
+ comment=f"qdisc backlog stuck on queue {target_queue} "
+ f"after ring reconfig")
+
+
def main() -> None:
""" Ksft boiler plate main """
- with NetDrvEpEnv(__file__) as cfg:
+ with NetDrvEpEnv(__file__, queue_count=2) as cfg:
cfg.eth = EthtoolFamily()
ksft_run([channels,
- ringparam],
+ ringparam,
+ reconfig_tx_stall],
args=(cfg, ))
ksft_exit()
diff --git a/tools/testing/selftests/drivers/net/shaper.py b/tools/testing/selftests/drivers/net/shaper.py
index e39d270e688d..a53316726f69 100755
--- a/tools/testing/selftests/drivers/net/shaper.py
+++ b/tools/testing/selftests/drivers/net/shaper.py
@@ -1,14 +1,54 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: GPL-2.0
+# pylint: disable=too-many-lines
import errno
+import glob
from lib.py import ksft_run, ksft_exit
-from lib.py import ksft_eq, ksft_raises, ksft_true, KsftSkipEx
+from lib.py import ksft_eq, ksft_true, ksft_raises, KsftSkipEx
from lib.py import EthtoolFamily, NetshaperFamily
from lib.py import NetDrvEnv
from lib.py import NlError
-from lib.py import cmd
+from lib.py import cmd, defer
+
+def _delete_shaper(cfg, nl_shaper, handle) -> None:
+ """ Delete the shaper identified by handle, ignoring a missing-shaper error. """
+ try:
+ nl_shaper.delete({'ifindex': cfg.ifindex,
+ 'handle': handle})
+ except NlError as e:
+ if e.error != errno.ENOENT:
+ raise
+
+def _require_queues(cfg, count):
+ """ Return the netdev TX queue count, skipping the test if fewer than count exist. """
+ qcnt = len(glob.glob(f"/sys/class/net/{cfg.ifname}/queues/tx-*"))
+ if qcnt < count:
+ raise KsftSkipEx(f"netdev has {qcnt} queues, {count} required")
+ return qcnt
+
+def _cap_get(cfg, nl_shaper, scope):
+ """ Return the shaper capabilities for the given scope, caching them on cfg. """
+ if not hasattr(cfg, 'cap_cache'):
+ cfg.cap_cache = {}
+ if scope not in cfg.cap_cache:
+ cfg.cap_cache[scope] = nl_shaper.cap_get({'ifindex': cfg.ifindex,
+ 'scope': scope})
+
+ return cfg.cap_cache[scope]
+
+def _require_caps(cfg, nl_shaper, scope, caps, msg) -> None:
+ """ Skip the test unless the given scope advertises all the required caps. """
+ try:
+ supported = _cap_get(cfg, nl_shaper, scope)
+ except NlError as e:
+ if e.error == errno.EOPNOTSUPP:
+ raise KsftSkipEx(f"{scope} scope shapers not supported by the device")
+ raise
+
+ if not set(caps).issubset(supported):
+ raise KsftSkipEx(msg)
def get_shapers(cfg, nl_shaper) -> None:
try:
@@ -44,17 +84,8 @@ def set_qshapers(cfg, nl_shaper) -> None:
if not 'support-bw-max' in caps or not 'support-metric-bps' in caps:
raise KsftSkipEx("device does not support queue scope shapers with bw_max and metric bps")
- cfg.queues = True;
- netnl = EthtoolFamily()
- channels = netnl.channels_get({'header': {'dev-index': cfg.ifindex}})
- if channels['combined-count'] == 0:
- cfg.rx_type = 'rx'
- cfg.nr_queues = channels['rx-count']
- else:
- cfg.rx_type = 'combined'
- cfg.nr_queues = channels['combined-count']
- if cfg.nr_queues < 3:
- raise KsftSkipEx(f"device does not support enough queues min 3 found {cfg.nr_queues}")
+ _require_queues(cfg, 3)
+ cfg.queues = True
nl_shaper.set({'ifindex': cfg.ifindex,
'handle': {'scope': 'queue', 'id': 1},
@@ -137,77 +168,370 @@ def del_nshapers(cfg, nl_shaper) -> None:
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
ksft_eq(len(shapers), 0)
-def basic_groups(cfg, nl_shaper) -> None:
- if not cfg.netdev:
- raise KsftSkipEx("netdev shaper not supported by the device")
- if cfg.nr_queues < 3:
- raise KsftSkipEx(f"netdev does not have enough queues min 3 reported {cfg.nr_queues}")
+def set_all_supported_attrs(cfg, nl_shaper) -> None:
+ """ Set every queue-scope attribute the device advertises and verify the read-back. """
+ _require_queues(cfg, 1)
- try:
- caps = nl_shaper.cap_get({'ifindex': cfg.ifindex,
- 'scope':'queue'})
- except NlError as e:
- if e.error == 95:
- raise KsftSkipEx("shapers not supported by the device")
- raise
- if not 'support-weight' in caps:
- raise KsftSkipEx("device does not support queue scope shapers with weight")
+ _require_caps(cfg, nl_shaper, 'queue', [],
+ "queue scope shapers not supported by the device")
+ caps = _cap_get(cfg, nl_shaper, 'queue')
+
+ attrs = {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}}
+ expected = {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'netdev'},
+ 'handle': {'scope': 'queue', 'id': 0}}
+
+ rate_attrs = {'support-bw-min': ('bw-min', 10000, 100),
+ 'support-bw-max': ('bw-max', 20000, 200),
+ 'support-burst': ('burst', 3000, 30)}
+ rate_attr_supported = any(cap in caps for cap in rate_attrs)
+ bps_supported = 'support-metric-bps' in caps
+ pps_supported = 'support-metric-pps' in caps
+
+ def add_rate_attrs(metric, value_idx) -> None:
+ attrs['metric'] = metric
+ expected['metric'] = metric
+ for cap, (attr, bps_value, pps_value) in rate_attrs.items():
+ if cap not in caps:
+ continue
+
+ value = bps_value if value_idx == 0 else pps_value
+ attrs[attr] = value
+ expected[attr] = value
+
+ if rate_attr_supported:
+ if bps_supported:
+ add_rate_attrs('bps', 0)
+ elif pps_supported:
+ add_rate_attrs('pps', 1)
+
+ if 'support-priority' in caps:
+ attrs['priority'] = 1
+ expected['priority'] = 1
+ if 'support-weight' in caps:
+ attrs['weight'] = 2
+ expected['weight'] = 2
+
+ if len(attrs) == 2:
+ raise KsftSkipEx("device does not advertise any supported queue shaper attributes")
+
+ nl_shaper.set(attrs)
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper, expected)
+
+ if rate_attr_supported and bps_supported and pps_supported:
+ add_rate_attrs('pps', 1)
+ nl_shaper.set(attrs)
+
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper, expected)
+
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def invalid_set_preserves_state(cfg, nl_shaper) -> None:
+ """ Verify a rejected .set leaves the existing shaper configuration unchanged. """
+ nq = _require_queues(cfg, 1)
+ _require_caps(cfg, nl_shaper, 'queue',
+ ['support-bw-max', 'support-metric-bps'],
+ "device does not support queue scope bw_max with bps metric")
+
+ initial = {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'netdev'},
+ 'handle': {'scope': 'queue', 'id': 0},
+ 'metric': 'bps',
+ 'bw-max': 10000}
+ nl_shaper.set({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0},
+ 'metric': 'bps',
+ 'bw-max': 10000})
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+
+ with ksft_raises(NlError):
+ nl_shaper.set({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': 0},
+ 'metric': 'bps',
+ 'bw-max': 20000})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper, initial)
+
+ with ksft_raises(NlError):
+ nl_shaper.set({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': nq},
+ 'metric': 'bps',
+ 'bw-max': 20000})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper, initial)
+
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def mixed_parent_group_requires_parent(cfg, nl_shaper) -> None:
+ r"""Grouping leaves from different nodes requires an explicit parent.
+
+ netdev netdev
+ / \ parent=netdev
+ N1 N2 group N
+ | | {Q0,Q1} / \
+ Q0 Q1 -------> Q0 Q1
+
+ Without an explicit parent the group is rejected; parent=netdev
+ collapses the leaves into one new node.
+ """
+ _require_queues(cfg, 2)
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps'],
+ "device does not support node scope shapers with bw_max and metric bps")
+ _require_caps(cfg, nl_shaper, 'queue',
+ ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ n1_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': 10000})
+ n1_id = n1_handle['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+
+ n2_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 2}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': 20000})
+ n2_id = n2_handle['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 1})
+
+ with ksft_raises(NlError):
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 3},
+ {'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 4}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': 30000})
+
+ shaper_q0 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper_q0, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n1_id},
+ 'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 1})
+ shaper_q1 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 1}})
+ ksft_eq(shaper_q1, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n2_id},
+ 'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 2})
node_handle = nl_shaper.group({
- 'ifindex': cfg.ifindex,
- 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
- 'weight': 1},
- {'handle': {'scope': 'queue', 'id': 2},
- 'weight': 2}],
- 'handle': {'scope':'netdev'},
- 'metric': 'bps',
- 'bw-max': 10000})
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 3},
+ {'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 4}],
+ 'handle': {'scope':'node'},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': 30000})
+ node_id = node_handle['handle']['id']
+
+ for old_id in (n1_id, n2_id):
+ with ksft_raises(NlError):
+ nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': old_id}})
+
+ shaper_q0 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper_q0, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node_id},
+ 'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 3})
+ shaper_q1 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 1}})
+ ksft_eq(shaper_q1, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node_id},
+ 'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 4})
+
+ for i in range(2):
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': i})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def recursive_empty_node_cleanup(cfg, nl_shaper) -> None:
+ r"""Deleting the last leaf recursively removes the emptied ancestors.
+
+ netdev netdev
+ | del Q0
+ N1 ------> (N1 and N2 removed too)
+ |
+ N2
+ |
+ Q0
+ """
+ _require_queues(cfg, 1)
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps', 'support-nesting'],
+ "device does not support nested node scope shapers")
+ _require_caps(cfg, nl_shaper, 'queue',
+ ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ n1_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': 10000})
+ n1_id = n1_handle['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+
+ n2_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'parent': {'scope': 'node', 'id': n1_id},
+ 'metric': 'bps',
+ 'bw-max': 5000})
+ n2_id = n2_handle['handle']['id']
+
+ shaper_q0 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+ ksft_eq(shaper_q0, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n2_id},
+ 'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 1})
+
+ nl_shaper.delete({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 0}})
+
+ for handle in ({'scope': 'queue', 'id': 0},
+ {'scope': 'node', 'id': n2_id},
+ {'scope': 'node', 'id': n1_id}):
+ with ksft_raises(NlError):
+ nl_shaper.get({'ifindex': cfg.ifindex, 'handle': handle})
+
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def _group_under_netdev(cfg, nl_shaper, bw_max=None):
+ r"""Group queues under a netdev-scope node; caller owns node teardown.
+
+ netdev netdev
+ / \ del Q1,Q2
+ Q1 Q2 -------> (netdev node persists)
+ """
+ group_args = {
+ 'ifindex': cfg.ifindex,
+ 'leaves': [{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1},
+ {'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 2}],
+ 'handle': {'scope': 'netdev'}}
+ if bw_max:
+ group_args['metric'] = 'bps'
+ group_args['bw-max'] = bw_max
+
+ node_handle = nl_shaper.group(group_args)
ksft_eq(node_handle, {'ifindex': cfg.ifindex,
'handle': {'scope': 'netdev'}})
+ del_node = defer(_delete_shaper, cfg, nl_shaper, {'scope': 'netdev'})
+ del_queues = [defer(_delete_shaper, cfg, nl_shaper,
+ {'scope': 'queue', 'id': qid})
+ for qid in (1, 2)]
+
shaper = nl_shaper.get({'ifindex': cfg.ifindex,
'handle': {'scope': 'queue', 'id': 1}})
ksft_eq(shaper, {'ifindex': cfg.ifindex,
'parent': {'scope': 'netdev'},
'handle': {'scope': 'queue', 'id': 1},
- 'weight': 1 })
+ 'weight': 1})
+ for dq in del_queues:
+ dq.exec()
- nl_shaper.delete({'ifindex': cfg.ifindex,
- 'handle': {'scope': 'queue', 'id': 2}})
- nl_shaper.delete({'ifindex': cfg.ifindex,
- 'handle': {'scope': 'queue', 'id': 1}})
+ # Caller owns the node teardown so it can verify the netdev-scope node
+ # survives leaf deletion before removing it.
+ return del_node
+
+def basic_groups(cfg, nl_shaper) -> None:
+ r"""Group queues under a netdev-scope node, then tear it down.
+
+ netdev
+ / \
+ Q1 Q2
+ """
+ _require_queues(cfg, 3)
+
+ _require_caps(cfg, nl_shaper, 'netdev', [], "netdev scope not supported by the device")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "queue scope not supported with nesting and weight")
+
+ del_node = _group_under_netdev(cfg, nl_shaper)
+
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(shapers, [{'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'netdev'}}])
+
+ del_node.exec()
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def basic_groups_with_rate(cfg, nl_shaper) -> None:
+ r"""Rate-limited netdev-scope node outlives deletion of its leaves.
+
+ netdev[10kbps] netdev[10kbps]
+ / \ del Q1,Q2
+ Q1 Q2 -------> (node persists)
+ """
+ bw_max = 10000
+
+ _require_queues(cfg, 3)
+
+ _require_caps(cfg, nl_shaper, 'netdev', ['support-bw-max', 'support-metric-bps'],
+ "device does not support netdev scope rate limiting")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support queue scope shapers with nesting and weight")
+
+ del_node = _group_under_netdev(cfg, nl_shaper, bw_max=bw_max)
# Deleting all the leaves shaper does not affect the node one
# when the latter has 'netdev' scope.
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
- ksft_eq(len(shapers), 1)
+ ksft_eq(shapers, [{'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': bw_max}])
- nl_shaper.delete({'ifindex': cfg.ifindex,
- 'handle': {'scope': 'netdev'}})
+ del_node.exec()
def qgroups(cfg, nl_shaper) -> None:
- if cfg.nr_queues < 4:
- raise KsftSkipEx(f"netdev does not have enough queues min 4 reported {cfg.nr_queues}")
- try:
- caps = nl_shaper.cap_get({'ifindex': cfg.ifindex,
- 'scope':'node'})
- except NlError as e:
- if e.error == 95:
- raise KsftSkipEx("shapers not supported by the device")
- raise
- if not 'support-bw-max' in caps or not 'support-metric-bps' in caps:
- raise KsftSkipEx("device does not support node scope shapers with bw_max and metric bps")
- try:
- caps = nl_shaper.cap_get({'ifindex': cfg.ifindex,
- 'scope':'queue'})
- except NlError as e:
- if e.error == 95:
- raise KsftSkipEx("shapers not supported by the device")
- raise
- if not 'support-nesting' in caps or not 'support-weight' in caps or not 'support-metric-bps' in caps:
- raise KsftSkipEx("device does not support nested queue scope shapers with weight")
+ _require_queues(cfg, 4)
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps'],
+ "device does not support node scope shapers with bw_max and metric bps")
+ _require_caps(cfg, nl_shaper, 'queue',
+ ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
- cfg.groups = True;
node_handle = nl_shaper.group({
'ifindex': cfg.ifindex,
'leaves':[{'handle': {'scope': 'queue', 'id': 1},
@@ -284,18 +608,116 @@ def qgroups(cfg, nl_shaper) -> None:
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
ksft_eq(len(shapers), 0)
+def set_node_shaper(cfg, nl_shaper) -> None:
+ """ Verify a node-scope shaper rate can be updated via .set. """
+ _require_queues(cfg, 2)
+ _require_caps(cfg, nl_shaper, 'node', ['support-bw-max', 'support-metric-bps'],
+ "device does not support node scope shapers with bw_max and metric bps")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ node_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': 10000})
+ node_id = node_handle['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 1})
+
+ # Update the node's rate via .set
+ nl_shaper.set({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id},
+ 'metric': 'bps',
+ 'bw-max': 20000})
+
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': 20000})
+
+ # Cleanup
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': 1})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def group_update_rate(cfg, nl_shaper) -> None:
+ """ Verify re-grouping a node updates its rate while leaving the leaves untouched. """
+ _require_queues(cfg, 3)
+ _require_caps(cfg, nl_shaper, 'node', ['support-bw-max', 'support-metric-bps'],
+ "device does not support node scope shapers with bw_max and metric bps")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ # Create node with Q1, Q2 at bw_max=10000
+ node_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1},
+ {'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': 10000})
+ node_id = node_handle['handle']['id']
+ for i in range(1, 3):
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': i})
+
+ # Update rate via .group on the same node
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1},
+ {'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1}],
+ 'handle': {'scope':'node', 'id': node_id},
+ 'metric': 'bps',
+ 'bw-max': 50000})
+
+ # Verify rate updated
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': 50000})
+
+ # Verify leaves unchanged
+ shaper_q1 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 1}})
+ ksft_eq(shaper_q1, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node_id},
+ 'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1})
+ shaper_q2 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 2}})
+ ksft_eq(shaper_q2, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node_id},
+ 'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1})
+
+ # Make sure we only have 3 shapers including 2 queues and the node
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 3)
+
+ # Cleanup
+ for i in range(1, 3):
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': i})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
def delegation(cfg, nl_shaper) -> None:
- if not cfg.groups:
- raise KsftSkipEx("device does not support node scope")
- try:
- caps = nl_shaper.cap_get({'ifindex': cfg.ifindex,
- 'scope':'node'})
- except NlError as e:
- if e.error == 95:
- raise KsftSkipEx("node scope shapers not supported by the device")
- raise
- if not 'support-nesting' in caps:
- raise KsftSkipEx("device does not support node scope shapers nesting")
+ _require_queues(cfg, 4)
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps', 'support-nesting'],
+ "device does not support node scope shapers with bw_max, metric bps and nesting")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
node_handle = nl_shaper.group({
'ifindex': cfg.ifindex,
@@ -375,20 +797,466 @@ def delegation(cfg, nl_shaper) -> None:
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
ksft_eq(len(shapers), 0)
+def nested_depth_limit(cfg, nl_shaper) -> None:
+ r"""Nest nodes as deep as the device allows to find the max depth.
+
+ netdev
+ |
+ N1 -- Q1
+ |
+ N2 -- Q2
+ |
+ N3 -- Q3
+ : (deepen until the driver rejects)
+ """
+ bw_max = 10000
+
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps', 'support-nesting'],
+ "device does not support node scope shapers with bw_max, metric bps and nesting")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ nq = _require_queues(cfg, 3)
+
+ node_ids = []
+ cleanups = []
+ queue_id = 1
+ max_depth = 0
+ limit_err = None
+
+ # Create initial node with a queue leaf
+ node_id = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves': [{'handle': {'scope': 'queue', 'id': queue_id},
+ 'weight': 1}],
+ 'handle': {'scope': 'node'},
+ 'metric': 'bps',
+ 'bw-max': bw_max})['handle']['id']
+ node_ids.append(node_id)
+ cleanups.append(defer(_delete_shaper, cfg, nl_shaper,
+ {'scope': 'node', 'id': node_id}))
+ cleanups.append(defer(_delete_shaper, cfg, nl_shaper,
+ {'scope': 'queue', 'id': queue_id}))
+ max_depth = 1
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': bw_max})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': queue_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node_id},
+ 'handle': {'scope': 'queue', 'id': queue_id},
+ 'weight': 1})
+ queue_id += 1
+
+ # Keep nesting deeper until the driver rejects or queues run out.
+ while queue_id < nq:
+ parent_id = node_ids[-1]
+ try:
+ node_id = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves': [{'handle': {'scope': 'queue',
+ 'id': queue_id},
+ 'weight': 1}],
+ 'handle': {'scope': 'node'},
+ 'parent': {'scope': 'node',
+ 'id': parent_id},
+ 'metric': 'bps',
+ 'bw-max': bw_max})['handle']['id']
+ except NlError as e:
+ # Only treat "cannot nest deeper" errors as the depth limit;
+ # drivers report it differently (EOPNOTSUPP/ENOSPC/E2BIG/EINVAL).
+ # Anything else (ENOMEM, EIO, EPERM, driver bug) is a real failure.
+ if e.error not in (errno.EOPNOTSUPP, errno.ENOSPC,
+ errno.E2BIG, errno.EINVAL):
+ raise
+ limit_err = e
+ break
+
+ node_ids.append(node_id)
+ cleanups.append(defer(_delete_shaper, cfg, nl_shaper,
+ {'scope': 'node', 'id': node_id}))
+ cleanups.append(defer(_delete_shaper, cfg, nl_shaper,
+ {'scope': 'queue', 'id': queue_id}))
+ max_depth += 1
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node_id},
+ 'parent': {'scope': 'node', 'id': parent_id},
+ 'metric': 'bps',
+ 'bw-max': bw_max})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue',
+ 'id': queue_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node_id},
+ 'handle': {'scope': 'queue', 'id': queue_id},
+ 'weight': 1})
+ queue_id += 1
+
+ if limit_err:
+ print(f"# max nesting depth supported: {max_depth} (errno {limit_err.error})")
+ else:
+ print(f"# max nesting depth tested: {max_depth}")
+ ksft_true(max_depth >= 2,
+ f"max nesting depth: {max_depth}")
+
+ # Cleanup: exec the deferred deletes in reverse creation order, so each
+ # queue leaf and deeper node is removed before its parent node.
+ for cleanup in reversed(cleanups):
+ cleanup.exec()
+ ksft_eq(len(nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)), 0)
+
+def delete_child_reparent(cfg, nl_shaper) -> None:
+ r"""Deleting a child node reparents its queue leaf to the parent.
+
+ netdev netdev
+ | |
+ N1 del N2 N1
+ / | \ -----> / | \
+ Q1 Q2 N2 Q1 Q2 Q3
+ |
+ Q3
+ """
+ n1_bw_max = 10000
+ n2_bw_max = 5000
+
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps', 'support-nesting'],
+ "device does not support node scope shapers with bw_max, metric bps and nesting")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ _require_queues(cfg, 4)
+
+ # Create parent node N1 with Q1, Q2
+ n1_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1},
+ {'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': n1_bw_max})
+ n1_id = n1_handle['handle']['id']
+ for i in range(1, 3):
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': i})
+
+ # Create child node N2 under N1 with Q3
+ n2_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'parent': {'scope': 'node', 'id': n1_id},
+ 'metric': 'bps',
+ 'bw-max': n2_bw_max})
+ n2_id = n2_handle['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 3})
+
+ # Delete child N2 - Q3 should reparent to N1
+ nl_shaper.delete({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n2_id}})
+
+ with ksft_raises(NlError):
+ nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n2_id}})
+
+ shaper_n1 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n1_id}})
+ ksft_eq(shaper_n1, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n1_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': n1_bw_max})
+ shaper_q3 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 3}})
+ ksft_eq(shaper_q3, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n1_id},
+ 'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1})
+
+ # Cleanup
+ for i in range(1, 4):
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': i})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def move_queue_between_nodes(cfg, nl_shaper) -> None:
+ r"""Move a queue between nodes by re-grouping the destination node.
+
+ netdev netdev
+ / \ .group N2 / \
+ N1 N2 {Q1,Q3} N1 N2
+ / \ | -------> | / \
+ Q1 Q2 Q3 Q2 Q1 Q3
+ """
+ n1_bw_max = 10000
+ n2_bw_max = 20000
+
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps', 'support-nesting'],
+ "device does not support node scope shapers with bw_max, metric bps and nesting")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ _require_queues(cfg, 4)
+
+ # Create N1 with Q1, Q2
+ n1_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1},
+ {'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': n1_bw_max})
+ n1_id = n1_handle['handle']['id']
+ for i in range(1, 3):
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': i})
+
+ # Create N2 with Q3
+ n2_handle = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': n2_bw_max})
+ n2_id = n2_handle['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 3})
+
+ # Move Q1 from N1 to N2 by re-grouping N2 with Q1, Q3
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 2},
+ {'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1}],
+ 'handle': {'scope':'node', 'id': n2_id},
+ 'metric': 'bps',
+ 'bw-max': n2_bw_max})
+
+ shaper_n1 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n1_id}})
+ ksft_eq(shaper_n1, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n1_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': n1_bw_max})
+ shaper_n2 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n2_id}})
+ ksft_eq(shaper_n2, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': n2_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': n2_bw_max})
+
+ # Verify Q1 moved to N2
+ shaper_q1 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 1}})
+ ksft_eq(shaper_q1, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n2_id},
+ 'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 2})
+
+ # Verify Q2 still under N1
+ shaper_q2 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 2}})
+ ksft_eq(shaper_q2, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n1_id},
+ 'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1})
+
+ # Verify Q3 remained under N2
+ shaper_q3 = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 3}})
+ ksft_eq(shaper_q3, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': n2_id},
+ 'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1})
+
+ # Cleanup
+ for i in range(1, 4):
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': i})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
+def reject_reparenting(cfg, nl_shaper) -> None:
+ r"""Reject reparenting an existing node; the hierarchy stays intact.
+
+ netdev
+ / \ rejected: N3 -> netdev
+ N1 N2 rejected: N1 -> N2
+ / \ | (both EOPNOTSUPP)
+ Q1 N3 Q2
+ |
+ Q3
+ """
+ node1_bw_max = 10000
+ node2_bw_max = 5000
+ node3_bw_max = 20000
+
+ _require_caps(cfg, nl_shaper, 'node',
+ ['support-bw-max', 'support-metric-bps', 'support-nesting'],
+ "device does not support node scope shapers with bw_max, metric bps and nesting")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+
+ _require_queues(cfg, 4)
+
+ # Create Node1 under netdev with Q1.
+ node1_id = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': node1_bw_max})['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 1})
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'node', 'id': node1_id})
+
+ # Create Node2 under netdev with Q2.
+ node2_id = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': node2_bw_max})['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 2})
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'node', 'id': node2_id})
+
+ # Create Node3 nested under Node1 with Q3.
+ node3_id = nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1}],
+ 'handle': {'scope':'node'},
+ 'metric': 'bps',
+ 'bw-max': node3_bw_max,
+ 'parent': {'scope': 'node', 'id': node1_id}})['handle']['id']
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'queue', 'id': 3})
+ defer(_delete_shaper, cfg, nl_shaper, {'scope': 'node', 'id': node3_id})
+
+ # Reparenting a nested node up to netdev must fail.
+ with ksft_raises(NlError) as cm:
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1}],
+ 'handle': {'scope':'node', 'id': node3_id},
+ 'parent': {'scope': 'netdev'}})
+ if cm.exception:
+ ksft_eq(cm.exception.error, errno.EOPNOTSUPP)
+
+ # Reparenting a node under another node must fail as well.
+ with ksft_raises(NlError) as cm:
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 1}],
+ 'handle': {'scope':'node', 'id': node1_id},
+ 'parent': {'scope': 'node', 'id': node2_id}})
+ if cm.exception:
+ ksft_eq(cm.exception.error, errno.EOPNOTSUPP)
+
+ # Updating a node with the same parent must succeed.
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 5}],
+ 'handle': {'scope':'node', 'id': node1_id},
+ 'parent': {'scope': 'netdev'}})
+
+ # Updating a node without specifying the parent must succeed.
+ nl_shaper.group({
+ 'ifindex': cfg.ifindex,
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 7}],
+ 'handle': {'scope':'node', 'id': node2_id}})
+
+ # The rejected reparents must have left the hierarchy intact.
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node1_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node1_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': node1_bw_max})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node2_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node2_id},
+ 'parent': {'scope': 'netdev'},
+ 'metric': 'bps',
+ 'bw-max': node2_bw_max})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node3_id}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'node', 'id': node3_id},
+ 'parent': {'scope': 'node', 'id': node1_id},
+ 'metric': 'bps',
+ 'bw-max': node3_bw_max})
+
+ # Verify the leaf weights were updated and parents unchanged.
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 1}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node1_id},
+ 'handle': {'scope': 'queue', 'id': 1},
+ 'weight': 5})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 2}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node2_id},
+ 'handle': {'scope': 'queue', 'id': 2},
+ 'weight': 7})
+ shaper = nl_shaper.get({'ifindex': cfg.ifindex,
+ 'handle': {'scope': 'queue', 'id': 3}})
+ ksft_eq(shaper, {'ifindex': cfg.ifindex,
+ 'parent': {'scope': 'node', 'id': node3_id},
+ 'handle': {'scope': 'queue', 'id': 3},
+ 'weight': 1})
+
+ # Cleanup. Delete the nodes explicitly instead of relying on the
+ # empty-node auto-delete: a kernel that wrongly accepts a reparent may
+ # mishandle the leaf accounting and leave a node behind. Removing them
+ # by handle keeps a failing run from leaking state into later tests.
+ for i in range(1, 4):
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': i})
+ for nid in (node1_id, node2_id, node3_id):
+ _delete_shaper(cfg, nl_shaper, {'scope': 'node', 'id': nid})
+ shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
+ ksft_eq(len(shapers), 0)
+
def queue_update(cfg, nl_shaper) -> None:
- if cfg.nr_queues < 4:
- raise KsftSkipEx(f"netdev does not have enough queues min 4 reported {cfg.nr_queues}")
+ nq = _require_queues(cfg, 4)
if not cfg.queues:
raise KsftSkipEx("device does not support queue scope")
+ netnl = EthtoolFamily()
+ channels = netnl.channels_get({'header': {'dev-index': cfg.ifindex}})
+ ch_type = 'combined' if channels['combined-count'] else 'tx'
+
for i in range(3):
nl_shaper.set({'ifindex': cfg.ifindex,
'handle': {'scope': 'queue', 'id': i},
'metric': 'bps',
'bw-max': (i + 1) * 1000})
+ defer(cmd, f"ethtool -L {cfg.dev['ifname']} {ch_type} {nq}")
+
# Delete a channel, with no shapers configured on top of the related
# queue: no changes expected
- cmd(f"ethtool -L {cfg.dev['ifname']} {cfg.rx_type} 3", timeout=10)
+ cmd(f"ethtool -L {cfg.dev['ifname']} {ch_type} 3")
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
ksft_eq(shapers, [{'ifindex': cfg.ifindex,
'parent': {'scope': 'netdev'},
@@ -408,7 +1276,7 @@ def queue_update(cfg, nl_shaper) -> None:
# Delete a channel, with a shaper configured on top of the related
# queue: the shaper must be deleted, too
- cmd(f"ethtool -L {cfg.dev['ifname']} {cfg.rx_type} 2", timeout=10)
+ cmd(f"ethtool -L {cfg.dev['ifname']} {ch_type} 2")
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
ksft_eq(shapers, [{'ifindex': cfg.ifindex,
@@ -423,7 +1291,7 @@ def queue_update(cfg, nl_shaper) -> None:
'bw-max': 2000}])
# Restore the original channels number, no expected changes
- cmd(f"ethtool -L {cfg.dev['ifname']} {cfg.rx_type} {cfg.nr_queues}", timeout=10)
+ cmd(f"ethtool -L {cfg.dev['ifname']} {ch_type} {nq}")
shapers = nl_shaper.get({'ifindex': cfg.ifindex}, dump=True)
ksft_eq(shapers, [{'ifindex': cfg.ifindex,
'parent': {'scope': 'netdev'},
@@ -443,34 +1311,57 @@ def queue_update(cfg, nl_shaper) -> None:
def dup_leaves(cfg, nl_shaper) -> None:
""" Ensure that the kernel rejects duplicate leaves. """
- if not cfg.groups:
- raise KsftSkipEx("device does not support node scope")
+ _require_caps(cfg, nl_shaper, 'node', ['support-bw-max', 'support-metric-bps'],
+ "device does not support node scope shapers with bw_max and metric bps")
+ _require_caps(cfg, nl_shaper, 'queue', ['support-nesting', 'support-weight'],
+ "device does not support nested queue scope shapers with weight")
+ node_handle = None
with ksft_raises(NlError) as cm:
- nl_shaper.group({
+ node_handle = nl_shaper.group({
'ifindex': cfg.ifindex,
- 'leaves':[{'handle': {'scope': 'queue', 'id': 0}},
- {'handle': {'scope': 'queue', 'id': 0}}],
+ 'leaves':[{'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 1},
+ {'handle': {'scope': 'queue', 'id': 0},
+ 'weight': 2}],
'handle': {'scope':'node'},
'metric': 'bps',
'bw-max': 10000})
+
+ # Clean up in case the kernel wrongly accepted the request.
+ if node_handle:
+ _delete_shaper(cfg, nl_shaper, node_handle['handle'])
+ _delete_shaper(cfg, nl_shaper, {'scope': 'queue', 'id': 0})
+
+ # ksft_raises() has already recorded the failure if nothing was raised.
+ if cm.exception is None:
+ return
ksft_eq(cm.exception.error, errno.EINVAL)
def main() -> None:
with NetDrvEnv(__file__, queue_count=4) as cfg:
cfg.queues = False
cfg.netdev = False
- cfg.groups = False
- cfg.nr_queues = 0
ksft_run([get_shapers,
get_caps,
set_qshapers,
del_qshapers,
set_nshapers,
del_nshapers,
+ set_all_supported_attrs,
+ invalid_set_preserves_state,
+ mixed_parent_group_requires_parent,
+ recursive_empty_node_cleanup,
basic_groups,
+ basic_groups_with_rate,
qgroups,
+ set_node_shaper,
+ group_update_rate,
delegation,
+ nested_depth_limit,
+ delete_child_reparent,
+ move_queue_between_nodes,
+ reject_reparenting,
dup_leaves,
queue_update],
args=(cfg, NetshaperFamily()))
diff --git a/tools/testing/selftests/drivers/net/so_txtime.py b/tools/testing/selftests/drivers/net/so_txtime.py
index adf6c848d6d8..a097fae0b335 100755
--- a/tools/testing/selftests/drivers/net/so_txtime.py
+++ b/tools/testing/selftests/drivers/net/so_txtime.py
@@ -12,6 +12,7 @@ import time
from lib.py import ksft_exit, ksft_run, ksft_variants
from lib.py import KsftNamedVariant, KsftSkipEx
from lib.py import NetDrvEpEnv, bkg, cmd, defer, tc
+from lib.py import CmdExitFailure
def test_so_txtime(cfg, clockid, ipver, args_tx, args_rx, expect_success):
@@ -27,7 +28,7 @@ def test_so_txtime(cfg, clockid, ipver, args_tx, args_rx, expect_success):
cmd_addr = f"-S {cfg.addr_v[ipver]} -D {cfg.remote_addr_v[ipver]}"
cmd_args = f"-{ipver} -c {clockid} -t {tstart} {cmd_addr}"
cmd_rx = f"{cfg.bin_remote} {cmd_args} {args_rx} -r"
- cmd_tx = f"{cfg.bin_local} {cmd_args} {args_tx}"
+ cmd_tx = f"{cfg.bin_local} -m 100 {cmd_args} {args_tx}"
expect_fail = not expect_success
if slow_machine:
@@ -45,7 +46,11 @@ def _qdisc_setup(ifname, qdisc, optargs=""):
"""
orig = tc(f"qdisc show dev {ifname} root", json=True)[0].get("kind", None)
defer(tc, f"qdisc replace dev {ifname} root {orig}")
- tc(f"qdisc replace dev {ifname} root {qdisc} {optargs}")
+ try:
+ tc(f"qdisc del dev {ifname} root")
+ except CmdExitFailure:
+ pass
+ tc(f"qdisc replace dev {ifname} root handle 1: {qdisc} {optargs}")
def _test_variants_fq():
@@ -96,11 +101,21 @@ def _test_variants_etf():
def test_so_txtime_etf(cfg, ipver, args_tx, args_rx, expect_fail):
"""Run all variants of etf tests."""
cfg.require_ipver(ipver)
+
+ # root qdisc for background traffic (e.g., bkg())
+ _qdisc_setup(cfg.ifname, "prio")
+
+ # leaf ETF qdisc only for intended packets
try:
- _qdisc_setup(cfg.ifname, "etf", "clockid CLOCK_TAI delta 400000")
+ etf_args = "clockid CLOCK_TAI delta 400000"
+ tc(f"qdisc add dev {cfg.ifname} parent 1:1 handle 10: etf {etf_args}")
except Exception as e:
raise KsftSkipEx("tc does not support qdisc etf. skipping") from e
+ # redirect mark 100 to leaf
+ filter_args = "protocol all handle 100 fw flowid 1:1"
+ tc(f"filter add dev {cfg.ifname} parent 1: {filter_args}")
+
test_so_txtime(cfg, "tai", ipver, args_tx, args_rx, expect_fail)
diff --git a/tools/testing/selftests/drivers/net/xdp.py b/tools/testing/selftests/drivers/net/xdp.py
index 2ad5932299e8..0369929f3c51 100755
--- a/tools/testing/selftests/drivers/net/xdp.py
+++ b/tools/testing/selftests/drivers/net/xdp.py
@@ -172,25 +172,45 @@ def _test_pass(cfg, bpf_info, msg_sz):
ksft_eq(stats[XDPStats.RX.value], stats[XDPStats.PASS.value], "RX and PASS stats mismatch")
-def test_xdp_native_pass_sb(cfg):
+_ipvers = [
+ KsftNamedVariant("ipv4", "4"),
+ KsftNamedVariant("ipv6", "6"),
+]
+
+
+def _set_ipver_defer_restore(cfg, ipver):
+ old_ipver = cfg.addr_ipver
+ cfg.set_ipver(ipver)
+ defer(cfg.set_ipver, old_ipver)
+
+
+@ksft_variants(_ipvers)
+def test_xdp_native_pass_sb(cfg, ipver):
"""
Tests the XDP_PASS action for single buffer case.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
bpf_info = BPFProgInfo("xdp_prog", "xdp_native.bpf.o", "xdp", 1500)
_test_pass(cfg, bpf_info, 256)
-def test_xdp_native_pass_mb(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_pass_mb(cfg, ipver):
"""
Tests the XDP_PASS action for a multi-buff size.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
bpf_info = BPFProgInfo("xdp_prog_frags", "xdp_native.bpf.o", "xdp.frags", 9000)
_test_pass(cfg, bpf_info, 8000)
@@ -219,25 +239,33 @@ def _test_drop(cfg, bpf_info, msg_sz):
ksft_eq(stats[XDPStats.RX.value], stats[XDPStats.DROP.value], "RX and DROP stats mismatch")
-def test_xdp_native_drop_sb(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_drop_sb(cfg, ipver):
"""
Tests the XDP_DROP action for a signle-buff case.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
bpf_info = BPFProgInfo("xdp_prog", "xdp_native.bpf.o", "xdp", 1500)
_test_drop(cfg, bpf_info, 256)
-def test_xdp_native_drop_mb(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_drop_mb(cfg, ipver):
"""
Tests the XDP_DROP action for a multi-buff case.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
bpf_info = BPFProgInfo("xdp_prog_frags", "xdp_native.bpf.o", "xdp.frags", 9000)
_test_drop(cfg, bpf_info, 8000)
@@ -287,13 +315,17 @@ def _test_xdp_native_tx(cfg, bpf_info, payload_lens):
ksft_eq(stats[XDPStats.TX.value], expected_pkts, "TX stats mismatch")
-def test_xdp_native_tx_sb(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_tx_sb(cfg, ipver):
"""
Tests the XDP_TX action for a single-buff case.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
bpf_info = BPFProgInfo("xdp_prog", "xdp_native.bpf.o", "xdp", 1500)
# Ensure there's enough room for an ETH / IP / UDP header
@@ -302,13 +334,17 @@ def test_xdp_native_tx_sb(cfg):
_test_xdp_native_tx(cfg, bpf_info, [0, 1500 // 2, 1500 - pkt_hdr_len])
-def test_xdp_native_tx_mb(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_tx_mb(cfg, ipver):
"""
Tests the XDP_TX action for a multi-buff case.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
bpf_info = BPFProgInfo("xdp_prog_frags", "xdp_native.bpf.o",
"xdp.frags", 9000)
# The first packet ensures we exercise the fragmented code path.
@@ -447,13 +483,17 @@ def _test_xdp_native_tail_adjst(cfg, pkt_sz_lst, offset_lst):
return {"status": "pass"}
-def test_xdp_native_adjst_tail_grow_data(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_adjst_tail_grow_data(cfg, ipver):
"""
Tests the XDP tail adjustment by growing packet data.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
pkt_sz_lst = [512, 1024, 2048]
offset_lst = [1, 16, 32, 64, 128, 256]
res = _test_xdp_native_tail_adjst(
@@ -465,13 +505,17 @@ def test_xdp_native_adjst_tail_grow_data(cfg):
_validate_res(res, offset_lst, pkt_sz_lst)
-def test_xdp_native_adjst_tail_shrnk_data(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_adjst_tail_shrnk_data(cfg, ipver):
"""
Tests the XDP tail adjustment by shrinking packet data.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
pkt_sz_lst = [512, 1024, 2048]
offset_lst = [-16, -32, -64, -128, -256]
res = _test_xdp_native_tail_adjst(
@@ -535,7 +579,7 @@ def _test_xdp_native_head_adjst(cfg, prog, pkt_sz_lst, offset_lst):
# after we eat into it. We send large-enough packets, but if HDS
# is enabled head will only contain headers. Don't try to eat
# more than 28 bytes (UDPv4 + eth hdr left: (14 + 20 + 8) - 14)
- l2_cut_off = 28 if cfg.addr_ipver == 4 else 48
+ l2_cut_off = 28 if cfg.addr_ipver == "4" else 48
if pkt_sz > hds_thresh and offset > l2_cut_off:
ksft_pr(
f"Failed run: pkt_sz ({pkt_sz}) > HDS threshold ({hds_thresh}) and "
@@ -579,18 +623,22 @@ def _test_xdp_native_head_adjst(cfg, prog, pkt_sz_lst, offset_lst):
return {"status": "pass"}
-def test_xdp_native_adjst_head_grow_data(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_adjst_head_grow_data(cfg, ipver):
"""
Tests the XDP headroom growth support.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
This function sets up the packet size and offset lists, then calls the
_test_xdp_native_head_adjst_mb function to perform the actual test. The
test is passed if the headroom is successfully extended for given packet
sizes and offsets.
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
pkt_sz_lst = [512, 1024, 2048]
# Negative values result in headroom shrinking, resulting in growing of payload
@@ -600,18 +648,22 @@ def test_xdp_native_adjst_head_grow_data(cfg):
_validate_res(res, offset_lst, pkt_sz_lst)
-def test_xdp_native_adjst_head_shrnk_data(cfg):
+@ksft_variants(_ipvers)
+def test_xdp_native_adjst_head_shrnk_data(cfg, ipver):
"""
Tests the XDP headroom shrinking support.
Args:
cfg: Configuration object containing network settings.
+ ipver: IP version to use ("4" or "6").
This function sets up the packet size and offset lists, then calls the
_test_xdp_native_head_adjst_mb function to perform the actual test. The
test is passed if the headroom is successfully shrunk for given packet
sizes and offsets.
"""
+ _set_ipver_defer_restore(cfg, ipver)
+
pkt_sz_lst = [512, 1024, 2048]
# Positive values result in headroom growing, resulting in shrinking of payload
@@ -621,12 +673,19 @@ def test_xdp_native_adjst_head_shrnk_data(cfg):
_validate_res(res, offset_lst, pkt_sz_lst)
-@ksft_variants([
- KsftNamedVariant("pass", XDPAction.PASS),
- KsftNamedVariant("drop", XDPAction.DROP),
- KsftNamedVariant("tx", XDPAction.TX),
-])
-def test_xdp_native_qstats(cfg, act):
+def _qstats_variants():
+ actions = [
+ ("pass", XDPAction.PASS),
+ ("drop", XDPAction.DROP),
+ ("tx", XDPAction.TX),
+ ]
+ for ipver in ["4", "6"]:
+ for name, act in actions:
+ yield KsftNamedVariant(f"{name}_ipv{ipver}", act, ipver)
+
+
+@ksft_variants(_qstats_variants())
+def test_xdp_native_qstats(cfg, act, ipver):
"""
Send 1000 messages. Expect XDP action specified in @act.
Make sure the packets were counted to interface level qstats
@@ -634,6 +693,7 @@ def test_xdp_native_qstats(cfg, act):
"""
cfg.require_cmd("socat")
+ _set_ipver_defer_restore(cfg, ipver)
bpf_info = BPFProgInfo("xdp_prog", "xdp_native.bpf.o", "xdp", 1500)
prog_info = _load_xdp_prog(cfg, bpf_info)
diff --git a/tools/testing/selftests/drivers/ntsync/ntsync.c b/tools/testing/selftests/drivers/ntsync/ntsync.c
index e6a37214aa46..1f0dc43bb4c0 100644
--- a/tools/testing/selftests/drivers/ntsync/ntsync.c
+++ b/tools/testing/selftests/drivers/ntsync/ntsync.c
@@ -8,12 +8,18 @@
#define _GNU_SOURCE
#include <sys/ioctl.h>
#include <sys/stat.h>
+#include <sys/wait.h>
#include <fcntl.h>
+#include <sched.h>
#include <time.h>
#include <pthread.h>
#include <linux/ntsync.h>
#include "kselftest_harness.h"
+#ifndef CLONE_NEWTIME
+#define CLONE_NEWTIME 0x00000080
+#endif
+
static int read_sem_state(int sem, __u32 *count, __u32 *max)
{
struct ntsync_sem_args args;
@@ -968,7 +974,7 @@ TEST(wake_all)
auto_event_args.manual = false;
auto_event_args.signaled = true;
objs[3] = ioctl(fd, NTSYNC_IOC_CREATE_EVENT, &auto_event_args);
- EXPECT_EQ(0, objs[3]);
+ EXPECT_LE(0, objs[3]);
wait_args.timeout = get_abs_timeout(1000);
wait_args.objs = (uintptr_t)objs;
@@ -1340,4 +1346,129 @@ TEST(stress_wait)
close(stress_device);
}
+TEST(wait_args_validation)
+{
+ struct ntsync_sem_args sem_args = { .count = 1, .max = 1 };
+ struct ntsync_wait_args wait_args = {0};
+ struct timespec timeout;
+ int fd, fd2, sem, ret;
+ __u32 index;
+
+ fd = open("/dev/ntsync", O_CLOEXEC | O_RDONLY);
+ ASSERT_GE(fd, 0);
+
+ fd2 = open("/dev/ntsync", O_CLOEXEC | O_RDONLY);
+ ASSERT_GE(fd2, 0);
+
+ sem = ioctl(fd, NTSYNC_IOC_CREATE_SEM, &sem_args);
+ EXPECT_GE(sem, 0);
+
+ ret = wait_any(fd, 1, &sem, 0, &index);
+ EXPECT_EQ(-1, ret);
+ EXPECT_EQ(EINVAL, errno);
+
+ ret = wait_all(fd, 1, &sem, 0, &index);
+ EXPECT_EQ(-1, ret);
+ EXPECT_EQ(EINVAL, errno);
+
+ clock_gettime(CLOCK_MONOTONIC, &timeout);
+ wait_args.timeout = timeout.tv_sec * 1000000000ULL + timeout.tv_nsec;
+ wait_args.count = 0;
+ wait_args.objs = 0;
+ wait_args.owner = 123;
+ wait_args.pad = 1;
+ ret = ioctl(fd, NTSYNC_IOC_WAIT_ANY, &wait_args);
+ EXPECT_EQ(-1, ret);
+ EXPECT_EQ(EINVAL, errno);
+
+ ret = wait_any(fd2, 1, &sem, 123, &index);
+ EXPECT_EQ(-1, ret);
+ EXPECT_EQ(EINVAL, errno);
+
+ close(sem);
+ close(fd2);
+ close(fd);
+}
+
+/*
+ * Absolute MONOTONIC timeouts must honour the caller's time namespace.
+ * With a negative monotonic offset, a 100 ms wait must still take ~100 ms
+ * of namespace time (not return immediately against the host clock).
+ */
+TEST(wait_any_monotonic_timens)
+{
+ struct ntsync_sem_args sem_args = {0};
+ struct ntsync_wait_args wait_args = {0};
+ struct timespec start, end;
+ char buf[64];
+ __u64 elapsed_ns;
+ int fd, offset_fd, sem, ret, status, len;
+ pid_t pid;
+
+ if (access("/proc/self/ns/time", F_OK))
+ SKIP(return, "Time namespaces are not supported");
+
+ fd = open("/dev/ntsync", O_CLOEXEC | O_RDONLY);
+ if (fd < 0)
+ SKIP(return, "/dev/ntsync is not available");
+
+ ret = unshare(CLONE_NEWTIME);
+ if (ret) {
+ close(fd);
+ if (errno == EPERM)
+ SKIP(return, "need CAP_SYS_ADMIN for CLONE_NEWTIME");
+ ASSERT_EQ(0, ret);
+ }
+
+ len = snprintf(buf, sizeof(buf), "%d %d 0", CLOCK_MONOTONIC, -10);
+ offset_fd = open("/proc/self/timens_offsets", O_WRONLY);
+ ASSERT_LE(0, offset_fd);
+ ASSERT_EQ(len, write(offset_fd, buf, len));
+ close(offset_fd);
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (!pid) {
+ int obj;
+
+ sem_args.count = 0;
+ sem_args.max = 1;
+ sem = ioctl(fd, NTSYNC_IOC_CREATE_SEM, &sem_args);
+ if (sem < 0)
+ _exit(1);
+
+ obj = sem;
+ wait_args.timeout = get_abs_timeout(100);
+ wait_args.objs = (uintptr_t)&obj;
+ wait_args.count = 1;
+ wait_args.owner = 123;
+ wait_args.index = 0xdeadbeef;
+
+ if (clock_gettime(CLOCK_MONOTONIC, &start))
+ _exit(2);
+ ret = ioctl(fd, NTSYNC_IOC_WAIT_ANY, &wait_args);
+ if (clock_gettime(CLOCK_MONOTONIC, &end))
+ _exit(2);
+
+ if (ret != -1 || errno != ETIMEDOUT)
+ _exit(3);
+
+ elapsed_ns = (end.tv_sec - start.tv_sec) * 1000000000ULL +
+ (end.tv_nsec - start.tv_nsec);
+ /* Without timens conversion this returns in ~0 ms. */
+ if (elapsed_ns < 50 * 1000000ULL)
+ _exit(4);
+ if (elapsed_ns > 1000 * 1000000ULL)
+ _exit(5);
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ EXPECT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ close(fd);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/.gitignore b/tools/testing/selftests/exec/.gitignore
index 7f3d1ae762ec..e42ecd4c908d 100644
--- a/tools/testing/selftests/exec/.gitignore
+++ b/tools/testing/selftests/exec/.gitignore
@@ -19,3 +19,14 @@ null-argv
xxxxxxxx*
pipe
S_I*.test
+binfmt_misc_bpf
+binfmt_misc_interplimit
+binfmt_bpf_interp
+binfmt_bpf_app
+binfmt_misc_transparent
+binfmt_transparent_interp
+binfmt_misc_loader
+binfmt_loader_payload
+binfmt_loader_payload_static
+*.bpf.o
+vmlinux.h
diff --git a/tools/testing/selftests/exec/Makefile b/tools/testing/selftests/exec/Makefile
index 45a3cfc435cf..b640af8f02b5 100644
--- a/tools/testing/selftests/exec/Makefile
+++ b/tools/testing/selftests/exec/Makefile
@@ -21,9 +21,56 @@ TEST_GEN_PROGS += recursion-depth
TEST_GEN_PROGS += null-argv
TEST_GEN_PROGS += check-exec
+# binfmt_misc must not be reachable as an exec source or as a stacking layer,
+# or an 'F' entry can pin the instance that owns it. Unprivileged, no bpf.
+TEST_GEN_PROGS += binfmt_misc_selfpin
+
+# The interpreters an 'F' or 'B' entry pre-opens are charged against
+# UCOUNT_BINFMT_MISC_INTERPRETERS. Unprivileged, no bpf.
+TEST_GEN_PROGS += binfmt_misc_interplimit
+
+# 'D' (register disabled) binfmt_misc test: an entry that exists but does
+# not dispatch until it is enabled. Static magic entry, no bpf toolchain.
+TEST_GEN_PROGS += binfmt_misc_disabled
+
+# Static ('T' flag) transparent binfmt_misc test; the asserting interpreter
+# is shared with the bpf harness's transparent case. No bpf toolchain needed.
+TEST_GEN_PROGS += binfmt_misc_transparent
+TEST_GEN_FILES += binfmt_transparent_interp
+
+# 'L' (loader substitution) binfmt_misc test: the payload runs as the main
+# image with a copy of the system loader substituted for its PT_INTERP and
+# asserts the native identity from inside; the static build proves the
+# override is dropped for a binary without PT_INTERP.
+TEST_GEN_PROGS += binfmt_misc_loader
+TEST_GEN_FILES += binfmt_loader_payload binfmt_loader_payload_static
+
+# binfmt_misc bpf-backed ('B') handler test: a libbpf harness plus its
+# struct_ops objects and the test interpreter/app it routes between. Only
+# built when clang, bpftool, the vmlinux BTF and libbpf are all present
+# (HAVE_BPF_TOOLCHAIN=y forces it) so the other exec selftests don't grow
+# a bpf toolchain dependency.
+CLANG ?= clang
+BPFTOOL ?= bpftool
+VMLINUX_BTF ?= /sys/kernel/btf/vmlinux
+HAVE_BPF_TOOLCHAIN ?= $(shell command -v $(CLANG) >/dev/null 2>&1 && \
+ command -v $(BPFTOOL) >/dev/null 2>&1 && \
+ test -r $(VMLINUX_BTF) && \
+ pkg-config --exists libbpf 2>/dev/null && echo y)
+ifeq ($(HAVE_BPF_TOOLCHAIN),y)
+TEST_GEN_PROGS += binfmt_misc_bpf
+TEST_GEN_FILES += bpf_interp.bpf.o nix_origin.bpf.o transparent.bpf.o
+TEST_GEN_FILES += loader.bpf.o interp_bind.bpf.o
+TEST_GEN_FILES += binfmt_bpf_interp binfmt_bpf_app binfmt_bind_interp
+else
+$(info exec selftests: skipping binfmt_misc_bpf, needs clang, bpftool, vmlinux BTF and libbpf)
+endif
+
EXTRA_CLEAN := $(OUTPUT)/subdir.moved $(OUTPUT)/execveat.moved $(OUTPUT)/xxxxx* \
$(OUTPUT)/S_I*.test
+LOCAL_HDRS += binfmt_misc_common.h
+
include ../lib.mk
CHECK_EXEC_SAMPLES := $(top_srcdir)/samples/check-exec
@@ -55,3 +102,49 @@ $(OUTPUT)/script-exec.inc: $(CHECK_EXEC_SAMPLES)/script-exec.inc
cp $< $@
$(OUTPUT)/script-noexec.inc: $(CHECK_EXEC_SAMPLES)/script-noexec.inc
cp $< $@
+
+# Reuses setup_userns()/write_file() from the filesystems selftests. Their
+# wrappers.h wants the uapi headers, so ask for them here rather than widening
+# CFLAGS for every program in this directory.
+$(OUTPUT)/binfmt_misc_selfpin: CFLAGS += $(TOOLS_INCLUDES)
+$(OUTPUT)/binfmt_misc_selfpin: ../filesystems/utils.c
+$(OUTPUT)/binfmt_misc_interplimit: CFLAGS += $(TOOLS_INCLUDES)
+$(OUTPUT)/binfmt_misc_interplimit: ../filesystems/utils.c
+
+# --- binfmt_misc bpf ('B') handler test ---------------------------------
+# The struct_ops bpf objects are compiled against the running kernel's BTF.
+# CLANG/BPFTOOL/VMLINUX_BTF are set above next to the toolchain check;
+# override LIBBPF_CFLAGS/LDLIBS to point at a libbpf install.
+BPF_CFLAGS ?= -I$(OUTPUT)
+LIBBPF_CFLAGS ?=
+LIBBPF_LDLIBS ?= -lbpf -lelf -lz
+
+$(OUTPUT)/vmlinux.h:
+ $(BPFTOOL) btf dump file $(VMLINUX_BTF) format c > $@
+
+# BPF_NO_KFUNC_PROTOTYPES: the programs declare the kfuncs they use themselves.
+$(OUTPUT)/%.bpf.o: %.bpf.c $(OUTPUT)/vmlinux.h
+ $(CLANG) -g -O2 -target bpf -mcpu=v3 -DBPF_NO_KFUNC_PROTOTYPES \
+ $(BPF_CFLAGS) $(LIBBPF_CFLAGS) -c $< -o $@
+
+$(OUTPUT)/binfmt_misc_bpf: binfmt_misc_bpf.c binfmt_misc_common.h
+ $(CC) $(CFLAGS) $(LIBBPF_CFLAGS) $(LDFLAGS) $< $(LIBBPF_LDLIBS) -o $@
+
+$(OUTPUT)/binfmt_bpf_interp: binfmt_bpf_interp.c
+ $(CC) $(CFLAGS) $(LDFLAGS) $< -o $@
+
+$(OUTPUT)/binfmt_bind_interp: binfmt_bind_interp.c
+ $(CC) $(CFLAGS) $(LDFLAGS) $< -o $@
+
+$(OUTPUT)/binfmt_loader_payload: binfmt_loader_payload.c binfmt_misc_common.h
+ $(CC) $(CFLAGS) $(LDFLAGS) -fPIE -pie $< -o $@
+
+$(OUTPUT)/binfmt_loader_payload_static: binfmt_loader_payload.c binfmt_misc_common.h
+ $(CC) $(CFLAGS) $(LDFLAGS) -static $< -o $@
+
+# PT_INTERP is set to the literal "$ORIGIN/binfmt_bpf_interp"; the nix_origin
+# handler resolves it relative to the binary at run time.
+$(OUTPUT)/binfmt_bpf_app: binfmt_bpf_app.c
+ $(CC) $(CFLAGS) $(LDFLAGS) -Wl,--dynamic-linker,'$$ORIGIN/binfmt_bpf_interp' $< -o $@
+
+EXTRA_CLEAN += $(OUTPUT)/vmlinux.h $(OUTPUT)/*.bpf.o
diff --git a/tools/testing/selftests/exec/binfmt_bind_interp.c b/tools/testing/selftests/exec/binfmt_bind_interp.c
new file mode 100644
index 000000000000..06d65062856b
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bind_interp.c
@@ -0,0 +1,14 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test interpreter for the bound-interpreter case of the binfmt_misc_bpf
+ * selftest. Two copies are installed at different paths and bound to one
+ * entry under different names; printing argv[0] - the path the kernel ran
+ * this copy under - tells the harness which of them the load program picked.
+ */
+#include <stdio.h>
+
+int main(int argc, char **argv)
+{
+ printf("BIND_RAN %s\n", argc > 0 ? argv[0] : "");
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_bpf_app.c b/tools/testing/selftests/exec/binfmt_bpf_app.c
new file mode 100644
index 000000000000..472270f148bc
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bpf_app.c
@@ -0,0 +1,12 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A relocatable binary for the binfmt_misc_bpf $ORIGIN case. The Makefile
+ * links it with PT_INTERP set to the literal "$ORIGIN/binfmt_bpf_interp"
+ * (-Wl,--dynamic-linker), which the kernel ELF loader cannot resolve. The
+ * nix_origin bpf handler resolves it relative to this binary's directory and
+ * routes execution to the co-located interpreter.
+ */
+int main(void)
+{
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_bpf_interp.c b/tools/testing/selftests/exec/binfmt_bpf_interp.c
new file mode 100644
index 000000000000..2db205f095b2
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bpf_interp.c
@@ -0,0 +1,15 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test interpreter for the binfmt_misc_bpf selftest. A bpf-backed 'B' handler
+ * routes a matched binary here; printing this marker proves the program's
+ * chosen interpreter actually ran.
+ */
+#include <unistd.h>
+
+int main(int argc, char **argv)
+{
+ (void)argc;
+ (void)argv;
+ write(1, "BPF_INTERP_RAN\n", 15);
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_loader_payload.c b/tools/testing/selftests/exec/binfmt_loader_payload.c
new file mode 100644
index 000000000000..272db8efb4b5
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_loader_payload.c
@@ -0,0 +1,146 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Payload for the binfmt_misc 'L' (loader substitution) selftest. It is
+ * executed as the MAIN image - a fully native exec - with the registered
+ * interpreter substituted for its PT_INTERP, and asserts the native
+ * identity from the inside. Exits 0 when every surface checks out.
+ *
+ * Modes, selected by the orchestrator via the environment:
+ * - default: full assertions, path-based ones included
+ * - BINFMT_TEST_MEMFD=1: executed from an inaccessible memfd, skip
+ * the path-based assertions
+ * - BINFMT_TEST_STATIC=1: static build; the override was dropped, so
+ * expect no interpreter at all
+ */
+#define _GNU_SOURCE
+#include <elf.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/auxv.h>
+#include <unistd.h>
+
+#include "binfmt_misc_common.h"
+
+/* Start of our own mapped image, courtesy of the linker. */
+extern const char __ehdr_start[];
+
+/* An image is never this large; used to bracket "within our image". */
+#define IMAGE_SPAN (16UL << 20)
+
+static int failed;
+
+static void check(int cond, const char *what)
+{
+ if (cond)
+ return;
+ fprintf(stderr, "[payload] FAILED: %s (errno %d)\n", what, errno);
+ failed = 1;
+}
+
+/* Return whether /proc/self/maps names a path starting with @prefix. */
+static int maps_has_prefix(const char *prefix)
+{
+ char *line = NULL;
+ size_t len = 0;
+ int found = 0;
+ FILE *f;
+
+ f = fopen("/proc/self/maps", "r");
+ if (!f)
+ return -1;
+ while (getline(&line, &len, f) > 0) {
+ char *path = strchr(line, '/');
+
+ if (path && !strncmp(path, prefix, strlen(prefix))) {
+ found = 1;
+ break;
+ }
+ }
+ free(line);
+ fclose(f);
+ return found;
+}
+
+int main(int argc, char *argv[])
+{
+ const char *binary = getenv("BINFMT_TEST_BINARY");
+ const char *interp = getenv("BINFMT_TEST_INTERP");
+ int memfd_mode = getenv("BINFMT_TEST_MEMFD") != NULL;
+ int static_mode = getenv("BINFMT_TEST_STATIC") != NULL;
+ unsigned long self = (unsigned long)__ehdr_start;
+ unsigned long base = getauxval(AT_BASE);
+ unsigned long phdr = getauxval(AT_PHDR);
+ unsigned long entry = getauxval(AT_ENTRY);
+ unsigned long start_code, end_code;
+
+ /* The argument vector is exactly what the caller built. */
+ check(argc == 3 && !strcmp(argv[0], PAYLOAD_ARGV0) &&
+ !strcmp(argv[1], PAYLOAD_ARG1) && !strcmp(argv[2], PAYLOAD_ARG2),
+ "argv was rewritten");
+
+ /* Native from birth: no execfd, no dispatch marker. */
+ check(getauxval(AT_EXECFD) == 0, "AT_EXECFD present");
+ check(getauxval(AT_FLAGS) == 0, "AT_FLAGS not native");
+
+ if (static_mode) {
+ /* The override was dropped: no interpreter was loaded. */
+ check(base == 0, "AT_BASE set for a static payload");
+ } else {
+ /* A loader is mapped in the interpreter slot, not our image. */
+ check(base != 0, "AT_BASE missing");
+ check(base < self || base >= self + IMAGE_SPAN,
+ "AT_BASE inside our own image");
+ }
+
+ /* We occupy the main-image slot. */
+ check(phdr >= self && phdr < self + IMAGE_SPAN,
+ "AT_PHDR outside our image");
+ check(entry >= self && entry < self + IMAGE_SPAN,
+ "AT_ENTRY outside our image");
+
+ /* The code statistics markers describe our image, natively placed. */
+ if (stat_codes(getpid(), &start_code, &end_code) == 0) {
+ check(start_code >= self && start_code < end_code &&
+ end_code < self + IMAGE_SPAN,
+ "stat start_code/end_code not our image");
+ check(entry >= start_code && entry < end_code,
+ "AT_ENTRY outside [start_code, end_code)");
+ } else {
+ check(0, "cannot parse /proc/self/stat");
+ }
+
+ if (!memfd_mode && binary) {
+ const char *execfn = (const char *)getauxval(AT_EXECFN);
+ const char *base_name = strrchr(binary, '/');
+
+ base_name = base_name ? base_name + 1 : binary;
+
+ /* exe link, AT_EXECFN and comm all follow the binary. */
+ check(exe_is(binary), "/proc/self/exe");
+ check(execfn && !strcmp(execfn, binary), "AT_EXECFN");
+ check(comm_is(base_name), "comm");
+
+ /* The running binary is write-denied, natively. */
+ check(write_denied(binary), "no ETXTBSY on the binary");
+ }
+
+ if (interp) {
+ int found = maps_has_prefix(interp);
+
+ if (static_mode)
+ /* Nothing was substituted, nothing may be mapped. */
+ check(found == 0, "loader mapped for a static payload");
+ else
+ /* The substituted loader shows under its real path. */
+ check(found == 1, "loader path not in /proc/self/maps");
+ }
+
+ if (failed)
+ return 1;
+ printf("[payload] native identity checks out\n");
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_misc_bpf.c b/tools/testing/selftests/exec/binfmt_misc_bpf.c
new file mode 100644
index 000000000000..b2a4518901b0
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_bpf.c
@@ -0,0 +1,638 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Selftest for binfmt_misc bpf-backed ('B') handlers.
+ *
+ * A handler is a struct binfmt_misc_ops struct_ops map with a sleepable match
+ * and a sleepable load program. Attaching it publishes it by name in the
+ * caller's user namespace; a 'B' entry referencing it by name in the
+ * interpreter field activates it:
+ *
+ * echo ':name:B::::<handler>:' > /proc/sys/fs/binfmt_misc/register
+ *
+ * Five self-contained cases are exercised:
+ *
+ * 1. bpf_interp: the match program matches a synthetic aarch64 ELF header
+ * from the prefetched bprm->buf and the load program routes it to a
+ * fixed interpreter of its choosing.
+ * 2. nix_origin: the match program reads the binary's program headers to
+ * commit only to a "$ORIGIN/..."-relative PT_INTERP and the load program
+ * resolves it to an interpreter co-located with the binary (the
+ * relocatable-loader case the kernel ELF loader cannot express).
+ * 3. transparent: the load program sets BPF_BINPRM_TRANSPARENT; the
+ * asserting interpreter (binfmt_transparent_interp) verifies the
+ * identity the kernel constructed (exe link, argv, cmdline, comm,
+ * AT_EXECFD, write denial) from inside the process.
+ * 4. loader: the load program sets BPF_BINPRM_LOADER; the payload
+ * (binfmt_loader_payload) runs as the main image with the selected
+ * interpreter substituted for its PT_INTERP and asserts the native
+ * identity from inside.
+ * 5. interp_bind: an entry registered disabled with 'D' is given its
+ * interpreters one write at a time, and the load program picks one by
+ * name per exec. Replacing what the path holds afterwards changes
+ * nothing, which is the point of binding a file rather than resolving
+ * a name at exec time. Enabling the entry seals it.
+ *
+ * The first two route to a test interpreter that prints BPF_INTERP_RAN,
+ * proving the program's chosen interpreter actually ran.
+ */
+#define _GNU_SOURCE
+#include <elf.h>
+#include <limits.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+
+#include <bpf/btf.h>
+#include <bpf/libbpf.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define INTERP_PATH "/tmp/binfmt_bpf_interp"
+#define AARCH64_PATH "/tmp/binfmt_bpf_aarch64"
+#define RELOC_TEMPLATE "/tmp/binfmt_relocXXXXXX"
+#define TRANS_INTERP "/tmp/binfmt_transparent_interp"
+#define TRANS_PATH "/tmp/binfmt_bpf_riscv"
+#define EXPECT "BPF_INTERP_RAN"
+#define TRANS_EXPECT "TRANSPARENT_OK"
+#define LOADER_INTERP "/tmp/binfmt_loader_interp"
+#define LOADER_PATH "/tmp/binfmt_bpf_loader.ldrtest"
+#define BIND_FIRST "/tmp/binfmt_bind_first"
+#define BIND_SECOND "/tmp/binfmt_bind_second"
+#define BIND_ARM_PATH "/tmp/binfmt_bind_arm"
+#define BIND_RISCV_PATH "/tmp/binfmt_bind_riscv"
+#define BIND_EXPECT "BIND_RAN "
+#define BIND_MAX 100
+#define INTERP_LIMIT "/proc/sys/user/max_binfmt_misc_interpreters"
+/* Exit status of the binding child when it cannot set up a budget of its own. */
+#define BIND_NO_BUDGET 200
+
+/* A minimal 64-bit little-endian ELF header, padded to the read size. */
+static int create_fake_elf(const char *path, unsigned short machine)
+{
+ unsigned char hdr[256] = {0};
+ int fd;
+
+ hdr[0] = 0x7f; hdr[1] = 'E'; hdr[2] = 'L'; hdr[3] = 'F';
+ hdr[4] = ELFCLASS64;
+ hdr[5] = ELFDATA2LSB;
+ hdr[6] = EV_CURRENT;
+ hdr[16] = ET_EXEC;
+ hdr[18] = machine & 0xff; /* e_machine, little-endian */
+ hdr[19] = machine >> 8;
+ hdr[20] = EV_CURRENT;
+
+ unlink(path);
+ fd = open(path, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, hdr, sizeof(hdr)) != (ssize_t)sizeof(hdr)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+/*
+ * Register a 'B' entry for @handler. With @flags "D" the entry is created
+ * disabled, which is what leaves it open to being given interpreters.
+ */
+static int register_entry(const char *name, const char *handler,
+ const char *flags)
+{
+ char rule[PATH_MAX];
+
+ snprintf(rule, sizeof(rule), ":%s:B::::%s:%s", name, handler,
+ flags ? flags : "");
+ return write_reg(rule);
+}
+
+static int check_output(const char *cmd, const char *expected)
+{
+ char buf[128];
+ FILE *fp;
+
+ fp = popen(cmd, "r");
+ if (!fp)
+ return -1;
+ if (!fgets(buf, sizeof(buf), fp)) {
+ pclose(fp);
+ return -1;
+ }
+ pclose(fp);
+ return strncmp(buf, expected, strlen(expected)) ? -1 : 0;
+}
+
+/* Does the kernel BTF know struct binfmt_misc_ops (CONFIG_BINFMT_MISC_BPF)? */
+static bool have_binfmt_misc_ops(void)
+{
+ struct btf *btf = btf__load_vmlinux_btf();
+ bool have;
+
+ have = btf && btf__find_by_name_kind(btf, "binfmt_misc_ops",
+ BTF_KIND_STRUCT) >= 0;
+ btf__free(btf);
+ return have;
+}
+
+/* The reason bpf handler cases cannot run here, NULL if they can. */
+static const char *bpf_handler_unsupported(void)
+{
+ if (getuid() != 0)
+ return "test must be run as root";
+ if (!have_binfmt_misc_ops())
+ return "no struct binfmt_misc_ops in the kernel BTF (CONFIG_BINFMT_MISC_BPF)";
+ if (!binfmt_misc_available())
+ return "no binfmt_misc";
+ return NULL;
+}
+
+/* An attached handler with its 'B' entry activated. */
+struct bpf_case {
+ struct bpf_object *obj;
+ struct bpf_link *link;
+ const char *entry;
+};
+
+/*
+ * Load @objfile, attach its struct_ops map @handler (which publishes the
+ * handler) and register a 'B' entry named @entry that references it, with
+ * @flags as the entry's register-string flags.
+ */
+static int bpf_case_start_flags(struct bpf_case *c, const char *objfile,
+ const char *handler, const char *entry,
+ const char *flags)
+{
+ struct bpf_map *map;
+
+ c->obj = NULL;
+ c->link = NULL;
+ c->entry = entry;
+
+ c->obj = bpf_object__open_file(objfile, NULL);
+ if (!c->obj || libbpf_get_error(c->obj)) {
+ fprintf(stderr, "open %s failed\n", objfile);
+ c->obj = NULL;
+ return -1;
+ }
+ if (bpf_object__load(c->obj)) {
+ fprintf(stderr, "load %s failed (check dmesg for the verifier log)\n",
+ objfile);
+ goto fail;
+ }
+ map = bpf_object__find_map_by_name(c->obj, handler);
+ if (!map) {
+ fprintf(stderr, "no struct_ops map '%s' in %s\n", handler, objfile);
+ goto fail;
+ }
+ c->link = bpf_map__attach_struct_ops(map);
+ if (!c->link || libbpf_get_error(c->link)) {
+ fprintf(stderr, "attach struct_ops '%s' failed\n", handler);
+ c->link = NULL;
+ goto fail;
+ }
+ if (register_entry(entry, handler, flags)) {
+ fprintf(stderr, "register 'B' entry '%s' failed\n", entry);
+ goto fail;
+ }
+ return 0;
+
+fail:
+ bpf_link__destroy(c->link);
+ bpf_object__close(c->obj);
+ c->obj = NULL;
+ c->link = NULL;
+ return -1;
+}
+
+static int bpf_case_start(struct bpf_case *c, const char *objfile,
+ const char *handler, const char *entry)
+{
+ return bpf_case_start_flags(c, objfile, handler, entry, NULL);
+}
+
+static void bpf_case_stop(struct bpf_case *c)
+{
+ unregister(c->entry);
+ bpf_link__destroy(c->link);
+ bpf_object__close(c->obj);
+}
+
+/* Activate @handler, run @target and check it produced @expect. */
+static int run_case(const char *objfile, const char *handler,
+ const char *entry, const char *target, const char *expect)
+{
+ struct bpf_case c;
+ int ret;
+
+ if (bpf_case_start(&c, objfile, handler, entry))
+ return -1;
+ ret = check_output(target, expect);
+ bpf_case_stop(&c);
+ return ret;
+}
+
+FIXTURE(bpf_handler) {
+ char obj[PATH_MAX]; /* struct_ops object of the case under test */
+};
+
+FIXTURE_SETUP(bpf_handler)
+{
+ char src[PATH_MAX];
+ const char *why = bpf_handler_unsupported();
+
+ if (why)
+ SKIP(return, "%s", why);
+
+ /* Shared test interpreter. */
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_bpf_interp"), 0);
+ ASSERT_EQ(copy_file(src, INTERP_PATH), 0);
+}
+
+FIXTURE_TEARDOWN(bpf_handler)
+{
+ unlink(INTERP_PATH);
+}
+
+/* The match program matches a synthetic header, the load program routes it. */
+TEST_F(bpf_handler, fixed_interpreter)
+{
+ ASSERT_EQ(create_fake_elf(AARCH64_PATH, EM_AARCH64), 0);
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "bpf_interp.bpf.o"), 0);
+ EXPECT_EQ(run_case(self->obj, "bpf_interp", "test_bpf_interp",
+ AARCH64_PATH, EXPECT), 0);
+ unlink(AARCH64_PATH);
+}
+
+/* A "$ORIGIN/..." PT_INTERP resolved to an interpreter next to the binary. */
+TEST_F(bpf_handler, origin_relative_interpreter)
+{
+ char src[PATH_MAX], app[PATH_MAX], interp[PATH_MAX];
+ char dir[] = RELOC_TEMPLATE;
+
+ ASSERT_NE(mkdtemp(dir), NULL);
+ snprintf(app, sizeof(app), "%s/app", dir);
+ snprintf(interp, sizeof(interp), "%s/binfmt_bpf_interp", dir);
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_bpf_app"), 0);
+ ASSERT_EQ(copy_file(src, app), 0);
+ ASSERT_EQ(copy_file(INTERP_PATH, interp), 0);
+
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "nix_origin.bpf.o"), 0);
+ EXPECT_EQ(run_case(self->obj, "nix_origin", "test_bpf_origin",
+ app, EXPECT), 0);
+
+ unlink(app);
+ unlink(interp);
+ rmdir(dir);
+}
+
+/* A transparent dispatch: the process presents as the binary, not the interp. */
+TEST_F(bpf_handler, transparent_dispatch)
+{
+ char src[PATH_MAX], cmd[PATH_MAX + 16];
+
+ /* Probe for transparent-mode support via its static counterpart. */
+ if (!binfmt_flag_supported('T'))
+ SKIP(return, "kernel without transparent mode");
+
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_transparent_interp"), 0);
+ ASSERT_EQ(copy_file(src, TRANS_INTERP), 0);
+ ASSERT_EQ(create_fake_elf(TRANS_PATH, EM_RISCV), 0);
+
+ setenv("BINFMT_TEST_BINARY", TRANS_PATH, 1);
+ snprintf(cmd, sizeof(cmd), "%s argone argtwo", TRANS_PATH);
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "transparent.bpf.o"), 0);
+ EXPECT_EQ(run_case(self->obj, "transparent", "test_bpf_transparent",
+ cmd, TRANS_EXPECT), 0);
+
+ unlink(TRANS_PATH);
+ unlink(TRANS_INTERP);
+}
+
+/* A per-exec loader substitution: the payload runs as a native exec. */
+TEST_F(bpf_handler, loader_substitution)
+{
+ char src[PATH_MAX], loader[PATH_MAX];
+ struct bpf_case c;
+ int status;
+
+ if (find_loader(loader, sizeof(loader)))
+ SKIP(return, "cannot determine own PT_INTERP");
+
+ ASSERT_EQ(copy_file(loader, LOADER_INTERP), 0);
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_loader_payload"), 0);
+ ASSERT_EQ(copy_file(src, LOADER_PATH), 0);
+ ASSERT_EQ(patch_file(LOADER_PATH, EI_PAD, LOADER_MARKER,
+ strlen(LOADER_MARKER)), 0);
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "loader.bpf.o"), 0);
+
+ setenv("BINFMT_TEST_BINARY", LOADER_PATH, 1);
+ setenv("BINFMT_TEST_INTERP", LOADER_INTERP, 1);
+
+ ASSERT_EQ(bpf_case_start(&c, self->obj, "loader", "test_bpf_loader"), 0);
+ status = run_payload(LOADER_PATH);
+ bpf_case_stop(&c);
+ EXPECT_EQ(status, 0);
+
+ unsetenv("BINFMT_TEST_INTERP");
+ unlink(LOADER_PATH);
+ unlink(LOADER_INTERP);
+}
+
+/* The errno an exec of @path fails with, 0 if it succeeded. */
+static int exec_errno(const char *path)
+{
+ int status;
+ pid_t pid;
+
+ pid = fork();
+ if (pid == 0) {
+ execl(path, path, (char *)NULL);
+ _exit(errno);
+ }
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/* Install a copy of the bound-interpreter test binary at @path. */
+static int install_interp(const char *path)
+{
+ char src[PATH_MAX];
+
+ if (artifact_path(src, sizeof(src), "binfmt_bind_interp"))
+ return -1;
+ return copy_file(src, path);
+}
+
+/* Bind @path to @entry under @name, the '+' command of a disabled entry. */
+static int entry_bind(const char *entry, const char *name, const char *path)
+{
+ char cmd[PATH_MAX];
+
+ snprintf(cmd, sizeof(cmd), "+%s %s\n", name, path);
+ return entry_command(entry, cmd);
+}
+
+/* Set the interpreter budget of this namespace. */
+static int write_interp_limit(const char *val)
+{
+ ssize_t n;
+ int fd;
+
+ fd = open(INTERP_LIMIT, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -1;
+ n = write(fd, val, strlen(val));
+ close(fd);
+ return n < 0 ? -1 : 0;
+}
+
+/*
+ * The errno a bind is refused with when the writer is a child that has spent
+ * the budget of a user namespace of its own, 0 if it succeeded and -1 if the
+ * child could not set itself up. The fd is opened here and inherited, so the
+ * interpreter is still opened with this process's credentials.
+ */
+static int bind_out_of_budget(const char *entry, const char *name,
+ const char *path)
+{
+ char cmd[PATH_MAX], file[PATH_MAX];
+ int fd, status, retval;
+ pid_t pid;
+
+ snprintf(file, sizeof(file), BINFMT_DIR "/%s", entry);
+ snprintf(cmd, sizeof(cmd), "+%s %s\n", name, path);
+
+ fd = open(file, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -1;
+
+ pid = fork();
+ if (pid == 0) {
+ ssize_t n;
+
+ /* A namespace of its own, with nothing left in it to spend. */
+ if (unshare(CLONE_NEWUSER) || write_interp_limit("0"))
+ _exit(BIND_NO_BUDGET);
+ n = write(fd, cmd, strlen(cmd));
+ _exit(n < 0 ? errno : 0);
+ }
+ close(fd);
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ retval = WEXITSTATUS(status);
+ return retval == BIND_NO_BUDGET ? -1 : retval;
+}
+
+FIXTURE(bound_interp) {
+ char obj[PATH_MAX];
+ struct bpf_case c;
+ bool started;
+};
+
+FIXTURE_SETUP(bound_interp)
+{
+ const char *why = bpf_handler_unsupported();
+
+ if (why)
+ SKIP(return, "%s", why);
+ if (!binfmt_flag_supported('D')) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'D' flag");
+ }
+
+ ASSERT_EQ(install_interp(BIND_FIRST), 0);
+ ASSERT_EQ(install_interp(BIND_SECOND), 0);
+
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "interp_bind.bpf.o"), 0);
+
+ /*
+ * Registered disabled, so it cannot be matched yet and can still be
+ * given interpreters. Each path is resolved once, by its write(2);
+ * from here on the entry holds the files themselves.
+ */
+ ASSERT_EQ(bpf_case_start_flags(&self->c, self->obj, "interp_bind",
+ "test_interp_bind", "D"), 0);
+ self->started = true;
+
+ ASSERT_EQ(entry_bind("test_interp_bind", "first", BIND_FIRST), 0);
+ ASSERT_EQ(entry_bind("test_interp_bind", "second", BIND_SECOND), 0);
+}
+
+FIXTURE_TEARDOWN(bound_interp)
+{
+ if (self->started)
+ bpf_case_stop(&self->c);
+ unlink(BIND_FIRST);
+ unlink(BIND_SECOND);
+ unlink(AARCH64_PATH);
+ unlink(BIND_RISCV_PATH);
+ unlink(BIND_ARM_PATH);
+}
+
+/* Enabling is what makes the configured entry matchable. */
+static int activate(const char *entry)
+{
+ return entry_command(entry, "1\n");
+}
+
+/* One entry, one interpreter per guest architecture, picked per exec. */
+TEST_F(bound_interp, selects_by_name)
+{
+ ASSERT_EQ(create_fake_elf(AARCH64_PATH, EM_AARCH64), 0);
+ ASSERT_EQ(create_fake_elf(BIND_RISCV_PATH, EM_RISCV), 0);
+
+ /* Disabled, so it does not match and no format claims the binary. */
+ EXPECT_EQ(exec_errno(AARCH64_PATH), ENOEXEC);
+
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+ EXPECT_EQ(check_output(AARCH64_PATH, BIND_EXPECT BIND_FIRST), 0);
+ EXPECT_EQ(check_output(BIND_RISCV_PATH, BIND_EXPECT BIND_SECOND), 0);
+}
+
+/* What was bound is what runs, whatever the path holds afterwards. */
+TEST_F(bound_interp, path_no_longer_decides)
+{
+ char other[PATH_MAX];
+
+ ASSERT_EQ(create_fake_elf(AARCH64_PATH, EM_AARCH64), 0);
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+
+ /* Bound interpreters are pinned against writes, exactly like 'F'. */
+ EXPECT_TRUE(write_denied(BIND_FIRST));
+
+ /* Replace the path with a different binary: a new file, new inode. */
+ ASSERT_EQ(artifact_path(other, sizeof(other), "binfmt_bpf_interp"), 0);
+ ASSERT_EQ(unlink(BIND_FIRST), 0);
+ ASSERT_EQ(copy_file(other, BIND_FIRST), 0);
+
+ EXPECT_EQ(check_output(AARCH64_PATH, BIND_EXPECT BIND_FIRST), 0);
+}
+
+/* The entry reports what it bound, under the names it bound them as. */
+TEST_F(bound_interp, entry_reports_bindings)
+{
+ EXPECT_TRUE(entry_shows("test_interp_bind",
+ "bpf-interpreter first " BIND_FIRST));
+ EXPECT_TRUE(entry_shows("test_interp_bind",
+ "bpf-interpreter second " BIND_SECOND));
+}
+
+/* Selecting a name the entry did not bind fails the exec. */
+TEST_F(bound_interp, unbound_name_fails)
+{
+ ASSERT_EQ(create_fake_elf(BIND_ARM_PATH, EM_ARM), 0);
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+
+ EXPECT_EQ(exec_errno(BIND_ARM_PATH), ENOENT);
+}
+
+/* Activating seals it: what can be matched cannot be changed. */
+TEST_F(bound_interp, sealed_once_active)
+{
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+
+ EXPECT_EQ(entry_bind("test_interp_bind", "third", BIND_SECOND), -EBUSY);
+ EXPECT_FALSE(entry_shows("test_interp_bind",
+ "bpf-interpreter third " BIND_SECOND));
+}
+
+/* The seal is for good: disabling the entry again reopens nothing. */
+TEST_F(bound_interp, disable_does_not_unseal)
+{
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+ ASSERT_EQ(entry_command("test_interp_bind", "0\n"), 0);
+
+ EXPECT_EQ(entry_bind("test_interp_bind", "third", BIND_SECOND), -EBUSY);
+}
+
+/* An entry registered without 'D' is sealed from the start. */
+TEST_F(bound_interp, born_sealed)
+{
+ /* A second entry for the handler the fixture already published. */
+ ASSERT_EQ(register_entry("test_born_sealed", "interp_bind", NULL), 0);
+
+ EXPECT_EQ(entry_bind("test_born_sealed", "first", BIND_FIRST), -EBUSY);
+ unregister("test_born_sealed");
+}
+
+/* A name is bound once; a second use of it is refused. */
+TEST_F(bound_interp, duplicate_name_refused)
+{
+ EXPECT_EQ(entry_bind("test_interp_bind", "first", BIND_SECOND), -EEXIST);
+}
+
+/* A name is a printable word: the entry file reports 'name path' lines. */
+TEST_F(bound_interp, name_must_be_printable)
+{
+ /* A control character would forge a line into the entry file. */
+ EXPECT_EQ(entry_bind("test_interp_bind", "a\tb", BIND_FIRST), -EINVAL);
+ EXPECT_EQ(entry_bind("test_interp_bind", "a\nb", BIND_FIRST), -EINVAL);
+
+ /* A space cannot even be spelled: the path starts after the first one. */
+ EXPECT_EQ(entry_bind("test_interp_bind", "a b", BIND_FIRST), -EINVAL);
+}
+
+/* The command ends at the write: bytes past an embedded nul are refused. */
+TEST_F(bound_interp, trailing_bytes_refused)
+{
+ char cmd[PATH_MAX];
+ size_t len;
+ int fd;
+
+ /* entry_command() cannot spell a nul, so write the buffer raw. */
+ snprintf(cmd, sizeof(cmd), "+nul %s", BIND_FIRST);
+ len = strlen(cmd) + 1;
+ memcpy(cmd + len, "junk", sizeof("junk"));
+ len += sizeof("junk");
+
+ fd = open(BINFMT_DIR "/test_interp_bind", O_WRONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(write(fd, cmd, len), -1);
+ EXPECT_EQ(errno, EINVAL);
+ close(fd);
+
+ EXPECT_FALSE(entry_shows("test_interp_bind",
+ "bpf-interpreter nul " BIND_FIRST));
+}
+
+/* An entry binds at most BIND_MAX interpreters. */
+TEST_F(bound_interp, capped_bindings)
+{
+ char name[16];
+ int i;
+
+ /* The fixture bound "first" and "second" already. */
+ for (i = 2; i < BIND_MAX; i++) {
+ snprintf(name, sizeof(name), "n%d", i);
+ ASSERT_EQ(entry_bind("test_interp_bind", name, BIND_FIRST), 0);
+ }
+ EXPECT_EQ(entry_bind("test_interp_bind", "over", BIND_FIRST), -ENOSPC);
+}
+
+/* A binding pins a file: it is charged, and refused once the budget is out. */
+TEST_F(bound_interp, bindings_are_charged)
+{
+ int err = bind_out_of_budget("test_interp_bind", "third", BIND_FIRST);
+
+ if (err < 0)
+ SKIP(return, "no user namespaces or no " INTERP_LIMIT);
+
+ /* The charge follows the writer, not the entry file it writes to. */
+ EXPECT_EQ(err, ENOSPC);
+
+ /* The budget was the only thing in the way. */
+ EXPECT_EQ(entry_bind("test_interp_bind", "third", BIND_FIRST), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_common.h b/tools/testing/selftests/exec/binfmt_misc_common.h
new file mode 100644
index 000000000000..745aff84dc78
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_common.h
@@ -0,0 +1,315 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/* Helpers shared by the binfmt_misc selftests. */
+#ifndef __SELFTESTS_EXEC_BINFMT_MISC_COMMON_H
+#define __SELFTESTS_EXEC_BINFMT_MISC_COMMON_H
+
+#include <elf.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <libgen.h>
+#include <limits.h>
+#include <link.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/types.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#define BINFMT_DIR "/proc/sys/fs/binfmt_misc"
+#define BINFMT_REG BINFMT_DIR "/register"
+
+/* comm holds 15 usable chars; a read of /proc/self/comm appends a newline. */
+#define TASK_COMM_LEN 16
+
+/* The canonical payload argv: run_payload() passes it, the payloads assert it. */
+#define PAYLOAD_ARGV0 "payload-argv0"
+#define PAYLOAD_ARG1 "argone"
+#define PAYLOAD_ARG2 "argtwo"
+
+/* Marker the loader tests poke into the payload's e_ident padding. */
+#define LOADER_MARKER "LDRTST"
+
+/* Exit status run_payload() reports when the exec was refused as unhandled. */
+#define RUN_ENOEXEC 42
+
+static inline int copy_file(const char *src, const char *dst)
+{
+ char buf[4096];
+ int in, out;
+ ssize_t n;
+
+ in = open(src, O_RDONLY);
+ if (in < 0)
+ return -1;
+ /* The tests share /tmp, so never write through a name they don't own. */
+ unlink(dst);
+ out = open(dst, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (out < 0) {
+ close(in);
+ return -1;
+ }
+ while ((n = read(in, buf, sizeof(buf))) > 0) {
+ if (write(out, buf, n) != n) {
+ close(in);
+ close(out);
+ return -1;
+ }
+ }
+ close(in);
+ close(out);
+ return n < 0 ? -1 : 0;
+}
+
+/* Write @rule to the register file, preserving the write's errno. */
+static inline int write_reg(const char *rule)
+{
+ int fd, saved;
+ ssize_t n;
+
+ fd = open(BINFMT_REG, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ n = write(fd, rule, strlen(rule));
+ saved = errno;
+ close(fd);
+ errno = saved;
+ return n < 0 ? -1 : 0;
+}
+
+static inline void unregister(const char *name)
+{
+ char path[PATH_MAX];
+ int fd;
+
+ snprintf(path, sizeof(path), BINFMT_DIR "/%s", name);
+ fd = open(path, O_WRONLY);
+ if (fd >= 0) {
+ if (write(fd, "-1", 2) < 0)
+ ; /* best effort */
+ close(fd);
+ }
+}
+
+/* Write @line to @entry's file, reporting the errno it was refused with. */
+static inline int entry_command(const char *entry, const char *line)
+{
+ char path[PATH_MAX];
+ int fd, retval = 0;
+ size_t len = strlen(line);
+
+ snprintf(path, sizeof(path), BINFMT_DIR "/%s", entry);
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -errno;
+ if (write(fd, line, len) != (ssize_t)len)
+ retval = -errno;
+ close(fd);
+ return retval;
+}
+
+/* Does @entry's file report @line? */
+static inline bool entry_shows(const char *entry, const char *line)
+{
+ char path[PATH_MAX], buf[PATH_MAX];
+ bool found = false;
+ FILE *fp;
+
+ snprintf(path, sizeof(path), BINFMT_DIR "/%s", entry);
+ fp = fopen(path, "r");
+ if (!fp)
+ return false;
+ while (fgets(buf, sizeof(buf), fp)) {
+ buf[strcspn(buf, "\n")] = '\0';
+ if (!strcmp(buf, line)) {
+ found = true;
+ break;
+ }
+ }
+ fclose(fp);
+ return found;
+}
+
+/* Mount binfmt_misc unless it already is, and report whether it is usable. */
+static inline bool binfmt_misc_available(void)
+{
+ if (access(BINFMT_REG, F_OK) < 0)
+ mount("binfmt_misc", BINFMT_DIR, "binfmt_misc", 0, NULL);
+ return access(BINFMT_REG, F_OK) == 0;
+}
+
+/* Absolute path of @name in the directory this test was built into. */
+static inline int artifact_path(char *out, size_t sz, const char *name)
+{
+ char exe[PATH_MAX];
+ ssize_t n;
+
+ n = readlink("/proc/self/exe", exe, sizeof(exe) - 1);
+ if (n < 0)
+ return -1;
+ exe[n] = '\0';
+ if ((size_t)snprintf(out, sz, "%s/%s", dirname(exe), name) >= sz)
+ return -1;
+ return 0;
+}
+
+/* Probe kernel support for a registration flag with a throwaway entry. */
+static inline bool binfmt_flag_supported(char flag)
+{
+ char rule[64];
+
+ snprintf(rule, sizeof(rule), ":bm_flag_probe:E::bmprobe::/bin/true:%c",
+ flag);
+ if (write_reg(rule))
+ return false;
+ unregister("bm_flag_probe");
+ return true;
+}
+
+/*
+ * Run @path with the canonical payload argv and return its exit status, or
+ * RUN_ENOEXEC when the exec itself was refused as unhandled.
+ */
+static inline int run_payload(const char *path)
+{
+ int status;
+ pid_t pid;
+
+ pid = fork();
+ if (pid == 0) {
+ execl(path, PAYLOAD_ARGV0, PAYLOAD_ARG1, PAYLOAD_ARG2,
+ (char *)NULL);
+ _exit(errno == ENOEXEC ? RUN_ENOEXEC : 126);
+ }
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/* Does the exe link name @path? */
+static inline bool exe_is(const char *path)
+{
+ char exe[PATH_MAX], real[PATH_MAX];
+ ssize_t n;
+
+ n = readlink("/proc/self/exe", exe, sizeof(exe) - 1);
+ if (n <= 0 || !realpath(path, real))
+ return false;
+ exe[n] = '\0';
+ return !strcmp(exe, real);
+}
+
+/* Is comm @name truncated to what a comm can hold? */
+static inline bool comm_is(const char *name)
+{
+ char comm[TASK_COMM_LEN + 2], expect[TASK_COMM_LEN];
+ ssize_t n;
+ int fd;
+
+ fd = open("/proc/self/comm", O_RDONLY);
+ if (fd < 0)
+ return false;
+ n = read(fd, comm, sizeof(comm) - 1);
+ close(fd);
+ if (n <= 0)
+ return false;
+ if (comm[n - 1] == '\n')
+ n--;
+ comm[n] = '\0';
+ snprintf(expect, sizeof(expect), "%s", name);
+ return !strcmp(comm, expect);
+}
+
+/* Opening @path for writing has to fail with ETXTBSY. */
+static inline bool write_denied(const char *path)
+{
+ int fd = open(path, O_WRONLY);
+
+ if (fd >= 0) {
+ close(fd);
+ return false;
+ }
+ return errno == ETXTBSY;
+}
+
+static inline int patch_file(const char *path, off_t off, const void *data, size_t len)
+{
+ ssize_t n;
+ int fd;
+
+ fd = open(path, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ n = pwrite(fd, data, len, off);
+ close(fd);
+ return n == (ssize_t)len ? 0 : -1;
+}
+
+/* start_code and end_code are the 26th and 27th fields of /proc/pid/stat. */
+static inline int stat_codes(pid_t pid, unsigned long *start_code,
+ unsigned long *end_code)
+{
+ char buf[4096], path[64], *p;
+ ssize_t n;
+ int fd, i;
+
+ snprintf(path, sizeof(path), "/proc/%d/stat", pid);
+ fd = open(path, O_RDONLY);
+ if (fd < 0)
+ return -1;
+ n = read(fd, buf, sizeof(buf) - 1);
+ close(fd);
+ if (n <= 0)
+ return -1;
+ buf[n] = '\0';
+
+ /* Skip "pid (comm)", then start_code is the 24th field after it. */
+ p = strrchr(buf, ')');
+ if (!p)
+ return -1;
+ p++;
+ for (i = 0; i < 23; i++) {
+ p = strchr(p + 1, ' ');
+ if (!p)
+ return -1;
+ }
+ if (sscanf(p, " %lu %lu", start_code, end_code) != 2)
+ return -1;
+ return 0;
+}
+
+/* Find the system loader through our own PT_INTERP. */
+static inline int find_loader(char *out, size_t sz)
+{
+ ElfW(Ehdr) eh;
+ ElfW(Phdr) ph;
+ int fd, i, ret = -1;
+
+ fd = open("/proc/self/exe", O_RDONLY);
+ if (fd < 0)
+ return -1;
+ if (pread(fd, &eh, sizeof(eh), 0) != sizeof(eh))
+ goto out;
+ for (i = 0; i < eh.e_phnum; i++) {
+ if (pread(fd, &ph, sizeof(ph),
+ eh.e_phoff + i * eh.e_phentsize) != sizeof(ph))
+ goto out;
+ if (ph.p_type != PT_INTERP)
+ continue;
+ if (!ph.p_filesz || ph.p_filesz > sz)
+ goto out;
+ if (pread(fd, out, ph.p_filesz, ph.p_offset) !=
+ (ssize_t)ph.p_filesz)
+ goto out;
+ out[ph.p_filesz - 1] = '\0';
+ ret = 0;
+ break;
+ }
+out:
+ close(fd);
+ return ret;
+}
+
+#endif /* __SELFTESTS_EXEC_BINFMT_MISC_COMMON_H */
diff --git a/tools/testing/selftests/exec/binfmt_misc_disabled.c b/tools/testing/selftests/exec/binfmt_misc_disabled.c
new file mode 100644
index 000000000000..47c9e8a4ee42
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_disabled.c
@@ -0,0 +1,172 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test the 'D' (register disabled) flag of binfmt_misc. An entry
+ * registered with it exists but cannot be matched until userspace enables
+ * it, which splits a registration into create and activate.
+ *
+ * Needs root for the registration; no bpf toolchain involved.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define MAGIC "#DISABLED-SELFTEST#"
+#define TARGET_PATH "/tmp/binfmt_disabled_target"
+#define INTERP_PATH "/tmp/binfmt_disabled_interp.sh"
+#define ENTRY "test_disabled"
+#define RULE(flags) ":" ENTRY ":M:0:" MAGIC "::" INTERP_PATH ":" flags
+
+/* The interpreter exits with a code the harness can recognise. */
+#define EXIT_INTERP 7
+
+/* The target only has to carry the magic; it is never actually loaded. */
+static int create_target(void)
+{
+ char buf[128] = MAGIC "\n";
+ int fd;
+
+ unlink(TARGET_PATH);
+ fd = open(TARGET_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, buf, sizeof(buf)) != (ssize_t)sizeof(buf)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+static int create_interp(void)
+{
+ char buf[64];
+ int fd;
+
+ unlink(INTERP_PATH);
+ fd = open(INTERP_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ snprintf(buf, sizeof(buf), "#!/bin/sh\nexit %d\n", EXIT_INTERP);
+ if (write(fd, buf, strlen(buf)) != (ssize_t)strlen(buf)) {
+ close(fd);
+ return -1;
+ }
+ return close(fd);
+}
+
+FIXTURE(disabled) {
+};
+
+FIXTURE_SETUP(disabled)
+{
+ if (getuid() != 0)
+ SKIP(return, "test must be run as root");
+ if (!binfmt_misc_available())
+ SKIP(return, "no binfmt_misc");
+
+ /* Skip the whole suite on a kernel that does not know 'D'. */
+ if (!binfmt_flag_supported('D')) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'D' flag");
+ }
+
+ ASSERT_EQ(create_interp(), 0);
+ ASSERT_EQ(create_target(), 0);
+}
+
+FIXTURE_TEARDOWN(disabled)
+{
+ unregister(ENTRY);
+ unlink(TARGET_PATH);
+ unlink(INTERP_PATH);
+}
+
+/* The entry exists but does not dispatch until it is enabled. */
+TEST_F(disabled, inert_until_enabled)
+{
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "disabled"));
+
+ /* Nothing matches it, so no binary format claims the target. */
+ EXPECT_EQ(run_payload(TARGET_PATH), RUN_ENOEXEC);
+
+ ASSERT_EQ(entry_command(ENTRY, "1\n"), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "enabled"));
+ EXPECT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+}
+
+/* Without 'D' an entry is matchable the moment it is registered. */
+TEST_F(disabled, enabled_without_the_flag)
+{
+ ASSERT_EQ(write_reg(RULE("")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "enabled"));
+ EXPECT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+}
+
+/* 'D' is spent on the registration: the entry does not report it back. */
+TEST_F(disabled, flag_not_reported)
+{
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ EXPECT_FALSE(entry_shows(ENTRY, "flags: D"));
+ EXPECT_TRUE(entry_shows(ENTRY, "flags: "));
+}
+
+/* A disabled entry can be disabled and enabled like any other. */
+TEST_F(disabled, toggles_like_any_entry)
+{
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+
+ ASSERT_EQ(entry_command(ENTRY, "1\n"), 0);
+ ASSERT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+ ASSERT_EQ(entry_command(ENTRY, "0\n"), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), RUN_ENOEXEC);
+ ASSERT_EQ(entry_command(ENTRY, "1\n"), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+}
+
+/* 'D' composes with the invocation flags a static entry can carry. */
+TEST_F(disabled, composes_with_invocation_flags)
+{
+ ASSERT_EQ(write_reg(RULE("PD")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "disabled"));
+ EXPECT_TRUE(entry_shows(ENTRY, "flags: P"));
+}
+
+/* '-1' to the status file sweeps a staged entry with everything else. */
+TEST_F(disabled, removed_by_remove_all)
+{
+ int fd;
+
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "disabled"));
+
+ fd = open(BINFMT_DIR "/status", O_WRONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(write(fd, "-1", 2), 2);
+ close(fd);
+
+ EXPECT_NE(access(BINFMT_DIR "/" ENTRY, F_OK), 0);
+}
+
+/* A file handle held across a removal cannot resurrect the entry. */
+TEST_F(disabled, no_resurrection_after_remove)
+{
+ int fd;
+
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ fd = open(BINFMT_DIR "/" ENTRY, O_WRONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+
+ ASSERT_EQ(write(fd, "-1", 2), 2);
+ EXPECT_NE(access(BINFMT_DIR "/" ENTRY, F_OK), 0);
+
+ /* Accepted like any toggle of a removed entry, but publishes nothing. */
+ EXPECT_EQ(write(fd, "1", 1), 1);
+ EXPECT_EQ(run_payload(TARGET_PATH), RUN_ENOEXEC);
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_interplimit.c b/tools/testing/selftests/exec/binfmt_misc_interplimit.c
new file mode 100644
index 000000000000..bf611c551784
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_interplimit.c
@@ -0,0 +1,232 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A pre-opened interpreter - what 'F' gives a static entry and what a 'B'
+ * entry binds - keeps a file open for as long as the entry lives, so it pins
+ * the mount it came from. It costs no file descriptor, and binfmt_misc is
+ * FS_USERNS_MOUNT, so an unprivileged user namespace can create them without
+ * bound. Check that UCOUNT_BINFMT_MISC_INTERPRETERS bounds it, that an entry
+ * that pre-opens nothing is not charged, that removing an entry gives the
+ * charge back, and that nesting a user namespace does not evade it.
+ *
+ * Runs unprivileged in a user namespace.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "../filesystems/utils.h"
+#include "kselftest_harness.h"
+
+#define MNT "/tmp/binfmt_interplimit"
+#define NESTED_MNT "/tmp/binfmt_interplimit_nested"
+#define LIMIT_SYSCTL "/proc/sys/user/max_binfmt_misc_interpreters"
+
+#define MAGIC "\\xde\\xad"
+/* Not on the instance, and unlike /bin/true it always exists. */
+#define INTERP "/proc/self/exe"
+
+/* Small enough to fill by hand, big enough that a refund is visible. */
+#define LIMIT 4
+
+/* What UCOUNT_ENTRY() lets a namespace raise its own limit to. */
+#define LIMIT_MAX "2147483647"
+
+static int ensure_dir(const char *path)
+{
+ if (mkdir(path, 0755) && errno != EEXIST)
+ return -1;
+ return 0;
+}
+
+/* Write @val to @path, preserving write(2)'s errno for the caller. */
+static int write_keep_errno(const char *path, const char *val)
+{
+ int fd, saved;
+ ssize_t n;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -1;
+ n = write(fd, val, strlen(val));
+ saved = errno;
+ close(fd);
+ errno = saved;
+ return n < 0 ? -1 : 0;
+}
+
+static int set_limit(const char *val)
+{
+ return write_keep_errno(LIMIT_SYSCTL, val);
+}
+
+static int register_at(const char *mnt, const char *rule)
+{
+ char path[PATH_MAX];
+
+ snprintf(path, sizeof(path), "%s/register", mnt);
+ return write_keep_errno(path, rule);
+}
+
+/* An 'F' entry: one interpreter pre-opened at registration, one charge. */
+static int register_fixed(const char *mnt, const char *name)
+{
+ char rule[PATH_MAX];
+
+ snprintf(rule, sizeof(rule), ":%s:M::" MAGIC "::" INTERP ":F", name);
+ return register_at(mnt, rule);
+}
+
+/* The same entry without 'F': the interpreter is opened per exec instead. */
+static int register_plain(const char *mnt, const char *name)
+{
+ char rule[PATH_MAX];
+
+ snprintf(rule, sizeof(rule), ":%s:M::" MAGIC "::" INTERP ":", name);
+ return register_at(mnt, rule);
+}
+
+static int remove_entry(const char *mnt, const char *name)
+{
+ char path[PATH_MAX];
+
+ snprintf(path, sizeof(path), "%s/%s", mnt, name);
+ return write_keep_errno(path, "-1\n");
+}
+
+static bool entry_exists(const char *mnt, const char *name)
+{
+ char path[PATH_MAX];
+
+ snprintf(path, sizeof(path), "%s/%s", mnt, name);
+ return access(path, F_OK) == 0;
+}
+
+/* Register @n 'F' entries, each with a name of its own. */
+static int fill_budget(const char *mnt, unsigned int n)
+{
+ char name[32];
+ unsigned int i;
+
+ for (i = 0; i < n; i++) {
+ snprintf(name, sizeof(name), "fixed%u", i);
+ if (register_fixed(mnt, name))
+ return -1;
+ }
+ return 0;
+}
+
+FIXTURE(interp_limit) {
+};
+
+FIXTURE_SETUP(interp_limit)
+{
+ /* setup_userns() exits rather than returns if this is not there. */
+ if (access("/proc/self/ns/user", F_OK))
+ SKIP(return, "kernel without user namespaces");
+ ASSERT_EQ(setup_userns(), 0);
+
+ /* CAP_SYS_RESOURCE in this namespace is what makes it writable. */
+ if (set_limit(LIMIT_MAX)) {
+ if (errno == ENOENT)
+ SKIP(return, "kernel without " LIMIT_SYSCTL);
+ SKIP(return, "cannot set the limit: %s", strerror(errno));
+ }
+
+ ASSERT_EQ(ensure_dir(MNT), 0);
+ if (mount("binfmt_misc", MNT, "binfmt_misc", 0, NULL)) {
+ int saved = errno;
+
+ /* Teardown doesn't run when setup skips, so clean up here. */
+ rmdir(MNT);
+ SKIP(return, "no binfmt_misc: %s", strerror(saved));
+ }
+}
+
+FIXTURE_TEARDOWN(interp_limit)
+{
+ /* The namespaces go with the process; just don't litter /tmp. */
+ umount2(NESTED_MNT, MNT_DETACH);
+ umount2(MNT, MNT_DETACH);
+ rmdir(NESTED_MNT);
+ rmdir(MNT);
+}
+
+/* Every pre-opened interpreter is charged, and the budget is a hard stop. */
+TEST_F(interp_limit, fixed_interpreters_are_charged)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%u", LIMIT);
+ ASSERT_EQ(set_limit(buf), 0);
+
+ ASSERT_EQ(fill_budget(MNT, LIMIT), 0);
+
+ EXPECT_NE(register_fixed(MNT, "over"), 0);
+ EXPECT_EQ(errno, ENOSPC);
+
+ /* A refused registration leaves nothing behind. */
+ EXPECT_FALSE(entry_exists(MNT, "over"));
+}
+
+/* An entry that pre-opens nothing pins nothing, so it is not charged. */
+TEST_F(interp_limit, plain_entries_are_not_charged)
+{
+ ASSERT_EQ(set_limit("0"), 0);
+
+ EXPECT_EQ(register_plain(MNT, "plain"), 0);
+ EXPECT_TRUE(entry_exists(MNT, "plain"));
+
+ /* ... while the same entry with 'F' has nothing to spend. */
+ EXPECT_NE(register_fixed(MNT, "fixed"), 0);
+ EXPECT_EQ(errno, ENOSPC);
+}
+
+/* Removing an entry closes its interpreters and gives the charge back. */
+TEST_F(interp_limit, removal_refunds_the_charge)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%u", LIMIT);
+ ASSERT_EQ(set_limit(buf), 0);
+
+ ASSERT_EQ(fill_budget(MNT, LIMIT), 0);
+ ASSERT_NE(register_fixed(MNT, "over"), 0);
+
+ ASSERT_EQ(remove_entry(MNT, "fixed0"), 0);
+ EXPECT_EQ(register_fixed(MNT, "over"), 0);
+}
+
+/*
+ * The charge walks the ancestors, so a namespace cannot buy itself budget by
+ * nesting: it may raise only its own limit, and the parent it was created
+ * from is charged for every binding made below it.
+ */
+TEST_F(interp_limit, nesting_does_not_evade_it)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%u", LIMIT);
+ ASSERT_EQ(set_limit(buf), 0);
+ ASSERT_EQ(fill_budget(MNT, LIMIT), 0);
+
+ ASSERT_EQ(setup_userns(), 0);
+ ASSERT_EQ(set_limit(LIMIT_MAX), 0);
+
+ ASSERT_EQ(ensure_dir(NESTED_MNT), 0);
+ ASSERT_EQ(mount("binfmt_misc", NESTED_MNT, "binfmt_misc", 0, NULL), 0);
+
+ /* A fresh instance with an unlimited budget of its own, and yet: */
+ EXPECT_NE(register_fixed(NESTED_MNT, "nested"), 0);
+ EXPECT_EQ(errno, ENOSPC);
+
+ /* The nested instance works for anything that pins no file. */
+ EXPECT_EQ(register_plain(NESTED_MNT, "nested_plain"), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_loader.c b/tools/testing/selftests/exec/binfmt_misc_loader.c
new file mode 100644
index 000000000000..1e14dcd274af
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_loader.c
@@ -0,0 +1,372 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test the 'L' (loader substitution) flag of binfmt_misc. A matched
+ * binary runs as the MAIN image - a fully native exec - with the
+ * registered interpreter substituted for its PT_INTERP. The payload
+ * (binfmt_loader_payload) asserts the native identity from inside.
+ *
+ * The substitute is a copy of the system loader found via our own
+ * PT_INTERP; magic matching pokes a marker into the ELF header's
+ * e_ident padding, which kernel and loader ignore.
+ *
+ * Needs root for the registration; no bpf toolchain involved.
+ */
+#define _GNU_SOURCE
+#include <elf.h>
+#include <link.h>
+#include <signal.h>
+#include <stddef.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <sys/mman.h>
+#include <sys/ptrace.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define ENTRY "test_loader"
+#define INTERP_PATH "/tmp/binfmt_loader_interp"
+#define MOVED_PATH INTERP_PATH ".moved"
+#define TARGET_PATH "/tmp/binfmt_loader_target.ldrtest"
+#define STATIC_PATH "/tmp/binfmt_loader_static.ldrtest"
+#define FOREIGN_PATH "/tmp/binfmt_loader_foreign.ldrtest"
+#define SCRIPT_PATH "/tmp/binfmt_loader_script.ldrtest"
+#define M_RULE ":" ENTRY ":M:9:" LOADER_MARKER "::" INTERP_PATH ":L"
+#define E_RULE ":" ENTRY ":E::ldrtest::" INTERP_PATH ":L"
+#define FL_RULE ":" ENTRY ":E::ldrtest::" INTERP_PATH ":FL"
+
+/* Execute the binary from an inaccessible O_CLOEXEC memfd. */
+static int run_memfd(const char *path)
+{
+ int status;
+ pid_t pid;
+
+ pid = fork();
+ if (pid == 0) {
+ char *argv[] = { PAYLOAD_ARGV0, PAYLOAD_ARG1, PAYLOAD_ARG2, NULL };
+ char buf[4096];
+ int in, mfd;
+ ssize_t n;
+
+ mfd = memfd_create("loader-test", MFD_CLOEXEC);
+ in = open(path, O_RDONLY);
+ if (mfd < 0 || in < 0)
+ _exit(125);
+ while ((n = read(in, buf, sizeof(buf))) > 0)
+ if (write(mfd, buf, n) != n)
+ _exit(125);
+ close(in);
+ setenv("BINFMT_TEST_MEMFD", "1", 1);
+ unsetenv("BINFMT_TEST_BINARY");
+ syscall(SYS_execveat, mfd, "", argv, environ, AT_EMPTY_PATH);
+ _exit(126);
+ }
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/*
+ * The differentiator against the transparent mode: at PTRACE_EVENT_EXEC
+ * the identity is already complete - exe, auxv and the stat code markers
+ * are mutually consistent with no window a debugger could observe.
+ */
+static int ptrace_probe(const char *target)
+{
+ unsigned long auxv[2 * 64], base = 0, entry = 0, at_flags = 0;
+ unsigned long start_code = 0, end_code = 0;
+ int status, fd, execfd_seen = 0, failed = 0;
+ char path[64], buf[PATH_MAX];
+ ssize_t n;
+ pid_t pid;
+ int i;
+
+ pid = fork();
+ if (pid == 0) {
+ ptrace(PTRACE_TRACEME, 0, NULL, NULL);
+ raise(SIGSTOP);
+ execl(target, PAYLOAD_ARGV0, PAYLOAD_ARG1, PAYLOAD_ARG2, (char *)NULL);
+ _exit(126);
+ }
+ if (pid < 0)
+ return -1;
+ if (waitpid(pid, &status, 0) != pid || !WIFSTOPPED(status))
+ goto fail_kill;
+ if (ptrace(PTRACE_SETOPTIONS, pid, NULL, (void *)PTRACE_O_TRACEEXEC))
+ goto fail_kill;
+ if (ptrace(PTRACE_CONT, pid, NULL, NULL))
+ goto fail_kill;
+ if (waitpid(pid, &status, 0) != pid || !WIFSTOPPED(status) ||
+ status >> 8 != (SIGTRAP | (PTRACE_EVENT_EXEC << 8))) {
+ fprintf(stderr, "no exec stop (status %#x)\n", status);
+ goto fail_kill;
+ }
+
+ snprintf(path, sizeof(path), "/proc/%d/exe", pid);
+ n = readlink(path, buf, sizeof(buf) - 1);
+ if (n <= 0) {
+ failed = 1;
+ } else {
+ buf[n] = '\0';
+ if (strcmp(buf, target)) {
+ fprintf(stderr, "exe at exec stop: %s\n", buf);
+ failed = 1;
+ }
+ }
+
+ snprintf(path, sizeof(path), "/proc/%d/auxv", pid);
+ fd = open(path, O_RDONLY);
+ if (fd < 0) {
+ n = -1;
+ } else {
+ n = read(fd, auxv, sizeof(auxv));
+ close(fd);
+ }
+ if (n <= 0) {
+ failed = 1;
+ n = 0;
+ }
+ for (i = 0; i + 1 < (int)(n / sizeof(unsigned long)); i += 2) {
+ switch (auxv[i]) {
+ case AT_BASE:
+ base = auxv[i + 1];
+ break;
+ case AT_ENTRY:
+ entry = auxv[i + 1];
+ break;
+ case AT_FLAGS:
+ at_flags = auxv[i + 1];
+ break;
+ case AT_EXECFD:
+ execfd_seen = 1;
+ break;
+ }
+ }
+
+ if (stat_codes(pid, &start_code, &end_code))
+ failed = 1;
+
+ if (!base || execfd_seen || at_flags) {
+ fprintf(stderr, "auxv at exec stop not native\n");
+ failed = 1;
+ }
+ if (!start_code || entry < start_code || entry >= end_code) {
+ fprintf(stderr, "auxv/stat inconsistent at exec stop\n");
+ failed = 1;
+ }
+
+ if (ptrace(PTRACE_CONT, pid, NULL, NULL))
+ goto fail_kill;
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status) ||
+ WEXITSTATUS(status))
+ failed = 1;
+ return failed ? -1 : 0;
+
+fail_kill:
+ kill(pid, SIGKILL);
+ waitpid(pid, &status, 0);
+ return -1;
+}
+
+FIXTURE(loader) {
+ bool have_static;
+};
+
+FIXTURE_SETUP(loader)
+{
+ unsigned short foreign_machine = 0xdead;
+ char src[PATH_MAX], loader[PATH_MAX];
+
+ if (getuid() != 0)
+ SKIP(return, "test must be run as root");
+ if (!binfmt_misc_available())
+ SKIP(return, "no binfmt_misc");
+ if (find_loader(loader, sizeof(loader)))
+ SKIP(return, "cannot determine own PT_INTERP");
+
+ ASSERT_EQ(copy_file(loader, INTERP_PATH), 0);
+
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_loader_payload"), 0);
+ ASSERT_EQ(copy_file(src, TARGET_PATH), 0);
+ ASSERT_EQ(patch_file(TARGET_PATH, EI_PAD, LOADER_MARKER,
+ strlen(LOADER_MARKER)), 0);
+
+ /* The same payload with a machine type this kernel cannot load. */
+ ASSERT_EQ(copy_file(src, FOREIGN_PATH), 0);
+ ASSERT_EQ(patch_file(FOREIGN_PATH, EI_PAD, LOADER_MARKER,
+ strlen(LOADER_MARKER)), 0);
+ ASSERT_EQ(patch_file(FOREIGN_PATH, offsetof(ElfW(Ehdr), e_machine),
+ &foreign_machine, sizeof(foreign_machine)), 0);
+
+ self->have_static =
+ artifact_path(src, sizeof(src), "binfmt_loader_payload_static") == 0 &&
+ copy_file(src, STATIC_PATH) == 0;
+
+ setenv("BINFMT_TEST_BINARY", TARGET_PATH, 1);
+ setenv("BINFMT_TEST_INTERP", INTERP_PATH, 1);
+
+ /* Everything below needs the flag; find out once. */
+ if (write_reg(E_RULE)) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'L' flag");
+ }
+ unregister(ENTRY);
+}
+
+FIXTURE_TEARDOWN(loader)
+{
+ unregister(ENTRY);
+ if (access(MOVED_PATH, F_OK) == 0)
+ rename(MOVED_PATH, INTERP_PATH);
+ unlink(TARGET_PATH);
+ unlink(STATIC_PATH);
+ unlink(FOREIGN_PATH);
+ unlink(SCRIPT_PATH);
+ unlink(INTERP_PATH);
+}
+
+/* Grammar sanity check: the same entry without 'L' has to register. */
+TEST_F(loader, plain_entry_registers)
+{
+ ASSERT_EQ(write_reg(":" ENTRY ":E::ldrtest::" INTERP_PATH ":"), 0);
+}
+
+/* 'L' is a native exec: every classic-dispatch flag is rejected. */
+TEST_F(loader, rejects_classic_flags)
+{
+ static const char * const combos[] = { "LT", "LP", "LC", "LO" };
+ char rule[PATH_MAX];
+ unsigned int i;
+
+ for (i = 0; i < ARRAY_SIZE(combos); i++) {
+ int rc;
+
+ snprintf(rule, sizeof(rule),
+ ":" ENTRY ":E::ldrtest::" INTERP_PATH ":%s", combos[i]);
+ rc = write_reg(rule);
+ EXPECT_EQ(rc, -1)
+ TH_LOG("'%s' was not rejected", combos[i]);
+ if (rc == 0) {
+ unregister(ENTRY);
+ continue;
+ }
+ EXPECT_EQ(errno, EINVAL);
+ }
+}
+
+/*
+ * Without 'F' the interpreter is opened when the binary is executed, so a
+ * relative path would be resolved against the caller's working directory.
+ */
+TEST_F(loader, rejects_relative_interpreter)
+{
+ static const char * const flags[] = { "L", "C" };
+ char rule[PATH_MAX];
+ unsigned int i;
+
+ for (i = 0; i < ARRAY_SIZE(flags); i++) {
+ int rc;
+
+ snprintf(rule, sizeof(rule),
+ ":" ENTRY ":E::ldrtest::binfmt_loader_interp:%s",
+ flags[i]);
+ rc = write_reg(rule);
+ EXPECT_EQ(rc, -1)
+ TH_LOG("'%s' accepted a relative interpreter", flags[i]);
+ if (rc == 0) {
+ unregister(ENTRY);
+ continue;
+ }
+ EXPECT_EQ(errno, EINVAL);
+ }
+}
+
+TEST_F(loader, extension_matched)
+{
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+TEST_F(loader, magic_matched)
+{
+ ASSERT_EQ(write_reg(M_RULE), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+/*
+ * The differentiator against the transparent mode: at PTRACE_EVENT_EXEC the
+ * identity is already complete, with no window a debugger could observe.
+ */
+TEST_F(loader, exec_stop_consistency)
+{
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ EXPECT_EQ(ptrace_probe(TARGET_PATH), 0);
+}
+
+/* A binary without PT_INTERP drops the override and runs natively. */
+TEST_F(loader, static_binary_runs_natively)
+{
+ if (!self->have_static)
+ SKIP(return, "no static payload built");
+
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ setenv("BINFMT_TEST_BINARY", STATIC_PATH, 1);
+ setenv("BINFMT_TEST_STATIC", "1", 1);
+ EXPECT_EQ(run_payload(STATIC_PATH), 0);
+ unsetenv("BINFMT_TEST_STATIC");
+ setenv("BINFMT_TEST_BINARY", TARGET_PATH, 1);
+}
+
+/*
+ * A '#!' file that matched an 'L' entry is claimed by binfmt_script, which
+ * sits ahead of binfmt_elf. The substitute the entry staged has to be
+ * released when the interpreter replaces the file, not leaked.
+ */
+TEST_F(loader, script_claims_the_file)
+{
+ static const char script[] = "#!/bin/sh\nexit 0\n";
+ int fd;
+
+ unlink(SCRIPT_PATH);
+ fd = open(SCRIPT_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(write(fd, script, sizeof(script) - 1),
+ (ssize_t)sizeof(script) - 1);
+ ASSERT_EQ(close(fd), 0);
+
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ EXPECT_EQ(run_payload(SCRIPT_PATH), 0);
+
+ /* A leaked substitute keeps its write denial on the loader. */
+ fd = open(INTERP_PATH, O_WRONLY);
+ EXPECT_GE(fd, 0)
+ TH_LOG("loader still write denied (errno %d)", errno);
+ if (fd >= 0)
+ close(fd);
+}
+
+/* Nothing needs the binary's path, so an inaccessible fd works. */
+TEST_F(loader, inaccessible_memfd)
+{
+ ASSERT_EQ(write_reg(M_RULE), 0);
+ EXPECT_EQ(run_memfd(TARGET_PATH), 0);
+}
+
+/* The whole exec of a wrong-arch binary fails as if unhandled. */
+TEST_F(loader, foreign_arch_enoexec)
+{
+ ASSERT_EQ(write_reg(M_RULE), 0);
+ EXPECT_EQ(run_payload(FOREIGN_PATH), RUN_ENOEXEC);
+}
+
+/* 'F' pre-opens the substitute, so it survives losing its path. */
+TEST_F(loader, fixed_interpreter_survives_rename)
+{
+ ASSERT_EQ(write_reg(FL_RULE), 0);
+ ASSERT_EQ(rename(INTERP_PATH, MOVED_PATH), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_selfpin.c b/tools/testing/selftests/exec/binfmt_misc_selfpin.c
new file mode 100644
index 000000000000..5286b0604eed
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_selfpin.c
@@ -0,0 +1,158 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An 'F' entry keeps its interpreter open for as long as the entry exists,
+ * and the entry only goes away when the binfmt_misc superblock is destroyed.
+ * An interpreter that lives on a mount which in turn keeps that superblock
+ * alive therefore pins the instance that owns it, and nothing can break the
+ * cycle. Check the two ways userspace could arrange for that: an interpreter
+ * on the binfmt_misc instance itself, and one on a filesystem stacked on it.
+ *
+ * Runs unprivileged in a user namespace; binfmt_misc is FS_USERNS_MOUNT.
+ */
+#define _GNU_SOURCE
+#include <fcntl.h>
+#include <limits.h>
+#include <sched.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+
+#include "../filesystems/utils.h"
+#include "kselftest_harness.h"
+
+#define MNT "/tmp/binfmt_selfpin"
+#define BACKING "/tmp/binfmt_selfpin_back"
+#define LOWER BACKING "/lower"
+#define MERGED "/tmp/binfmt_selfpin_merged"
+
+#define MAGIC "\\xde\\xad"
+#define RULE(interp) ":selfpin:M::" MAGIC "::" interp ":F"
+/* Not on the instance, and unlike /bin/true it always exists. */
+#define INTERP "/proc/self/exe"
+
+#define OPTS_MAX (3 * PATH_MAX + 64)
+
+static int ensure_dir(const char *path)
+{
+ if (mkdir(path, 0755) && errno != EEXIST)
+ return -1;
+ return 0;
+}
+
+/* Write @rule to this instance's register file, preserving write(2)'s errno. */
+static int register_at(struct __test_metadata *_metadata, const char *rule)
+{
+ int fd, saved;
+ ssize_t n;
+
+ fd = open(MNT "/register", O_WRONLY);
+ ASSERT_GE(fd, 0);
+ n = write(fd, rule, strlen(rule));
+ saved = errno;
+ close(fd);
+ errno = saved;
+ return n < 0 ? -1 : 0;
+}
+
+/*
+ * Mount an overlay over @lower using a private upper/work pair, so the two
+ * mounts this test performs cannot interfere with each other and neither
+ * overlaps the lower layer.
+ */
+static int mount_overlay(const char *lower, int nr)
+{
+ char opts[OPTS_MAX], upper[PATH_MAX], work[PATH_MAX];
+
+ snprintf(upper, sizeof(upper), "%s/upper%d", BACKING, nr);
+ snprintf(work, sizeof(work), "%s/work%d", BACKING, nr);
+ if (mkdir(upper, 0755) || mkdir(work, 0755))
+ return -1;
+
+ snprintf(opts, sizeof(opts), "lowerdir=%s,upperdir=%s,workdir=%s",
+ lower, upper, work);
+ return mount("ovl", MERGED, "overlay", 0, opts);
+}
+
+FIXTURE(selfpin) {
+};
+
+FIXTURE_SETUP(selfpin)
+{
+ /* setup_userns() exits rather than returns if this is not there. */
+ if (access("/proc/self/ns/user", F_OK))
+ SKIP(return, "kernel without user namespaces");
+ ASSERT_EQ(setup_userns(), 0);
+
+ ASSERT_EQ(ensure_dir(MNT), 0);
+ if (mount("binfmt_misc", MNT, "binfmt_misc", 0, NULL)) {
+ int saved = errno;
+
+ /* Teardown doesn't run when setup skips, so clean up here. */
+ rmdir(MNT);
+ SKIP(return, "no binfmt_misc: %s", strerror(saved));
+ }
+}
+
+FIXTURE_TEARDOWN(selfpin)
+{
+ /* The namespaces go with the process; just don't litter /tmp. */
+ umount2(MERGED, MNT_DETACH);
+ umount2(BACKING, MNT_DETACH);
+ umount2(MNT, MNT_DETACH);
+ rmdir(MERGED);
+ rmdir(BACKING);
+ rmdir(MNT);
+}
+
+/*
+ * The instance's own files are regular files the mounter owns, so they can be
+ * made executable. Opening one for exec still has to fail, otherwise the entry
+ * pins the very superblock it lives in.
+ */
+TEST_F(selfpin, interpreter_on_the_instance)
+{
+ ASSERT_EQ(chmod(MNT "/status", 0755), 0);
+
+ ASSERT_NE(register_at(_metadata, RULE(MNT "/status")), 0);
+ EXPECT_EQ(errno, EACCES);
+}
+
+/* Same for an entry file rather than one of the control files. */
+TEST_F(selfpin, interpreter_on_an_entry)
+{
+ ASSERT_EQ(register_at(_metadata, ":victim:M::" MAGIC "::" INTERP ":"), 0);
+ ASSERT_EQ(chmod(MNT "/victim", 0755), 0);
+
+ ASSERT_NE(register_at(_metadata, RULE(MNT "/victim")), 0);
+ EXPECT_EQ(errno, EACCES);
+}
+
+/*
+ * A stacking filesystem holds a private clone of each layer for its whole
+ * lifetime, so an instance used as a layer can be pinned by an interpreter
+ * that does not live on it at all. Refuse to be a layer.
+ */
+TEST_F(selfpin, refuses_to_be_stacked_on)
+{
+ ASSERT_EQ(ensure_dir(BACKING), 0);
+ ASSERT_EQ(mount("tmpfs", BACKING, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mkdir(LOWER, 0755), 0);
+ ASSERT_EQ(ensure_dir(MERGED), 0);
+
+ /* Nothing to prove unless overlayfs works here at all. */
+ if (mount_overlay(LOWER, 1)) {
+ if (errno == ENODEV || errno == EPERM)
+ SKIP(return, "no unprivileged overlayfs");
+ SKIP(return, "overlayfs unusable here: %s", strerror(errno));
+ }
+ ASSERT_EQ(umount(MERGED), 0);
+
+ EXPECT_NE(mount_overlay(MNT, 2), 0);
+}
+
+/* An ordinary interpreter still registers with 'F'. */
+TEST_F(selfpin, ordinary_interpreter_still_works)
+{
+ EXPECT_EQ(register_at(_metadata, RULE(INTERP)), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_transparent.c b/tools/testing/selftests/exec/binfmt_misc_transparent.c
new file mode 100644
index 000000000000..2ebf73de8018
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_transparent.c
@@ -0,0 +1,95 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test the static transparent flag 'T' of binfmt_misc. A magic-matched
+ * binary is dispatched to an interpreter with the argument vector left
+ * untouched, the binary passed through AT_EXECFD and mm->exe_file labeled
+ * with the binary. The asserting interpreter (binfmt_transparent_interp)
+ * verifies the constructed identity from inside the process and exits 0.
+ *
+ * Needs root for the registration; no bpf toolchain involved.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define MAGIC "#TRANSPARENT-SELFTEST#"
+#define TARGET_PATH "/tmp/binfmt_transparent_target"
+#define INTERP_PATH "/tmp/binfmt_transparent_interp"
+#define ENTRY "test_transparent"
+#define RULE(flags) ":" ENTRY ":M:0:" MAGIC "::" INTERP_PATH ":" flags
+
+/* The target only has to carry the magic; it is never actually loaded. */
+static int create_target(void)
+{
+ char buf[128] = MAGIC "\n";
+ int fd;
+
+ unlink(TARGET_PATH);
+ fd = open(TARGET_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, buf, sizeof(buf)) != (ssize_t)sizeof(buf)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+FIXTURE(transparent) {
+};
+
+FIXTURE_SETUP(transparent)
+{
+ char src[PATH_MAX];
+
+ if (getuid() != 0)
+ SKIP(return, "test must be run as root");
+ if (!binfmt_misc_available())
+ SKIP(return, "no binfmt_misc");
+
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_transparent_interp"), 0);
+ ASSERT_EQ(copy_file(src, INTERP_PATH), 0);
+ ASSERT_EQ(create_target(), 0);
+
+ /* Skip the whole suite on a kernel that does not know 'T'. */
+ if (!binfmt_flag_supported('T')) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'T' flag");
+ }
+}
+
+FIXTURE_TEARDOWN(transparent)
+{
+ unregister(ENTRY);
+ unlink(TARGET_PATH);
+ unlink(INTERP_PATH);
+}
+
+/* Grammar sanity check: the same entry without 'T' has to register. */
+TEST_F(transparent, plain_entry_registers)
+{
+ ASSERT_EQ(write_reg(RULE("")), 0);
+}
+
+/* 'T' preserves the whole argv, so combining it with 'P' is rejected. */
+TEST_F(transparent, rejects_preserve_argv0)
+{
+ ASSERT_NE(write_reg(RULE("TP")), 0);
+ EXPECT_EQ(errno, EINVAL);
+}
+
+/* The interpreter asserts the identity the kernel built for it. */
+TEST_F(transparent, dispatch)
+{
+ ASSERT_EQ(write_reg(RULE("T")), 0);
+
+ setenv("BINFMT_TEST_BINARY", TARGET_PATH, 1);
+ setenv("BINFMT_TEST_ARGV0", PAYLOAD_ARGV0, 1);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_transparent_interp.c b/tools/testing/selftests/exec/binfmt_transparent_interp.c
new file mode 100644
index 000000000000..d4c4a538c9aa
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_transparent_interp.c
@@ -0,0 +1,112 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Asserting interpreter for the transparent binfmt_misc mode. It runs in
+ * place of the dispatched binary and verifies the identity the kernel
+ * constructed: the aux vector contract, the exe link, argv, cmdline, comm
+ * and the write denial on the binary. BINFMT_TEST_BINARY names the binary;
+ * the harness execs it with the arguments "argone argtwo". Prints
+ * TRANSPARENT_OK and exits 0 when every check holds.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/auxv.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest.h"
+
+#ifndef AT_FLAGS_TRANSPARENT_INTERP
+#define AT_FLAGS_TRANSPARENT_INTERP (1 << 1)
+#endif
+
+static int fail;
+
+static void ok(int cond, const char *what)
+{
+ if (!cond) {
+ fprintf(stderr, "TRANSPARENT_FAIL: %s (errno %d)\n", what, errno);
+ fail = 1;
+ }
+}
+
+int main(int argc, char **argv)
+{
+ const char *binary = getenv("BINFMT_TEST_BINARY");
+ const char *argv0 = getenv("BINFMT_TEST_ARGV0");
+ char expect[PATH_MAX + 32], buf[PATH_MAX];
+ unsigned long execfd;
+ struct stat stb, stfd;
+ const char *want[3];
+ const char *base;
+ size_t expect_len, i;
+ int fd, have_stb, have_stfd;
+ ssize_t n;
+
+ if (!binary) {
+ fprintf(stderr, "TRANSPARENT_FAIL: BINFMT_TEST_BINARY unset\n");
+ return 1;
+ }
+ /* Distinct from the binary path, so a classic argv splice is caught. */
+ want[0] = argv0 ? argv0 : binary;
+ want[1] = PAYLOAD_ARG1;
+ want[2] = PAYLOAD_ARG2;
+
+ /* The aux vector announces the transparent contract. */
+ ok(getauxval(AT_FLAGS) & AT_FLAGS_TRANSPARENT_INTERP,
+ "AT_FLAGS lacks AT_FLAGS_TRANSPARENT_INTERP");
+
+ /* AT_EXECFD refers to the very file that was executed. */
+ execfd = getauxval(AT_EXECFD);
+ ok(execfd > 2, "no AT_EXECFD");
+ have_stb = !stat(binary, &stb);
+ ok(have_stb, "cannot stat the binary");
+ have_stfd = !fstat((int)execfd, &stfd);
+ ok(have_stfd, "cannot fstat AT_EXECFD");
+ ok(have_stb && have_stfd && stb.st_dev == stfd.st_dev &&
+ stb.st_ino == stfd.st_ino, "AT_EXECFD is not the binary");
+
+ /* The exe link names the binary, not this interpreter. */
+ ok(exe_is(binary), "/proc/self/exe is not the binary");
+
+ /* argv arrived unspliced. */
+ ok(argc == (int)ARRAY_SIZE(want), "argv was rewritten");
+ for (i = 0; i < ARRAY_SIZE(want) && i < (size_t)argc; i++)
+ ok(!strcmp(argv[i], want[i]), "argv was rewritten");
+
+ /* And so did the kernel's copy of it: the same strings, NUL separated. */
+ for (i = 0, expect_len = 0; i < ARRAY_SIZE(want); i++) {
+ size_t len = strlen(want[i]) + 1;
+
+ if (expect_len + len > sizeof(expect)) {
+ ok(0, "argv does not fit the expectation buffer");
+ break;
+ }
+ memcpy(expect + expect_len, want[i], len);
+ expect_len += len;
+ }
+ fd = open("/proc/self/cmdline", O_RDONLY);
+ n = fd >= 0 ? read(fd, buf, sizeof(buf)) : -1;
+ if (fd >= 0)
+ close(fd);
+ ok(n == (ssize_t)expect_len && !memcmp(buf, expect, expect_len),
+ "/proc/self/cmdline was rewritten");
+
+ /* comm is the binary's basename. */
+ base = strrchr(binary, '/');
+ base = base ? base + 1 : binary;
+ ok(comm_is(base), "comm is not the binary's basename");
+
+ /* The binary is write-denied while it runs, like a direct exec. */
+ ok(write_denied(binary), "binary is writable while running");
+ ok(write_denied("/proc/self/exe"), "exe link is writable while running");
+
+ if (!fail)
+ printf("TRANSPARENT_OK\n");
+ return fail;
+}
diff --git a/tools/testing/selftests/exec/bpf_interp.bpf.c b/tools/testing/selftests/exec/bpf_interp.bpf.c
new file mode 100644
index 000000000000..8df2d2d01e25
--- /dev/null
+++ b/tools/testing/selftests/exec/bpf_interp.bpf.c
@@ -0,0 +1,61 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the selftest's fixed-interpreter case: match a
+ * 64-bit aarch64 ELF header from the prefetched buffer and route it to a fixed
+ * interpreter chosen by the program. This is the portable, self-contained
+ * equivalent of routing a foreign binary to an emulator: it matches
+ * programmatically and computes the interpreter, but points at a test binary
+ * the harness installs rather than a system emulator.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define EM_AARCH64 183
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+
+/*
+ * A magic-style decision needs nothing beyond the prefetched bprm->buf,
+ * even though the match program could read the file.
+ */
+SEC("struct_ops.s/match")
+bool BPF_PROG(bpf_interp_match, struct linux_binprm *bprm)
+{
+ __u16 machine;
+
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* e_machine is a 16-bit little-endian field at offset 18. */
+ machine = (__u8)bprm->buf[18] | ((__u16)(__u8)bprm->buf[19] << 8);
+ return machine == EM_AARCH64;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(bpf_interp_load, struct linux_binprm *bprm)
+{
+ /*
+ * Keep the path on the (writable) stack: bpf_binprm_set_interp() takes
+ * a sized memory arg and the verifier rejects a read-only .rodata
+ * buffer for it. The harness installs the interpreter at this path.
+ */
+ char interp[] = "/tmp/binfmt_bpf_interp";
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops bpf_interp = {
+ .match = (void *)bpf_interp_match,
+ .load = (void *)bpf_interp_load,
+ .name = "bpf_interp",
+};
diff --git a/tools/testing/selftests/exec/config b/tools/testing/selftests/exec/config
index c308079867b3..ea359a929ae8 100644
--- a/tools/testing/selftests/exec/config
+++ b/tools/testing/selftests/exec/config
@@ -1,2 +1,12 @@
CONFIG_BLK_DEV=y
CONFIG_BLK_DEV_LOOP=y
+CONFIG_BINFMT_MISC=y
+CONFIG_BINFMT_MISC_BPF=y
+CONFIG_BPF_JIT=y
+CONFIG_BPF_SYSCALL=y
+CONFIG_DEBUG_INFO=y
+CONFIG_DEBUG_INFO_BTF=y
+CONFIG_DEBUG_INFO_DWARF4=y
+CONFIG_OVERLAY_FS=y
+CONFIG_TMPFS=y
+CONFIG_USER_NS=y
diff --git a/tools/testing/selftests/exec/interp_bind.bpf.c b/tools/testing/selftests/exec/interp_bind.bpf.c
new file mode 100644
index 000000000000..1ce45cca215f
--- /dev/null
+++ b/tools/testing/selftests/exec/interp_bind.bpf.c
@@ -0,0 +1,76 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the selftest's bound-interpreter case: one
+ * handler, one entry, an interpreter per guest architecture - each bound to
+ * a file when the entry was registered rather than to a path resolved at
+ * exec time. The load program names the one it wants; a name the entry did
+ * not bind fails the exec, which the harness checks too.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define E_MACHINE_OFF 18
+#define EM_ARM 40
+#define EM_AARCH64 183
+#define EM_RISCV 243
+
+extern int bpf_binprm_select_interp(struct linux_binprm *bprm,
+ const char *name, size_t name__sz) __ksym;
+
+/* The guest architecture of a 64-bit ELF, or zero if it is not one. */
+static __u16 elf_machine(struct linux_binprm *bprm)
+{
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return 0;
+
+ /* Little-endian 16-bit field, read byte-wise for the verifier. */
+ return (__u8)bprm->buf[E_MACHINE_OFF] |
+ ((__u16)(__u8)bprm->buf[E_MACHINE_OFF + 1] << 8);
+}
+
+SEC("struct_ops.s/match")
+bool BPF_PROG(interp_bind_match, struct linux_binprm *bprm)
+{
+ __u16 machine = elf_machine(bprm);
+
+ return machine == EM_AARCH64 || machine == EM_RISCV ||
+ machine == EM_ARM;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(interp_bind_load, struct linux_binprm *bprm)
+{
+ /*
+ * Names, not paths: each one selects a file the entry pre-opened, so
+ * nothing is resolved here or later, in any namespace. The buffers
+ * are on the stack because the verifier rejects .rodata for a sized
+ * memory argument.
+ */
+ char first[] = "first";
+ char second[] = "second";
+ char unbound[] = "unbound";
+
+ switch (elf_machine(bprm)) {
+ case EM_AARCH64:
+ return bpf_binprm_select_interp(bprm, first, sizeof(first));
+ case EM_RISCV:
+ return bpf_binprm_select_interp(bprm, second, sizeof(second));
+ }
+
+ /* The entry bound nothing under this name: -ENOENT fails the exec. */
+ return bpf_binprm_select_interp(bprm, unbound, sizeof(unbound));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops interp_bind = {
+ .match = (void *)interp_bind_match,
+ .load = (void *)interp_bind_load,
+ .name = "interp_bind",
+};
diff --git a/tools/testing/selftests/exec/loader.bpf.c b/tools/testing/selftests/exec/loader.bpf.c
new file mode 100644
index 000000000000..108e51dd4961
--- /dev/null
+++ b/tools/testing/selftests/exec/loader.bpf.c
@@ -0,0 +1,56 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the loader-substitution case: match the
+ * marker the harness poked into the payload's e_ident padding and ask for
+ * the selected interpreter to be substituted for the binary's PT_INTERP,
+ * so the binary itself runs as a fully native exec.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define EI_PAD 9
+#define ELFCLASS64 2
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+extern int bpf_binprm_set_flags(struct linux_binprm *bprm,
+ enum bpf_binprm_flags flags) __ksym;
+
+SEC("struct_ops.s/match")
+bool BPF_PROG(loader_match, struct linux_binprm *bprm)
+{
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* The harness marks the payload with "LDRTST" at EI_PAD. */
+ return bprm->buf[EI_PAD + 0] == 'L' && bprm->buf[EI_PAD + 1] == 'D' &&
+ bprm->buf[EI_PAD + 2] == 'R' && bprm->buf[EI_PAD + 3] == 'T' &&
+ bprm->buf[EI_PAD + 4] == 'S' && bprm->buf[EI_PAD + 5] == 'T';
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(loader_load, struct linux_binprm *bprm)
+{
+ char interp[] = "/tmp/binfmt_loader_interp";
+ int err;
+
+ err = bpf_binprm_set_flags(bprm, BPF_BINPRM_LOADER);
+ if (err)
+ return err;
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops loader = {
+ .match = (void *)loader_match,
+ .load = (void *)loader_load,
+ .name = "loader",
+};
diff --git a/tools/testing/selftests/exec/nix_origin.bpf.c b/tools/testing/selftests/exec/nix_origin.bpf.c
new file mode 100644
index 000000000000..378e22a4c43b
--- /dev/null
+++ b/tools/testing/selftests/exec/nix_origin.bpf.c
@@ -0,0 +1,224 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * nix_origin.bpf.c - $ORIGIN-relative PT_INTERP resolution
+ *
+ * A binfmt_misc_ops handler that makes relocatable (Nix-style) ELF
+ * binaries work: if PT_INTERP starts with "$ORIGIN/", the loader is
+ * resolved relative to the directory of the binary being executed and
+ * selected via bpf_binprm_set_interp(). The match program reads the
+ * program headers itself, so anything else never commits to this
+ * handler and passes through untouched.
+ *
+ * Activate with:
+ * bpftool struct_ops register nix_origin.bpf.o /sys/fs/bpf
+ * echo ':nix-origin:B::::nix_origin:' > /proc/sys/fs/binfmt_misc/register
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define PATH_MAX 4096
+#define EI_CLASS 4
+#define ELFCLASSXX 2 /* ELFCLASS64; flip to 1 for 32-bit */
+#define PT_INTERP 3
+#define MAX_PHDRS 64
+
+#define ORIGIN "$ORIGIN"
+#define ORIGIN_LEN (sizeof(ORIGIN) - 1)
+
+#define ENOENT 2
+#define ENOEXEC 8
+#define ENAMETOOLONG 36
+
+extern int bpf_dynptr_from_file(struct file *file, __u32 flags,
+ struct bpf_dynptr *ptr__uninit) __ksym;
+extern int bpf_dynptr_file_discard(struct bpf_dynptr *dynptr) __ksym;
+extern int bpf_path_d_path(const struct path *path, char *buf,
+ size_t buf__sz) __ksym;
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+
+struct scratch {
+ char interp[PATH_MAX]; /* PT_INTERP as embedded in the binary */
+ char path[PATH_MAX]; /* d_path of the binary, becomes the result */
+};
+
+/* Keyed by pid: execs run concurrently and the programs can sleep. */
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(max_entries, 512);
+ __type(key, __u64);
+ __type(value, struct scratch);
+} scratch_map SEC(".maps");
+
+static const struct scratch zero_scratch;
+
+/* An ELF64 binary per the prefetched header? */
+static bool is_elf64(struct linux_binprm *bprm)
+{
+ return bprm->buf[0] == 0x7f && bprm->buf[1] == 'E' &&
+ bprm->buf[2] == 'L' && bprm->buf[3] == 'F' &&
+ bprm->buf[EI_CLASS] == ELFCLASSXX;
+}
+
+/* Locate PT_INTERP; false if the file has none or looks malformed. */
+static bool find_pt_interp(struct bpf_dynptr *dp, struct elf64_phdr *phdr)
+{
+ struct elf64_hdr ehdr;
+ bool found = false;
+ int i;
+
+ if (bpf_dynptr_read(&ehdr, sizeof(ehdr), dp, 0, 0))
+ return false;
+ if (ehdr.e_phentsize != sizeof(struct elf64_phdr))
+ return false;
+
+ bpf_for(i, 0, ehdr.e_phnum) {
+ if (i >= MAX_PHDRS)
+ break;
+ if (bpf_dynptr_read(phdr, sizeof(*phdr), dp,
+ ehdr.e_phoff + i * sizeof(*phdr), 0))
+ return false;
+ if (phdr->p_type == PT_INTERP) {
+ found = true;
+ break;
+ }
+ }
+ return found;
+}
+
+/*
+ * An ELF64 binary whose PT_INTERP starts with "$ORIGIN/" is ours. The
+ * match can sleep and read the file, so the decision is made here and
+ * regular binaries never commit to this handler: later binfmt_misc
+ * entries and binfmt_elf see them as if we did not exist.
+ */
+SEC("struct_ops.s/match")
+bool BPF_PROG(nix_origin_match, struct linux_binprm *bprm)
+{
+ char prefix[ORIGIN_LEN + 1] = {};
+ struct elf64_phdr phdr;
+ struct bpf_dynptr dp;
+ bool ours = false;
+
+ if (!is_elf64(bprm))
+ return false;
+
+ /* The dynptr must be discarded on every path once requested. */
+ if (bpf_dynptr_from_file(bprm->file, 0, &dp))
+ goto out;
+ if (find_pt_interp(&dp, &phdr) &&
+ phdr.p_filesz > ORIGIN_LEN + 1 &&
+ !bpf_dynptr_read(prefix, sizeof(prefix), &dp, phdr.p_offset, 0))
+ ours = !bpf_strncmp(prefix, sizeof(prefix), ORIGIN "/");
+out:
+ bpf_dynptr_file_discard(&dp);
+ return ours;
+}
+
+/*
+ * The match is committed and already vetted the "$ORIGIN/" prefix, so
+ * everything here reads the file again from scratch: -ENOEXEC only
+ * covers a binary that changed under us and stopped being ours.
+ */
+SEC("struct_ops.s/load")
+int BPF_PROG(nix_origin_load, struct linux_binprm *bprm)
+{
+ __u32 isz, sfx, rsz, slash;
+ struct elf64_phdr phdr;
+ struct bpf_dynptr dp;
+ struct scratch *sc;
+ __u64 id;
+ int ret = -ENOEXEC, len, i;
+
+ if (bpf_dynptr_from_file(bprm->file, 0, &dp))
+ goto out;
+
+ if (!find_pt_interp(&dp, &phdr))
+ goto out;
+
+ isz = phdr.p_filesz;
+ if (isz <= ORIGIN_LEN + 1 || isz >= sizeof(sc->interp))
+ goto out;
+ /*
+ * The range check above compiles to a test on a zero-extended copy of
+ * the u64 p_filesz, so the verifier does not carry the bound to the
+ * dynptr_read() length below ("unbounded memory access"). Mask isz to
+ * the buffer size (a power of two) and force the masked value to be
+ * materialized with a barrier so the read uses the bounded register.
+ */
+ isz &= sizeof(sc->interp) - 1;
+ barrier_var(isz);
+
+ id = bpf_get_current_pid_tgid();
+ if (bpf_map_update_elem(&scratch_map, &id, &zero_scratch, BPF_ANY))
+ goto out;
+ sc = bpf_map_lookup_elem(&scratch_map, &id);
+ if (!sc)
+ goto out_del;
+
+ if (bpf_dynptr_read(sc->interp, isz, &dp, phdr.p_offset, 0))
+ goto out_del;
+ if (sc->interp[isz - 1] != '\0')
+ goto out_del;
+
+ /* Not "$ORIGIN/..." anymore? Then it is not ours anymore either. */
+ if (sc->interp[0] != '$' || sc->interp[1] != 'O' ||
+ sc->interp[2] != 'R' || sc->interp[3] != 'I' ||
+ sc->interp[4] != 'G' || sc->interp[5] != 'I' ||
+ sc->interp[6] != 'N' || sc->interp[7] != '/')
+ goto out_del;
+
+ /*
+ * From here on resolution failures fail the exec instead of falling
+ * back to binfmt_elf, which would resolve the literal "$ORIGIN/..."
+ * relative to the caller's cwd.
+ */
+ ret = -ENOENT;
+ len = bpf_path_d_path(&bprm->file->f_path, sc->path, sizeof(sc->path));
+ if (len <= 0 || len > sizeof(sc->path))
+ goto out_del;
+ /* Unreachable or unlinked ("... (deleted)") binaries can't resolve. */
+ if (sc->path[0] != '/')
+ goto out_del;
+
+ /* $ORIGIN = dirname of the binary. */
+ slash = 0;
+ bpf_for(i, 1, len - 1) {
+ if (i >= sizeof(sc->path))
+ break;
+ if (sc->path[i] == '/')
+ slash = i;
+ }
+
+ /* Splice the suffix (leading '/' and NUL included) onto the dir. */
+ sfx = isz - ORIGIN_LEN;
+ rsz = slash + sfx;
+ if (rsz > sizeof(sc->path)) {
+ ret = -ENAMETOOLONG;
+ goto out_del;
+ }
+ bpf_for(i, 0, sfx) {
+ __u32 s = ORIGIN_LEN + i, d = slash + i;
+
+ if (s >= sizeof(sc->interp) || d >= sizeof(sc->path))
+ break;
+ sc->path[d] = sc->interp[s];
+ }
+
+ ret = bpf_binprm_set_interp(bprm, sc->path, rsz);
+out_del:
+ bpf_map_delete_elem(&scratch_map, &id);
+out:
+ bpf_dynptr_file_discard(&dp);
+ return ret;
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops nix_origin = {
+ .match = (void *)nix_origin_match,
+ .load = (void *)nix_origin_load,
+ .name = "nix_origin",
+};
diff --git a/tools/testing/selftests/exec/transparent.bpf.c b/tools/testing/selftests/exec/transparent.bpf.c
new file mode 100644
index 000000000000..7632019ebe69
--- /dev/null
+++ b/tools/testing/selftests/exec/transparent.bpf.c
@@ -0,0 +1,57 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the transparent-mode case: match a synthetic
+ * riscv ELF header and run the asserting interpreter transparently - the
+ * argument vector untouched, the binary in AT_EXECFD and mm->exe_file
+ * labeled with the binary.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define EM_RISCV 243
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+extern int bpf_binprm_set_flags(struct linux_binprm *bprm,
+ enum bpf_binprm_flags flags) __ksym;
+
+SEC("struct_ops.s/match")
+bool BPF_PROG(transparent_match, struct linux_binprm *bprm)
+{
+ __u16 machine;
+
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* e_machine is a 16-bit little-endian field at offset 18. */
+ machine = (__u8)bprm->buf[18] | ((__u16)(__u8)bprm->buf[19] << 8);
+ return machine == EM_RISCV;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(transparent_load, struct linux_binprm *bprm)
+{
+ char interp[] = "/tmp/binfmt_transparent_interp";
+ int err;
+
+ err = bpf_binprm_set_flags(bprm, BPF_BINPRM_TRANSPARENT);
+ if (err)
+ return err;
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops transparent = {
+ .match = (void *)transparent_match,
+ .load = (void *)transparent_load,
+ .name = "transparent",
+};
diff --git a/tools/testing/selftests/filesystems/.gitignore b/tools/testing/selftests/filesystems/.gitignore
index a78f894157de..9eb185fb2f9d 100644
--- a/tools/testing/selftests/filesystems/.gitignore
+++ b/tools/testing/selftests/filesystems/.gitignore
@@ -6,3 +6,4 @@ file_stressor
anon_inode_test
kernfs_test
idmapped_tmpfile
+ustat_test
diff --git a/tools/testing/selftests/filesystems/Makefile b/tools/testing/selftests/filesystems/Makefile
index a7ec2ba2dd83..03be337c1f35 100644
--- a/tools/testing/selftests/filesystems/Makefile
+++ b/tools/testing/selftests/filesystems/Makefile
@@ -1,7 +1,7 @@
# SPDX-License-Identifier: GPL-2.0
CFLAGS += $(KHDR_INCLUDES)
-TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog
+TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog ustat_test
TEST_GEN_PROGS += idmapped_tmpfile
TEST_GEN_PROGS_EXTENDED := dnotify_test
diff --git a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c
index f6f1a7ff01b0..81a994943e12 100644
--- a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c
+++ b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c
@@ -3538,4 +3538,27 @@ TEST(epoll65)
close(ctx.efd[1]);
}
+TEST(epoll66)
+{
+ struct epoll_event event;
+ int pfd[2], efd;
+
+ ASSERT_EQ(pipe(pfd), 0);
+
+ efd = epoll_create1(0);
+ ASSERT_GE(efd, 0);
+
+ event.events = EPOLLIN | EPOLLET;
+ ASSERT_EQ(epoll_ctl(efd, EPOLL_CTL_ADD, pfd[0], &event), 0);
+
+ for (int i = 0; i < 2; ++i) {
+ ASSERT_EQ(write(pfd[1], "", 1), 1);
+ EXPECT_EQ(epoll_wait(efd, &event, 1, 0), 1);
+ }
+
+ close(pfd[0]);
+ close(pfd[1]);
+ close(efd);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/failfs/.gitignore b/tools/testing/selftests/filesystems/failfs/.gitignore
new file mode 100644
index 000000000000..cd3b5d884d7e
--- /dev/null
+++ b/tools/testing/selftests/filesystems/failfs/.gitignore
@@ -0,0 +1,2 @@
+# SPDX-License-Identifier: GPL-2.0-only
+failfs_test
diff --git a/tools/testing/selftests/filesystems/failfs/Makefile b/tools/testing/selftests/filesystems/failfs/Makefile
new file mode 100644
index 000000000000..3c5d98b4fe72
--- /dev/null
+++ b/tools/testing/selftests/filesystems/failfs/Makefile
@@ -0,0 +1,5 @@
+# SPDX-License-Identifier: GPL-2.0
+CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
+TEST_GEN_PROGS := failfs_test
+
+include ../../lib.mk
diff --git a/tools/testing/selftests/filesystems/failfs/failfs_test.c b/tools/testing/selftests/filesystems/failfs/failfs_test.c
new file mode 100644
index 000000000000..29a3c294127e
--- /dev/null
+++ b/tools/testing/selftests/filesystems/failfs/failfs_test.c
@@ -0,0 +1,585 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <link.h>
+#include <sched.h>
+#include <signal.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/types.h>
+#include <sys/vfs.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef __NR_fchroot
+#define __NR_fchroot 472
+#endif
+
+#ifndef FD_PIDFS_ROOT
+#define FD_PIDFS_ROOT -10002
+#endif
+
+#ifndef FD_NSFS_ROOT
+#define FD_NSFS_ROOT -10003
+#endif
+
+#ifndef FD_FAILFS_ROOT
+#define FD_FAILFS_ROOT -10004
+#endif
+
+#define NOBODY_UID 65534
+
+/* Child sentinel exit code: the exec was blocked as expected. */
+#define FAILFS_EXEC_BLOCKED 99
+
+/* Stack for the CLONE_FS helper in fchroot_sentinel_shared_fs_struct. */
+#define FAILFS_CLONE_STACK (64 * 1024)
+
+static int sys_fchroot(int fd, unsigned int flags)
+{
+ return syscall(__NR_fchroot, fd, flags);
+}
+
+/*
+ * Raw syscall: glibc's getcwd() rejects the kernel's "(unreachable)"
+ * result and falls back to a generic implementation.
+ */
+static long sys_getcwd(char *buf, size_t size)
+{
+ return syscall(__NR_getcwd, buf, size);
+}
+
+static int drop_to_nobody(void)
+{
+ return setresuid(NOBODY_UID, NOBODY_UID, NOBODY_UID);
+}
+
+/* Parked CLONE_FS child; dies with its parent so it never leaks. */
+static int failfs_park(void *arg)
+{
+ pid_t parent = (pid_t)(long)arg;
+
+ prctl(PR_SET_PDEATHSIG, SIGKILL);
+ /* The parent may have died before the death signal was armed. */
+ if (getppid() != parent)
+ _exit(0);
+ pause();
+ return 0;
+}
+
+/* Is fd a dynamically linked ELF with an absolute PT_INTERP interpreter? */
+static int elf_has_absolute_interp(int fd)
+{
+ ElfW(Ehdr) ehdr;
+ ElfW(Phdr) phdr;
+ char interp;
+ int i;
+
+ if (pread(fd, &ehdr, sizeof(ehdr), 0) != sizeof(ehdr))
+ return 0;
+ if (memcmp(ehdr.e_ident, ELFMAG, SELFMAG) != 0)
+ return 0;
+
+ for (i = 0; i < ehdr.e_phnum; i++) {
+ if (pread(fd, &phdr, sizeof(phdr),
+ ehdr.e_phoff + i * sizeof(phdr)) != sizeof(phdr))
+ return 0;
+ if (phdr.p_type != PT_INTERP)
+ continue;
+ if (pread(fd, &interp, 1, phdr.p_offset) != 1)
+ return 0;
+ return interp == '/';
+ }
+
+ return 0;
+}
+
+TEST(fchdir_sentinel)
+{
+ char buf[PATH_MAX];
+ int fd;
+
+ ASSERT_EQ(fchdir(FD_FAILFS_ROOT), 0);
+
+ /* The working directory is unreachable from the process root. */
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strncmp(buf, "(unreachable)", 13), 0);
+
+ /* Every AT_FDCWD-relative lookup fails. */
+ ASSERT_EQ(openat(AT_FDCWD, "foo", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(openat(AT_FDCWD, ".", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(openat(AT_FDCWD, "..", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(openat(AT_FDCWD, "foo", O_WRONLY | O_CREAT, 0600), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The cwd cannot be pinned by following /proc/self/cwd into it. */
+ ASSERT_EQ(open("/proc/self/cwd", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The root is untouched so absolute lookups keep working... */
+ fd = open("/", O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ /* ... and the working directory can be recovered. */
+ ASSERT_EQ(chdir("/"), 0);
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strcmp(buf, "/"), 0);
+}
+
+TEST(fchdir_rejects_other_sentinels)
+{
+ ASSERT_EQ(fchdir(FD_PIDFS_ROOT), -1);
+ ASSERT_EQ(errno, EBADF);
+ ASSERT_EQ(fchdir(FD_NSFS_ROOT), -1);
+ ASSERT_EQ(errno, EBADF);
+ ASSERT_EQ(fchdir(-10009), -1);
+ ASSERT_EQ(errno, EBADF);
+}
+
+TEST(fchroot_flags)
+{
+ int fd;
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 1), -1);
+ ASSERT_EQ(errno, EINVAL);
+
+ fd = open("/", O_PATH | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(sys_fchroot(fd, 1), -1);
+ ASSERT_EQ(errno, EINVAL);
+ ASSERT_EQ(close(fd), 0);
+}
+
+TEST(fchroot_bad_fd)
+{
+ ASSERT_EQ(sys_fchroot(-1, 0), -1);
+ ASSERT_EQ(errno, EBADF);
+
+ /* Only FD_FAILFS_ROOT is a valid sentinel. */
+ ASSERT_EQ(sys_fchroot(FD_PIDFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EBADF);
+ ASSERT_EQ(sys_fchroot(FD_NSFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EBADF);
+}
+
+TEST(fchroot_notdir)
+{
+ int fd;
+
+ fd = open("/proc/self/status", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(sys_fchroot(fd, 0), -1);
+ ASSERT_EQ(errno, ENOTDIR);
+ ASSERT_EQ(close(fd), 0);
+}
+
+TEST(fchroot_realfd_requires_cap)
+{
+ int fd;
+
+ if (geteuid() == 0)
+ ASSERT_EQ(drop_to_nobody(), 0);
+
+ fd = open("/", O_PATH | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(sys_fchroot(fd, 0), -1);
+ ASSERT_EQ(errno, EPERM);
+ ASSERT_EQ(close(fd), 0);
+}
+
+TEST(fchroot_realfd)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ char path[PATH_MAX];
+ struct stat st;
+ int tmpfd, dfd, fd;
+
+ if (geteuid() != 0)
+ SKIP(return, "fchroot() with a regular fd requires CAP_SYS_CHROOT");
+
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+
+ ASSERT_NE(mkdtemp(template), NULL);
+ snprintf(path, sizeof(path), "%s/canary", template);
+ fd = open(path, O_WRONLY | O_CREAT, 0600);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ dfd = open(template, O_PATH | O_DIRECTORY);
+ ASSERT_GE(dfd, 0);
+ ASSERT_EQ(sys_fchroot(dfd, 0), 0);
+ ASSERT_EQ(close(dfd), 0);
+
+ ASSERT_EQ(stat("/canary", &st), 0);
+
+ /* Best-effort cleanup: dirfd-anchored I/O works with the new root. */
+ snprintf(path, sizeof(path), "%s/canary", template + strlen("/tmp/"));
+ unlinkat(tmpfd, path, 0);
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+}
+
+TEST(fchroot_sentinel)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ struct stat realroot, st;
+ struct statfs sfs;
+ char buf[PATH_MAX];
+ int procfd, tmpfd, dfd, fd;
+ struct {
+ struct file_handle handle;
+ unsigned char f_handle[MAX_HANDLE_SZ];
+ } fh;
+ int mntid;
+ ssize_t ret;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ ASSERT_EQ(stat("/", &realroot), 0);
+ procfd = open("/proc", O_PATH | O_DIRECTORY);
+ ASSERT_GE(procfd, 0);
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+ ASSERT_NE(mkdtemp(template), NULL);
+ dfd = open(template, O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(dfd, 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /* Absolute lookups fail. */
+ ASSERT_EQ(open("/etc/passwd", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(mkdir("/foo", 0700), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /*
+ * The root cannot be referenced at all - not even an O_PATH open,
+ * which skips ->permission(), because it lands on the root as a
+ * jumped walk terminal that ->d_weak_revalidate() refuses.
+ */
+ ASSERT_EQ(open("/", O_RDONLY | O_DIRECTORY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(open("/", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(statfs("/", &sfs), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /*
+ * It cannot be pinned by following /proc/self/root into it either
+ * (only the root is in failfs here, so self/cwd is still real).
+ */
+ ASSERT_EQ(openat(procfd, "self/root", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* Nor encoded into a file handle. */
+ fh.handle.handle_bytes = MAX_HANDLE_SZ;
+ ASSERT_EQ(name_to_handle_at(AT_FDCWD, "/", &fh.handle, &mntid, 0), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The working directory is now unreachable from the root. */
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strncmp(buf, "(unreachable)", 13), 0);
+
+ /* Lookups anchored at real directories keep working. */
+ fd = openat(AT_FDCWD, ".", O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+ fd = openat(dfd, "canary", O_WRONLY | O_CREAT, 0600);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(write(fd, "x", 1), 1);
+ ASSERT_EQ(close(fd), 0);
+ fd = openat(dfd, "canary", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ /* ".." walks clamp at the top of the mount tree, not at failfs. */
+ fd = openat(AT_FDCWD, "../../../../../../../../../..", O_PATH);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstat(fd, &st), 0);
+ ASSERT_EQ(st.st_dev, realroot.st_dev);
+ ASSERT_EQ(st.st_ino, realroot.st_ino);
+ ASSERT_EQ(close(fd), 0);
+
+ /* readlink of the magic link still works: it does not follow. */
+ ret = readlinkat(procfd, "self/root", buf, sizeof(buf) - 1);
+ ASSERT_GT(ret, 0);
+ buf[ret] = '\0';
+ TH_LOG("/proc/self/root points to '%s'", buf);
+ /* d_path() names the failfs root synthetically, never as a real path. */
+ ASSERT_EQ(strcmp(buf, "failfs:/"), 0);
+
+ /* But following it into failfs is refused. */
+ ASSERT_EQ(fstatat(procfd, "self/root", &st, 0), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* Best-effort cleanup via the pre-opened dirfds. */
+ unlinkat(dfd, "canary", 0);
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+}
+
+TEST(fchroot_sentinel_absolute_symlink)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ int tmpfd, dfd, fd;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+ ASSERT_NE(mkdtemp(template), NULL);
+ dfd = open(template, O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(dfd, 0);
+
+ fd = openat(dfd, "target", O_WRONLY | O_CREAT, 0600);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+ ASSERT_EQ(symlinkat("target", dfd, "rel"), 0);
+ ASSERT_EQ(symlinkat("/etc", dfd, "abs"), 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /* Relative symlinks keep resolving within the dirfd-anchored walk... */
+ fd = openat(dfd, "rel", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ /* ... absolute symlinks restart the walk at the failfs root. */
+ ASSERT_EQ(openat(dfd, "abs", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* Best-effort cleanup via the pre-opened dirfds. */
+ unlinkat(dfd, "abs", 0);
+ unlinkat(dfd, "rel", 0);
+ unlinkat(dfd, "target", 0);
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+}
+
+TEST(fchroot_sentinel_unprivileged)
+{
+ char buf[PATH_MAX];
+
+ if (geteuid() == 0)
+ ASSERT_EQ(drop_to_nobody(), 0);
+
+ /* Without no_new_privs entering failfs is not allowed... */
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EPERM);
+
+ /* ... with no_new_privs set it is allowed. */
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ ASSERT_EQ(open("/etc/passwd", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The task counts as chrooted: no user namespaces anymore. */
+ ASSERT_EQ(unshare(CLONE_NEWUSER), -1);
+ ASSERT_EQ(errno, EPERM);
+
+ /* With both root and cwd in failfs getcwd() reports "/". */
+ ASSERT_EQ(fchdir(FD_FAILFS_ROOT), 0);
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strcmp(buf, "/"), 0);
+}
+
+TEST(fchroot_sentinel_rejected_when_chrooted)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ int tmpfd;
+
+ if (geteuid() != 0)
+ SKIP(return, "chroot() requires CAP_SYS_CHROOT");
+
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+ ASSERT_NE(mkdtemp(template), NULL);
+ ASSERT_EQ(chroot(template), 0);
+ ASSERT_EQ(chdir("/"), 0);
+
+ /* Remove the jail while still privileged; sticky /tmp blocks nobody. */
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+
+ ASSERT_EQ(drop_to_nobody(), 0);
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+
+ /* An unprivileged chrooted task must not lift its ".." barrier. */
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EPERM);
+}
+
+TEST(fchroot_sentinel_shared_fs_struct)
+{
+ char stack[FAILFS_CLONE_STACK];
+ pid_t pid;
+
+ if (geteuid() == 0)
+ ASSERT_EQ(drop_to_nobody(), 0);
+
+ /* A CLONE_FS sibling shares the fs_struct: bump fs->users to 2. */
+ pid = clone(failfs_park, stack + sizeof(stack), CLONE_FS | SIGCHLD,
+ (void *)(long)getpid());
+ ASSERT_GE(pid, 0);
+
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+
+ /*
+ * A sibling without no_new_privs could exec a setuid binary with
+ * the failfs root, so a shared fs_struct is refused even with
+ * no_new_privs set.
+ */
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EINVAL);
+
+ ASSERT_EQ(kill(pid, SIGKILL), 0);
+ ASSERT_EQ(waitpid(pid, NULL, 0), pid);
+}
+
+TEST(fchroot_sentinel_no_overmount)
+{
+ if (geteuid() != 0)
+ SKIP(return, "mounting requires privileges");
+
+ /*
+ * Contain the blast radius: if failfs ever regressed and "/"
+ * resolved to the real root, the tmpfs mount below must not touch
+ * the host. A private mount namespace keeps it local to this child.
+ */
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /*
+ * Nothing can be mounted on top of the failfs root. It cannot even
+ * be named as a mount target: resolving "/" is refused before the
+ * mount machinery (which, failfs being in no mount namespace, would
+ * reject it anyway) is ever reached. open_tree(OPEN_TREE_CLONE) is
+ * likewise moot since no fd to the root can be obtained.
+ */
+ ASSERT_EQ(mount("none", "/", "tmpfs", 0, NULL), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+}
+
+TEST(fchroot_sentinel_setns_escape)
+{
+ struct stat realroot, st;
+ int nsfd;
+
+ if (geteuid() != 0)
+ SKIP(return, "setns() to a mount namespace requires privileges");
+
+ ASSERT_EQ(stat("/", &realroot), 0);
+ nsfd = open("/proc/self/ns/mnt", O_RDONLY);
+ ASSERT_GE(nsfd, 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+ ASSERT_EQ(open("/etc", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* A mount namespace fd is the key out: it resets root and cwd. */
+ ASSERT_EQ(setns(nsfd, CLONE_NEWNS), 0);
+ ASSERT_EQ(close(nsfd), 0);
+
+ ASSERT_EQ(stat("/", &st), 0);
+ ASSERT_EQ(st.st_dev, realroot.st_dev);
+ ASSERT_EQ(st.st_ino, realroot.st_ino);
+}
+
+TEST(fchroot_sentinel_exec)
+{
+ pid_t pid;
+ int status;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /*
+ * Exec in a child: a wrongly successful exec would replace the test
+ * image and its exit code would not match the sentinel below.
+ */
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ execl("/bin/true", "true", NULL);
+ _exit(errno == EOPNOTSUPP ? FAILFS_EXEC_BLOCKED : 1);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), FAILFS_EXEC_BLOCKED);
+}
+
+TEST(fchroot_sentinel_exec_interpreter)
+{
+ static const char * const argv[] = { "failfs_test", NULL };
+ static const char * const envp[] = { NULL };
+ pid_t pid;
+ int status, exefd;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ /* Exec ourselves: the one binary guaranteed to be around. */
+ exefd = open("/proc/self/exe", O_RDONLY);
+ ASSERT_GE(exefd, 0);
+ if (!elf_has_absolute_interp(exefd))
+ SKIP(return, "test binary has no absolute PT_INTERP interpreter");
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /*
+ * The binary itself needs no path lookup - it is executed by fd -
+ * but loading it fails on opening the absolute PT_INTERP
+ * interpreter. Run it in a child so a wrongly successful exec does
+ * not replace the test image and masquerade as a pass.
+ */
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ syscall(__NR_execveat, exefd, "", argv, envp, AT_EMPTY_PATH);
+ _exit(errno == EOPNOTSUPP ? FAILFS_EXEC_BLOCKED : 1);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), FAILFS_EXEC_BLOCKED);
+}
+
+TEST(fchroot_sentinel_inherited)
+{
+ pid_t pid;
+ int status;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ if (open("/etc", O_PATH) != -1 || errno != EOPNOTSUPP)
+ _exit(1);
+ _exit(0);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/fscontext_ns/Makefile b/tools/testing/selftests/filesystems/fscontext_ns/Makefile
new file mode 100644
index 000000000000..7e3506294757
--- /dev/null
+++ b/tools/testing/selftests/filesystems/fscontext_ns/Makefile
@@ -0,0 +1,10 @@
+# SPDX-License-Identifier: GPL-2.0
+TEST_GEN_PROGS := fscontext_ns_test
+
+CFLAGS += -Wall -O0 -g $(KHDR_INCLUDES) $(TOOLS_INCLUDES)
+LDLIBS := -lcap
+
+include ../../lib.mk
+
+$(OUTPUT)/fscontext_ns_test: fscontext_ns_test.c ../utils.c
+ $(CC) $(CFLAGS) -o $@ $^ $(LDLIBS)
diff --git a/tools/testing/selftests/filesystems/fscontext_ns/fscontext_ns_test.c b/tools/testing/selftests/filesystems/fscontext_ns/fscontext_ns_test.c
new file mode 100644
index 000000000000..0f30a3e6e197
--- /dev/null
+++ b/tools/testing/selftests/filesystems/fscontext_ns/fscontext_ns_test.c
@@ -0,0 +1,239 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (C) 2026 Christian Brauner <brauner@kernel.org>
+ *
+ * Test that completing a filesystem context from another user namespace
+ * doesn't warn.
+ *
+ * fsopen() records the caller's user namespace in fc->user_ns and hands
+ * back an ordinary file descriptor. The task that issues
+ * FSCONFIG_CMD_CREATE need not be the one that created the context: the fd
+ * is inherited across fork() and exec() and it can be passed over a unix
+ * socket. vfs_cmd_create() authorizes the create with mount_capable(),
+ * which for FS_USERNS_MOUNT checks ns_capable(fc->user_ns, CAP_SYS_ADMIN),
+ * and that succeeds for a task holding CAP_SYS_ADMIN in an ancestor of
+ * fc->user_ns.
+ *
+ * binfmt_misc and overlayfs used to WARN_ON() that mismatch, which let an
+ * unprivileged user taint the kernel, flood the log and panic a kernel
+ * booted with panic_on_warn. The mount must still be refused, but it must
+ * not warn.
+ */
+#define _GNU_SOURCE
+
+#include <errno.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/socket.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include "../wrappers.h"
+#include "../utils.h"
+#include "../../kselftest_harness.h"
+
+#ifndef FSCONFIG_CMD_CREATE
+#define FSCONFIG_CMD_CREATE 6
+#endif
+
+/* TAINT_WARN, i.e. bit 9 of /proc/sys/kernel/tainted. */
+#define TAINT_WARN_BIT 9
+
+static bool taint_warn_set(void)
+{
+ unsigned long taint = 0;
+ FILE *f;
+
+ f = fopen("/proc/sys/kernel/tainted", "r");
+ if (!f)
+ return false;
+ if (fscanf(f, "%lu", &taint) != 1)
+ taint = 0;
+ fclose(f);
+
+ return taint & (1UL << TAINT_WARN_BIT);
+}
+
+static int send_fd(int sock, int fd)
+{
+ char cmsgbuf[CMSG_SPACE(sizeof(int))] = {};
+ char b[1] = { 'x' };
+ struct iovec iov = { .iov_base = b, .iov_len = sizeof(b) };
+ struct msghdr msg = {
+ .msg_iov = &iov,
+ .msg_iovlen = 1,
+ .msg_control = cmsgbuf,
+ .msg_controllen = sizeof(cmsgbuf),
+ };
+ struct cmsghdr *cmsg;
+
+ cmsg = CMSG_FIRSTHDR(&msg);
+ cmsg->cmsg_level = SOL_SOCKET;
+ cmsg->cmsg_type = SCM_RIGHTS;
+ cmsg->cmsg_len = CMSG_LEN(sizeof(int));
+ memcpy(CMSG_DATA(cmsg), &fd, sizeof(int));
+
+ return sendmsg(sock, &msg, 0) < 0 ? -1 : 0;
+}
+
+static int recv_fd(int sock)
+{
+ char cmsgbuf[CMSG_SPACE(sizeof(int))] = {};
+ char b[1];
+ struct iovec iov = { .iov_base = b, .iov_len = sizeof(b) };
+ struct msghdr msg = {
+ .msg_iov = &iov,
+ .msg_iovlen = 1,
+ .msg_control = cmsgbuf,
+ .msg_controllen = sizeof(cmsgbuf),
+ };
+ struct cmsghdr *cmsg;
+ int fd = -1;
+
+ if (recvmsg(sock, &msg, 0) <= 0)
+ return -1;
+
+ cmsg = CMSG_FIRSTHDR(&msg);
+ if (!cmsg || cmsg->cmsg_type != SCM_RIGHTS)
+ return -1;
+ memcpy(&fd, CMSG_DATA(cmsg), sizeof(int));
+
+ return fd;
+}
+
+/*
+ * Create a context for @fsname in a child and complete it here. With @nest
+ * the child first creates its own user namespace, so that the context is
+ * created in a descendant of the namespace completing it. The child needs a
+ * mount namespace of its own as well: fsopen() gates on may_mount(), which
+ * asks for CAP_SYS_ADMIN in the user namespace owning the caller's mount
+ * namespace.
+ *
+ * Returns the result of FSCONFIG_CMD_CREATE with errno set, or -ENODATA if
+ * the child could not create the context at all.
+ */
+static int create_from_child(const char *fsname, bool nest)
+{
+ int sock[2], fd, ret, status;
+ pid_t pid;
+
+ if (socketpair(AF_UNIX, SOCK_STREAM, 0, sock))
+ return -ENODATA;
+
+ pid = fork();
+ if (pid < 0) {
+ close(sock[0]);
+ close(sock[1]);
+ return -ENODATA;
+ }
+
+ if (pid == 0) {
+ close(sock[0]);
+
+ if (nest && unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ _exit(1);
+
+ fd = sys_fsopen(fsname, 0);
+ if (fd < 0)
+ _exit(1);
+ if (send_fd(sock[1], fd))
+ _exit(1);
+ _exit(0);
+ }
+
+ close(sock[1]);
+ fd = recv_fd(sock[0]);
+ close(sock[0]);
+ wait_for_pid(pid);
+ waitpid(pid, &status, WNOHANG);
+
+ if (fd < 0)
+ return -ENODATA;
+
+ errno = 0;
+ ret = sys_fsconfig(fd, FSCONFIG_CMD_CREATE, NULL, NULL, 0);
+ status = errno;
+ close(fd);
+ errno = status;
+
+ return ret;
+}
+
+FIXTURE(fscontext_ns) {
+ bool warn_before;
+};
+
+FIXTURE_SETUP(fscontext_ns)
+{
+ self->warn_before = taint_warn_set();
+
+ if (setup_userns() != 0)
+ SKIP(return, "setup_userns failed");
+}
+
+FIXTURE_TEARDOWN(fscontext_ns)
+{
+}
+
+/*
+ * The condition the kernel used to WARN about. It has to be refused, and it
+ * has to be refused quietly: an unprivileged task reaches this.
+ */
+FIXTURE_VARIANT(fscontext_ns) {
+ const char *fsname;
+ int expected_errno;
+};
+
+FIXTURE_VARIANT_ADD(fscontext_ns, binfmt_misc) {
+ .fsname = "binfmt_misc",
+ .expected_errno = EINVAL,
+};
+
+FIXTURE_VARIANT_ADD(fscontext_ns, overlay) {
+ .fsname = "overlay",
+ .expected_errno = EIO,
+};
+
+TEST_F(fscontext_ns, create_from_descendant_userns)
+{
+ int ret;
+
+ ret = create_from_child(variant->fsname, true);
+ if (ret == -ENODATA)
+ SKIP(return, "%s unavailable", variant->fsname);
+
+ ASSERT_EQ(-1, ret);
+ ASSERT_EQ(variant->expected_errno, errno);
+
+ /*
+ * Only meaningful if nothing had warned before us. Note that an
+ * unrelated warning racing this test would look like a failure.
+ */
+ if (self->warn_before)
+ TH_LOG("TAINT_WARN already set, not checking for a new warning");
+ else
+ ASSERT_FALSE(taint_warn_set());
+}
+
+/*
+ * The same handover within one user namespace is a supported thing to do and
+ * has to keep working. binfmt_misc takes no options, so the create succeeds
+ * outright and this also shows the test really drives the create path.
+ */
+TEST(create_from_same_userns)
+{
+ int ret;
+
+ if (setup_userns() != 0)
+ SKIP(return, "setup_userns failed");
+
+ ret = create_from_child("binfmt_misc", false);
+ if (ret == -ENODATA)
+ SKIP(return, "binfmt_misc unavailable");
+
+ ASSERT_EQ(0, ret);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/fuse/.gitignore b/tools/testing/selftests/filesystems/fuse/.gitignore
index 3e72e742d08e..fb51603fe419 100644
--- a/tools/testing/selftests/filesystems/fuse/.gitignore
+++ b/tools/testing/selftests/filesystems/fuse/.gitignore
@@ -1,3 +1,4 @@
# SPDX-License-Identifier: GPL-2.0-only
fuse_mnt
fusectl_test
+write_extend_eof_test
diff --git a/tools/testing/selftests/filesystems/fuse/Makefile b/tools/testing/selftests/filesystems/fuse/Makefile
index f47141484275..95a1ee947ca7 100644
--- a/tools/testing/selftests/filesystems/fuse/Makefile
+++ b/tools/testing/selftests/filesystems/fuse/Makefile
@@ -3,6 +3,7 @@
CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
TEST_GEN_PROGS := fusectl_test
+TEST_GEN_PROGS += write_extend_eof_test
TEST_GEN_FILES := fuse_mnt
# fuse_acl_cache_test requires libfuse3; add it only when the library is present.
@@ -14,6 +15,8 @@ endif
include ../../lib.mk
+$(OUTPUT)/write_extend_eof_test: LDLIBS += -lpthread
+
VAR_CFLAGS := $(shell pkg-config fuse --cflags 2>/dev/null)
ifeq ($(VAR_CFLAGS),)
VAR_CFLAGS := -D_FILE_OFFSET_BITS=64 -I/usr/include/fuse
diff --git a/tools/testing/selftests/filesystems/fuse/write_extend_eof_test.c b/tools/testing/selftests/filesystems/fuse/write_extend_eof_test.c
new file mode 100644
index 000000000000..ca6ce6eca382
--- /dev/null
+++ b/tools/testing/selftests/filesystems/fuse/write_extend_eof_test.c
@@ -0,0 +1,368 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Regression test for the fuse write-extend partial-EOF-page zeroing bug.
+ *
+ * A buffered write that extends i_size past a non-page-aligned EOF must zero
+ * the tail of the old last page. If an application has mmap'd that page and
+ * stored into the post-EOF region (undefined until the file grows), the
+ * now-in-bounds tail must read back as zero, not as the stale stored bytes.
+ *
+ * The bug is exposed on a non-writeback_cache server that keeps the page cache
+ * across the write (FOPEN_KEEP_CACHE without FOPEN_DIRECT_IO). This test is a
+ * raw /dev/fuse server in that mode; the backing data is always zero in the
+ * hole, so any non-zero byte a read sees is stale page-cache data.
+ *
+ * Requires root to mount fuse.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <linux/falloc.h>
+#include <pthread.h>
+#include <stdint.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mman.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/uio.h>
+#include <linux/fuse.h>
+
+#include "../../kselftest_harness.h"
+
+#define FUSE_ROOT_ID 1
+#define FILE_INO 2
+#define MAX_WRITE (128 * 1024)
+#define BACKING_SIZE (4 * 1024 * 1024)
+#define POLLUTE 0xee
+
+/* Server-side state, shared with the responder thread. */
+struct server {
+ int fd;
+ unsigned char backing[BACKING_SIZE]; /* authoritative bytes */
+ uint64_t size;
+};
+
+static void reply(int fd, uint64_t unique, int error, void *data, size_t len)
+{
+ struct fuse_out_header oh = {
+ .len = sizeof(oh) + (data ? len : 0),
+ .error = error,
+ .unique = unique,
+ };
+ struct iovec iov[2] = { { &oh, sizeof(oh) }, { data, len } };
+
+ /* Errors here are teardown races (device closed on unmount); ignore. */
+ if (writev(fd, iov, data ? 2 : 1) < 0)
+ return;
+}
+
+static void fill_attr(struct fuse_attr *a, uint64_t ino, uint32_t mode,
+ uint64_t size)
+{
+ memset(a, 0, sizeof(*a));
+ a->ino = ino;
+ a->mode = mode;
+ a->nlink = 1;
+ a->size = size;
+ a->blksize = sysconf(_SC_PAGESIZE);
+}
+
+static void *server_thread(void *arg)
+{
+ struct server *s = arg;
+ static char buf[MAX_WRITE + 4096];
+
+ for (;;) {
+ ssize_t n = read(s->fd, buf, sizeof(buf));
+ struct fuse_in_header *ih = (void *)buf;
+
+ if (n < 0) {
+ if (errno == EINTR || errno == EAGAIN)
+ continue;
+ return NULL; /* device closed on unmount */
+ }
+ if (n < (ssize_t)sizeof(*ih))
+ continue;
+
+ switch (ih->opcode) {
+ case FUSE_INIT: {
+ struct fuse_init_in *in = (void *)(ih + 1);
+ struct fuse_init_out out = {0};
+
+ /* No FUSE_WRITEBACK_CACHE: the exposed configuration. */
+ out.major = FUSE_KERNEL_VERSION;
+ out.minor = FUSE_KERNEL_MINOR_VERSION;
+ out.max_readahead = in->max_readahead;
+ out.max_write = MAX_WRITE;
+ out.max_background = 16;
+ out.congestion_threshold = 12;
+ out.flags = FUSE_MAX_PAGES;
+ out.max_pages = MAX_WRITE / sysconf(_SC_PAGESIZE);
+ reply(s->fd, ih->unique, 0, &out, sizeof(out));
+ break;
+ }
+ case FUSE_GETATTR: {
+ struct fuse_attr_out out = {0};
+ int root = ih->nodeid == FUSE_ROOT_ID;
+
+ out.attr_valid = 3600;
+ fill_attr(&out.attr, ih->nodeid,
+ root ? (S_IFDIR | 0755) : (S_IFREG | 0644),
+ root ? 0 : s->size);
+ reply(s->fd, ih->unique, 0, &out, sizeof(out));
+ break;
+ }
+ case FUSE_LOOKUP: {
+ struct fuse_entry_out out = {0};
+
+ out.nodeid = FILE_INO;
+ out.attr_valid = 3600;
+ out.entry_valid = 3600;
+ fill_attr(&out.attr, FILE_INO, S_IFREG | 0644, s->size);
+ reply(s->fd, ih->unique, 0, &out, sizeof(out));
+ break;
+ }
+ case FUSE_OPEN:
+ case FUSE_OPENDIR: {
+ struct fuse_open_out out = {0};
+
+ /* Keep the cache across the write, but not direct I/O. */
+ out.open_flags = FOPEN_KEEP_CACHE;
+ reply(s->fd, ih->unique, 0, &out, sizeof(out));
+ break;
+ }
+ case FUSE_READ: {
+ struct fuse_read_in *in = (void *)(ih + 1);
+ uint64_t off = in->offset;
+ uint32_t size = in->size;
+
+ if (off >= BACKING_SIZE)
+ size = 0;
+ else if (off + size > BACKING_SIZE)
+ size = BACKING_SIZE - off;
+ reply(s->fd, ih->unique, 0, s->backing + off, size);
+ break;
+ }
+ case FUSE_WRITE: {
+ struct fuse_write_in *in = (void *)(ih + 1);
+ struct fuse_write_out out = {0};
+ uint64_t off = in->offset;
+ uint32_t size = in->size;
+
+ if (off < BACKING_SIZE) {
+ uint32_t c = size;
+
+ if (off + c > BACKING_SIZE)
+ c = BACKING_SIZE - off;
+ memcpy(s->backing + off, in + 1, c);
+ if (off + c > s->size)
+ s->size = off + c;
+ }
+ out.size = size;
+ reply(s->fd, ih->unique, 0, &out, sizeof(out));
+ break;
+ }
+ case FUSE_SETATTR: {
+ struct fuse_setattr_in *in = (void *)(ih + 1);
+ struct fuse_attr_out out = {0};
+
+ if ((in->valid & FATTR_SIZE) && in->size <= BACKING_SIZE) {
+ if (in->size > s->size)
+ memset(s->backing + s->size, 0,
+ in->size - s->size);
+ s->size = in->size;
+ }
+ out.attr_valid = 3600;
+ fill_attr(&out.attr, ih->nodeid, S_IFREG | 0644, s->size);
+ reply(s->fd, ih->unique, 0, &out, sizeof(out));
+ break;
+ }
+ case FUSE_FALLOCATE: {
+ struct fuse_fallocate_in *in = (void *)(ih + 1);
+ uint64_t end = in->offset + in->length;
+
+ /* Only plain (size-extending) fallocate is used here. */
+ if (!(in->mode & FALLOC_FL_KEEP_SIZE) &&
+ end <= BACKING_SIZE && end > s->size) {
+ memset(s->backing + s->size, 0, end - s->size);
+ s->size = end;
+ }
+ reply(s->fd, ih->unique, 0, NULL, 0);
+ break;
+ }
+ case FUSE_FLUSH:
+ case FUSE_RELEASE:
+ case FUSE_RELEASEDIR:
+ case FUSE_FSYNC:
+ case FUSE_ACCESS:
+ reply(s->fd, ih->unique, 0, NULL, 0);
+ break;
+ case FUSE_FORGET:
+ break;
+ default:
+ reply(s->fd, ih->unique, -EOPNOTSUPP, NULL, 0);
+ break;
+ }
+ }
+}
+
+FIXTURE(fuse)
+{
+ struct server *srv;
+ pthread_t thread;
+ char dir[64];
+ long page; /* runtime page size */
+ off_t eof; /* mid-page EOF, page-relative */
+ int fd; /* open test file */
+ char *map; /* mmap of the EOF page */
+ int mounted;
+};
+
+FIXTURE_SETUP(fuse)
+{
+ char opts[128];
+ pthread_t t;
+
+ if (geteuid() != 0)
+ SKIP(return, "need root to mount fuse");
+
+ self->page = sysconf(_SC_PAGESIZE);
+ self->fd = -1;
+ self->map = MAP_FAILED;
+
+ self->srv = mmap(NULL, sizeof(*self->srv), PROT_READ | PROT_WRITE,
+ MAP_SHARED | MAP_ANONYMOUS, -1, 0);
+ ASSERT_NE(MAP_FAILED, self->srv);
+
+ self->srv->fd = open("/dev/fuse", O_RDWR);
+ ASSERT_GE(self->srv->fd, 0);
+
+ strcpy(self->dir, "/tmp/fuse_weof_XXXXXX");
+ ASSERT_NE(NULL, mkdtemp(self->dir));
+
+ snprintf(opts, sizeof(opts),
+ "fd=%d,rootmode=40000,user_id=0,group_id=0",
+ self->srv->fd);
+ ASSERT_EQ(0, mount("fuse", self->dir, "fuse", 0, opts));
+ self->mounted = 1;
+
+ ASSERT_EQ(0, pthread_create(&t, NULL, server_thread, self->srv));
+ self->thread = t;
+}
+
+FIXTURE_TEARDOWN(fuse)
+{
+ if (self->map != MAP_FAILED)
+ munmap(self->map, self->page);
+ if (self->fd >= 0)
+ close(self->fd);
+ if (self->mounted)
+ umount2(self->dir, MNT_DETACH);
+ if (self->srv && self->srv != MAP_FAILED) {
+ if (self->srv->fd > 0)
+ close(self->srv->fd);
+ munmap(self->srv, sizeof(*self->srv));
+ }
+ if (self->dir[0])
+ rmdir(self->dir);
+}
+
+/*
+ * Create the test file with a mid-page EOF and mmap-store POLLUTE into its
+ * post-EOF tail (a legal store, undefined until the file grows). Leaves the
+ * file open and the EOF page mapped in the fixture for the caller to extend.
+ */
+static void pollute_eof_tail(struct __test_metadata *_metadata,
+ FIXTURE_DATA(fuse) * self)
+{
+ off_t eof = 2 * self->page + self->page / 4;
+ char path[128];
+ char *buf;
+
+ snprintf(path, sizeof(path), "%s/file", self->dir);
+ self->fd = open(path, O_RDWR | O_CREAT | O_TRUNC, 0644);
+ ASSERT_GE(self->fd, 0);
+ self->eof = eof;
+
+ buf = malloc(eof);
+ ASSERT_NE(NULL, buf);
+ memset(buf, 'A', eof);
+ ASSERT_EQ(eof, pwrite(self->fd, buf, eof, 0));
+ free(buf);
+
+ self->map = mmap(NULL, self->page, PROT_READ | PROT_WRITE, MAP_SHARED,
+ self->fd, eof & ~(self->page - 1));
+ ASSERT_NE(MAP_FAILED, self->map);
+ memset(self->map + (eof & (self->page - 1)), POLLUTE,
+ self->page - (eof & (self->page - 1)));
+}
+
+/* Assert the old post-EOF tail [eof, end of its page) now reads back as zero. */
+static void assert_tail_zeroed(struct __test_metadata *_metadata,
+ FIXTURE_DATA(fuse) * self)
+{
+ off_t base = self->eof & ~(self->page - 1);
+ char *tail = malloc(self->page);
+ int i;
+
+ ASSERT_NE(NULL, tail);
+ ASSERT_EQ(self->page, pread(self->fd, tail, self->page, base));
+ for (i = self->eof & (self->page - 1); i < self->page; i++)
+ ASSERT_EQ(0, tail[i]);
+ free(tail);
+}
+
+/* Basic: pollute the post-EOF tail, extend past it by a later write. */
+TEST_F(fuse, write_extend)
+{
+ pollute_eof_tail(_metadata, self);
+ ASSERT_EQ(4, pwrite(self->fd, "data", 4, 5 * self->page + self->page / 3));
+ assert_tail_zeroed(_metadata, self);
+}
+
+/* Extend via ftruncate() rather than a write. */
+TEST_F(fuse, ftruncate_extend)
+{
+ pollute_eof_tail(_metadata, self);
+ ASSERT_EQ(0, ftruncate(self->fd, 8 * self->page));
+ assert_tail_zeroed(_metadata, self);
+}
+
+/* Extend via fallocate() starting at the old EOF. */
+TEST_F(fuse, fallocate_extend)
+{
+ pollute_eof_tail(_metadata, self);
+ ASSERT_EQ(0, fallocate(self->fd, 0, self->eof, 4 * self->page));
+ assert_tail_zeroed(_metadata, self);
+}
+
+/* A write landing inside the old EOF page must not clobber its own data. */
+TEST_F(fuse, extend_into_eof_page_preserves_data)
+{
+ off_t base, wr;
+ char *buf, *rd;
+ int i;
+
+ pollute_eof_tail(_metadata, self);
+ base = self->eof & ~(self->page - 1);
+ wr = base + 3 * self->page / 4; /* starts in the EOF page */
+
+ buf = malloc(2 * self->page);
+ ASSERT_NE(NULL, buf);
+ memset(buf, 'B', 2 * self->page);
+ ASSERT_EQ(2 * self->page, pwrite(self->fd, buf, 2 * self->page, wr));
+ free(buf);
+
+ rd = malloc(self->page);
+ ASSERT_NE(NULL, rd);
+ ASSERT_EQ(self->page, pread(self->fd, rd, self->page, base));
+ /* [eof, wr) is hole -> zero; [wr, page) is written data -> 'B'. */
+ for (i = self->eof & (self->page - 1); i < wr - base; i++)
+ ASSERT_EQ(0, rd[i]);
+ for (i = wr - base; i < self->page; i++)
+ ASSERT_EQ('B', rd[i]);
+ free(rd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mntns_cleanup/.gitignore b/tools/testing/selftests/filesystems/mntns_cleanup/.gitignore
new file mode 100644
index 000000000000..493fbcf8d9ec
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mntns_cleanup/.gitignore
@@ -0,0 +1,2 @@
+# SPDX-License-Identifier: GPL-2.0-only
+mntns_cleanup_test
diff --git a/tools/testing/selftests/filesystems/mntns_cleanup/Makefile b/tools/testing/selftests/filesystems/mntns_cleanup/Makefile
new file mode 100644
index 000000000000..0e09e7030a5c
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mntns_cleanup/Makefile
@@ -0,0 +1,6 @@
+# SPDX-License-Identifier: GPL-2.0
+TEST_GEN_PROGS := mntns_cleanup_test
+
+CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
+
+include ../../lib.mk
diff --git a/tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c b/tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c
new file mode 100644
index 000000000000..5209712568b1
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c
@@ -0,0 +1,58 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "../../kselftest_harness.h"
+
+FIXTURE(mntns_cleanup) {
+};
+
+FIXTURE_SETUP(mntns_cleanup)
+{
+ if (geteuid() != 0)
+ SKIP(return, "test requires CAP_SYS_ADMIN");
+
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+ rmdir("/mnt_dir");
+ ASSERT_EQ(mkdir("/mnt_dir", 0755), 0);
+ ASSERT_EQ(mount("tmpfs", "/mnt_dir", "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mkdir("/mnt_dir/hidden", 0755), 0);
+ ASSERT_EQ(mkdir("/mnt_dir/hidden/secret", 0755), 0);
+ ASSERT_EQ(mount("tmpfs", "/mnt_dir/hidden", "tmpfs", 0, NULL), 0);
+}
+
+FIXTURE_TEARDOWN(mntns_cleanup)
+{
+}
+
+/* Mounts must stay connected when a mount namespace is cleaned up. */
+TEST_F(mntns_cleanup, keeps_mounts_connected)
+{
+ int fd, sfd, err;
+
+ fd = open("/mnt_dir", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+
+ /* Destroy the namespace; the fd keeps /mnt_dir alive. */
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+
+ sfd = openat(fd, "hidden/secret", O_RDONLY);
+ err = errno;
+ if (sfd >= 0)
+ close(sfd);
+ close(fd);
+
+ ASSERT_LT(sfd, 0)
+ TH_LOG("mount namespace teardown revealed what the overmount covered");
+ ASSERT_EQ(err, ENOENT);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/overlayfs/.gitignore b/tools/testing/selftests/filesystems/overlayfs/.gitignore
index e23a18c8b37f..077f7a128168 100644
--- a/tools/testing/selftests/filesystems/overlayfs/.gitignore
+++ b/tools/testing/selftests/filesystems/overlayfs/.gitignore
@@ -1,3 +1,4 @@
# SPDX-License-Identifier: GPL-2.0-only
dev_in_maps
set_layers_via_fds
+idmapped_mounts
diff --git a/tools/testing/selftests/filesystems/overlayfs/Makefile b/tools/testing/selftests/filesystems/overlayfs/Makefile
index d3ad4a77db9b..b3185f684add 100644
--- a/tools/testing/selftests/filesystems/overlayfs/Makefile
+++ b/tools/testing/selftests/filesystems/overlayfs/Makefile
@@ -8,7 +8,9 @@ LOCAL_HDRS += ../wrappers.h log.h
TEST_GEN_PROGS := dev_in_maps
TEST_GEN_PROGS += set_layers_via_fds
+TEST_GEN_PROGS += idmapped_mounts
include ../../lib.mk
$(OUTPUT)/set_layers_via_fds: ../utils.c
+$(OUTPUT)/idmapped_mounts: ../utils.c
diff --git a/tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c b/tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c
new file mode 100644
index 000000000000..44a75839f4ed
--- /dev/null
+++ b/tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c
@@ -0,0 +1,501 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+
+#include <fcntl.h>
+#include <limits.h>
+#include <sched.h>
+#include <stdio.h>
+#include <unistd.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+
+#include <linux/mount.h>
+#include <linux/types.h>
+
+#include "kselftest_harness.h"
+#include "../wrappers.h"
+#include "../utils.h"
+
+/*
+ * An idmapping that maps the mount-visible id range [0, ID_RANGE) onto the
+ * host/overlay-final id range [ID_HOST, ID_HOST + ID_RANGE). Through such an
+ * idmapped overlay mount, an overlay-final id of ID_HOST + n is reported as n,
+ * and an id of n requested through the mount is stored as ID_HOST + n.
+ */
+#define ID_NS 0
+#define ID_HOST 10000
+#define ID_RANGE 10000
+
+/*
+ * For the composition test the lower layer's on-disk ids live in a
+ * separate range and are mapped by an idmapped lower layer onto the
+ * overlay-final range [ID_HOST, ID_HOST + ID_RANGE).
+ */
+#define LAYER_HOST 20000
+
+#ifndef MOUNT_ATTR_IDMAP
+#define MOUNT_ATTR_IDMAP 0x00100000
+#endif
+
+#ifndef __NR_mount_setattr
+#define __NR_mount_setattr 442
+#endif
+
+static inline int sys_mount_setattr(int dfd, const char *path,
+ unsigned int flags,
+ struct mount_attr *attr, size_t size)
+{
+ return syscall(__NR_mount_setattr, dfd, path, flags, attr, size);
+}
+
+static bool ovl_supported(void)
+{
+ int fd = sys_fsopen("overlay", 0);
+
+ if (fd < 0)
+ return false;
+ close(fd);
+ return true;
+}
+
+/* base/{l,u,w} owned by ID_HOST so they map to ID_NS through the idmap. */
+static int setup_layers(const char *base)
+{
+ static const char *sub[] = { "", "/l", "/u", "/w" };
+ char path[PATH_MAX];
+
+ for (size_t i = 0; i < ARRAY_SIZE(sub); i++) {
+ snprintf(path, sizeof(path), "%s%s", base, sub[i]);
+ if (mkdir(path, 0755) && errno != EEXIST)
+ return -1;
+ if (i && chown(path, ID_HOST, ID_HOST))
+ return -1;
+ }
+ return 0;
+}
+
+static int ovl_mount(const char *base, bool nfs_export)
+{
+ char lower[PATH_MAX], upper[PATH_MAX], work[PATH_MAX];
+ int fsfd, ovl;
+
+ snprintf(lower, sizeof(lower), "%s/l", base);
+ snprintf(upper, sizeof(upper), "%s/u", base);
+ snprintf(work, sizeof(work), "%s/w", base);
+
+ fsfd = sys_fsopen("overlay", 0);
+ if (fsfd < 0)
+ return -1;
+
+ if (sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", "test", 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "lowerdir", lower, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "upperdir", upper, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "workdir", work, 0))
+ goto err;
+ if (nfs_export &&
+ (sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "index", "on", 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "nfs_export", "on", 0)))
+ goto err;
+ if (sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0))
+ goto err;
+
+ ovl = sys_fsmount(fsfd, 0, 0);
+ close(fsfd);
+ return ovl;
+err:
+ close(fsfd);
+ return -1;
+}
+
+/* Idmap the (still detached, not yet visible) overlay mount @mfd. */
+static int ovl_idmap(int mfd)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_IDMAP,
+ };
+ int ret, userns_fd;
+
+ /*
+ * get_userns_fd(fs_id, mount_id, range): a file whose filesystem id
+ * is fs_id + n is shown through the idmapped mount as mount_id + n.
+ * Here the overlay-final (fs side) range is [ID_HOST, ..) and the
+ * caller-visible (mount side) range is [ID_NS, ..).
+ */
+ userns_fd = get_userns_fd(ID_HOST, ID_NS, ID_RANGE);
+ if (userns_fd < 0)
+ return -1;
+
+ attr.userns_fd = userns_fd;
+ ret = sys_mount_setattr(mfd, "", AT_EMPTY_PATH, &attr, sizeof(attr));
+ close(userns_fd);
+ return ret;
+}
+
+/* Clone @path into a detached, idmapped mount usable as an overlay layer. */
+static int idmapped_layer_fd(const char *path, int nsid, int hostid, int range)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_IDMAP,
+ };
+ int fd_tree, userns_fd;
+
+ fd_tree = sys_open_tree(AT_FDCWD, path,
+ OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ if (fd_tree < 0)
+ return -1;
+ userns_fd = get_userns_fd(nsid, hostid, range);
+ if (userns_fd < 0) {
+ close(fd_tree);
+ return -1;
+ }
+ attr.userns_fd = userns_fd;
+ if (sys_mount_setattr(fd_tree, "", AT_EMPTY_PATH, &attr,
+ sizeof(attr))) {
+ close(userns_fd);
+ close(fd_tree);
+ return -1;
+ }
+ close(userns_fd);
+ return fd_tree;
+}
+
+/* Overlay with a layer passed by fd (idmapped) plus a plain upper/work. */
+static int ovl_mount_lower_fd(const char *upper, const char *work, int fd_lower)
+{
+ int fsfd, ovl;
+
+ fsfd = sys_fsopen("overlay", 0);
+ if (fsfd < 0)
+ return -1;
+
+ if (sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", "test", 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "upperdir", upper, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "workdir", work, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_FD, "lowerdir+", NULL, fd_lower) ||
+ sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0))
+ goto err;
+
+ ovl = sys_fsmount(fsfd, 0, 0);
+ close(fsfd);
+ return ovl;
+err:
+ close(fsfd);
+ return -1;
+}
+
+/*
+ * Mount an overlay inside user namespace @u1 (so the overlay sb's s_user_ns is
+ * not the initial namespace) and idmap that overlay mount with @u2. Runs in a
+ * child that joins @u1; returns 0 on success.
+ */
+static int userns_overlay_child(int u1)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_IDMAP,
+ };
+ struct stat st;
+ int ovl, u2;
+
+ /* Become root in the overlay sb's user namespace u1. */
+ if (!switch_userns(u1, 0, 0, false))
+ return fprintf(stderr, "userns: switch_userns: %m\n"), -1;
+ if (unshare(CLONE_NEWNS) ||
+ sys_mount(NULL, "/", NULL, MS_SLAVE | MS_REC, NULL))
+ return fprintf(stderr, "userns: unshare/slave: %m\n"), -1;
+ if (sys_mount("tmpfs", "/tmp", "tmpfs", 0, NULL))
+ return fprintf(stderr, "userns: mount tmpfs: %m\n"), -1;
+ if (setup_layers("/tmp/ovl"))
+ return fprintf(stderr, "userns: setup_layers: %m\n"), -1;
+ if (mknod("/tmp/ovl/l/file", S_IFREG | 0644, 0) ||
+ chown("/tmp/ovl/l/file", ID_HOST + 5, ID_HOST + 5))
+ return fprintf(stderr, "userns: lower file: %m\n"), -1;
+
+ ovl = ovl_mount("/tmp/ovl", false);
+ if (ovl < 0)
+ return fprintf(stderr, "userns: ovl_mount: %m\n"), -1;
+
+ /*
+ * mount_setattr() requires CAP_SYS_ADMIN over the idmap user
+ * namespace, so it must be a child of u1. Create it now, from
+ * inside u1.
+ */
+ u2 = get_userns_fd(ID_HOST, ID_NS, ID_RANGE);
+ if (u2 < 0)
+ return fprintf(stderr, "userns: get_userns_fd: %m\n"), -1;
+ attr.userns_fd = u2;
+ if (sys_mount_setattr(ovl, "", AT_EMPTY_PATH, &attr, sizeof(attr)))
+ return fprintf(stderr, "userns: mount_setattr: %m\n"), -1;
+ close(u2);
+
+ if (fstatat(ovl, "file", &st, 0))
+ return fprintf(stderr, "userns: fstatat: %m\n"), -1;
+ if (st.st_uid != ID_NS + 5 || st.st_gid != ID_NS + 5) {
+ fprintf(stderr, "userns: got %u:%u expected %u:%u\n",
+ st.st_uid, st.st_gid, ID_NS + 5, ID_NS + 5);
+ return -1;
+ }
+ return 0;
+}
+
+FIXTURE(idmapped_overlay) {
+ char base[64];
+};
+
+FIXTURE_SETUP(idmapped_overlay)
+{
+ /* Private mount namespace so test mounts need no cleanup. */
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(sys_mount(NULL, "/", NULL, MS_SLAVE | MS_REC, NULL), 0);
+
+ /* tmpfs for the layers so we can chown them to arbitrary ids. */
+ ASSERT_EQ(sys_mount("tmpfs", "/tmp", "tmpfs", 0, NULL), 0);
+
+ snprintf(self->base, sizeof(self->base), "/tmp/ovl");
+ ASSERT_EQ(setup_layers(self->base), 0);
+}
+
+FIXTURE_TEARDOWN(idmapped_overlay)
+{
+}
+
+/* A file owned by ID_HOST + 5 is reported as ID_NS + 5 through the idmap. */
+TEST_F(idmapped_overlay, getattr)
+{
+ char path[PATH_MAX];
+ struct stat st;
+ int ovl;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ snprintf(path, sizeof(path), "%s/l/file", self->base);
+ ASSERT_EQ(mknod(path, S_IFREG | 0644, 0), 0);
+ ASSERT_EQ(chown(path, ID_HOST + 5, ID_HOST + 5), 0);
+
+ ovl = ovl_mount(self->base, false);
+ ASSERT_GE(ovl, 0);
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ ASSERT_EQ(fstatat(ovl, "file", &st, 0), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 5);
+ EXPECT_EQ(st.st_gid, ID_NS + 5);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/*
+ * Every creation path initializes the new owner through the mount idmap:
+ * created as caller id ID_NS, stored on the upper layer as overlay-final
+ * ID_HOST. Covers ovl_create() (regular file), ovl_mkdir(), ovl_mknod()
+ * and ovl_symlink() (which share ovl_create_object()), plus the separate
+ * ovl_tmpfile() path.
+ */
+TEST_F(idmapped_overlay, create)
+{
+ static const char *names[] = { "reg", "dir", "fifo", "lnk" };
+ char path[PATH_MAX];
+ struct stat st;
+ int ovl, fd;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ ovl = ovl_mount(self->base, false);
+ ASSERT_GE(ovl, 0);
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ /* One object per creation operation, all as caller id ID_NS. */
+ fd = openat(ovl, "reg", O_CREAT | O_WRONLY | O_EXCL, 0644);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(close(fd), 0);
+ ASSERT_EQ(mkdirat(ovl, "dir", 0755), 0);
+ ASSERT_EQ(mknodat(ovl, "fifo", S_IFIFO | 0644, 0), 0);
+ ASSERT_EQ(symlinkat("target", ovl, "lnk"), 0);
+
+ for (size_t i = 0; i < ARRAY_SIZE(names); i++) {
+ /* Reported as ID_NS through the idmapped mount ... */
+ ASSERT_EQ(fstatat(ovl, names[i], &st, AT_SYMLINK_NOFOLLOW), 0);
+ EXPECT_EQ(st.st_uid, ID_NS);
+ EXPECT_EQ(st.st_gid, ID_NS);
+ /* ... and stored as ID_HOST on the upper layer. */
+ snprintf(path, sizeof(path), "%s/u/%s", self->base, names[i]);
+ ASSERT_EQ(lstat(path, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_HOST);
+ EXPECT_EQ(st.st_gid, ID_HOST);
+ }
+
+ /* O_TMPFILE goes through the separate ovl_tmpfile() path. */
+ fd = openat(ovl, ".", O_TMPFILE | O_WRONLY, 0644);
+ ASSERT_GE(fd, 0);
+ /* Inside the mount: caller id ID_NS. */
+ ASSERT_EQ(fstat(fd, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_NS);
+ EXPECT_EQ(st.st_gid, ID_NS);
+ /* Link it in so the upper backing file can be inspected too. */
+ ASSERT_EQ(linkat(fd, "", ovl, "tmp", AT_EMPTY_PATH), 0);
+ EXPECT_EQ(close(fd), 0);
+ snprintf(path, sizeof(path), "%s/u/tmp", self->base);
+ ASSERT_EQ(lstat(path, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_HOST);
+ EXPECT_EQ(st.st_gid, ID_HOST);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/* chown through the idmapped mount round-trips: ID_NS + 5 <-> ID_HOST + 5. */
+TEST_F(idmapped_overlay, chown)
+{
+ char path[PATH_MAX];
+ struct stat st;
+ int ovl, fd;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ ovl = ovl_mount(self->base, false);
+ ASSERT_GE(ovl, 0);
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ fd = openat(ovl, "f", O_CREAT | O_WRONLY | O_EXCL, 0644);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(close(fd), 0);
+
+ ASSERT_EQ(fchownat(ovl, "f", ID_NS + 5, ID_NS + 5, 0), 0);
+
+ ASSERT_EQ(fstatat(ovl, "f", &st, 0), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 5);
+ EXPECT_EQ(st.st_gid, ID_NS + 5);
+
+ snprintf(path, sizeof(path), "%s/u/f", self->base);
+ ASSERT_EQ(stat(path, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_HOST + 5);
+ EXPECT_EQ(st.st_gid, ID_HOST + 5);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/*
+ * Composition: an idmapped lower layer underneath an idmapped overlay mount.
+ * An on-disk id is mapped by the layer idmap into the overlay-final range and
+ * then by the mount idmap into the caller's range:
+ *
+ * on-disk LAYER_HOST+7 --layer--> ID_HOST+7 --mount--> ID_NS+7
+ */
+TEST_F(idmapped_overlay, composition)
+{
+ char lower[PATH_MAX], upper[PATH_MAX], work[PATH_MAX], path[PATH_MAX];
+ struct stat st;
+ int ovl, fd_lower;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ snprintf(lower, sizeof(lower), "%s/l", self->base);
+ snprintf(upper, sizeof(upper), "%s/u", self->base);
+ snprintf(work, sizeof(work), "%s/w", self->base);
+
+ /* Put the lower layer's ids in the on-disk [LAYER_HOST, ..) range. */
+ ASSERT_EQ(chown(lower, LAYER_HOST, LAYER_HOST), 0);
+ snprintf(path, sizeof(path), "%s/l/file", self->base);
+ ASSERT_EQ(mknod(path, S_IFREG | 0644, 0), 0);
+ ASSERT_EQ(chown(path, LAYER_HOST + 7, LAYER_HOST + 7), 0);
+
+ /* Idmapped lower: on-disk LAYER_HOST <-> overlay-final ID_HOST. */
+ fd_lower = idmapped_layer_fd(lower, LAYER_HOST, ID_HOST, ID_RANGE);
+ ASSERT_GE(fd_lower, 0);
+
+ ovl = ovl_mount_lower_fd(upper, work, fd_lower);
+ ASSERT_GE(ovl, 0);
+ EXPECT_EQ(close(fd_lower), 0);
+
+ /* Idmap the overlay mount: overlay-final ID_HOST <-> caller ID_NS. */
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ ASSERT_EQ(fstatat(ovl, "file", &st, 0), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 7);
+ EXPECT_EQ(st.st_gid, ID_NS + 7);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/* An idmapped overlay mount whose sb lives inside a user namespace. */
+TEST_F(idmapped_overlay, userns)
+{
+ int u1;
+ pid_t pid;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ /* u1 backs the overlay sb: identity-mapped, but not the init ns. */
+ u1 = get_userns_fd(0, 0, 65536);
+ if (u1 < 0)
+ SKIP(return, "user namespaces not available");
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ int ret = userns_overlay_child(u1);
+
+ _exit(ret ? EXIT_FAILURE : EXIT_SUCCESS);
+ }
+ EXPECT_EQ(wait_for_pid(pid), 0);
+
+ EXPECT_EQ(close(u1), 0);
+}
+
+/*
+ * An nfs_export overlay can be idmapped, and decodable file handles round-trip
+ * through the idmapped mount with correctly mapped ownership. Overlay file
+ * handles encode object identity, not ownership, so the mount idmap does not
+ * affect them; it only maps the owner reported once a handle is reopened.
+ */
+TEST_F(idmapped_overlay, nfs_export_handles)
+{
+ char path[PATH_MAX], mnt[128];
+ union {
+ struct file_handle fh;
+ char buf[sizeof(struct file_handle) + MAX_HANDLE_SZ];
+ } fhu;
+ struct file_handle *fh = &fhu.fh;
+ struct stat st;
+ int ovl, mfd, fd, mount_id;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ snprintf(path, sizeof(path), "%s/l/file", self->base);
+ ASSERT_EQ(mknod(path, S_IFREG | 0644, 0), 0);
+ ASSERT_EQ(chown(path, ID_HOST + 7, ID_HOST + 7), 0);
+
+ /* nfs_export=on gives decodable overlay file handles. */
+ ovl = ovl_mount(self->base, true);
+ if (ovl < 0)
+ SKIP(return, "overlayfs nfs_export not supported");
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ /* Attach the idmapped mount so handles can be resolved against it. */
+ snprintf(mnt, sizeof(mnt), "%s/mnt", self->base);
+ ASSERT_EQ(mkdir(mnt, 0755), 0);
+ ASSERT_EQ(sys_move_mount(ovl, "", AT_FDCWD, mnt,
+ MOVE_MOUNT_F_EMPTY_PATH), 0);
+
+ snprintf(path, sizeof(path), "%s/file", mnt);
+ fh->handle_bytes = MAX_HANDLE_SZ;
+ ASSERT_EQ(name_to_handle_at(AT_FDCWD, path, fh, &mount_id, 0), 0);
+
+ mfd = open(mnt, O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(mfd, 0);
+ fd = open_by_handle_at(mfd, fh, O_RDONLY);
+ EXPECT_EQ(close(mfd), 0);
+ ASSERT_GE(fd, 0);
+
+ ASSERT_EQ(fstat(fd, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 7);
+ EXPECT_EQ(st.st_gid, ID_NS + 7);
+
+ EXPECT_EQ(close(fd), 0);
+ EXPECT_EQ(close(ovl), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c b/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c
index 3c0b93183348..7a293544233d 100644
--- a/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c
+++ b/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c
@@ -624,7 +624,7 @@ TEST_F(set_layers_via_fds, set_layers_via_detached_mount_fds)
ASSERT_EQ(sys_move_mount(fd_tmpfs, "", -EBADF, "/set_layers_via_fds_tmpfs", MOVE_MOUNT_F_EMPTY_PATH), 0);
- fd_tmp = open_tree(fd_tmpfs, "u", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ fd_tmp = sys_open_tree(fd_tmpfs, "u", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(fd_tmp, 0);
layer_fds[0] = openat(fd_tmp, "upper", O_CLOEXEC | O_DIRECTORY | O_PATH);
@@ -633,25 +633,25 @@ TEST_F(set_layers_via_fds, set_layers_via_detached_mount_fds)
layer_fds[1] = openat(fd_tmp, "work", O_CLOEXEC | O_DIRECTORY | O_PATH);
ASSERT_GE(layer_fds[1], 0);
- layer_fds[2] = open_tree(fd_tmpfs, "l1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[2] = sys_open_tree(fd_tmpfs, "l1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[2], 0);
- layer_fds[3] = open_tree(fd_tmpfs, "l2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[3] = sys_open_tree(fd_tmpfs, "l2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[3], 0);
- layer_fds[4] = open_tree(fd_tmpfs, "l3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[4] = sys_open_tree(fd_tmpfs, "l3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[4], 0);
- layer_fds[5] = open_tree(fd_tmpfs, "l4", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[5] = sys_open_tree(fd_tmpfs, "l4", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[5], 0);
- layer_fds[6] = open_tree(fd_tmpfs, "d1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[6] = sys_open_tree(fd_tmpfs, "d1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[6], 0);
- layer_fds[7] = open_tree(fd_tmpfs, "d2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[7] = sys_open_tree(fd_tmpfs, "d2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[7], 0);
- layer_fds[8] = open_tree(fd_tmpfs, "d3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[8] = sys_open_tree(fd_tmpfs, "d3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[8], 0);
ASSERT_EQ(close(fd_tmpfs), 0);
diff --git a/tools/testing/selftests/filesystems/statmount/statmount_test.c b/tools/testing/selftests/filesystems/statmount/statmount_test.c
index 8dc018d47a93..60c2c544db6a 100644
--- a/tools/testing/selftests/filesystems/statmount/statmount_test.c
+++ b/tools/testing/selftests/filesystems/statmount/statmount_test.c
@@ -82,6 +82,9 @@ static void cleanup_namespace(void)
{
int ret;
+ if (f_mountinfo)
+ fclose(f_mountinfo);
+
ret = fchdir(orig_root);
if (ret == -1)
ksft_perror("fchdir to original root");
@@ -515,7 +518,7 @@ static void test_statmount_mnt_opts(void)
return;
}
- ksft_test_result_fail("didnt't find mount entry\n");
+ ksft_test_result_fail("didn't find mount entry\n");
free(sm);
free(line);
}
diff --git a/tools/testing/selftests/filesystems/ustat_test.c b/tools/testing/selftests/filesystems/ustat_test.c
new file mode 100644
index 000000000000..d429fd18d779
--- /dev/null
+++ b/tools/testing/selftests/filesystems/ustat_test.c
@@ -0,0 +1,135 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test ustat(2): looking up superblocks by device number.
+ *
+ * ustat() resolves a device number to a mounted superblock via
+ * user_get_super(). Check that the device number of a mounted tmpfs (an
+ * anonymous device) resolves, that it stops resolving once the filesystem
+ * is unmounted and that bogus device numbers report EINVAL.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <unistd.h>
+
+#include "../kselftest_harness.h"
+
+/* struct ustat is not exported through UAPI, mirror include/linux/types.h. */
+struct ustat_buf {
+ int f_tfree;
+ unsigned long f_tinode;
+ char f_fname[6];
+ char f_fpack[6];
+ /* slack in case an architecture lays the struct out differently */
+ char pad[64];
+};
+
+#ifdef __NR_ustat
+
+/*
+ * The kernel decodes @dev with new_decode_dev(), which matches the low 32
+ * bits of the st_dev encoding stat(2) returns for any major below 4096.
+ */
+static int sys_ustat(unsigned int dev, struct ustat_buf *buf)
+{
+ return syscall(__NR_ustat, dev, buf);
+}
+
+static int write_string(const char *path, const char *string)
+{
+ ssize_t len = strlen(string);
+ int fd;
+
+ fd = open(path, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ if (write(fd, string, len) != len) {
+ close(fd);
+ return -1;
+ }
+ return close(fd);
+}
+
+/* Enter namespaces in which mounting a tmpfs instance is allowed. */
+static int setup_namespaces(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[64];
+
+ if (unshare(CLONE_NEWNS | (uid ? CLONE_NEWUSER : 0)))
+ return -1;
+
+ if (uid) {
+ if (write_string("/proc/self/setgroups", "deny"))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_string("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_string("/proc/self/gid_map", map))
+ return -1;
+ }
+
+ return mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+TEST(resolves_mounted_superblock)
+{
+ char dir[] = "/tmp/ustat_test.XXXXXX";
+ struct ustat_buf ub;
+ struct stat st;
+
+ ASSERT_NE(NULL, mkdtemp(dir));
+
+ if (setup_namespaces()) {
+ rmdir(dir);
+ SKIP(return, "cannot set up namespaces: %s", strerror(errno));
+ }
+
+ ASSERT_EQ(0, mount("ustat_test", dir, "tmpfs", 0, NULL));
+ ASSERT_EQ(0, stat(dir, &st));
+
+ memset(&ub, 0xff, sizeof(ub));
+ ASSERT_EQ(0, sys_ustat(st.st_dev, &ub))
+ TH_LOG("ustat(%u): %s", (unsigned int)st.st_dev,
+ strerror(errno));
+
+ ASSERT_EQ(0, umount(dir));
+
+ /* The unmount removed the superblock, the device is gone. */
+ ASSERT_EQ(-1, sys_ustat(st.st_dev, &ub));
+ ASSERT_EQ(EINVAL, errno);
+
+ rmdir(dir);
+}
+
+TEST(bogus_device_numbers)
+{
+ struct ustat_buf ub;
+
+ ASSERT_EQ(-1, sys_ustat(0, &ub));
+ ASSERT_EQ(EINVAL, errno);
+
+ /* major 4095, minor 1048575: nothing plausible lives there */
+ ASSERT_EQ(-1, sys_ustat((0xfffu << 8) | 0xffu | (0xfff00u << 12), &ub));
+ ASSERT_EQ(EINVAL, errno);
+}
+
+#else /* !__NR_ustat */
+
+TEST(unsupported)
+{
+ SKIP(return, "ustat(2) is not available on this architecture");
+}
+
+#endif /* __NR_ustat */
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/ftrace/ftracetest b/tools/testing/selftests/ftrace/ftracetest
index 8ad2c385407e..246d7e1d015c 100755
--- a/tools/testing/selftests/ftrace/ftracetest
+++ b/tools/testing/selftests/ftrace/ftracetest
@@ -7,6 +7,9 @@
# Written by Masami Hiramatsu <masami.hiramatsu.pt@hitachi.com>
#
+# Keep command output parsing stable regardless of the user's locale.
+export LC_ALL=C
+
usage() { # errno [message]
[ ! -z "$2" ] && echo $2
echo "Usage: ftracetest [options] [testcase(s)] [testcase-directory(s)]"
diff --git a/tools/testing/selftests/ftrace/poll.c b/tools/testing/selftests/ftrace/poll.c
index 53258f7515e7..4da86a20dc85 100644
--- a/tools/testing/selftests/ftrace/poll.c
+++ b/tools/testing/selftests/ftrace/poll.c
@@ -66,7 +66,7 @@ int main(int argc, char *argv[])
}
close(pfd.fd);
- /* If timeout happned (ret == 0), exit code is 1 */
+ /* If timeout happened (ret == 0), exit code is 1 */
if (ret == 0)
return 1;
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_uprobe.tc b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_uprobe.tc
index f2048c244526..19430bd5864c 100644
--- a/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_uprobe.tc
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_uprobe.tc
@@ -12,9 +12,32 @@ echo 0 > events/enable
echo > dynamic_events
REALBIN=`readlink -f /bin/sh`
-ENTRYPOINT=`readelf -h ${REALBIN} | grep Entry | sed -e 's/[^0]*//'`
-echo "p:myevent ${REALBIN}:${ENTRYPOINT}" >> uprobe_events
+# Get the entry point virtual address from ELF header
+ENTRY=`readelf -hW ${REALBIN} | grep "Entry point" | awk '{print $NF}'`
+
+# Convert virtual address to file offset: find the LOAD segment containing
+# the entry point, then compute file_offset = e_entry - p_vaddr + p_offset.
+# For PIE binaries this is a no-op (vaddr == file offset), but for non-PIE
+# executables the virtual address is much larger than the file size and
+# must be converted, otherwise uprobe_register() rejects it with -EINVAL.
+ENTRY_DEC=$(printf '%d' "$ENTRY")
+OFFSET=$ENTRY
+while IFS= read -r line; do
+ set -- $line
+ [ "$1" = "LOAD" ] || continue
+ VA_DEC=$(printf '%d' "$3")
+ OFF_DEC=$(printf '%d' "$2")
+ FSZ_DEC=$(printf '%d' "$5")
+ if [ "$ENTRY_DEC" -ge "$VA_DEC" ] && [ "$ENTRY_DEC" -lt "$((VA_DEC + FSZ_DEC))" ]; then
+ OFFSET=$(printf '0x%x' "$((ENTRY_DEC - VA_DEC + OFF_DEC))")
+ break
+ fi
+done << EOF
+$(readelf -lW ${REALBIN} | grep LOAD)
+EOF
+
+echo "p:myevent ${REALBIN}:${OFFSET}" >> uprobe_events
grep -q myevent uprobe_events
test -d events/uprobes/myevent
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/btf_probe_event.tc b/tools/testing/selftests/ftrace/test.d/dynevent/btf_probe_event.tc
new file mode 100644
index 000000000000..bf71368c31a4
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/btf_probe_event.tc
@@ -0,0 +1,51 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: BTF event with typecast and percpu access
+# requires: dynamic_events "this_cpu_read(<fetcharg>)":README "[(structname[,field])]<argname>[->field[->field|.field...]]":README
+
+# Check if the sample module is loaded
+if ! lsmod | grep -q trace_events_sample; then
+ modprobe trace-events-sample || exit_unresolved
+fi
+
+echo 0 > events/enable
+echo > dynamic_events
+
+# The sample_timer_cb(struct timer_list *t) is called.
+# We want to check (STRUCT,FIELD)VAR typecast and this_cpu_read() access.
+# (foo_timer_data,timer)t converts t to struct foo_timer_data * using container_of.
+# data->counter is a per-cpu pointer to int.
+# this_cpu_read(data->counter) should give the value of the counter.
+
+echo 'f:mysample/myevent sample_timer_cb name=(foo_timer_data,timer)t->name:string count=this_cpu_read((foo_timer_data,timer)t->counter)' >> dynamic_events
+
+echo 1 > events/mysample/myevent/enable
+echo 1 > events/sample-trace/foo_timer_fn/enable
+
+sleep 2
+
+echo 0 > events/mysample/myevent/enable
+echo 0 > events/sample-trace/foo_timer_fn/enable
+
+# Compare the values.
+MATCH=0
+while read line; do
+ if echo $line | grep -q "foo_timer_fn:"; then
+ NAME=`echo $line | sed 's/.*name=\([^ ]*\) .*/\1/'`
+ COUNT=`echo $line | sed 's/.*count=\([^ ]*\).*/\1/'`
+ if grep -q "myevent:.*name=\"${NAME}\" count=$COUNT" trace; then
+ MATCH=$((MATCH+1))
+ fi
+ fi
+done < trace
+
+if [ $MATCH -eq 0 ]; then
+ echo "No matching events found"
+ exit_fail
+fi
+
+# Clean up
+echo 0 > events/mysample/myevent/enable
+echo 0 > events/sample-trace/foo_timer_fn/enable
+echo > dynamic_events
+clear_trace
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/btf_typecast_accepted.tc b/tools/testing/selftests/ftrace/test.d/dynevent/btf_typecast_accepted.tc
new file mode 100644
index 000000000000..dd5552727054
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/btf_typecast_accepted.tc
@@ -0,0 +1,103 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: BTF typecast and percpu access syntax validation
+# requires: dynamic_events "this_cpu_read(<fetcharg>)":README "[(structname[,field])]<argname>[->field[->field|.field...]]":README
+
+KPROBES=
+FPROBES=
+
+if grep -qF "p[:[<group>/][<event>]] <place> [<args>]" README ; then
+ KPROBES=yes
+fi
+if grep -qF "f[:[<group>/][<event>]] <func-name>[%return] [<args>]" README ; then
+ FPROBES=yes
+fi
+
+if [ -z "$KPROBES" -a -z "$FPROBES" ] ; then
+ exit_unsupported
+fi
+
+echo 0 > events/enable
+echo > dynamic_events
+
+# Load trace-events-sample module if available to have per-CPU counter structure defined
+if ! lsmod | grep -q trace_events_sample; then
+ modprobe trace-events-sample || exit_unresolved
+fi
+
+if [ "$FPROBES" ] ; then
+ # 1. Test basic typecast on fprobe
+ echo 'f:fpevent1 vfs_read name=(file)file->f_path.dentry->d_name.name:string' >> dynamic_events
+ # 2. Test parenthesized typecast target on fprobe
+ echo 'f:fpevent2 vfs_read name=(file)(file)->f_path.dentry->d_name.name:string' >> dynamic_events
+ # 3. Test nested typecasts on fprobe
+ echo 'f:fpevent3 vfs_read name=(dentry)((file)file->f_path.dentry)->d_name.name:string' >> dynamic_events
+ # 4. Test container_of-style typecast with field option on fprobe
+ echo 'f:fpevent4 vfs_read name=(file,f_path)file->f_mode' >> dynamic_events
+ # 5. Test typecast on return value on fprobe
+ echo 'f:fpevent5 vfs_read%return name=(file)$retval->f_path.dentry->d_name.name:string' >> dynamic_events
+ # 6. Test $current variable support on fprobe
+ echo 'f:fpevent6 vfs_read pid=$current->pid' >> dynamic_events
+ echo 'f:fpevent7 vfs_read pid=(task_struct)$current->pid' >> dynamic_events
+ echo 'f:fpevent8 vfs_read pid=(task_struct,group_leader)$current->pid' >> dynamic_events
+
+ # Test this_cpu_read and this_cpu_ptr on fprobe
+ echo 'f:fpevent9 sample_timer_cb name=(foo_timer_data,timer)t->name:string count=this_cpu_read((foo_timer_data,timer)t->counter)' >> dynamic_events
+ echo 'f:fpevent10 sample_timer_cb ptr=this_cpu_ptr((foo_timer_data,timer)t->counter)' >> dynamic_events
+fi
+
+if [ "$KPROBES" ] ; then
+ # 7. Test basic typecast on kprobe
+ echo 'p:kpevent1 vfs_read name=(file)file->f_path.dentry->d_name.name:string' >> dynamic_events
+ # 8. Test parenthesized typecast target on kprobe
+ echo 'p:kpevent2 vfs_read name=(file)(file)->f_path.dentry->d_name.name:string' >> dynamic_events
+ # 9. Test nested typecasts on kprobe
+ echo 'p:kpevent3 vfs_read name=(dentry)((file)file->f_path.dentry)->d_name.name:string' >> dynamic_events
+ # 10. Test container_of-style typecast with field option on kprobe
+ echo 'p:kpevent4 vfs_read name=(file,f_path)file->f_mode' >> dynamic_events
+ # 11. Test typecast on return value on kretprobe
+ echo 'r:kpevent5 vfs_read name=(file)$retval->f_path.dentry->d_name.name:string' >> dynamic_events
+ # 12. Test $current variable support on kprobe
+ echo 'p:kpevent6 vfs_read pid=$current->pid' >> dynamic_events
+ echo 'p:kpevent7 vfs_read pid=(task_struct)$current->pid' >> dynamic_events
+ echo 'p:kpevent8 vfs_read pid=(task_struct,group_leader)$current->pid' >> dynamic_events
+
+ # Test this_cpu_read and this_cpu_ptr on kprobe
+ echo 'p:kpevent9 sample_timer_cb name=(foo_timer_data,timer)t->name:string count=this_cpu_read((foo_timer_data,timer)t->counter)' >> dynamic_events
+ echo 'p:kpevent10 sample_timer_cb ptr=this_cpu_ptr((foo_timer_data,timer)t->counter)' >> dynamic_events
+fi
+
+# Verify the events exist in dynamic_events
+if [ "$FPROBES" ] ; then
+ grep -q "fpevent1 " dynamic_events
+ grep -q "fpevent2 " dynamic_events
+ grep -q "fpevent3 " dynamic_events
+ grep -q "fpevent4 " dynamic_events
+ grep -q "fpevent5 " dynamic_events
+ grep -q "fpevent6 " dynamic_events
+ grep -q "fpevent7 " dynamic_events
+ grep -q "fpevent8 " dynamic_events
+ if lsmod | grep -q trace_events_sample; then
+ grep -q "fpevent9 " dynamic_events
+ grep -q "fpevent10 " dynamic_events
+ fi
+fi
+
+if [ "$KPROBES" ] ; then
+ grep -q "kpevent1 " dynamic_events
+ grep -q "kpevent2 " dynamic_events
+ grep -q "kpevent3 " dynamic_events
+ grep -q "kpevent4 " dynamic_events
+ grep -q "kpevent5 " dynamic_events
+ grep -q "kpevent6 " dynamic_events
+ grep -q "kpevent7 " dynamic_events
+ grep -q "kpevent8 " dynamic_events
+ if lsmod | grep -q trace_events_sample; then
+ grep -q "kpevent9 " dynamic_events
+ grep -q "kpevent10 " dynamic_events
+ fi
+fi
+
+# Clean up
+echo > dynamic_events
+clear_trace
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/eprobes_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/dynevent/eprobes_syntax_errors.tc
index 2a680c086047..c2e3f9d19f13 100644
--- a/tools/testing/selftests/ftrace/test.d/dynevent/eprobes_syntax_errors.tc
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/eprobes_syntax_errors.tc
@@ -10,7 +10,7 @@ check_error() { # command-with-error-pos-by-^
check_error 'e ^a.' # NO_EVENT_INFO
check_error 'e ^.b' # NO_EVENT_INFO
check_error 'e ^a.b' # BAD_ATTACH_EVENT
-check_error 'e syscalls/sys_enter_openat ^foo' # BAD_ATTACH_ARG
+check_error 'e syscalls/sys_enter_openat ^foo' # NO_EVENT_FIELD
check_error 'e:^/bar syscalls/sys_enter_openat' # NO_GROUP_NAME
check_error 'e:^12345678901234567890123456789012345678901234567890123456789012345/bar syscalls/sys_enter_openat' # GROUP_TOO_LONG
@@ -19,11 +19,19 @@ check_error 'e:^ syscalls/sys_enter_openat' # NO_EVENT_NAME
check_error 'e:foo/^12345678901234567890123456789012345678901234567890123456789012345 syscalls/sys_enter_openat' # EVENT_TOO_LONG
check_error 'e:foo/^bar.1 syscalls/sys_enter_openat' # BAD_EVENT_NAME
-check_error 'e:foo/bar syscalls/sys_enter_openat arg=^dfd' # BAD_FETCH_ARG
check_error 'e:foo/bar syscalls/sys_enter_openat arg=^$foo' # BAD_ATTACH_ARG
+check_error 'e:foo/bar syscalls/sys_enter_openat arg=^COMM' # NO_EVENT_FIELD
+if grep -q "\$current.*" README; then
+ check_error 'e:foo/bar syscalls/sys_enter_openat arg=^current' # NO_EVENT_FIELD
+fi
+
if grep -q '<attached-group>\.<attached-event>.*\[if <filter>\]' README; then
check_error 'e:foo/bar syscalls/sys_enter_openat if ^' # NO_EP_FILTER
fi
+if grep -q 'this_cpu_read(<fetcharg>)' README; then
+ check_error 'e:foo/bar syscalls/sys_enter_openat arg=^this_cpu_read(file)' # NOSUP_PERCPU
+fi
+
exit 0
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/fprobe_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/dynevent/fprobe_syntax_errors.tc
index fee479295e2f..384209968325 100644
--- a/tools/testing/selftests/ftrace/test.d/dynevent/fprobe_syntax_errors.tc
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/fprobe_syntax_errors.tc
@@ -60,7 +60,7 @@ check_error 'f vfs_read ^&1' # BAD_FETCH_ARG
# We've introduced this limitation with array support
if grep -q ' <type>\\\[<array-size>\\\]' README; then
-check_error 'f vfs_read +0(^+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(@0))))))))))))))' # TOO_MANY_OPS?
+check_error 'f vfs_read +0(+0(+0(+0(+0(+0(+0(+0(^+0(@0)))))))))' # TOO_MANY_NESTED
check_error 'f vfs_read +0(@11):u8[10^' # ARRAY_NO_CLOSE
check_error 'f vfs_read +0(@11):u8[10]^a' # BAD_ARRAY_SUFFIX
check_error 'f vfs_read +0(@11):u8[^10a]' # BAD_ARRAY_NUM
@@ -75,7 +75,7 @@ check_error 'f vfs_read ^arg123456789012345678901234567890=@11' # ARG_NAME_TOO_L
check_error 'f vfs_read ^=@11' # NO_ARG_NAME
check_error 'f vfs_read ^var.1=@11' # BAD_ARG_NAME
check_error 'f vfs_read var1=@11 ^var1=@12' # USED_ARG_NAME
-check_error 'f vfs_read ^+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(@1234))))))' # ARG_TOO_LONG
+check_error 'f vfs_read ^+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(@1234))))))))))))))))))))))))))' # ARG_TOO_LONG
check_error 'f vfs_read arg1=^' # NO_ARG_BODY
@@ -112,6 +112,18 @@ check_error 'f vfs_read%return $retval->^foo' # NO_PTR_STRCT
check_error 'f vfs_read file->^foo' # NO_BTF_FIELD
check_error 'f vfs_read file^-.foo' # BAD_HYPHEN
check_error 'f vfs_read ^file:string' # BAD_TYPE4STR
+if grep -qF "[(structname" README ; then
+check_error 'f vfs_read arg1=(task_struct)file^' # TYPECAST_REQ_FIELD
+check_error 'f vfs_read arg1=(a)((b)((c)((d)((e)((f)((g)((h)(^(i)file->i)->h)->g)->f)->e)->d)->c)->b)->a' # TOO_MANY_NESTED
+check_error 'f vfs_read arg1=(task_struct,^in_execve)file->comm' # TYPECAST_NOT_ALIGNED
+check_error 'f vfs_read arg1=(task_struct,^foo_bar)file->pid' # NO_BTF_FIELD
+check_error 'f vfs_read arg1=(^task_struct1234)file->pid' # NO_PTR_STRCT
+check_error 'f vfs_read arg1=(task_struct,se^->group_node)file->comm' # TYPECAST_BAD_ARROW
+check_error 'f vfs_read arg1=(task_struct,^->pid)file->comm' # NO_BTF_FIELD
+check_error 'f vfs_read arg1=(task_struct,^.pid)file->comm' # NO_BTF_FIELD
+check_error 'f vfs_read arg1=(task_struct,^.)file->comm' # NO_BTF_FIELD
+check_error 'f vfs_read arg1=(task_struct)^@symbol+10->comm' # TYPECAST_SYM_OFFSET
+fi
fi
else
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/tprobe_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/dynevent/tprobe_syntax_errors.tc
index ffe8ffef4027..72b8652df9ba 100644
--- a/tools/testing/selftests/ftrace/test.d/dynevent/tprobe_syntax_errors.tc
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/tprobe_syntax_errors.tc
@@ -46,7 +46,7 @@ check_error 't kfree ^&1' # BAD_FETCH_ARG
# We've introduced this limitation with array support
if grep -q ' <type>\\\[<array-size>\\\]' README; then
-check_error 't kfree +0(^+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(@0))))))))))))))' # TOO_MANY_OPS?
+check_error 't kfree +0(+0(+0(+0(+0(+0(+0(+0(^+0(@0)))))))))' # TOO_MANY_NESTED
check_error 't kfree +0(@11):u8[10^' # ARRAY_NO_CLOSE
check_error 't kfree +0(@11):u8[10]^a' # BAD_ARRAY_SUFFIX
check_error 't kfree +0(@11):u8[^10a]' # BAD_ARRAY_NUM
@@ -61,7 +61,7 @@ check_error 't kfree ^arg123456789012345678901234567890=@11' # ARG_NAME_TOO_LOG
check_error 't kfree ^=@11' # NO_ARG_NAME
check_error 't kfree ^var.1=@11' # BAD_ARG_NAME
check_error 't kfree var1=@11 ^var1=@12' # USED_ARG_NAME
-check_error 't kfree ^+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(@1234))))))' # ARG_TOO_LONG
+check_error 't kfree ^+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(@1234))))))))))))))))))))))))))' # ARG_TOO_LONG
check_error 't kfree arg1=^' # NO_ARG_BODY
diff --git a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_module.tc b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_module.tc
index 7e74ee11edf9..4b32e1b9a8d3 100644
--- a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_module.tc
+++ b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_module.tc
@@ -13,6 +13,12 @@ fi
MOD=trace_printk
FUNC=trace_printk_irq_work
+:;: "Add an event on a module function without module name" ;:
+
+echo "p:event0 $FUNC" > kprobe_events
+test -d events/kprobes/event0 || exit_failure
+echo "-:kprobes/event0" >> kprobe_events
+
:;: "Add an event on a module function without specifying event name" ;:
echo "p $MOD:$FUNC" > kprobe_events
diff --git a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_syntax_errors.tc
index 8f1c58f0c239..b0e6b80ccb01 100644
--- a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_syntax_errors.tc
+++ b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_syntax_errors.tc
@@ -56,7 +56,7 @@ check_error 'p vfs_read ^&1' # BAD_FETCH_ARG
# We've introduced this limitation with array support
if grep -q ' <type>\\\[<array-size>\\\]' README; then
-check_error 'p vfs_read +0(^+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(+0(@0))))))))))))))' # TOO_MANY_OPS?
+check_error 'p vfs_read +0(+0(+0(+0(+0(+0(+0(+0(^+0(@0)))))))))' # TOO_MANY_NESTED
check_error 'p vfs_read +0(@11):u8[10^' # ARRAY_NO_CLOSE
check_error 'p vfs_read +0(@11):u8[10]^a' # BAD_ARRAY_SUFFIX
check_error 'p vfs_read +0(@11):u8[^10a]' # BAD_ARRAY_NUM
@@ -71,7 +71,7 @@ check_error 'p vfs_read ^arg123456789012345678901234567890=@11' # ARG_NAME_TOO_L
check_error 'p vfs_read ^=@11' # NO_ARG_NAME
check_error 'p vfs_read ^var.1=@11' # BAD_ARG_NAME
check_error 'p vfs_read var1=@11 ^var1=@12' # USED_ARG_NAME
-check_error 'p vfs_read ^+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(@1234))))))' # ARG_TOO_LONG
+check_error 'p vfs_read ^+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(+1234567(@1234))))))))))))))))))))))))))' # ARG_TOO_LONG
check_error 'p vfs_read arg1=^' # NO_ARG_BODY
# instruction boundary check is valid on x86 (at this moment)
@@ -115,6 +115,18 @@ check_error 'p vfs_read+20 ^$arg*' # NOFENTRY_ARGS
check_error 'p vfs_read ^hoge' # NO_BTFARG
check_error 'p kfree ^$arg10' # NO_BTFARG (exceed the number of parameters)
check_error 'r kfree ^$retval' # NO_RETVAL
+if grep -qF "[(structname" README ; then
+check_error 'p vfs_read arg1=(task_struct)file^' # TYPECAST_REQ_FIELD
+check_error 'p vfs_read arg1=(a)((b)((c)((d)((e)((f)((g)((h)(^(i)file->i)->h)->g)->f)->e)->d)->c)->b)->a' # TOO_MANY_NESTED
+check_error 'p vfs_read arg1=(task_struct,^in_execve)file->comm' # TYPECAST_NOT_ALIGNED
+check_error 'p vfs_read arg1=(task_struct,^foo_bar)file->pid' # NO_BTF_FIELD
+check_error 'p vfs_read arg1=(^task_struct1234)file->pid' # NO_PTR_STRCT
+check_error 'p vfs_read arg1=(task_struct,se^->group_node)file->comm' # TYPECAST_BAD_ARROW
+check_error 'p vfs_read arg1=(task_struct,^->pid)file->comm' # NO_BTF_FIELD
+check_error 'p vfs_read arg1=(task_struct,^.pid)file->comm' # NO_BTF_FIELD
+check_error 'p vfs_read arg1=(task_struct,^.)file->comm' # NO_BTF_FIELD
+check_error 'p vfs_read arg1=(task_struct)^@symbol+10->comm' # TYPECAST_SYM_OFFSET
+fi
else
check_error 'p vfs_read ^$arg*' # NOSUP_BTFARG
fi
diff --git a/tools/testing/selftests/ftrace/test.d/kprobe/uprobe_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/kprobe/uprobe_syntax_errors.tc
index c817158b99db..e12dc967ec76 100644
--- a/tools/testing/selftests/ftrace/test.d/kprobe/uprobe_syntax_errors.tc
+++ b/tools/testing/selftests/ftrace/test.d/kprobe/uprobe_syntax_errors.tc
@@ -28,4 +28,9 @@ if grep -q ".*symstr.*" README; then
check_error 'p /bin/sh:10 $stack0:^symstr' # BAD_TYPE
fi
+# $current is not supported by uprobe
+if grep -q "\$current.*" README; then
+check_error 'p /bin/sh:10 ^$current:u8' # BAD_VAR
+fi
+
exit 0
diff --git a/tools/testing/selftests/futex/functional/.gitignore b/tools/testing/selftests/futex/functional/.gitignore
index 23b9fea8d190..7c39d10b38e4 100644
--- a/tools/testing/selftests/futex/functional/.gitignore
+++ b/tools/testing/selftests/futex/functional/.gitignore
@@ -1,4 +1,5 @@
# SPDX-License-Identifier: GPL-2.0-only
+futex_lock_pi_exiting
futex_numa_mpol
futex_priv_hash
futex_requeue
diff --git a/tools/testing/selftests/futex/functional/Makefile b/tools/testing/selftests/futex/functional/Makefile
index 5c1c824f9740..a03bd5acba50 100644
--- a/tools/testing/selftests/futex/functional/Makefile
+++ b/tools/testing/selftests/futex/functional/Makefile
@@ -11,7 +11,8 @@ endif
LOCAL_HDRS := \
../include/futextest.h \
- ../include/atomic.h
+ ../include/atomic.h \
+ ../include/futex_thread.h
TEST_GEN_PROGS := \
futex_wait_timeout \
futex_wait_wouldblock \
@@ -26,7 +27,8 @@ TEST_GEN_PROGS := \
futex_numa_mpol \
futex_waitv \
futex_numa \
- robust_list
+ robust_list \
+ futex_lock_pi_exiting
TEST_PROGS := run.sh
diff --git a/tools/testing/selftests/futex/functional/futex_lock_pi_exiting.c b/tools/testing/selftests/futex/functional/futex_lock_pi_exiting.c
new file mode 100644
index 000000000000..623c5f3a1836
--- /dev/null
+++ b/tools/testing/selftests/futex/functional/futex_lock_pi_exiting.c
@@ -0,0 +1,263 @@
+// SPDX-License-Identifier: GPL-2.0-or-later
+/******************************************************************************
+ *
+ * futex_lock_pi_exiting.c
+ *
+ * Coverage for the FUTEX_LOCK_PI owner-exiting path. futex_wait_timeout.c
+ * already covers FUTEX_LOCK_PI timeout semantics and robust_list.c covers
+ * owner death via the robust list, but nothing exercises FUTEX_LOCK_PI when a
+ * non-robust PI owner exits while holding the lock, nor the basic ownership /
+ * EDEADLK / unlock word semantics.
+ *
+ * DESCRIPTION
+ * Three tests:
+ *
+ * 1. lock_unlock_basic - uncontended FUTEX_LOCK_PI semantics: the futex
+ * word carries the owner TID, a recursive lock by the owner returns
+ * EDEADLK, and FUTEX_UNLOCK_PI clears the word.
+ *
+ * 2. owner_dies_with_blocked_waiter - a thread acquires a PI futex and
+ * exits while holding it. do_exit() runs futex_cleanup_begin() (which
+ * flips the task's futex state to FUTEX_STATE_EXITING) and
+ * exit_pi_state_list() (which hands off / tears down the pi_state). A
+ * contending FUTEX_LOCK_PI waiter must end up in one of:
+ *
+ * 0 - ownership was transferred to / acquired by the waiter
+ * EOWNERDEAD - previous owner died holding the lock; the caller is
+ * now the owner and must acknowledge by unlocking
+ * ESRCH - the owner encoded in the futex word is already gone
+ *
+ * and on the first two it must actually own the lock afterwards.
+ *
+ * 3. stress_owner_exits - hammer that same exiting-owner path. This is
+ * where the following bug lived: the 'exiting' task pointer was not
+ * reset at the retry label, so after wait_for_owner_exiting() dropped
+ * its reference a subsequent retry that returned a non-EBUSY error fed
+ * the stale pointer back in and tripped WARN_ON_ONCE(exiting). That
+ * warning is invisible to user space, so this test cannot observe it
+ * through a syscall return value; it only becomes a visible failure
+ * (crash) on a kernel booted with panic_on_warn=1 (or built with
+ * CONFIG_BUG_ON_DATA_CORRUPTION). The loop drives the path so that
+ * such a kernel trips on it - the canonical way fuzz/CI catch these.
+ *
+ * Fix: 210d36d892de ("futex: Clear stale exiting pointer in
+ * futex_lock_pi() retry path")
+ * Fixes: 3ef240eaff36 ("futex: Prevent exit livelock")
+ *
+ * AUTHOR
+ * Based on futex test boilerplate by Darren Hart <dvhart@linux.intel.com>
+ *
+ *****************************************************************************/
+
+#define _GNU_SOURCE
+
+#include <errno.h>
+#include <pthread.h>
+#include <stdint.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/syscall.h>
+
+#include "futextest.h"
+#include "kselftest_harness.h"
+
+/*
+ * Iterations for the stress variant. Enough to repeatedly land in the narrow
+ * EXITING window while keeping the test fast.
+ */
+#define STRESS_ITERS 1000
+
+static futex_t pi_futex;
+static pthread_barrier_t locked_barrier;
+static pthread_barrier_t release_barrier;
+
+static pid_t sys_gettid(void)
+{
+ return syscall(SYS_gettid);
+}
+
+/*
+ * Owner thread: acquire the PI futex and exit while still holding it. Two
+ * modes:
+ * park == 0: signal that we hold the lock, then exit immediately (racy; the
+ * waiter races against our exit path).
+ * park == 1: signal that we hold the lock and keep holding until released
+ * via release_barrier, so a waiter has time to contend as a real
+ * PI waiter before we die.
+ */
+static void *owner_thread(void *arg)
+{
+ long park = (long)arg;
+
+ if (futex_lock_pi(&pi_futex, NULL, 0, FUTEX_PRIVATE_FLAG) != 0)
+ return (void *)(intptr_t)-errno;
+
+ pthread_barrier_wait(&locked_barrier);
+
+ if (park)
+ pthread_barrier_wait(&release_barrier);
+
+ /* Die while still holding the lock. */
+ pthread_exit((void *)0);
+}
+
+/*
+ * Block on the PI futex as a waiter. Returns 0 on acquisition, otherwise the
+ * positive errno.
+ */
+static int waiter_lock_pi(void)
+{
+ int ret = futex_lock_pi(&pi_futex, NULL, 0, FUTEX_PRIVATE_FLAG);
+
+ return ret == 0 ? 0 : errno;
+}
+
+static int outcome_ok(int outcome)
+{
+ return outcome == 0 || outcome == EOWNERDEAD || outcome == ESRCH;
+}
+
+/* Results published by waiter_thread() for the owning thread to assert on. */
+static int waiter_outcome;
+static int waiter_owns;
+
+/*
+ * Waiter thread for the blocked-waiter test. Contends for the lock and, when
+ * it acquires, records whether the futex word actually carries its TID and
+ * releases the lock itself (FUTEX_UNLOCK_PI must run in the owning thread).
+ */
+static void *waiter_thread(void *arg)
+{
+ pid_t tid = sys_gettid();
+
+ waiter_outcome = waiter_lock_pi();
+ if (waiter_outcome == 0 || waiter_outcome == EOWNERDEAD) {
+ waiter_owns = (pi_futex & FUTEX_TID_MASK) == (futex_t)tid;
+ futex_unlock_pi(&pi_futex, FUTEX_PRIVATE_FLAG);
+ }
+ return NULL;
+}
+
+FIXTURE(lock_pi_exiting) {
+};
+
+FIXTURE_SETUP(lock_pi_exiting) {
+}
+
+FIXTURE_TEARDOWN(lock_pi_exiting) {
+}
+
+/*
+ * Uncontended FUTEX_LOCK_PI semantics, fully deterministic.
+ */
+TEST_F(lock_pi_exiting, lock_unlock_basic)
+{
+ pid_t tid = sys_gettid();
+ int ret;
+
+ pi_futex = FUTEX_INITIALIZER;
+
+ /* Acquire: we become the owner, our TID lands in the futex word. */
+ ret = futex_lock_pi(&pi_futex, NULL, 0, FUTEX_PRIVATE_FLAG);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("lock failed: errno=%d (%s)", errno, strerror(errno));
+ ASSERT_EQ(pi_futex & FUTEX_TID_MASK, (futex_t)tid)
+ TH_LOG("owner TID not in futex word: 0x%08x", pi_futex);
+
+ /* A recursive lock by the owner must be refused, not deadlock. */
+ errno = 0;
+ ret = futex_lock_pi(&pi_futex, NULL, 0, FUTEX_PRIVATE_FLAG);
+ ASSERT_EQ(ret, -1);
+ ASSERT_EQ(errno, EDEADLK)
+ TH_LOG("recursive lock: expected EDEADLK, got errno=%d", errno);
+
+ /* Release: the futex word is handed back clean. */
+ ret = futex_unlock_pi(&pi_futex, FUTEX_PRIVATE_FLAG);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("unlock failed: errno=%d", errno);
+ ASSERT_EQ(pi_futex, (futex_t)0)
+ TH_LOG("futex word not cleared after unlock: 0x%08x", pi_futex);
+}
+
+/*
+ * A PI waiter inherits the lock when the owner dies holding it.
+ *
+ * The owner parks while holding the lock, this thread contends for it, then
+ * the owner exits. The waiter must come out cleanly (no hang, no unexpected
+ * error) and, when it acquires, must actually own the lock.
+ */
+TEST_F(lock_pi_exiting, owner_dies_with_blocked_waiter)
+{
+ pthread_t owner, waiter;
+
+ pthread_barrier_init(&locked_barrier, NULL, 2);
+ pthread_barrier_init(&release_barrier, NULL, 2);
+ pi_futex = FUTEX_INITIALIZER;
+ waiter_outcome = -1;
+ waiter_owns = 0;
+
+ ASSERT_EQ(pthread_create(&owner, NULL, owner_thread, (void *)1), 0);
+
+ /* Wait until the owner actually holds the lock. */
+ pthread_barrier_wait(&locked_barrier);
+
+ /* Start the waiter and give it time to block as a real PI waiter. */
+ ASSERT_EQ(pthread_create(&waiter, NULL, waiter_thread, NULL), 0);
+ usleep(1000);
+
+ /* Release the owner so it dies while the waiter is queued on it. */
+ pthread_barrier_wait(&release_barrier);
+
+ pthread_join(waiter, NULL);
+ pthread_join(owner, NULL);
+
+ ASSERT_TRUE(outcome_ok(waiter_outcome)) {
+ TH_LOG("unexpected FUTEX_LOCK_PI outcome: %d (%s)",
+ waiter_outcome, strerror(waiter_outcome));
+ }
+ if (waiter_outcome == 0 || waiter_outcome == EOWNERDEAD) {
+ ASSERT_TRUE(waiter_owns)
+ TH_LOG("waiter acquired but futex word lacks its TID");
+ }
+
+ pthread_barrier_destroy(&locked_barrier);
+ pthread_barrier_destroy(&release_barrier);
+}
+
+/*
+ * Stress: repeatedly let an owner exit while a waiter contends for the lock.
+ *
+ * Each iteration drives the FUTEX_STATE_EXITING -> -EBUSY -> retry path that
+ * the stale-'exiting'-pointer bug lived on (210d36d892de). The warning it
+ * fixed is invisible to user space, so on a normally-configured kernel both
+ * the buggy and fixed kernels pass here; the point is to make a kernel booted
+ * with panic_on_warn=1 trip during one of these iterations.
+ */
+TEST_F(lock_pi_exiting, stress_owner_exits)
+{
+ for (int i = 0; i < STRESS_ITERS; i++) {
+ pthread_t owner;
+ int outcome;
+
+ pthread_barrier_init(&locked_barrier, NULL, 2);
+ pi_futex = FUTEX_INITIALIZER;
+
+ ASSERT_EQ(pthread_create(&owner, NULL, owner_thread, (void *)0), 0);
+
+ /* Owner holds the lock; race FUTEX_LOCK_PI against its exit. */
+ pthread_barrier_wait(&locked_barrier);
+
+ outcome = waiter_lock_pi();
+ ASSERT_TRUE(outcome_ok(outcome)) {
+ TH_LOG("iter %d: unexpected outcome %d (%s)",
+ i, outcome, strerror(outcome));
+ }
+ if (outcome == 0 || outcome == EOWNERDEAD)
+ futex_unlock_pi(&pi_futex, FUTEX_PRIVATE_FLAG);
+
+ pthread_join(owner, NULL);
+ pthread_barrier_destroy(&locked_barrier);
+ }
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_numa_mpol.c b/tools/testing/selftests/futex/functional/futex_numa_mpol.c
index 78c0f7a59e17..4ffcf41efe1f 100644
--- a/tools/testing/selftests/futex/functional/futex_numa_mpol.c
+++ b/tools/testing/selftests/futex/functional/futex_numa_mpol.c
@@ -9,6 +9,7 @@
#include <pthread.h>
#include <stdio.h>
#include <stdlib.h>
+#include <string.h>
#include <unistd.h>
#ifdef LIBNUMA_VER_SUFFICIENT
#include <numa.h>
@@ -28,9 +29,9 @@ static pthread_barrier_t barrier_main;
static pthread_t threads[MAX_THREADS];
struct thread_args {
- void *futex_ptr;
- unsigned int flags;
- int result;
+ void *futex_ptr;
+ unsigned int flags;
+ int result;
};
static struct thread_args thread_args[MAX_THREADS];
@@ -54,7 +55,7 @@ static void *thread_lock_fn(void *arg)
return NULL;
}
-static void create_max_threads(void *futex_ptr)
+static void create_max_threads(struct __test_metadata *_metadata, void *futex_ptr)
{
int i, ret;
@@ -63,28 +64,29 @@ static void create_max_threads(void *futex_ptr)
thread_args[i].flags = FUTEX2_SIZE_U32 | FUTEX_PRIVATE_FLAG | FUTEX2_NUMA;
thread_args[i].result = 0;
ret = pthread_create(&threads[i], NULL, thread_lock_fn, &thread_args[i]);
- if (ret)
- ksft_exit_fail_msg("pthread_create failed\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_create failed");
}
}
-static void join_max_threads(void)
+static void join_max_threads(struct __test_metadata *_metadata)
{
int i, ret;
for (i = 0; i < MAX_THREADS; i++) {
ret = pthread_join(threads[i], NULL);
- if (ret)
- ksft_exit_fail_msg("pthread_join failed for thread %d\n", i);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_join failed for thread %d", i);
}
}
-static void __test_futex(void *futex_ptr, int err_value, unsigned int futex_flags)
+static void __test_futex(struct __test_metadata *_metadata, void *futex_ptr, int err_value,
+ unsigned int futex_flags)
{
- int to_wake, ret, i, need_exit = 0;
+ int to_wake, ret, i;
pthread_barrier_init(&barrier_main, NULL, MAX_THREADS + 1);
- create_max_threads(futex_ptr);
+ create_max_threads(_metadata, futex_ptr);
pthread_barrier_wait(&barrier_main);
to_wake = MAX_THREADS;
@@ -92,45 +94,50 @@ static void __test_futex(void *futex_ptr, int err_value, unsigned int futex_flag
ret = futex2_wake(futex_ptr, to_wake, futex_flags);
if (err_value) {
- if (ret >= 0)
- ksft_exit_fail_msg("futex2_wake(%d, 0x%x) should fail, but didn't\n",
- to_wake, futex_flags);
+ EXPECT_LT(ret, 0) {
+ TH_LOG("futex2_wake(%d, 0x%x) should fail, but didn't",
+ to_wake, futex_flags);
+ }
- if (errno != err_value)
- ksft_exit_fail_msg("futex2_wake(%d, 0x%x) expected error was %d, but returned %d (%s)\n",
- to_wake, futex_flags, err_value, errno, strerror(errno));
+ EXPECT_EQ(errno, err_value) {
+ TH_LOG("futex2_wake(%d, 0x%x) expected error was %d, but returned %d (%s)",
+ to_wake, futex_flags, err_value, errno, strerror(errno));
+ }
break;
}
if (ret < 0) {
- ksft_exit_fail_msg("Failed futex2_wake(%d, 0x%x): %m\n",
- to_wake, futex_flags);
+ if (errno == ENOSYS || (errno == EINVAL && (futex_flags & FUTEX2_NUMA)))
+ SKIP(return, "futex2 or FUTEX2_NUMA not supported by kernel");
+
+ ASSERT_GE(ret, 0) {
+ TH_LOG("Failed futex2_wake(%d, 0x%x): %s",
+ to_wake, futex_flags, strerror(errno));
+ }
}
if (!ret)
usleep(50);
to_wake -= ret;
} while (to_wake);
- join_max_threads();
+ join_max_threads(_metadata);
for (i = 0; i < MAX_THREADS; i++) {
- if (err_value && thread_args[i].result != -1) {
- ksft_print_msg("Thread %d should fail but succeeded (%d)\n",
+ if (err_value) {
+ EXPECT_EQ(thread_args[i].result, -1) {
+ TH_LOG("Thread %d should fail but succeeded (%d)",
i, thread_args[i].result);
- need_exit = 1;
- }
- if (!err_value && thread_args[i].result != 0) {
- ksft_print_msg("Thread %d failed (%d)\n", i, thread_args[i].result);
- need_exit = 1;
+ }
+ } else {
+ EXPECT_EQ(thread_args[i].result, 0)
+ TH_LOG("Thread %d failed (%d)", i, thread_args[i].result);
}
}
- if (need_exit)
- ksft_exit_fail_msg("Aborting due to earlier errors.\n");
}
-static void test_futex(void *futex_ptr, int err_value)
+static void test_futex(struct __test_metadata *_metadata, void *futex_ptr, int err_value)
{
- __test_futex(futex_ptr, err_value, FUTEX2_SIZE_U32 | FUTEX_PRIVATE_FLAG | FUTEX2_NUMA);
+ __test_futex(_metadata, futex_ptr, err_value, FUTEX2_SIZE_U32 | FUTEX_PRIVATE_FLAG | FUTEX2_NUMA);
}
TEST(futex_numa_mpol)
@@ -141,43 +148,41 @@ TEST(futex_numa_mpol)
mem_size = sysconf(_SC_PAGE_SIZE);
futex_ptr = mmap(NULL, mem_size * 2, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, 0, 0);
- if (futex_ptr == MAP_FAILED)
- ksft_exit_fail_msg("mmap() for %d bytes failed\n", mem_size);
+ ASSERT_NE(futex_ptr, MAP_FAILED)
+ TH_LOG("mmap() for %d bytes failed: %s", mem_size, strerror(errno));
/* Create an invalid memory region for the "Memory out of range" test */
mprotect(futex_ptr + mem_size, mem_size, PROT_NONE);
futex_numa = futex_ptr;
- ksft_print_msg("Regular test\n");
+ TH_LOG("Regular test");
futex_numa->futex = 0;
futex_numa->numa = FUTEX_NO_NODE;
- test_futex(futex_ptr, 0);
+ test_futex(_metadata, futex_ptr, 0);
- if (futex_numa->numa == FUTEX_NO_NODE)
- ksft_exit_fail_msg("NUMA node is left uninitialized\n");
+ EXPECT_NE(futex_numa->numa, FUTEX_NO_NODE)
+ TH_LOG("NUMA node is left uninitialized");
/* FUTEX2_NUMA futex must be 8-byte aligned */
- ksft_print_msg("Mis-aligned futex\n");
- test_futex(futex_ptr + mem_size - 4, EINVAL);
+ TH_LOG("Mis-aligned futex");
+ test_futex(_metadata, futex_ptr + mem_size - 4, EINVAL);
- ksft_print_msg("Memory out of range\n");
- test_futex(futex_ptr + mem_size, EFAULT);
+ TH_LOG("Memory out of range");
+ test_futex(_metadata, futex_ptr + mem_size, EFAULT);
futex_numa->numa = FUTEX_NO_NODE;
mprotect(futex_ptr, mem_size, PROT_READ);
- ksft_print_msg("Memory, RO\n");
- test_futex(futex_ptr, EFAULT);
+ TH_LOG("Memory, RO");
+ test_futex(_metadata, futex_ptr, EFAULT);
mprotect(futex_ptr, mem_size, PROT_NONE);
- ksft_print_msg("Memory, no access\n");
- test_futex(futex_ptr, EFAULT);
+ TH_LOG("Memory, no access");
+ test_futex(_metadata, futex_ptr, EFAULT);
mprotect(futex_ptr, mem_size, PROT_READ | PROT_WRITE);
- ksft_print_msg("Memory back to RW\n");
- test_futex(futex_ptr, 0);
-
- ksft_test_result_pass("futex2 memory boundary tests passed\n");
+ TH_LOG("Memory back to RW");
+ test_futex(_metadata, futex_ptr, 0);
/* MPOL test. Does not work as expected */
#ifdef LIBNUMA_VER_SUFFICIENT
@@ -190,25 +195,23 @@ TEST(futex_numa_mpol)
sizeof(nodemask) * 8, 0);
if (ret == 0) {
ret = numa_set_mempolicy_home_node(futex_ptr, mem_size, i, 0);
- if (ret != 0)
- ksft_exit_fail_msg("Failed to set home node: %m, %d\n", errno);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("Failed to set home node: %s, %d", strerror(errno), errno);
- ksft_print_msg("Node %d test\n", i);
+ TH_LOG("Node %d test", i);
futex_numa->futex = 0;
futex_numa->numa = FUTEX_NO_NODE;
- ret = futex2_wake(futex_ptr, 0, FUTEX2_SIZE_U32 | FUTEX_PRIVATE_FLAG | FUTEX2_NUMA | FUTEX2_MPOL);
- if (ret < 0)
- ksft_test_result_fail("Failed to wake 0 with MPOL: %m\n");
- if (futex_numa->numa != i) {
- ksft_exit_fail_msg("Returned NUMA node is %d expected %d\n",
- futex_numa->numa, i);
- }
+ ret = futex2_wake(futex_ptr, 0, FUTEX2_SIZE_U32 | FUTEX_PRIVATE_FLAG |
+ FUTEX2_NUMA | FUTEX2_MPOL);
+ EXPECT_GE(ret, 0)
+ TH_LOG("Failed to wake 0 with MPOL: %s", strerror(errno));
+ EXPECT_EQ(futex_numa->numa, i)
+ TH_LOG("Returned NUMA node is %d expected %d", futex_numa->numa, i);
}
}
- ksft_test_result_pass("futex2 MPOL hints test passed\n");
#else
- ksft_test_result_skip("futex2 MPOL hints test requires libnuma 2.0.18+\n");
+ SKIP(return, "futex2 MPOL hints test requires libnuma 2.0.18+");
#endif
munmap(futex_ptr, mem_size * 2);
}
diff --git a/tools/testing/selftests/futex/functional/futex_priv_hash.c b/tools/testing/selftests/futex/functional/futex_priv_hash.c
index e8079d7c65e8..a8742e204540 100644
--- a/tools/testing/selftests/futex/functional/futex_priv_hash.c
+++ b/tools/testing/selftests/futex/functional/futex_priv_hash.c
@@ -9,6 +9,7 @@
#include <pthread.h>
#include <stdio.h>
#include <stdlib.h>
+#include <string.h>
#include <unistd.h>
#include <linux/prctl.h>
@@ -39,31 +40,28 @@ static int futex_hash_slots_get(void)
return prctl(PR_FUTEX_HASH, PR_FUTEX_HASH_GET_SLOTS);
}
-static void futex_hash_slots_set_verify(int slots)
+static void futex_hash_slots_set_verify(struct __test_metadata *_metadata, int slots)
{
int ret;
ret = futex_hash_slots_set(slots);
- if (ret != 0) {
- ksft_test_result_fail("Failed to set slots to %d: %m\n", slots);
- ksft_finished();
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("Failed to set slots to %d: %s", slots, strerror(errno));
+
ret = futex_hash_slots_get();
- if (ret != slots) {
- ksft_test_result_fail("Set %d slots but PR_FUTEX_HASH_GET_SLOTS returns: %d, %m\n",
- slots, ret);
- ksft_finished();
+ ASSERT_EQ(ret, slots) {
+ TH_LOG("Set %d slots but PR_FUTEX_HASH_GET_SLOTS returns: %d, %s",
+ slots, ret, strerror(errno));
}
- ksft_test_result_pass("SET and GET slots %d passed\n", slots);
}
-static void futex_hash_slots_set_must_fail(int slots)
+static void futex_hash_slots_set_must_fail(struct __test_metadata *_metadata, int slots)
{
int ret;
ret = futex_hash_slots_set(slots);
- ksft_test_result(ret < 0, "futex_hash_slots_set(%d)\n",
- slots);
+ EXPECT_LT(ret, 0)
+ TH_LOG("futex_hash_slots_set(%d) should fail but succeeded", slots);
}
static void *thread_return_fn(void *arg)
@@ -82,32 +80,32 @@ static void *thread_lock_fn(void *arg)
return NULL;
}
-static void create_max_threads(void *(*thread_fn)(void *))
+static void create_max_threads(struct __test_metadata *_metadata, void *(*thread_fn)(void *))
{
int i, ret;
for (i = 0; i < MAX_THREADS; i++) {
ret = pthread_create(&threads[i], NULL, thread_fn, NULL);
- if (ret)
- ksft_exit_fail_msg("pthread_create failed: %m\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_create failed: %s", strerror(errno));
}
}
-static void join_max_threads(void)
+static void join_max_threads(struct __test_metadata *_metadata)
{
int i, ret;
for (i = 0; i < MAX_THREADS; i++) {
ret = pthread_join(threads[i], NULL);
- if (ret)
- ksft_exit_fail_msg("pthread_join failed for thread %d\n", i);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_join failed for thread %d: %s", i, strerror(errno));
}
}
#define SEC_IN_NSEC 1000000000
#define MSEC_IN_NSEC 1000000
-static void futex_dummy_op(void)
+static void futex_dummy_op(struct __test_metadata *_metadata)
{
pthread_mutex_t lock = PTHREAD_MUTEX_INITIALIZER;
struct timespec timeout;
@@ -121,11 +119,11 @@ static void futex_dummy_op(void)
timeout.tv_sec++;
}
ret = pthread_mutex_timedlock(&lock, &timeout);
- if (ret == 0)
- ksft_exit_fail_msg("Successfully locked an already locked mutex.\n");
+ ASSERT_NE(ret, 0)
+ TH_LOG("Successfully locked an already locked mutex");
- if (ret != ETIMEDOUT)
- ksft_exit_fail_msg("pthread_mutex_timedlock() did not timeout: %d.\n", ret);
+ ASSERT_EQ(ret, ETIMEDOUT)
+ TH_LOG("pthread_mutex_timedlock() did not timeout: %d", ret);
}
static const char *test_msg_auto_create = "Automatic hash bucket init on thread creation.\n";
@@ -140,50 +138,48 @@ TEST(priv_hash)
ret = pthread_mutexattr_init(&mutex_attr_pi);
ret |= pthread_mutexattr_setprotocol(&mutex_attr_pi, PTHREAD_PRIO_INHERIT);
ret |= pthread_mutex_init(&global_lock, &mutex_attr_pi);
- if (ret != 0) {
- ksft_exit_fail_msg("Failed to initialize pthread mutex.\n");
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("Failed to initialize pthread mutex");
+
/* First thread, expect to be 0, not yet initialized */
ret = futex_hash_slots_get();
- if (ret != 0)
- ksft_exit_fail_msg("futex_hash_slots_get() failed: %d, %m\n", ret);
+ if (ret < 0 && errno == EINVAL)
+ SKIP(return, "PR_FUTEX_HASH not supported by kernel");
+
+ ASSERT_EQ(ret, 0)
+ TH_LOG("futex_hash_slots_get() failed: %d, %s", ret, strerror(errno));
- ksft_test_result_pass("Basic get slots and immutable status.\n");
ret = pthread_create(&threads[0], NULL, thread_return_fn, NULL);
- if (ret != 0)
- ksft_exit_fail_msg("pthread_create() failed: %d, %m\n", ret);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_create() failed: %d, %s", ret, strerror(errno));
ret = pthread_join(threads[0], NULL);
- if (ret != 0)
- ksft_exit_fail_msg("pthread_join() failed: %d, %m\n", ret);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_join() failed: %d, %s", ret, strerror(errno));
/* First thread, has to initialize private hash */
futex_slots1 = futex_hash_slots_get();
- if (futex_slots1 <= 0) {
- ksft_print_msg("Current hash buckets: %d\n", futex_slots1);
- ksft_exit_fail_msg("%s", test_msg_auto_create);
- }
-
- ksft_test_result_pass("%s", test_msg_auto_create);
+ EXPECT_GT(futex_slots1, 0)
+ TH_LOG("Current hash buckets: %d. %s", futex_slots1, test_msg_auto_create);
online_cpus = sysconf(_SC_NPROCESSORS_ONLN);
ret = pthread_barrier_init(&barrier_main, NULL, MAX_THREADS + 1);
- if (ret != 0)
- ksft_exit_fail_msg("pthread_barrier_init failed: %m.\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_barrier_init failed: %s", strerror(errno));
ret = pthread_mutex_lock(&global_lock);
- if (ret != 0)
- ksft_exit_fail_msg("pthread_mutex_lock failed: %m.\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_mutex_lock failed: %s", strerror(errno));
counter = 0;
- create_max_threads(thread_lock_fn);
+ create_max_threads(_metadata, thread_lock_fn);
pthread_barrier_wait(&barrier_main);
/*
* The current default size of hash buckets is 16. The auto increase
* works only if more than 16 CPUs are available.
*/
- ksft_print_msg("Online CPUs: %d\n", online_cpus);
+ TH_LOG("Online CPUs: %d", online_cpus);
if (online_cpus > 16) {
retry_getslots:
futex_slotsn = futex_hash_slots_get();
@@ -200,71 +196,76 @@ retry_getslots:
* sleep for 100ms and issue a futex operation.
*/
if (retry > 0) {
- futex_dummy_op();
+ futex_dummy_op(_metadata);
goto retry_getslots;
}
- ksft_print_msg("Expected increase of hash buckets but got: %d -> %d\n",
- futex_slots1, futex_slotsn);
- ksft_exit_fail_msg("%s", test_msg_auto_inc);
+ EXPECT_NE(futex_slots1, futex_slotsn) {
+ TH_LOG("Expected increase of hash buckets but got: %d -> %d. %s",
+ futex_slots1, futex_slotsn, test_msg_auto_inc);
+ }
}
- ksft_test_result_pass("%s", test_msg_auto_inc);
} else {
- ksft_test_result_skip("%s", test_msg_auto_inc);
+ SKIP(return, "Automatic increase with more than 16 CPUs (only %d online)", online_cpus);
}
ret = pthread_mutex_unlock(&global_lock);
/* Once the user changes it, it has to be what is set */
- futex_hash_slots_set_verify(2);
- futex_hash_slots_set_verify(4);
- futex_hash_slots_set_verify(8);
- futex_hash_slots_set_verify(32);
- futex_hash_slots_set_verify(16);
+ futex_hash_slots_set_verify(_metadata, 2);
+ futex_hash_slots_set_verify(_metadata, 4);
+ futex_hash_slots_set_verify(_metadata, 8);
+ futex_hash_slots_set_verify(_metadata, 32);
+ futex_hash_slots_set_verify(_metadata, 16);
ret = futex_hash_slots_set(15);
- ksft_test_result(ret < 0, "Use 15 slots\n");
+ EXPECT_LT(ret, 0)
+ TH_LOG("Use 15 slots should fail but succeeded");
+
+ futex_hash_slots_set_verify(_metadata, 2);
+ join_max_threads(_metadata);
+
+ EXPECT_EQ(counter, MAX_THREADS)
+ TH_LOG("Created and waited for %d of %d threads", counter, MAX_THREADS);
- futex_hash_slots_set_verify(2);
- join_max_threads();
- ksft_test_result(counter == MAX_THREADS, "Created and waited for %d of %d threads\n",
- counter, MAX_THREADS);
counter = 0;
/* Once the user set something, auto resize must be disabled */
ret = pthread_barrier_init(&barrier_main, NULL, MAX_THREADS);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_barrier_init failed: %s", strerror(errno));
- create_max_threads(thread_lock_fn);
- join_max_threads();
+ create_max_threads(_metadata, thread_lock_fn);
+ join_max_threads(_metadata);
ret = futex_hash_slots_get();
- ksft_test_result(ret == 2, "No more auto-resize after manual setting, got %d\n",
- ret);
+ EXPECT_EQ(ret, 2)
+ TH_LOG("No more auto-resize after manual setting, got %d", ret);
- futex_hash_slots_set_must_fail(1 << 29);
- futex_hash_slots_set_verify(4);
+ futex_hash_slots_set_must_fail(_metadata, 1 << 29);
+ futex_hash_slots_set_verify(_metadata, 4);
/*
* Once the global hash has been requested, then this requested can not
* be undone.
*/
ret = futex_hash_slots_set(0);
- ksft_test_result(ret == 0, "Global hash request\n");
- if (ret != 0)
- return;
+ ASSERT_EQ(ret, 0)
+ TH_LOG("Global hash request failed: %s", strerror(errno));
- futex_hash_slots_set_must_fail(4);
- futex_hash_slots_set_must_fail(8);
- futex_hash_slots_set_must_fail(8);
- futex_hash_slots_set_must_fail(0);
- futex_hash_slots_set_must_fail(6);
+ futex_hash_slots_set_must_fail(_metadata, 4);
+ futex_hash_slots_set_must_fail(_metadata, 8);
+ futex_hash_slots_set_must_fail(_metadata, 8);
+ futex_hash_slots_set_must_fail(_metadata, 0);
+ futex_hash_slots_set_must_fail(_metadata, 6);
ret = pthread_barrier_init(&barrier_main, NULL, MAX_THREADS);
- if (ret != 0)
- ksft_exit_fail_msg("pthread_barrier_init failed: %m\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_barrier_init failed: %s", strerror(errno));
- create_max_threads(thread_lock_fn);
- join_max_threads();
+ create_max_threads(_metadata, thread_lock_fn);
+ join_max_threads(_metadata);
ret = futex_hash_slots_get();
- ksft_test_result(ret == 0, "Continue to use global hash\n");
+ EXPECT_EQ(ret, 0)
+ TH_LOG("Continue to use global hash failed");
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_requeue.c b/tools/testing/selftests/futex/functional/futex_requeue.c
index dcf0d5f2f312..cc31f051765d 100644
--- a/tools/testing/selftests/futex/functional/futex_requeue.c
+++ b/tools/testing/selftests/futex/functional/futex_requeue.c
@@ -5,55 +5,70 @@
* futex cmp requeue test by André Almeida <andrealmeid@collabora.com>
*/
-#include <pthread.h>
#include <limits.h>
+#include <pthread.h>
+#include <string.h>
#include "futextest.h"
+#include "futex_thread.h"
#include "kselftest_harness.h"
-#define timeout_ns 30000000
-#define WAKE_WAIT_US 10000
+struct waiter_args {
+ struct __test_metadata *_metadata;
+ unsigned int n_threads;
+};
volatile futex_t *f1;
-void *waiterfn(void *arg)
+static int waiterfn(void *arg)
{
- struct timespec to;
+ struct __test_metadata *_metadata;
+ struct waiter_args *wargs = arg;
+ struct timespec to = { };
+ int res;
- to.tv_sec = 0;
- to.tv_nsec = timeout_ns;
+ _metadata = wargs->_metadata;
+ to.tv_sec = (wargs->n_threads + 1) * WAIT_FOR_THREAD_SECS;
- if (futex_wait(f1, *f1, &to, 0))
- printf("waiter failed errno %d\n", errno);
+ res = futex_wait(f1, *f1, &to, 0);
+ if (res) {
+ EXPECT_EQ(res, 0)
+ TH_LOG("waiter failed errno %d: %s", errno, strerror(errno));
+ }
- return NULL;
+ return 0;
}
TEST(requeue_single)
{
+ struct waiter_args wargs = { ._metadata = _metadata, .n_threads = 1 };
+ struct futex_thread waiter;
volatile futex_t _f1 = 0;
volatile futex_t f2 = 0;
- pthread_t waiter[10];
f1 = &_f1;
/*
* Requeue a waiter from f1 to f2, and wake f2.
*/
- ASSERT_EQ(0, pthread_create(&waiter[0], NULL, waiterfn, NULL));
+ ASSERT_EQ(futex_thread_create(&waiter, waiterfn, &wargs), 0)
+ TH_LOG("pthread_create failed");
- usleep(WAKE_WAIT_US);
+ ASSERT_EQ(futex_wait_for_thread(&waiter, _metadata), 0)
+ TH_LOG("Wait for thread failed");
- EXPECT_EQ(1, futex_cmp_requeue(f1, 0, &f2, 0, 1, 0));
- EXPECT_EQ(1, futex_wake(&f2, 1, 0));
+ EXPECT_EQ(futex_cmp_requeue(f1, 0, &f2, 0, 1, 0), 1);
+ EXPECT_EQ(futex_wake(&f2, 1, 0), 1);
+
+ EXPECT_EQ(futex_thread_destroy(&waiter), 0);
}
TEST(requeue_multiple)
{
+ struct waiter_args wargs = { ._metadata = _metadata, .n_threads = 10 };
+ struct futex_thread waiter[10];
volatile futex_t _f1 = 0;
volatile futex_t f2 = 0;
- pthread_t waiter[10];
- int i;
f1 = &_f1;
@@ -61,13 +76,21 @@ TEST(requeue_multiple)
* Create 10 waiters at f1. At futex_requeue, wake 3 and requeue 7.
* At futex_wake, wake INT_MAX (should be exactly 7).
*/
- for (i = 0; i < 10; i++)
- ASSERT_EQ(0, pthread_create(&waiter[i], NULL, waiterfn, NULL));
+ for (int i = 0; i < 10; i++) {
+ ASSERT_EQ(futex_thread_create(&waiter[i], waiterfn, &wargs), 0)
+ TH_LOG("pthread_create failed for waiter %d", i);
+ }
+
+ for (int i = 0; i < 10; i++) {
+ ASSERT_EQ(futex_wait_for_thread(&waiter[i], _metadata), 0)
+ TH_LOG("Wait for waiter thread %d failed", i);
+ }
- usleep(WAKE_WAIT_US);
+ EXPECT_EQ(futex_cmp_requeue(f1, 0, &f2, 3, 7, 0), 10);
+ EXPECT_EQ(futex_wake(&f2, INT_MAX, 0), 7);
- EXPECT_EQ(10, futex_cmp_requeue(f1, 0, &f2, 3, 7, 0));
- EXPECT_EQ(7, futex_wake(&f2, INT_MAX, 0));
+ for (int i = 0; i < 10; i++)
+ EXPECT_EQ(futex_thread_destroy(&waiter[i]), 0);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_requeue_pi.c b/tools/testing/selftests/futex/functional/futex_requeue_pi.c
index 46d2858e15a8..65326cbef3a4 100644
--- a/tools/testing/selftests/futex/functional/futex_requeue_pi.c
+++ b/tools/testing/selftests/futex/functional/futex_requeue_pi.c
@@ -43,12 +43,13 @@ futex_t f2 = FUTEX_INITIALIZER;
futex_t wake_complete = FUTEX_INITIALIZER;
struct thread_arg {
- long id;
- struct timespec *timeout;
- int lock;
- int ret;
+ struct __test_metadata *_metadata;
+ long id;
+ struct timespec *timeout;
+ int lock;
+ int ret;
};
-#define THREAD_ARG_INITIALIZER { 0, NULL, 0, 0 }
+#define THREAD_ARG_INITIALIZER { NULL, 0, NULL, 0, 0 }
FIXTURE(args)
{
@@ -118,7 +119,7 @@ FIXTURE_VARIANT_ADD_TIMEOUT(5000);
FIXTURE_VARIANT_ADD_TIMEOUT(500000);
FIXTURE_VARIANT_ADD_TIMEOUT(2000000000);
-int create_rt_thread(pthread_t *pth, void*(*func)(void *), void *arg,
+int create_rt_thread(struct __test_metadata *_metadata, pthread_t *pth, void*(*func)(void *), void *arg,
int policy, int prio)
{
int ret;
@@ -129,29 +130,22 @@ int create_rt_thread(pthread_t *pth, void*(*func)(void *), void *arg,
memset(&schedp, 0, sizeof(schedp));
ret = pthread_attr_setinheritsched(&attr, PTHREAD_EXPLICIT_SCHED);
- if (ret) {
- ksft_exit_fail_msg("pthread_attr_setinheritsched\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_attr_setinheritsched failed");
ret = pthread_attr_setschedpolicy(&attr, policy);
- if (ret) {
- ksft_exit_fail_msg("pthread_attr_setschedpolicy\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_attr_setschedpolicy failed");
schedp.sched_priority = prio;
ret = pthread_attr_setschedparam(&attr, &schedp);
- if (ret) {
- ksft_exit_fail_msg("pthread_attr_setschedparam\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_attr_setschedparam failed");
ret = pthread_create(pth, &attr, func, arg);
- if (ret) {
- ksft_exit_fail_msg("pthread_create\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_create failed");
+
return 0;
}
@@ -159,70 +153,76 @@ int create_rt_thread(pthread_t *pth, void*(*func)(void *), void *arg,
void *waiterfn(void *arg)
{
struct thread_arg *args = (struct thread_arg *)arg;
+ struct __test_metadata *_metadata = args->_metadata;
futex_t old_val;
- ksft_print_dbg_msg("Waiter %ld: running\n", args->id);
+ TH_LOG("Waiter %ld: running", args->id);
/* Each thread sleeps for a different amount of time
* This is to avoid races, because we don't lock the
- * external mutex here */
+ * external mutex here
+ */
usleep(1000 * (long)args->id);
old_val = f1;
atomic_inc(&waiters_blocked);
- ksft_print_dbg_msg("Calling futex_wait_requeue_pi: %p (%u) -> %p\n",
- &f1, f1, &f2);
+ TH_LOG("Calling futex_wait_requeue_pi: %p (%u) -> %p", &f1, f1, &f2);
args->ret = futex_wait_requeue_pi(&f1, old_val, &f2, args->timeout,
FUTEX_PRIVATE_FLAG);
- ksft_print_dbg_msg("waiter %ld woke with %d %s\n", args->id, args->ret,
- args->ret < 0 ? strerror(errno) : "");
+ TH_LOG("waiter %ld woke with %d %s", args->id, args->ret,
+ args->ret < 0 ? strerror(errno) : "");
atomic_inc(&waiters_woken);
if (args->ret < 0) {
- if (args->timeout && errno == ETIMEDOUT)
+ if (args->timeout && errno == ETIMEDOUT) {
args->ret = 0;
- else {
- ksft_exit_fail_msg("futex_wait_requeue_pi\n");
+ } else {
+ ASSERT_EQ(args->ret, 0)
+ TH_LOG("futex_wait_requeue_pi failed: %s", strerror(errno));
}
futex_lock_pi(&f2, NULL, 0, FUTEX_PRIVATE_FLAG);
}
futex_unlock_pi(&f2, FUTEX_PRIVATE_FLAG);
- ksft_print_dbg_msg("Waiter %ld: exiting with %d\n", args->id, args->ret);
+ TH_LOG("Waiter %ld: exiting with %d", args->id, args->ret);
pthread_exit((void *)&args->ret);
}
void *broadcast_wakerfn(void *arg)
{
struct thread_arg *args = (struct thread_arg *)arg;
+ struct __test_metadata *_metadata = args->_metadata;
int nr_requeue = INT_MAX;
int task_count = 0;
futex_t old_val;
int nr_wake = 1;
int i = 0;
- ksft_print_dbg_msg("Waker: waiting for waiters to block\n");
+ TH_LOG("Waker: waiting for waiters to block");
while (waiters_blocked.val < THREAD_MAX)
usleep(1000);
usleep(1000);
- ksft_print_dbg_msg("Waker: Calling broadcast\n");
+ TH_LOG("Waker: Calling broadcast");
if (args->lock) {
- ksft_print_dbg_msg("Calling FUTEX_LOCK_PI on mutex=%x @ %p\n", f2, &f2);
+ TH_LOG("Calling FUTEX_LOCK_PI on mutex=%x @ %p", f2, &f2);
futex_lock_pi(&f2, NULL, 0, FUTEX_PRIVATE_FLAG);
}
continue_requeue:
old_val = f1;
args->ret = futex_cmp_requeue_pi(&f1, old_val, &f2, nr_wake, nr_requeue,
- FUTEX_PRIVATE_FLAG);
+ FUTEX_PRIVATE_FLAG);
if (args->ret < 0) {
- ksft_exit_fail_msg("FUTEX_CMP_REQUEUE_PI failed\n");
+ ASSERT_GE(args->ret, 0)
+ TH_LOG("FUTEX_CMP_REQUEUE_PI failed: %s", strerror(errno));
} else if (++i < MAX_WAKE_ITERS) {
task_count += args->ret;
if (task_count < THREAD_MAX - waiters_woken.val)
goto continue_requeue;
} else {
- ksft_exit_fail_msg("max broadcast iterations (%d) reached with %d/%d tasks woken or requeued\n",
- MAX_WAKE_ITERS, task_count, THREAD_MAX);
+ ASSERT_TRUE(0) {
+ TH_LOG("max broadcast iterations (%d) reached with %d/%d tasks woken or requeued",
+ MAX_WAKE_ITERS, task_count, THREAD_MAX);
+ }
}
futex_wake(&wake_complete, 1, FUTEX_PRIVATE_FLAG);
@@ -233,33 +233,33 @@ void *broadcast_wakerfn(void *arg)
if (args->ret > 0)
args->ret = task_count;
- ksft_print_dbg_msg("Waker: exiting with %d\n", args->ret);
+ TH_LOG("Waker: exiting with %d", args->ret);
pthread_exit((void *)&args->ret);
}
void *signal_wakerfn(void *arg)
{
struct thread_arg *args = (struct thread_arg *)arg;
+ struct __test_metadata *_metadata = args->_metadata;
unsigned int old_val;
int nr_requeue = 0;
int task_count = 0;
int nr_wake = 1;
int i = 0;
- ksft_print_dbg_msg("Waker: waiting for waiters to block\n");
+ TH_LOG("Waker: waiting for waiters to block");
while (waiters_blocked.val < THREAD_MAX)
usleep(1000);
usleep(1000);
while (task_count < THREAD_MAX && waiters_woken.val < THREAD_MAX) {
- ksft_print_dbg_msg("task_count: %d, waiters_woken: %d\n",
+ TH_LOG("task_count: %d, waiters_woken: %d",
task_count, waiters_woken.val);
if (args->lock) {
- ksft_print_dbg_msg("Calling FUTEX_LOCK_PI on mutex=%x @ %p\n",
- f2, &f2);
+ TH_LOG("Calling FUTEX_LOCK_PI on mutex=%x @ %p", f2, &f2);
futex_lock_pi(&f2, NULL, 0, FUTEX_PRIVATE_FLAG);
}
- ksft_print_dbg_msg("Waker: Calling signal\n");
+ TH_LOG("Waker: Calling signal");
/* cond_signal */
old_val = f1;
args->ret = futex_cmp_requeue_pi(&f1, old_val, &f2,
@@ -267,23 +267,27 @@ void *signal_wakerfn(void *arg)
FUTEX_PRIVATE_FLAG);
if (args->ret < 0)
args->ret = -errno;
- ksft_print_dbg_msg("futex: %x\n", f2);
+ TH_LOG("futex: %x", f2);
if (args->lock) {
- ksft_print_dbg_msg("Calling FUTEX_UNLOCK_PI on mutex=%x @ %p\n",
+ TH_LOG("Calling FUTEX_UNLOCK_PI on mutex=%x @ %p",
f2, &f2);
futex_unlock_pi(&f2, FUTEX_PRIVATE_FLAG);
}
- ksft_print_dbg_msg("futex: %x\n", f2);
- if (args->ret < 0)
- ksft_exit_fail_msg("FUTEX_CMP_REQUEUE_PI failed\n");
+ TH_LOG("futex: %x", f2);
+ if (args->ret < 0) {
+ ASSERT_GE(args->ret, 0)
+ TH_LOG("FUTEX_CMP_REQUEUE_PI failed: %s", strerror(-args->ret));
+ }
task_count += args->ret;
usleep(SIGNAL_PERIOD_US);
i++;
/* we have to loop at least THREAD_MAX times */
if (i > MAX_WAKE_ITERS + THREAD_MAX) {
- ksft_exit_fail_msg("max signaling iterations (%d) reached, giving up on pending waiters.\n",
- MAX_WAKE_ITERS + THREAD_MAX);
+ ASSERT_TRUE(0) {
+ TH_LOG("max signaling iterations (%d) reached, giving up on pending waiters.",
+ MAX_WAKE_ITERS + THREAD_MAX);
+ }
}
}
@@ -292,14 +296,15 @@ void *signal_wakerfn(void *arg)
if (args->ret >= 0)
args->ret = task_count;
- ksft_print_dbg_msg("Waker: exiting with %d\n", args->ret);
- ksft_print_dbg_msg("Waker: waiters_woken: %d\n", waiters_woken.val);
+ TH_LOG("Waker: exiting with %d", args->ret);
+ TH_LOG("Waker: waiters_woken: %d", waiters_woken.val);
pthread_exit((void *)&args->ret);
}
void *third_party_blocker(void *arg)
{
struct thread_arg *args = (struct thread_arg *)arg;
+ struct __test_metadata *_metadata = args->_metadata;
int ret2 = 0;
args->ret = futex_lock_pi(&f2, NULL, 0, FUTEX_PRIVATE_FLAG);
@@ -310,8 +315,10 @@ void *third_party_blocker(void *arg)
ret2 = futex_unlock_pi(&f2, FUTEX_PRIVATE_FLAG);
out:
- if (args->ret || ret2)
- ksft_exit_fail_msg("third_party_blocker() futex error");
+ if (args->ret || ret2) {
+ ASSERT_TRUE(0)
+ TH_LOG("%s() futex error", __func__);
+ }
pthread_exit((void *)&args->ret);
}
@@ -330,20 +337,19 @@ TEST_F(args, futex_requeue_pi)
bool lock = variant->locked;
int *waiter_ret, i, ret = 0;
- ksft_print_msg(
- "\tArguments: broadcast=%d locked=%d owner=%d timeout=%ldns\n",
+ TH_LOG("Arguments: broadcast=%d locked=%d owner=%d timeout=%ldns",
broadcast, lock, third_party_owner, timeout_ns);
if (timeout_ns) {
time_t secs;
- ksft_print_dbg_msg("timeout_ns = %ld\n", timeout_ns);
+ TH_LOG("timeout_ns = %ld", timeout_ns);
ret = clock_gettime(CLOCK_MONOTONIC, &ts);
secs = (ts.tv_nsec + timeout_ns) / 1000000000;
ts.tv_nsec = ((int64_t)ts.tv_nsec + timeout_ns) % 1000000000;
ts.tv_sec += secs;
- ksft_print_dbg_msg("ts.tv_sec = %ld\n", ts.tv_sec);
- ksft_print_dbg_msg("ts.tv_nsec = %ld\n", ts.tv_nsec);
+ TH_LOG("ts.tv_sec = %ld", ts.tv_sec);
+ TH_LOG("ts.tv_nsec = %ld", ts.tv_nsec);
tsp = &ts;
}
@@ -351,34 +357,29 @@ TEST_F(args, futex_requeue_pi)
wakerfn = broadcast_wakerfn;
if (third_party_owner) {
- if (create_rt_thread(&blocker, third_party_blocker,
- (void *)&blocker_arg, SCHED_FIFO, 1)) {
- ksft_exit_fail_msg("Creating third party blocker thread failed\n");
- }
+ blocker_arg._metadata = _metadata;
+ create_rt_thread(_metadata, &blocker, third_party_blocker,
+ (void *)&blocker_arg, SCHED_FIFO, 1);
}
atomic_set(&waiters_woken, 0);
for (i = 0; i < THREAD_MAX; i++) {
+ args[i]._metadata = _metadata;
args[i].id = i;
args[i].timeout = tsp;
- ksft_print_dbg_msg("Starting thread %d\n", i);
- if (create_rt_thread(&waiter[i], waiterfn, (void *)&args[i],
- SCHED_FIFO, 1)) {
- ksft_exit_fail_msg("Creating waiting thread failed\n");
- }
+ TH_LOG("Starting thread %d", i);
+ create_rt_thread(_metadata, &waiter[i], waiterfn, (void *)&args[i],
+ SCHED_FIFO, 1);
}
+ waker_arg._metadata = _metadata;
waker_arg.lock = lock;
- if (create_rt_thread(&waker, wakerfn, (void *)&waker_arg,
- SCHED_FIFO, 1)) {
- ksft_exit_fail_msg("Creating waker thread failed\n");
- }
+ create_rt_thread(_metadata, &waker, wakerfn, (void *)&waker_arg, SCHED_FIFO, 1);
/* Wait for threads to finish */
/* Store the first error or failure encountered in waiter_ret */
waiter_ret = &args[0].ret;
for (i = 0; i < THREAD_MAX; i++)
- pthread_join(waiter[i],
- *waiter_ret ? NULL : (void **)&waiter_ret);
+ pthread_join(waiter[i], *waiter_ret ? NULL : (void **)&waiter_ret);
if (third_party_owner)
pthread_join(blocker, NULL);
@@ -393,8 +394,8 @@ TEST_F(args, futex_requeue_pi)
ret = blocker_arg.ret;
}
- if (ret)
- ksft_test_result_fail("fail");
+ EXPECT_EQ(ret, 0)
+ TH_LOG("Test failed with error code: %d", ret);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_requeue_pi_mismatched_ops.c b/tools/testing/selftests/futex/functional/futex_requeue_pi_mismatched_ops.c
index f686e605359c..35bb8a807bb9 100644
--- a/tools/testing/selftests/futex/functional/futex_requeue_pi_mismatched_ops.c
+++ b/tools/testing/selftests/futex/functional/futex_requeue_pi_mismatched_ops.c
@@ -29,14 +29,17 @@
futex_t f1 = FUTEX_INITIALIZER;
futex_t f2 = FUTEX_INITIALIZER;
-int child_ret = 0;
+int child_ret;
void *blocking_child(void *arg)
{
+ struct __test_metadata *_metadata = (struct __test_metadata *)arg;
+
child_ret = futex_wait(&f1, f1, NULL, FUTEX_PRIVATE_FLAG);
if (child_ret < 0) {
child_ret = -errno;
- ksft_exit_fail_msg("futex_wait\n");
+ ASSERT_EQ(child_ret, 0)
+ TH_LOG("futex_wait failed: %s", strerror(errno));
}
return (void *)&child_ret;
}
@@ -46,8 +49,8 @@ TEST(requeue_pi_mismatched_ops)
pthread_t child;
int ret;
- if (pthread_create(&child, NULL, blocking_child, NULL))
- ksft_exit_fail_msg("pthread_create\n");
+ ASSERT_EQ(pthread_create(&child, NULL, blocking_child, _metadata), 0)
+ TH_LOG("pthread_create failed");
/* Allow the child to block in the kernel. */
sleep(1);
@@ -67,27 +70,33 @@ TEST(requeue_pi_mismatched_ops)
* FUTEX_WAKE.
*/
ret = futex_wake(&f1, 1, FUTEX_PRIVATE_FLAG);
- if (ret == 1)
+ if (ret == 1) {
ret = 0;
- else if (ret < 0)
- ksft_exit_fail_msg("futex_wake\n");
- else
- ksft_exit_fail_msg("futex_wake did not wake the child\n");
+ } else if (ret < 0) {
+ ASSERT_GE(ret, 0)
+ TH_LOG("futex_wake failed: %s", strerror(errno));
+ } else {
+ ASSERT_TRUE(0)
+ TH_LOG("futex_wake did not wake the child");
+ }
} else {
- ksft_exit_fail_msg("futex_cmp_requeue_pi\n");
+ ASSERT_TRUE(0)
+ TH_LOG("futex_cmp_requeue_pi failed with unexpected errno: %s", strerror(errno));
}
} else if (ret > 0) {
- ksft_test_result_fail("futex_cmp_requeue_pi failed to detect the mismatch\n");
+ EXPECT_EQ(ret, 0)
+ TH_LOG("futex_cmp_requeue_pi failed to detect the mismatch");
} else {
- ksft_exit_fail_msg("futex_cmp_requeue_pi found no waiters\n");
+ ASSERT_TRUE(0)
+ TH_LOG("futex_cmp_requeue_pi found no waiters");
}
pthread_join(child, NULL);
- if (!ret && !child_ret)
- ksft_test_result_pass("futex_requeue_pi_mismatched_ops passed\n");
- else
- ksft_test_result_pass("futex_requeue_pi_mismatched_ops failed\n");
+ EXPECT_EQ(ret, 0)
+ TH_LOG("Test failed: ret=%d", ret);
+ EXPECT_EQ(child_ret, 0)
+ TH_LOG("Child failed: child_ret=%d", child_ret);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_requeue_pi_signal_restart.c b/tools/testing/selftests/futex/functional/futex_requeue_pi_signal_restart.c
index a18ccae73eb1..4933612a7b55 100644
--- a/tools/testing/selftests/futex/functional/futex_requeue_pi_signal_restart.c
+++ b/tools/testing/selftests/futex/functional/futex_requeue_pi_signal_restart.c
@@ -35,10 +35,10 @@ futex_t f1 = FUTEX_INITIALIZER;
futex_t f2 = FUTEX_INITIALIZER;
atomic_t requeued = ATOMIC_INITIALIZER;
-int waiter_ret = 0;
+int waiter_ret;
-int create_rt_thread(pthread_t *pth, void*(*func)(void *), void *arg,
- int policy, int prio)
+int create_rt_thread(struct __test_metadata *_metadata, pthread_t *pth, void*(*func)(void *),
+ void *arg, int policy, int prio)
{
struct sched_param schedp;
pthread_attr_t attr;
@@ -48,45 +48,47 @@ int create_rt_thread(pthread_t *pth, void*(*func)(void *), void *arg,
memset(&schedp, 0, sizeof(schedp));
ret = pthread_attr_setinheritsched(&attr, PTHREAD_EXPLICIT_SCHED);
- if (ret)
- ksft_exit_fail_msg("pthread_attr_setinheritsched\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_attr_setinheritsched failed");
ret = pthread_attr_setschedpolicy(&attr, policy);
- if (ret)
- ksft_exit_fail_msg("pthread_attr_setschedpolicy\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_attr_setschedpolicy failed");
schedp.sched_priority = prio;
ret = pthread_attr_setschedparam(&attr, &schedp);
- if (ret)
- ksft_exit_fail_msg("pthread_attr_setschedparam\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_attr_setschedparam failed");
ret = pthread_create(pth, &attr, func, arg);
- if (ret)
- ksft_exit_fail_msg("pthread_create\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_create failed");
return 0;
}
void handle_signal(int signo)
{
- ksft_print_dbg_msg("signal received %s requeue\n",
- requeued.val ? "after" : "prior to");
+ printf("INFO: signal received %s requeue\n", requeued.val ? "after" : "prior to");
}
void *waiterfn(void *arg)
{
+ struct __test_metadata *_metadata = (struct __test_metadata *)arg;
unsigned int old_val;
int res;
- ksft_print_dbg_msg("Waiter running\n");
- ksft_print_dbg_msg("Calling FUTEX_LOCK_PI on f2=%x @ %p\n", f2, &f2);
+ TH_LOG("Waiter running");
+ TH_LOG("Calling FUTEX_LOCK_PI on f2=%x @ %p", f2, &f2);
old_val = f1;
res = futex_wait_requeue_pi(&f1, old_val, &(f2), NULL,
FUTEX_PRIVATE_FLAG);
if (!requeued.val || errno != EWOULDBLOCK) {
- ksft_test_result_fail("unexpected return from futex_wait_requeue_pi: %d (%s)\n",
- res, strerror(errno));
- ksft_print_dbg_msg("w2:futex: %x\n", f2);
+ EXPECT_TRUE(0) {
+ TH_LOG("unexpected return from futex_wait_requeue_pi: %d (%s)",
+ res, strerror(errno));
+ }
+ TH_LOG("w2:futex: %x", f2);
if (!res)
futex_unlock_pi(&f2, FUTEX_PRIVATE_FLAG);
}
@@ -94,7 +96,6 @@ void *waiterfn(void *arg)
pthread_exit(NULL);
}
-
TEST(futex_requeue_pi_signal_restart)
{
unsigned int old_val;
@@ -105,19 +106,17 @@ TEST(futex_requeue_pi_signal_restart)
sa.sa_handler = handle_signal;
sigemptyset(&sa.sa_mask);
sa.sa_flags = 0;
- if (sigaction(SIGUSR1, &sa, NULL))
- ksft_exit_fail_msg("sigaction\n");
+ ASSERT_EQ(sigaction(SIGUSR1, &sa, NULL), 0)
+ TH_LOG("sigaction failed");
- ksft_print_dbg_msg("m1:f2: %x\n", f2);
- ksft_print_dbg_msg("Creating waiter\n");
- res = create_rt_thread(&waiter, waiterfn, NULL, SCHED_FIFO, 1);
- if (res)
- ksft_exit_fail_msg("Creating waiting thread failed");
+ TH_LOG("m1:f2: %x", f2);
+ TH_LOG("Creating waiter");
+ create_rt_thread(_metadata, &waiter, waiterfn, _metadata, SCHED_FIFO, 1);
- ksft_print_dbg_msg("Calling FUTEX_LOCK_PI on f2=%x @ %p\n", f2, &f2);
- ksft_print_dbg_msg("m2:f2: %x\n", f2);
+ TH_LOG("Calling FUTEX_LOCK_PI on f2=%x @ %p", f2, &f2);
+ TH_LOG("m2:f2: %x", f2);
futex_lock_pi(&f2, 0, 0, FUTEX_PRIVATE_FLAG);
- ksft_print_dbg_msg("m3:f2: %x\n", f2);
+ TH_LOG("m3:f2: %x", f2);
while (1) {
/*
@@ -125,11 +124,11 @@ TEST(futex_requeue_pi_signal_restart)
* restart futex_wait_requeue_pi() in the kernel. Wait for the
* waiter to block on f1 again.
*/
- ksft_print_dbg_msg("Issuing SIGUSR1 to waiter\n");
+ TH_LOG("Issuing SIGUSR1 to waiter");
pthread_kill(waiter, SIGUSR1);
usleep(DELAY_US);
- ksft_print_dbg_msg("Requeueing waiter via FUTEX_CMP_REQUEUE_PI\n");
+ TH_LOG("Requeueing waiter via FUTEX_CMP_REQUEUE_PI");
old_val = f1;
res = futex_cmp_requeue_pi(&f1, old_val, &(f2), 1, 0,
FUTEX_PRIVATE_FLAG);
@@ -143,10 +142,11 @@ TEST(futex_requeue_pi_signal_restart)
atomic_set(&requeued, 1);
break;
} else if (res < 0) {
- ksft_exit_fail_msg("FUTEX_CMP_REQUEUE_PI failed\n");
+ ASSERT_GE(res, 0)
+ TH_LOG("FUTEX_CMP_REQUEUE_PI failed: %s", strerror(errno));
}
}
- ksft_print_dbg_msg("m4:f2: %x\n", f2);
+ TH_LOG("m4:f2: %x", f2);
/*
* Signal the waiter after requeue, waiter should return from
@@ -154,14 +154,14 @@ TEST(futex_requeue_pi_signal_restart)
* futex_unlock_pi() can't happen before the signal wakeup is detected
* in the kernel.
*/
- ksft_print_dbg_msg("Issuing SIGUSR1 to waiter\n");
+ TH_LOG("Issuing SIGUSR1 to waiter");
pthread_kill(waiter, SIGUSR1);
- ksft_print_dbg_msg("Waiting for waiter to return\n");
+ TH_LOG("Waiting for waiter to return");
pthread_join(waiter, NULL);
- ksft_print_dbg_msg("Calling FUTEX_UNLOCK_PI on mutex=%x @ %p\n", f2, &f2);
+ TH_LOG("Calling FUTEX_UNLOCK_PI on mutex=%x @ %p", f2, &f2);
futex_unlock_pi(&f2, FUTEX_PRIVATE_FLAG);
- ksft_print_dbg_msg("m5:f2: %x\n", f2);
+ TH_LOG("m5:f2: %x", f2);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_wait.c b/tools/testing/selftests/futex/functional/futex_wait.c
index 7b8879409007..5292f5d2c3b5 100644
--- a/tools/testing/selftests/futex/functional/futex_wait.c
+++ b/tools/testing/selftests/futex/functional/futex_wait.c
@@ -5,10 +5,11 @@
* futex cmp requeue test by André Almeida <andrealmeid@collabora.com>
*/
+#include <fcntl.h>
#include <pthread.h>
+#include <stdlib.h>
#include <sys/shm.h>
#include <sys/mman.h>
-#include <fcntl.h>
#include "futextest.h"
#include "kselftest_harness.h"
@@ -19,125 +20,157 @@
void *futex;
+struct waiter_args {
+ struct __test_metadata *_metadata;
+ unsigned int flags;
+};
+
static void *waiterfn(void *arg)
{
+ struct waiter_args *args = (struct waiter_args *)arg;
+ struct __test_metadata *_metadata = args->_metadata;
struct timespec to;
- unsigned int flags = 0;
-
- if (arg)
- flags = *((unsigned int *) arg);
+ int res;
to.tv_sec = 0;
to.tv_nsec = timeout_ns;
- if (futex_wait(futex, 0, &to, flags))
- printf("waiter failed errno %d\n", errno);
+ res = futex_wait(futex, 0, &to, args->flags);
+ if (res) {
+ EXPECT_EQ(res, 0)
+ TH_LOG("waiter failed errno %d: %s", errno, strerror(errno));
+ }
+ free(args);
return NULL;
}
TEST(private_futex)
{
- unsigned int flags = FUTEX_PRIVATE_FLAG;
+ struct waiter_args *args = malloc(sizeof(*args));
u_int32_t f_private = 0;
pthread_t waiter;
int res;
+ args->_metadata = _metadata;
+ args->flags = FUTEX_PRIVATE_FLAG;
futex = &f_private;
/* Testing a private futex */
- ksft_print_dbg_msg("Calling private futex_wait on futex: %p\n", futex);
- if (pthread_create(&waiter, NULL, waiterfn, (void *) &flags))
- ksft_exit_fail_msg("pthread_create failed\n");
+ TH_LOG("Calling private futex_wait on futex: %p", futex);
+ ASSERT_EQ(pthread_create(&waiter, NULL, waiterfn, args), 0)
+ TH_LOG("pthread_create failed");
usleep(WAKE_WAIT_US);
- ksft_print_dbg_msg("Calling private futex_wake on futex: %p\n", futex);
+ TH_LOG("Calling private futex_wake on futex: %p", futex);
res = futex_wake(futex, 1, FUTEX_PRIVATE_FLAG);
- if (res != 1) {
- ksft_test_result_fail("futex_wake private returned: %d %s\n",
- errno, strerror(errno));
- } else {
- ksft_test_result_pass("futex_wake private succeeds\n");
- }
+ EXPECT_EQ(res, 1)
+ TH_LOG("futex_wake private returned: %d %s", res, res < 0 ? strerror(errno) : "");
+
+ pthread_join(waiter, NULL);
}
TEST(anon_page)
{
+ struct waiter_args *args = malloc(sizeof(*args));
u_int32_t *shared_data;
pthread_t waiter;
int res, shm_id;
+ args->_metadata = _metadata;
+ args->flags = 0;
+
/* Testing an anon page shared memory */
shm_id = shmget(IPC_PRIVATE, 4096, IPC_CREAT | 0666);
if (shm_id < 0) {
- if (errno == ENOSYS)
- ksft_exit_skip("shmget syscall not supported\n");
- perror("shmget");
- exit(1);
+ if (errno == ENOSYS) {
+ free(args);
+ SKIP(return, "shmget syscall not supported");
+ }
+ ASSERT_GE(shm_id, 0)
+ TH_LOG("shmget failed: %s", strerror(errno));
}
shared_data = shmat(shm_id, NULL, 0);
+ if (shared_data == (void *)-1) {
+ free(args);
+ ASSERT_NE(shared_data, (void *)-1)
+ TH_LOG("shmat failed: %s", strerror(errno));
+ }
*shared_data = 0;
futex = shared_data;
- ksft_print_dbg_msg("Calling shared (page anon) futex_wait on futex: %p\n", futex);
- if (pthread_create(&waiter, NULL, waiterfn, NULL))
- ksft_exit_fail_msg("pthread_create failed\n");
+ TH_LOG("Calling shared (page anon) futex_wait on futex: %p", futex);
+ ASSERT_EQ(pthread_create(&waiter, NULL, waiterfn, args), 0)
+ TH_LOG("pthread_create failed");
usleep(WAKE_WAIT_US);
- ksft_print_dbg_msg("Calling shared (page anon) futex_wake on futex: %p\n", futex);
+ TH_LOG("Calling shared (page anon) futex_wake on futex: %p", futex);
res = futex_wake(futex, 1, 0);
- if (res != 1) {
- ksft_test_result_fail("futex_wake shared (page anon) returned: %d %s\n",
- errno, strerror(errno));
- } else {
- ksft_test_result_pass("futex_wake shared (page anon) succeeds\n");
+ EXPECT_EQ(res, 1) {
+ TH_LOG("futex_wake shared (page anon) returned: %d %s",
+ res, res < 0 ? strerror(errno) : "");
}
+ pthread_join(waiter, NULL);
shmdt(shared_data);
}
TEST(file_backed)
{
+ struct waiter_args *args = malloc(sizeof(*args));
u_int32_t f_private = 0;
pthread_t waiter;
int res, fd;
void *shm;
+ args->_metadata = _metadata;
+ args->flags = 0;
+
/* Testing a file backed shared memory */
- fd = open(SHM_PATH, O_RDWR | O_CREAT, S_IRUSR | S_IWUSR);
- if (fd < 0)
- ksft_exit_fail_msg("open\n");
+ fd = open(SHM_PATH, O_RDWR | O_CREAT, 0600);
+ if (fd < 0) {
+ free(args);
+ ASSERT_GE(fd, 0)
+ TH_LOG("open failed: %s", strerror(errno));
+ }
- if (ftruncate(fd, sizeof(f_private)))
- ksft_exit_fail_msg("ftruncate\n");
+ if (ftruncate(fd, sizeof(f_private))) {
+ free(args);
+ close(fd);
+ ASSERT_TRUE(0)
+ TH_LOG("ftruncate failed: %s", strerror(errno));
+ }
shm = mmap(NULL, sizeof(f_private), PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0);
- if (shm == MAP_FAILED)
- ksft_exit_fail_msg("mmap\n");
+ if (shm == MAP_FAILED) {
+ free(args);
+ close(fd);
+ ASSERT_NE(shm, MAP_FAILED)
+ TH_LOG("mmap failed: %s", strerror(errno));
+ }
memcpy(shm, &f_private, sizeof(f_private));
futex = shm;
- ksft_print_dbg_msg("Calling shared (file backed) futex_wait on futex: %p\n", futex);
- if (pthread_create(&waiter, NULL, waiterfn, NULL))
- ksft_exit_fail_msg("pthread_create failed\n");
+ TH_LOG("Calling shared (file backed) futex_wait on futex: %p", futex);
+ ASSERT_EQ(pthread_create(&waiter, NULL, waiterfn, args), 0)
+ TH_LOG("pthread_create failed");
usleep(WAKE_WAIT_US);
- ksft_print_dbg_msg("Calling shared (file backed) futex_wake on futex: %p\n", futex);
+ TH_LOG("Calling shared (file backed) futex_wake on futex: %p", futex);
res = futex_wake(shm, 1, 0);
- if (res != 1) {
- ksft_test_result_fail("futex_wake shared (file backed) returned: %d %s\n",
- errno, strerror(errno));
- } else {
- ksft_test_result_pass("futex_wake shared (file backed) succeeds\n");
+ EXPECT_EQ(res, 1) {
+ TH_LOG("futex_wake shared (file backed) returned: %d %s",
+ res, res < 0 ? strerror(errno) : "");
}
+ pthread_join(waiter, NULL);
munmap(shm, sizeof(f_private));
remove(SHM_PATH);
close(fd);
diff --git a/tools/testing/selftests/futex/functional/futex_wait_private_mapped_file.c b/tools/testing/selftests/futex/functional/futex_wait_private_mapped_file.c
index 2a749f9b14eb..a9f7a02e3a0b 100644
--- a/tools/testing/selftests/futex/functional/futex_wait_private_mapped_file.c
+++ b/tools/testing/selftests/futex/functional/futex_wait_private_mapped_file.c
@@ -26,6 +26,7 @@
#include <pthread.h>
#include <libgen.h>
#include <signal.h>
+#include <string.h>
#include "futextest.h"
#include "kselftest_harness.h"
@@ -41,17 +42,22 @@ struct timespec wait_timeout = { .tv_sec = 5, .tv_nsec = 0};
void *thr_futex_wait(void *arg)
{
+ struct __test_metadata *_metadata = (struct __test_metadata *)arg;
int ret;
- ksft_print_dbg_msg("futex wait\n");
+ TH_LOG("futex wait");
ret = futex_wait(&val, 1, &wait_timeout, 0);
- if (ret && errno != EWOULDBLOCK && errno != ETIMEDOUT)
- ksft_exit_fail_msg("futex error.\n");
+ if (ret && errno != EWOULDBLOCK && errno != ETIMEDOUT) {
+ ASSERT_TRUE(0)
+ TH_LOG("futex error: %s", strerror(errno));
+ }
- if (ret && errno == ETIMEDOUT)
- ksft_exit_fail_msg("waiter timedout\n");
+ if (ret && errno == ETIMEDOUT) {
+ ASSERT_TRUE(0)
+ TH_LOG("waiter timedout");
+ }
- ksft_print_dbg_msg("futex_wait: ret = %d, errno = %d\n", ret, errno);
+ TH_LOG("futex_wait: ret = %d, errno = %d", ret, errno);
return NULL;
}
@@ -61,22 +67,20 @@ TEST(wait_private_mapped_file)
pthread_t thr;
int res;
- res = pthread_create(&thr, NULL, thr_futex_wait, NULL);
- if (res < 0)
- ksft_exit_fail_msg("pthread_create error\n");
+ res = pthread_create(&thr, NULL, thr_futex_wait, _metadata);
+ ASSERT_EQ(res, 0)
+ TH_LOG("pthread_create error");
- ksft_print_dbg_msg("wait a while\n");
+ TH_LOG("wait a while");
usleep(WAKE_WAIT_US);
val = 2;
res = futex_wake(&val, 1, 0);
- ksft_print_dbg_msg("futex_wake %d\n", res);
- if (res != 1)
- ksft_exit_fail_msg("FUTEX_WAKE didn't find the waiting thread.\n");
+ TH_LOG("futex_wake %d", res);
+ EXPECT_EQ(res, 1)
+ TH_LOG("FUTEX_WAKE didn't find the waiting thread");
- ksft_print_dbg_msg("join\n");
+ TH_LOG("join");
pthread_join(thr, NULL);
-
- ksft_test_result_pass("wait_private_mapped_file");
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_wait_timeout.c b/tools/testing/selftests/futex/functional/futex_wait_timeout.c
index 674dd13af421..6e6e770fe96a 100644
--- a/tools/testing/selftests/futex/functional/futex_wait_timeout.c
+++ b/tools/testing/selftests/futex/functional/futex_wait_timeout.c
@@ -31,53 +31,47 @@ static pthread_barrier_t barrier;
*/
void *get_pi_lock(void *arg)
{
+ struct __test_metadata *_metadata = (struct __test_metadata *)arg;
int ret;
volatile futex_t lock = 0;
ret = futex_lock_pi(&futex_pi, NULL, 0, 0);
- if (ret != 0)
- ksft_exit_fail_msg("futex_lock_pi failed\n");
+ ASSERT_EQ(ret, 0)
+ TH_LOG("futex_lock_pi failed");
pthread_barrier_wait(&barrier);
/* Blocks forever */
ret = futex_wait(&lock, 0, NULL, 0);
- ksft_exit_fail_msg("futex_wait failed\n");
+ ASSERT_TRUE(0)
+ TH_LOG("futex_wait returned unexpectedly: %d", ret);
return NULL;
}
-/*
- * Check if the function returned the expected error
- */
-static void test_timeout(int res, char *test_name, int err)
-{
- if (!res || errno != err) {
- ksft_test_result_fail("%s returned %d\n", test_name,
- res < 0 ? errno : res);
- } else {
- ksft_test_result_pass("%s succeeds\n", test_name);
- }
-}
-
-/*
- * Calculate absolute timeout and correct overflow
- */
-static int futex_get_abs_timeout(clockid_t clockid, struct timespec *to,
- long timeout_ns)
-{
- if (clock_gettime(clockid, to))
- ksft_exit_fail_msg("clock_gettime failed\n");
-
- to->tv_nsec += timeout_ns;
-
- if (to->tv_nsec >= 1000000000) {
- to->tv_sec++;
- to->tv_nsec -= 1000000000;
- }
-
- return 0;
-}
+#define TEST_TIMEOUT(_res, _test_name, _err) do { \
+ if ((_res) < 0 && errno == ENOSYS && (_err) != ENOSYS) { \
+ SKIP(return, "%s is not supported (ENOSYS)", _test_name); \
+ } \
+ EXPECT_EQ((_res), -1) \
+ TH_LOG("%s returned unexpected result: %d", _test_name, (_res));\
+ if ((_res) == -1) { \
+ EXPECT_EQ(errno, (_err)) { \
+ TH_LOG("%s returned unexpected errno: %d (expected %d)",\
+ _test_name, errno, (_err)); \
+ } \
+ } \
+} while (0)
+
+#define GET_ABS_TIMEOUT(_clockid, _to, _timeout_ns) do { \
+ ASSERT_EQ(clock_gettime((_clockid), (_to)), 0) \
+ TH_LOG("clock_gettime failed"); \
+ (_to)->tv_nsec += (_timeout_ns); \
+ if ((_to)->tv_nsec >= 1000000000) { \
+ (_to)->tv_sec++; \
+ (_to)->tv_nsec -= 1000000000; \
+ } \
+} while (0)
TEST(wait_bitset)
{
@@ -90,19 +84,17 @@ TEST(wait_bitset)
to.tv_nsec = timeout_ns;
res = futex_wait(&f1, f1, &to, 0);
- test_timeout(res, "futex_wait relative", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_wait relative", ETIMEDOUT);
/* FUTEX_WAIT_BITSET with CLOCK_REALTIME */
- if (futex_get_abs_timeout(CLOCK_REALTIME, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_REALTIME, &to, timeout_ns);
res = futex_wait_bitset(&f1, f1, &to, 1, FUTEX_CLOCK_REALTIME);
- test_timeout(res, "futex_wait_bitset realtime", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_wait_bitset realtime", ETIMEDOUT);
/* FUTEX_WAIT_BITSET with CLOCK_MONOTONIC */
- if (futex_get_abs_timeout(CLOCK_MONOTONIC, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_MONOTONIC, &to, timeout_ns);
res = futex_wait_bitset(&f1, f1, &to, 1, 0);
- test_timeout(res, "futex_wait_bitset monotonic", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_wait_bitset monotonic", ETIMEDOUT);
}
TEST(requeue_pi)
@@ -112,17 +104,14 @@ TEST(requeue_pi)
int res;
/* FUTEX_WAIT_REQUEUE_PI with CLOCK_REALTIME */
- if (futex_get_abs_timeout(CLOCK_REALTIME, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_REALTIME, &to, timeout_ns);
res = futex_wait_requeue_pi(&f1, f1, &futex_pi, &to, FUTEX_CLOCK_REALTIME);
- test_timeout(res, "futex_wait_requeue_pi realtime", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_wait_requeue_pi realtime", ETIMEDOUT);
/* FUTEX_WAIT_REQUEUE_PI with CLOCK_MONOTONIC */
- if (futex_get_abs_timeout(CLOCK_MONOTONIC, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_MONOTONIC, &to, timeout_ns);
res = futex_wait_requeue_pi(&f1, f1, &futex_pi, &to, 0);
- test_timeout(res, "futex_wait_requeue_pi monotonic", ETIMEDOUT);
-
+ TEST_TIMEOUT(res, "futex_wait_requeue_pi monotonic", ETIMEDOUT);
}
TEST(lock_pi)
@@ -133,7 +122,8 @@ TEST(lock_pi)
/* Create a thread that will lock forever so any waiter will timeout */
pthread_barrier_init(&barrier, NULL, 2);
- pthread_create(&thread, NULL, get_pi_lock, NULL);
+ ASSERT_EQ(pthread_create(&thread, NULL, get_pi_lock, _metadata), 0)
+ TH_LOG("pthread_create failed");
/* Wait until the other thread calls futex_lock_pi() */
pthread_barrier_wait(&barrier);
@@ -149,14 +139,13 @@ TEST(lock_pi)
* time or your time machine) the monotonic clock value is always
* smaller than realtime and the syscall will timeout immediately.
*/
- if (futex_get_abs_timeout(CLOCK_REALTIME, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_REALTIME, &to, timeout_ns);
res = futex_lock_pi(&futex_pi, &to, 0, 0);
- test_timeout(res, "futex_lock_pi realtime", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_lock_pi realtime", ETIMEDOUT);
/* Test operations that don't support FUTEX_CLOCK_REALTIME */
res = futex_lock_pi(&futex_pi, NULL, 0, FUTEX_CLOCK_REALTIME);
- test_timeout(res, "futex_lock_pi invalid timeout flag", ENOSYS);
+ TEST_TIMEOUT(res, "futex_lock_pi invalid timeout flag", ENOSYS);
}
TEST(waitv)
@@ -171,17 +160,18 @@ TEST(waitv)
struct timespec to;
int res;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
/* futex_waitv with CLOCK_MONOTONIC */
- if (futex_get_abs_timeout(CLOCK_MONOTONIC, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_MONOTONIC, &to, timeout_ns);
res = futex_waitv(&waitv, 1, 0, &to, CLOCK_MONOTONIC);
- test_timeout(res, "futex_waitv monotonic", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_waitv monotonic", ETIMEDOUT);
/* futex_waitv with CLOCK_REALTIME */
- if (futex_get_abs_timeout(CLOCK_REALTIME, &to, timeout_ns))
- ksft_test_result_error("get_time error");
+ GET_ABS_TIMEOUT(CLOCK_REALTIME, &to, timeout_ns);
res = futex_waitv(&waitv, 1, 0, &to, CLOCK_REALTIME);
- test_timeout(res, "futex_waitv realtime", ETIMEDOUT);
+ TEST_TIMEOUT(res, "futex_waitv realtime", ETIMEDOUT);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_wait_uninitialized_heap.c b/tools/testing/selftests/futex/functional/futex_wait_uninitialized_heap.c
index b07d68a67f31..bbffc23e0006 100644
--- a/tools/testing/selftests/futex/functional/futex_wait_uninitialized_heap.c
+++ b/tools/testing/selftests/futex/functional/futex_wait_uninitialized_heap.c
@@ -17,17 +17,18 @@
*
*****************************************************************************/
+#include <errno.h>
+#include <libgen.h>
#include <pthread.h>
#include <stdio.h>
#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <linux/futex.h>
#include <sys/mman.h>
#include <syscall.h>
#include <sys/types.h>
#include <sys/stat.h>
-#include <unistd.h>
-#include <errno.h>
-#include <linux/futex.h>
-#include <libgen.h>
#include "futextest.h"
#include "kselftest_harness.h"
@@ -40,6 +41,7 @@ void *buf;
void *wait_thread(void *arg)
{
+ struct __test_metadata *_metadata = (struct __test_metadata *)arg;
int res;
child_ret = true;
@@ -47,7 +49,8 @@ void *wait_thread(void *arg)
child_blocked = 0;
if (res != 0 && errno != EWOULDBLOCK) {
- ksft_exit_fail_msg("futex failure\n");
+ EXPECT_EQ(res, 0)
+ TH_LOG("futex failure: %s", strerror(errno));
child_ret = false;
}
pthread_exit(NULL);
@@ -63,21 +66,23 @@ TEST(futex_wait_uninitialized_heap)
buf = mmap(NULL, page_size, PROT_READ|PROT_WRITE,
MAP_PRIVATE|MAP_ANONYMOUS, 0, 0);
- if (buf == (void *)-1)
- ksft_exit_fail_msg("mmap\n");
+ ASSERT_NE(buf, MAP_FAILED)
+ TH_LOG("mmap failed: %s", strerror(errno));
- ret = pthread_create(&thr, NULL, wait_thread, NULL);
- if (ret)
- ksft_exit_fail_msg("pthread_create\n");
+ ret = pthread_create(&thr, NULL, wait_thread, _metadata);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("pthread_create failed");
- ksft_print_dbg_msg("waiting %dus for child to return\n", WAIT_US);
+ TH_LOG("waiting %dus for child to return", WAIT_US);
usleep(WAIT_US);
- if (child_blocked)
- ksft_test_result_fail("child blocked in kernel\n");
+ EXPECT_EQ(child_blocked, 0)
+ TH_LOG("child blocked in kernel");
+ EXPECT_TRUE(child_ret)
+ TH_LOG("child error");
- if (!child_ret)
- ksft_test_result_fail("child error\n");
+ pthread_join(thr, NULL);
+ munmap(buf, page_size);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/futex/functional/futex_wait_wouldblock.c b/tools/testing/selftests/futex/functional/futex_wait_wouldblock.c
index 9ff936ecf164..ab039dda3e51 100644
--- a/tools/testing/selftests/futex/functional/futex_wait_wouldblock.c
+++ b/tools/testing/selftests/futex/functional/futex_wait_wouldblock.c
@@ -28,20 +28,20 @@
#define timeout_ns 100000
+
TEST(futex_wait_wouldblock)
{
struct timespec to = {.tv_sec = 0, .tv_nsec = timeout_ns};
futex_t f1 = FUTEX_INITIALIZER;
int res;
- ksft_print_dbg_msg("Calling futex_wait on f1: %u @ %p with val=%u\n", f1, &f1, f1+1);
+ TH_LOG("Calling futex_wait on f1: %u @ %p with val=%u", f1, &f1, f1+1);
res = futex_wait(&f1, f1+1, &to, FUTEX_PRIVATE_FLAG);
- if (!res || errno != EWOULDBLOCK) {
- ksft_test_result_fail("futex_wait returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_wait\n");
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_wait returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EWOULDBLOCK)
+ TH_LOG("futex_wait returned unexpected errno: %d", errno);
}
}
@@ -57,8 +57,11 @@ TEST(futex_waitv_wouldblock)
};
int res;
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("clock_gettime failed %d\n", errno);
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("clock_gettime failed");
to.tv_nsec += timeout_ns;
@@ -67,14 +70,13 @@ TEST(futex_waitv_wouldblock)
to.tv_nsec -= 1000000000;
}
- ksft_print_dbg_msg("Calling futex_waitv on f1: %u @ %p with val=%u\n", f1, &f1, f1+1);
+ TH_LOG("Calling futex_waitv on f1: %u @ %p with val=%u", f1, &f1, f1+1);
res = futex_waitv(&waitv, 1, 0, &to, CLOCK_MONOTONIC);
- if (!res || errno != EWOULDBLOCK) {
- ksft_test_result_fail("futex_waitv returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv\n");
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_waitv returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EWOULDBLOCK)
+ TH_LOG("futex_waitv returned unexpected errno: %d", errno);
}
}
diff --git a/tools/testing/selftests/futex/functional/futex_waitv.c b/tools/testing/selftests/futex/functional/futex_waitv.c
index b5ada9fdb26f..4858d5faeecf 100644
--- a/tools/testing/selftests/futex/functional/futex_waitv.c
+++ b/tools/testing/selftests/futex/functional/futex_waitv.c
@@ -25,24 +25,26 @@
static struct futex_waitv waitv[NR_FUTEXES];
u_int32_t futexes[NR_FUTEXES] = {0};
+
void *waiterfn(void *arg)
{
+ struct __test_metadata *_metadata = (struct __test_metadata *)arg;
struct timespec to;
int res;
/* setting absolute timeout for futex2 */
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("gettime64 failed\n");
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("gettime64 failed");
to.tv_sec++;
res = futex_waitv(waitv, NR_FUTEXES, 0, &to, CLOCK_MONOTONIC);
if (res < 0) {
- ksft_test_result_fail("futex_waitv returned: %d %s\n",
- errno, strerror(errno));
- } else if (res != NR_FUTEXES - 1) {
- ksft_test_result_fail("futex_waitv returned: %d, expecting %d\n",
- res, NR_FUTEXES - 1);
+ EXPECT_EQ(res, NR_FUTEXES - 1)
+ TH_LOG("futex_waitv failed: %s", strerror(errno));
+ } else {
+ EXPECT_EQ(res, NR_FUTEXES - 1)
+ TH_LOG("futex_waitv returned %d, expected %d", res, NR_FUTEXES - 1);
}
return NULL;
@@ -53,6 +55,9 @@ TEST(private_waitv)
pthread_t waiter;
int res, i;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
for (i = 0; i < NR_FUTEXES; i++) {
waitv[i].uaddr = (uintptr_t)&futexes[i];
waitv[i].flags = FUTEX_32 | FUTEX_PRIVATE_FLAG;
@@ -61,19 +66,14 @@ TEST(private_waitv)
}
/* Private waitv */
- if (pthread_create(&waiter, NULL, waiterfn, NULL))
- ksft_exit_fail_msg("pthread_create failed\n");
+ ASSERT_EQ(pthread_create(&waiter, NULL, waiterfn, _metadata), 0)
+ TH_LOG("pthread_create failed");
usleep(WAKE_WAIT_US);
res = futex_wake(u64_to_ptr(waitv[NR_FUTEXES - 1].uaddr), 1, FUTEX_PRIVATE_FLAG);
- if (res != 1) {
- ksft_test_result_fail("futex_wake private returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv private\n");
- }
+ EXPECT_EQ(res, 1)
+ TH_LOG("futex_wake private returned: %d %s", res, res < 0 ? strerror(errno) : "");
}
TEST(shared_waitv)
@@ -81,15 +81,18 @@ TEST(shared_waitv)
pthread_t waiter;
int res, i;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
/* Shared waitv */
for (i = 0; i < NR_FUTEXES; i++) {
int shm_id = shmget(IPC_PRIVATE, 4096, IPC_CREAT | 0666);
if (shm_id < 0) {
if (errno == ENOSYS)
- ksft_exit_skip("shmget syscall not supported\n");
- perror("shmget");
- exit(1);
+ SKIP(return, "shmget syscall not supported");
+ ASSERT_GE(shm_id, 0)
+ TH_LOG("shmget failed");
}
unsigned int *shared_data = shmat(shm_id, NULL, 0);
@@ -101,19 +104,14 @@ TEST(shared_waitv)
waitv[i].__reserved = 0;
}
- if (pthread_create(&waiter, NULL, waiterfn, NULL))
- ksft_exit_fail_msg("pthread_create failed\n");
+ ASSERT_EQ(pthread_create(&waiter, NULL, waiterfn, _metadata), 0)
+ TH_LOG("pthread_create failed");
usleep(WAKE_WAIT_US);
res = futex_wake(u64_to_ptr(waitv[NR_FUTEXES - 1].uaddr), 1, 0);
- if (res != 1) {
- ksft_test_result_fail("futex_wake shared returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv shared\n");
- }
+ EXPECT_EQ(res, 1)
+ TH_LOG("futex_wake shared returned: %d %s", res, res < 0 ? strerror(errno) : "");
for (i = 0; i < NR_FUTEXES; i++)
shmdt(u64_to_ptr(waitv[i].uaddr));
@@ -124,21 +122,24 @@ TEST(invalid_flag)
struct timespec to;
int res;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
/* Testing a waiter without FUTEX_32 flag */
waitv[0].flags = FUTEX_PRIVATE_FLAG;
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("gettime64 failed\n");
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("gettime64 failed");
to.tv_sec++;
res = futex_waitv(waitv, NR_FUTEXES, 0, &to, CLOCK_MONOTONIC);
- if (res == EINVAL) {
- ksft_test_result_fail("futex_waitv private returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv without FUTEX_32\n");
+
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_waitv returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EINVAL)
+ TH_LOG("futex_waitv returned unexpected errno: %d", errno);
}
}
@@ -147,22 +148,25 @@ TEST(unaligned_address)
struct timespec to;
int res;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
/* Testing a waiter with an unaligned address */
waitv[0].flags = FUTEX_PRIVATE_FLAG | FUTEX_32;
waitv[0].uaddr = 1;
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("gettime64 failed\n");
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("gettime64 failed");
to.tv_sec++;
res = futex_waitv(waitv, NR_FUTEXES, 0, &to, CLOCK_MONOTONIC);
- if (res == EINVAL) {
- ksft_test_result_fail("futex_wake private returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv with an unaligned address\n");
+
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_waitv returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EINVAL)
+ TH_LOG("futex_waitv returned unexpected errno: %d", errno);
}
}
@@ -171,36 +175,39 @@ TEST(null_address)
struct timespec to;
int res;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
/* Testing a NULL address for waiters.uaddr */
waitv[0].uaddr = 0x00000000;
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("gettime64 failed\n");
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("gettime64 failed");
to.tv_sec++;
res = futex_waitv(waitv, NR_FUTEXES, 0, &to, CLOCK_MONOTONIC);
- if (res == EINVAL) {
- ksft_test_result_fail("futex_waitv private returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv NULL address in waitv.uaddr\n");
+
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_waitv returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EINVAL)
+ TH_LOG("futex_waitv returned unexpected errno: %d", errno);
}
/* Testing a NULL address for *waiters */
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("gettime64 failed\n");
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("gettime64 failed");
to.tv_sec++;
res = futex_waitv(NULL, NR_FUTEXES, 0, &to, CLOCK_MONOTONIC);
- if (res == EINVAL) {
- ksft_test_result_fail("futex_waitv private returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv NULL address in *waiters\n");
+
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_waitv returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EINVAL)
+ TH_LOG("futex_waitv returned unexpected errno: %d", errno);
}
}
@@ -209,19 +216,22 @@ TEST(invalid_clockid)
struct timespec to;
int res;
+ if (!is_futex_waitv_supported())
+ SKIP(return, "futex_waitv syscall not supported");
+
/* Testing an invalid clockid */
- if (clock_gettime(CLOCK_MONOTONIC, &to))
- ksft_exit_fail_msg("gettime64 failed\n");
+ ASSERT_EQ(clock_gettime(CLOCK_MONOTONIC, &to), 0)
+ TH_LOG("gettime64 failed");
to.tv_sec++;
res = futex_waitv(NULL, NR_FUTEXES, 0, &to, CLOCK_TAI);
- if (res == EINVAL) {
- ksft_test_result_fail("futex_waitv private returned: %d %s\n",
- res ? errno : res,
- res ? strerror(errno) : "");
- } else {
- ksft_test_result_pass("futex_waitv invalid clockid\n");
+
+ EXPECT_EQ(res, -1)
+ TH_LOG("futex_waitv returned unexpected result: %d", res);
+ if (res == -1) {
+ EXPECT_EQ(errno, EINVAL)
+ TH_LOG("futex_waitv returned unexpected errno: %d", errno);
}
}
diff --git a/tools/testing/selftests/futex/functional/robust_list.c b/tools/testing/selftests/futex/functional/robust_list.c
index b3fab60181d5..87217c549361 100644
--- a/tools/testing/selftests/futex/functional/robust_list.c
+++ b/tools/testing/selftests/futex/functional/robust_list.c
@@ -25,24 +25,24 @@
#define _GNU_SOURCE
#include "futextest.h"
-#include "../../kselftest_harness.h"
+#include "kselftest_harness.h"
#include <dlfcn.h>
#include <errno.h>
#include <pthread.h>
#include <signal.h>
-#include <stdint.h>
#include <stdatomic.h>
#include <stdbool.h>
#include <stddef.h>
+#include <stdint.h>
+#include <stdlib.h>
+#include <string.h>
#include <sys/auxv.h>
#include <sys/mman.h>
#include <sys/wait.h>
#define STACK_SIZE (1024 * 1024)
-
#define FUTEX_TIMEOUT 3
-
#define SLEEP_US 100
#if __SIZEOF_LONG__ == 8
@@ -71,30 +71,46 @@ static int sys_futex_robust_unlock(_Atomic(uint32_t) *uaddr, unsigned int op, in
* Basic lock struct, contains just the futex word and the robust list element
* Real implementations have also a *prev to easily walk in the list
*/
+typedef _Atomic(unsigned int) atomic_futex_t;
+
struct lock_struct {
- _Atomic(unsigned int) futex;
+ atomic_futex_t futex;
struct robust_list list;
};
+struct child_args {
+ struct __test_metadata *_metadata;
+ void *arg;
+};
+
/*
* Helper function to spawn a child thread. Returns -1 on error, pid on success
*/
-static int create_child(int (*fn)(void *arg), void *arg)
+static int create_child(struct __test_metadata *_metadata, int (*fn)(void *arg), void *arg)
{
+ struct child_args *cargs = malloc(sizeof(*cargs));
char *stack;
pid_t pid;
+ if (!cargs)
+ return -1;
+ cargs->_metadata = _metadata;
+ cargs->arg = arg;
+
stack = mmap(NULL, STACK_SIZE, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK, -1, 0);
- if (stack == MAP_FAILED)
+ if (stack == MAP_FAILED) {
+ free(cargs);
return -1;
+ }
stack += STACK_SIZE;
- pid = clone(fn, stack, CLONE_VM | SIGCHLD, arg);
-
- if (pid == -1)
+ pid = clone(fn, stack, CLONE_VM | SIGCHLD, cargs);
+ if (pid == -1) {
+ free(cargs);
return -1;
+ }
return pid;
}
@@ -123,7 +139,7 @@ static int set_list(struct robust_list_head *head)
*/
static int mutex_lock(struct lock_struct *lock, struct robust_list_head *head, bool error_inject)
{
- _Atomic(unsigned int) *futex = &lock->futex;
+ atomic_futex_t *futex = &lock->futex;
unsigned int zero = 0;
pid_t tid = gettid();
int ret = -1;
@@ -183,21 +199,21 @@ static int mutex_lock(struct lock_struct *lock, struct robust_list_head *head, b
*/
static int child_fn_lock(void *arg)
{
- struct lock_struct *lock = arg;
+ struct child_args *cargs = arg;
+ struct __test_metadata *_metadata = cargs->_metadata;
+ struct lock_struct *lock = cargs->arg;
struct robust_list_head head;
int ret;
+ free(cargs);
+
ret = set_list(&head);
- if (ret) {
- ksft_test_result_fail("set_robust_list error\n");
- return ret;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("set_robust_list error");
ret = mutex_lock(lock, &head, false);
- if (ret) {
- ksft_test_result_fail("mutex_lock error\n");
- return ret;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("mutex_lock error");
pthread_barrier_wait(&barrier);
@@ -220,7 +236,7 @@ static int child_fn_lock(void *arg)
TEST(test_robustness)
{
struct lock_struct lock = { .futex = 0 };
- _Atomic(unsigned int) *futex = &lock.futex;
+ atomic_futex_t *futex = &lock.futex;
struct robust_list_head head;
int ret, pid, wstatus;
@@ -234,7 +250,7 @@ TEST(test_robustness)
ret = pthread_barrier_init(&barrier, NULL, 2);
ASSERT_EQ(ret, 0);
- pid = create_child(&child_fn_lock, &lock);
+ pid = create_child(_metadata, &child_fn_lock, &lock);
ASSERT_NE(pid, -1);
pthread_barrier_wait(&barrier);
@@ -251,9 +267,8 @@ TEST(test_robustness)
wait(&wstatus);
pthread_barrier_destroy(&barrier);
- /* Pass only if the child hasn't return error */
- if (!WEXITSTATUS(wstatus))
- ksft_test_result_pass("%s\n", __func__);
+ EXPECT_EQ(WEXITSTATUS(wstatus), 0)
+ TH_LOG("child failed");
}
/*
@@ -279,8 +294,6 @@ TEST(test_set_robust_list_invalid_size)
ret = set_robust_list(&head, 0);
ASSERT_EQ(ret, -1);
ASSERT_EQ(errno, EINVAL);
-
- ksft_test_result_pass("%s\n", __func__);
}
/*
@@ -307,20 +320,20 @@ TEST(test_get_robust_list_self)
ASSERT_EQ(ret, 0);
ASSERT_EQ(get_head, &head2);
ASSERT_EQ(head_size, len_ptr);
-
- ksft_test_result_pass("%s\n", __func__);
}
static int child_list(void *arg)
{
- struct robust_list_head *head = arg;
+ struct child_args *cargs = arg;
+ struct __test_metadata *_metadata = cargs->_metadata;
+ struct robust_list_head *head = cargs->arg;
int ret;
+ free(cargs);
+
ret = set_robust_list(head, sizeof(*head));
- if (ret) {
- ksft_test_result_fail("set_robust_list error\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("set_robust_list error");
/*
* After setting the list head, wait until the main thread can call
@@ -350,7 +363,7 @@ TEST(test_get_robust_list_child)
ret = pthread_barrier_init(&barrier2, NULL, 2);
ASSERT_EQ(ret, 0);
- tid = create_child(&child_list, &head);
+ tid = create_child(_metadata, &child_list, &head);
ASSERT_NE(tid, -1);
pthread_barrier_wait(&barrier);
@@ -365,28 +378,27 @@ TEST(test_get_robust_list_child)
pthread_barrier_destroy(&barrier);
pthread_barrier_destroy(&barrier2);
- /* Pass only if the child hasn't return error */
- if (!WEXITSTATUS(wstatus))
- ksft_test_result_pass("%s\n", __func__);
+ EXPECT_EQ(WEXITSTATUS(wstatus), 0)
+ TH_LOG("child failed");
}
static int child_fn_lock_with_error(void *arg)
{
- struct lock_struct *lock = arg;
+ struct child_args *cargs = arg;
+ struct __test_metadata *_metadata = cargs->_metadata;
+ struct lock_struct *lock = cargs->arg;
struct robust_list_head head;
int ret;
+ free(cargs);
+
ret = set_list(&head);
- if (ret) {
- ksft_test_result_fail("set_robust_list error\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("set_robust_list error");
ret = mutex_lock(lock, &head, true);
- if (ret) {
- ksft_test_result_fail("mutex_lock error\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("mutex_lock error");
pthread_barrier_wait(&barrier);
@@ -404,7 +416,7 @@ static int child_fn_lock_with_error(void *arg)
TEST(test_set_list_op_pending)
{
struct lock_struct lock = { .futex = 0 };
- _Atomic(unsigned int) *futex = &lock.futex;
+ atomic_futex_t *futex = &lock.futex;
struct robust_list_head head;
int ret, wstatus;
@@ -414,7 +426,7 @@ TEST(test_set_list_op_pending)
ret = pthread_barrier_init(&barrier, NULL, 2);
ASSERT_EQ(ret, 0);
- ret = create_child(&child_fn_lock_with_error, &lock);
+ ret = create_child(_metadata, &child_fn_lock_with_error, &lock);
ASSERT_NE(ret, -1);
pthread_barrier_wait(&barrier);
@@ -427,21 +439,21 @@ TEST(test_set_list_op_pending)
wait(&wstatus);
pthread_barrier_destroy(&barrier);
- /* Pass only if the child hasn't return error */
- if (!WEXITSTATUS(wstatus))
- ksft_test_result_pass("%s\n", __func__);
- else
- ksft_test_result_fail("%s\n", __func__);
+ EXPECT_EQ(WEXITSTATUS(wstatus), 0)
+ TH_LOG("child failed");
}
#define CHILD_NR 10
static int child_lock_holder(void *arg)
{
- struct lock_struct *locks = arg;
+ struct child_args *cargs = arg;
+ struct lock_struct *locks = cargs->arg;
struct robust_list_head head;
int i;
+ free(cargs);
+
set_list(&head);
for (i = 0; i < CHILD_NR; i++) {
@@ -460,22 +472,21 @@ static int child_lock_holder(void *arg)
static int child_wait_lock(void *arg)
{
- struct lock_struct *lock = arg;
+ struct child_args *cargs = arg;
+ struct __test_metadata *_metadata = cargs->_metadata;
+ struct lock_struct *lock = cargs->arg;
struct robust_list_head head;
int ret;
+ free(cargs);
+
pthread_barrier_wait(&barrier2);
ret = mutex_lock(lock, &head, false);
+ ASSERT_EQ(ret, 0)
+ TH_LOG("mutex_lock error");
- if (ret) {
- ksft_test_result_fail("mutex_lock error\n");
- return -1;
- }
-
- if (!(lock->futex & FUTEX_OWNER_DIED)) {
- ksft_test_result_fail("futex not marked with FUTEX_OWNER_DIED\n");
- return -1;
- }
+ ASSERT_TRUE(lock->futex & FUTEX_OWNER_DIED)
+ TH_LOG("futex not marked with FUTEX_OWNER_DIED");
return 0;
}
@@ -495,18 +506,20 @@ TEST(test_robust_list_multiple_elements)
ret = pthread_barrier_init(&barrier2, NULL, CHILD_NR + 1);
ASSERT_EQ(ret, 0);
- pids[0] = create_child(&child_lock_holder, &locks);
+ pids[0] = create_child(_metadata, &child_lock_holder, &locks);
+ ASSERT_NE(pids[0], -1);
/* Wait until the locker thread takes the look */
pthread_barrier_wait(&barrier);
- for (i = 0; i < CHILD_NR; i++)
- pids[i+1] = create_child(&child_wait_lock, &locks[i]);
+ for (i = 0; i < CHILD_NR; i++) {
+ pids[i+1] = create_child(_metadata, &child_wait_lock, &locks[i]);
+ ASSERT_NE(pids[i+1], -1);
+ }
- /* Wait for all children to return */
+ /* Wait for all children to return (holder + all waiters) */
ret = 0;
-
- for (i = 0; i < CHILD_NR; i++) {
+ for (i = 0; i < CHILD_NR + 1; i++) {
waitpid(pids[i], &wstatus, 0);
if (WEXITSTATUS(wstatus))
ret = -1;
@@ -515,22 +528,23 @@ TEST(test_robust_list_multiple_elements)
pthread_barrier_destroy(&barrier);
pthread_barrier_destroy(&barrier2);
- /* Pass only if the child hasn't return error */
- if (!ret)
- ksft_test_result_pass("%s\n", __func__);
+ EXPECT_EQ(ret, 0)
+ TH_LOG("One or more children failed");
}
static int child_circular_list(void *arg)
{
- static struct robust_list_head head;
- struct lock_struct a, b, c;
+ struct child_args *cargs = arg;
+ struct __test_metadata *_metadata = cargs->_metadata;
+ static struct lock_struct a, b, c;
+ struct robust_list_head head;
int ret;
+ free(cargs);
+
ret = set_list(&head);
- if (ret) {
- ksft_test_result_fail("set_list error\n");
- return -1;
- }
+ ASSERT_EQ(ret, 0)
+ TH_LOG("set_list error");
head.list.next = &a.list;
@@ -552,14 +566,15 @@ static int child_circular_list(void *arg)
TEST(test_circular_list)
{
int wstatus;
+ pid_t pid;
- create_child(child_circular_list, NULL);
+ pid = create_child(_metadata, child_circular_list, NULL);
+ ASSERT_NE(pid, -1);
wait(&wstatus);
- /* Pass only if the child hasn't return error */
- if (!WEXITSTATUS(wstatus))
- ksft_test_result_pass("%s\n", __func__);
+ EXPECT_EQ(WEXITSTATUS(wstatus), 0)
+ TH_LOG("child failed");
}
/*
diff --git a/tools/testing/selftests/futex/functional/run.sh b/tools/testing/selftests/futex/functional/run.sh
index e88545c06d57..d1a681b798bd 100755
--- a/tools/testing/selftests/futex/functional/run.sh
+++ b/tools/testing/selftests/futex/functional/run.sh
@@ -51,3 +51,6 @@ echo
echo
./futex_numa_mpol
+
+echo
+./futex_lock_pi_exiting
diff --git a/tools/testing/selftests/futex/include/futex2test.h b/tools/testing/selftests/futex/include/futex2test.h
index 1f625b39948a..53e88b60ac6d 100644
--- a/tools/testing/selftests/futex/include/futex2test.h
+++ b/tools/testing/selftests/futex/include/futex2test.h
@@ -5,7 +5,9 @@
* Copyright 2021 Collabora Ltd.
*/
#include <linux/time_types.h>
+#include <errno.h>
#include <stdint.h>
+#include <stdbool.h>
#define u64_to_ptr(x) ((void *)(uintptr_t)(x))
@@ -96,3 +98,11 @@ static inline int futex2_wake(void *uaddr, int nr, unsigned int flags)
{
return syscall(__NR_futex_wake, uaddr, ~0U, nr, flags);
}
+
+static inline bool is_futex_waitv_supported(void)
+{
+ struct timespec ts = {0, 0};
+ int res = futex_waitv(NULL, 0, 0, &ts, CLOCK_MONOTONIC);
+
+ return !(res < 0 && errno == ENOSYS);
+}
diff --git a/tools/testing/selftests/futex/include/futex_thread.h b/tools/testing/selftests/futex/include/futex_thread.h
new file mode 100644
index 000000000000..a90882960264
--- /dev/null
+++ b/tools/testing/selftests/futex/include/futex_thread.h
@@ -0,0 +1,117 @@
+/* SPDX-License-Identifier: GPL-2.0-or-later */
+
+#ifndef _FUTEX_THREAD_H
+#define _FUTEX_THREAD_H
+#include <errno.h>
+#include <pthread.h>
+#include <stdio.h>
+#include <string.h>
+#include <unistd.h>
+
+#include "kselftest_harness.h"
+
+#define USEC_PER_SEC 1000000L
+#define WAIT_FOR_THREAD_SECS 1
+#define WAIT_FOR_THREAD_USECS (WAIT_FOR_THREAD_SECS * USEC_PER_SEC)
+#define WAIT_THREAD_RETRIES 100
+
+struct futex_thread {
+ pthread_t thread;
+ pthread_barrier_t barrier;
+ pid_t tid;
+ int (*threadfn)(void *arg);
+ void *arg;
+ int retval;
+};
+
+static inline int __wait_for_thread(FILE *fp, struct __test_metadata *_metadata)
+{
+ unsigned int sleep_time_us = WAIT_FOR_THREAD_USECS / WAIT_THREAD_RETRIES;
+ char buf[80] = "";
+
+ for (int i = 0; i < WAIT_THREAD_RETRIES; i++) {
+ if (!fgets(buf, sizeof(buf), fp))
+ return EIO;
+ if (!strncmp(buf, "futex", 5))
+ return 0;
+ usleep(sleep_time_us);
+ rewind(fp);
+ }
+
+ TH_LOG("/proc/$PID/wchan contains \"%s\". Trying to continue.", buf);
+ return 0;
+}
+
+static void *__futex_thread_fn(void *arg)
+{
+ struct futex_thread *t = arg;
+
+ t->tid = gettid();
+ pthread_barrier_wait(&t->barrier);
+ t->retval = t->threadfn(t->arg);
+ return NULL;
+}
+
+/**
+ * futex_wait_for_thread - Wait for the child thread to sleep in the futex context
+ * @t: Thread handle.
+ * @_metadata: Test metadata for TH_LOG() context
+ */
+static inline int futex_wait_for_thread(struct futex_thread *t, struct __test_metadata *_metadata)
+{
+ char fname[80];
+ FILE *fp;
+ int res;
+
+ snprintf(fname, sizeof(fname), "/proc/%d/wchan", t->tid);
+ fp = fopen(fname, "r");
+ if (!fp) {
+ /* If /proc/... is not available, sleep */
+ if (errno != ENOENT)
+ return errno;
+ TH_LOG("/proc/$PID/wchan not accessible, continue with sleep()");
+ sleep(WAIT_FOR_THREAD_SECS);
+ return 0;
+ }
+
+ res = __wait_for_thread(fp, _metadata);
+ fclose(fp);
+ return res;
+}
+
+/**
+ * futex_thread_create - Create a new thread for testing.
+ * @t: The handle of the newly created thread.
+ * @threadfn: The new thread starts execution by invoking threadfn
+ * @arg: The parameters passed to threadfn.
+ */
+static inline int futex_thread_create(struct futex_thread *t, int (*threadfn)(void *), void *arg)
+{
+ pthread_barrier_init(&t->barrier, NULL, 2);
+
+ t->tid = 0;
+ t->threadfn = threadfn;
+ t->arg = arg;
+
+ if (pthread_create(&t->thread, NULL, __futex_thread_fn, t) < 0) {
+ int ret = errno;
+ pthread_barrier_destroy(&t->barrier);
+ return ret;
+ }
+
+ pthread_barrier_wait(&t->barrier);
+ return 0;
+}
+
+/**
+ * futex_thread_destroy - Wait for and reclaim the resources of the thread.
+ * @t: Thread handle.
+ */
+static inline int futex_thread_destroy(struct futex_thread *t)
+{
+ pthread_join(t->thread, NULL);
+ pthread_barrier_destroy(&t->barrier);
+ return t->retval;
+}
+
+#endif
diff --git a/tools/testing/selftests/hid/hid_bpf.c b/tools/testing/selftests/hid/hid_bpf.c
index b851339308c2..32d81ba15a25 100644
--- a/tools/testing/selftests/hid/hid_bpf.c
+++ b/tools/testing/selftests/hid/hid_bpf.c
@@ -5,7 +5,7 @@
#include <bpf/bpf.h>
struct hid_hw_request_syscall_args {
- __u8 data[10];
+ __u8 data[MAX_BUF_SIZE];
unsigned int hid;
int retval;
size_t size;
@@ -54,11 +54,27 @@ FIXTURE_TEARDOWN(hid_bpf) {
hid_bpf_teardown(_metadata, self, variant); \
} while (0)
+FIXTURE_VARIANT(hid_bpf) {
+ __u8 *rdesc;
+ size_t rdesc_size;
+};
+
+FIXTURE_VARIANT_ADD(hid_bpf, numbered) {
+ .rdesc = rdesc,
+ .rdesc_size = sizeof(rdesc),
+};
+
+FIXTURE_VARIANT_ADD(hid_bpf, unnumbered) {
+ .rdesc = fido2_rdesc,
+ .rdesc_size = sizeof(fido2_rdesc),
+};
+
FIXTURE_SETUP(hid_bpf)
{
int err;
- err = setup_uhid(_metadata, &self->hid, BUS_USB, 0x0001, 0x0a36, rdesc, sizeof(rdesc));
+ err = setup_uhid(_metadata, &self->hid, BUS_USB, 0x0001, 0x0a36,
+ variant->rdesc, variant->rdesc_size);
ASSERT_OK(err);
}
@@ -67,14 +83,17 @@ struct test_program {
int insert_head;
};
#define LOAD_PROGRAMS(progs) \
- load_programs(progs, ARRAY_SIZE(progs), _metadata, self, variant)
+ load_programs(progs, ARRAY_SIZE(progs), false, _metadata, self, variant)
+#define LOAD_PROGRAMS_MAY_FAIL(progs) \
+ load_programs(progs, ARRAY_SIZE(progs), true, _metadata, self, variant)
#define LOAD_BPF \
- load_programs(NULL, 0, _metadata, self, variant)
-static void load_programs(const struct test_program programs[],
- const size_t progs_count,
- struct __test_metadata *_metadata,
- FIXTURE_DATA(hid_bpf) * self,
- const FIXTURE_VARIANT(hid_bpf) * variant)
+ load_programs(NULL, 0, false, _metadata, self, variant)
+static int load_programs(const struct test_program programs[],
+ const size_t progs_count,
+ bool load_may_fail,
+ struct __test_metadata *_metadata,
+ FIXTURE_DATA(hid_bpf) * self,
+ const FIXTURE_VARIANT(hid_bpf) * variant)
{
struct bpf_map *iter_map;
int err = -EINVAL;
@@ -128,6 +147,9 @@ static void load_programs(const struct test_program programs[],
}
err = hid__load(self->skel);
+ if (err && load_may_fail)
+ return err;
+
ASSERT_OK(err) TH_LOG("hid_skel_load failed: %d", err);
for (int i = 0; i < progs_count; i++) {
@@ -147,6 +169,7 @@ static void load_programs(const struct test_program programs[],
self->hidraw_fd = open_hidraw(&self->hid);
ASSERT_GE(self->hidraw_fd, 0) TH_LOG("open_hidraw");
+ return 0;
}
/*
@@ -168,7 +191,7 @@ TEST_F(hid_bpf, raw_event)
const struct test_program progs[] = {
{ .name = "hid_first_event" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -219,7 +242,7 @@ TEST_F(hid_bpf, subprog_raw_event)
const struct test_program progs[] = {
{ .name = "hid_subprog_first_event" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -277,7 +300,7 @@ TEST_F(hid_bpf, test_attach_detach)
{ .name = "hid_second_event" },
};
struct bpf_link *link;
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err, link_fd;
LOAD_PROGRAMS(progs);
@@ -362,7 +385,7 @@ TEST_F(hid_bpf, test_hid_change_report)
const struct test_program progs[] = {
{ .name = "hid_change_report_id" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -389,21 +412,24 @@ TEST_F(hid_bpf, test_hid_user_input_report_call)
{
struct hid_hw_request_syscall_args args = {
.retval = -1,
- .size = 10,
+ .size = MAX_BUF_SIZE,
};
DECLARE_LIBBPF_OPTS(bpf_test_run_opts, tattrs,
.ctx_in = &args,
.ctx_size_in = sizeof(args),
);
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err, prog_fd;
LOAD_BPF;
args.hid = self->hid.hid_id;
args.data[0] = 1; /* report ID */
- args.data[1] = 2; /* report ID */
- args.data[2] = 42; /* report ID */
+ args.data[1] = 2;
+ args.data[2] = 42;
+
+ if (variant->rdesc == fido2_rdesc)
+ args.data[0] = 0;
prog_fd = bpf_program__fd(self->skel->progs.hid_user_input_report);
@@ -421,8 +447,13 @@ TEST_F(hid_bpf, test_hid_user_input_report_call)
/* read the data from hidraw */
memset(buf, 0, sizeof(buf));
err = read(self->hidraw_fd, buf, sizeof(buf));
- ASSERT_EQ(err, 6) TH_LOG("read_hidraw");
- ASSERT_EQ(buf[0], 1);
+ if (variant->rdesc == rdesc) {
+ ASSERT_EQ(err, 6) TH_LOG("read_hidraw");
+ } else {
+ ASSERT_EQ(err, 64)
+ TH_LOG("read_hidraw");
+ }
+ ASSERT_EQ(buf[0], args.data[0]);
ASSERT_EQ(buf[1], 2);
ASSERT_EQ(buf[2], 42);
}
@@ -435,7 +466,7 @@ TEST_F(hid_bpf, test_hid_user_output_report_call)
{
struct hid_hw_request_syscall_args args = {
.retval = -1,
- .size = 10,
+ .size = MAX_BUF_SIZE,
};
DECLARE_LIBBPF_OPTS(bpf_test_run_opts, tattrs,
.ctx_in = &args,
@@ -448,8 +479,11 @@ TEST_F(hid_bpf, test_hid_user_output_report_call)
args.hid = self->hid.hid_id;
args.data[0] = 1; /* report ID */
- args.data[1] = 2; /* report ID */
- args.data[2] = 42; /* report ID */
+ args.data[1] = 2;
+ args.data[2] = 42;
+
+ if (variant->rdesc == fido2_rdesc)
+ args.data[0] = 0;
prog_fd = bpf_program__fd(self->skel->progs.hid_user_output_report);
@@ -465,9 +499,14 @@ TEST_F(hid_bpf, test_hid_user_output_report_call)
ASSERT_OK(err) TH_LOG("error while calling bpf_prog_test_run_opts");
ASSERT_OK(cond_err) TH_LOG("error while calling waiting for the condition");
- ASSERT_EQ(args.retval, 3);
+ if (variant->rdesc == rdesc) {
+ ASSERT_EQ(args.retval, 3);
+ } else if (variant->rdesc == fido2_rdesc) {
+ ASSERT_EQ(args.retval, 65)
+ TH_LOG("report size error, should have 64 + 1 extra byte for the report ID 0");
+ }
- ASSERT_EQ(output_report[0], 1);
+ ASSERT_EQ(output_report[0], args.data[0]);
ASSERT_EQ(output_report[1], 2);
ASSERT_EQ(output_report[2], 42);
@@ -484,7 +523,7 @@ TEST_F(hid_bpf, test_hid_user_raw_request_call)
.retval = -1,
.type = HID_FEATURE_REPORT,
.request_type = HID_REQ_GET_REPORT,
- .size = 10,
+ .size = MAX_BUF_SIZE,
};
DECLARE_LIBBPF_OPTS(bpf_test_run_opts, tattrs,
.ctx_in = &args,
@@ -517,7 +556,7 @@ TEST_F(hid_bpf, test_hid_filter_raw_request_call)
const struct test_program progs[] = {
{ .name = "hid_test_filter_raw_request" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -570,7 +609,7 @@ TEST_F(hid_bpf, test_hid_change_raw_request_call)
const struct test_program progs[] = {
{ .name = "hid_test_hidraw_raw_request" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -596,7 +635,7 @@ TEST_F(hid_bpf, test_hid_infinite_loop_raw_request_call)
const struct test_program progs[] = {
{ .name = "hid_test_infinite_loop_raw_request" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -619,7 +658,7 @@ TEST_F(hid_bpf, test_hid_filter_output_report_call)
const struct test_program progs[] = {
{ .name = "hid_test_filter_output_report" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -672,7 +711,7 @@ TEST_F(hid_bpf, test_hid_change_output_report_call)
const struct test_program progs[] = {
{ .name = "hid_test_hidraw_output_report" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -696,7 +735,7 @@ TEST_F(hid_bpf, test_hid_infinite_loop_output_report_call)
const struct test_program progs[] = {
{ .name = "hid_test_infinite_loop_output_report" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -722,7 +761,7 @@ TEST_F(hid_bpf, test_multiply_events_wq)
const struct test_program progs[] = {
{ .name = "hid_test_multiply_events_wq" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -760,7 +799,7 @@ TEST_F(hid_bpf, test_multiply_events)
const struct test_program progs[] = {
{ .name = "hid_test_multiply_events" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -794,7 +833,7 @@ TEST_F(hid_bpf, test_hid_infinite_loop_input_report_call)
const struct test_program progs[] = {
{ .name = "hid_test_infinite_loop_input_report" },
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -848,7 +887,7 @@ TEST_F(hid_bpf, test_hid_attach_flags)
.insert_head = 0,
},
};
- __u8 buf[10] = {0};
+ __u8 buf[MAX_BUF_SIZE] = {0};
int err;
LOAD_PROGRAMS(progs);
@@ -879,6 +918,9 @@ TEST_F(hid_bpf, test_rdesc_fixup)
};
int err, desc_size;
+ if (variant->rdesc != rdesc)
+ SKIP(return, "not compatible report descriptor");
+
LOAD_PROGRAMS(progs);
/* check that hid_rdesc_fixup() was executed */
@@ -904,11 +946,39 @@ TEST_F(hid_bpf, test_rdesc_fixup_get_data_overflow)
{ .name = "hid_rdesc_fixup_get_data_overflow" },
};
- LOAD_PROGRAMS(progs);
+ /* newer verifier can detect the overflow at load time */
+ if (LOAD_PROGRAMS_MAY_FAIL(progs))
+ return;
ASSERT_EQ(self->skel->bss->get_data_overflow_check, 1);
}
+TEST_F(hid_bpf, test_rdesc_fixup_change_uniq_name_phys)
+{
+ const struct test_program progs[] = {
+ { .name = "hid_rdesc_fixup_change_uniq_name_phys" },
+ };
+ char expected[256], buf[256] = {};
+ int err;
+
+ LOAD_PROGRAMS(progs);
+
+ err = ioctl(self->hidraw_fd, HIDIOCGRAWNAME(sizeof(buf)), buf);
+ ASSERT_GE(err, 0) TH_LOG("HIDIOCGRAWNAME");
+ ASSERT_STREQ("name coming from bpf", buf);
+
+ snprintf(expected, sizeof(expected), "%d phys:coming:from:bpf", self->hid.dev_id);
+
+ err = ioctl(self->hidraw_fd, HIDIOCGRAWPHYS(sizeof(buf)), buf);
+ ASSERT_GE(err, 0) TH_LOG("HIDIOCGRAWPHYS");
+ ASSERT_STREQ(expected, buf);
+
+ err = ioctl(self->hidraw_fd, HIDIOCGRAWUNIQ(sizeof(buf)), buf);
+ ASSERT_GE(err, 0) TH_LOG("HIDIOCGRAWUNIQ");
+ ASSERT_STREQ("uniq:coming:from:bpf", buf);
+
+}
+
static int libbpf_print_fn(enum libbpf_print_level level,
const char *format, va_list args)
{
diff --git a/tools/testing/selftests/hid/hid_common.h b/tools/testing/selftests/hid/hid_common.h
index e3b267446fa0..b7890ba2878f 100644
--- a/tools/testing/selftests/hid/hid_common.h
+++ b/tools/testing/selftests/hid/hid_common.h
@@ -13,6 +13,7 @@
#include <linux/uhid.h>
#define SHOW_UHID_DEBUG 0
+#define MAX_BUF_SIZE 128
#define min(a, b) \
({ __typeof__(a) _a = (a); \
@@ -97,6 +98,28 @@ static unsigned char rdesc[] = {
static __u8 feature_data[] = { 1, 2 };
+static __maybe_unused unsigned char fido2_rdesc[] = {
+ 0x06, 0xd0, 0xf1, /* Usage Page (FIDO Alliance) */
+ 0x09, 0x01, /* Usage (U2F Authenticator Device) */
+ 0xa1, 0x01, /* Collection (Application) */
+ 0x09, 0x20, /* Usage (Input Report Data) */
+ 0x15, 0x00, /* Logical Minimum (0) */
+ 0x26, 0xff, 0x00, /* Logical Maximum (255) */
+ 0x75, 0x08, /* Report Size (8) */
+ 0x95, 0x40, /* Report Count (64) */
+ 0x81, 0x02, /* Input (Data,Var,Abs) */
+ 0x09, 0x21, /* Usage (Output Report Data) */
+ 0x15, 0x00, /* Logical Minimum (0) */
+ 0x26, 0xff, 0x00, /* Logical Maximum (255) */
+ 0x75, 0x08, /* Report Size (8) */
+ 0x95, 0x40, /* Report Count (64) */
+ 0x91, 0x02, /* Output (Data,Var,Abs) */
+ 0x06, 0x00, 0xff, /* Usage Page (Vendor Defined Page 1) */
+ 0x09, 0x22, /* Usage (Vendor Usage 0x22) */
+ 0xb1, 0x02, /* Feature (Data,Var,Abs) */
+ 0xc0, /* End Collection */
+};
+
#define ASSERT_OK(data) ASSERT_FALSE(data)
#define ASSERT_OK_PTR(ptr) ASSERT_NE(NULL, ptr)
@@ -110,7 +133,7 @@ static pthread_cond_t uhid_started = PTHREAD_COND_INITIALIZER;
static pthread_mutex_t uhid_output_mtx = PTHREAD_MUTEX_INITIALIZER;
static pthread_cond_t uhid_output_cond = PTHREAD_COND_INITIALIZER;
-static unsigned char output_report[10];
+static unsigned char output_report[MAX_BUF_SIZE];
/* no need to protect uhid_stopped, only one thread accesses it */
static bool uhid_stopped;
diff --git a/tools/testing/selftests/hid/progs/hid.c b/tools/testing/selftests/hid/progs/hid.c
index b21fbb13c926..48aa8088cc53 100644
--- a/tools/testing/selftests/hid/progs/hid.c
+++ b/tools/testing/selftests/hid/progs/hid.c
@@ -98,7 +98,7 @@ struct hid_bpf_ops change_report_id = {
struct hid_hw_request_syscall_args {
/* data needs to come at offset 0 so we can use it in calls */
- __u8 data[10];
+ __u8 data[128];
unsigned int hid;
int retval;
size_t size;
@@ -255,6 +255,32 @@ struct hid_bpf_ops rdesc_fixup_get_data_overflow = {
.hid_rdesc_fixup = (void *)hid_rdesc_fixup_get_data_overflow,
};
+SEC("?struct_ops.s/hid_rdesc_fixup")
+int BPF_PROG(hid_rdesc_fixup_change_uniq_name_phys, struct hid_bpf_ctx *hid_ctx)
+{
+#define HID_BPF_MEMCPY(target, str) \
+ __builtin_memcpy(target, str, sizeof(str))
+
+ HID_BPF_MEMCPY(hid_ctx->hid->name, "name coming from bpf");
+ HID_BPF_MEMCPY(hid_ctx->hid->uniq, "uniq:coming:from:bpf");
+ /* hid_bpf relies on a phys being a rand % 1024 */
+ for (int i = 0; i < 5; i++) {
+ if (!hid_ctx->hid->phys[i]) {
+ HID_BPF_MEMCPY(hid_ctx->hid->phys + i, " phys:coming:from:bpf");
+ break;
+ }
+ }
+
+#undef HID_BPF_MEMCPY
+
+ return 0;
+}
+
+SEC(".struct_ops.link")
+struct hid_bpf_ops rdesc_fixup_change_uniq_name_phys = {
+ .hid_rdesc_fixup = (void *)hid_rdesc_fixup_change_uniq_name_phys,
+};
+
SEC("?struct_ops/hid_device_event")
int BPF_PROG(hid_test_insert1, struct hid_bpf_ctx *hid_ctx, enum hid_report_type type)
{
diff --git a/tools/testing/selftests/hid/progs/hid_bpf_helpers.h b/tools/testing/selftests/hid/progs/hid_bpf_helpers.h
index cdca912f3afd..05698793762a 100644
--- a/tools/testing/selftests/hid/progs/hid_bpf_helpers.h
+++ b/tools/testing/selftests/hid/progs/hid_bpf_helpers.h
@@ -61,6 +61,9 @@ enum hid_report_type {
struct hid_device {
unsigned int id;
+ char name[128];
+ char phys[64];
+ char uniq[64];
} __attribute__((preserve_access_index));
struct bpf_wq {
diff --git a/tools/testing/selftests/ipc/msgque.c b/tools/testing/selftests/ipc/msgque.c
index 82f73cdae120..4b4c1abfac0f 100644
--- a/tools/testing/selftests/ipc/msgque.c
+++ b/tools/testing/selftests/ipc/msgque.c
@@ -161,7 +161,7 @@ int dump_queue(struct msgque_data *msgque)
ret = msgrcv(msgque->msq_id, &msgque->messages[i].mtype,
MAX_MSG_SIZE, i, IPC_NOWAIT | MSG_COPY);
if (ret < 0) {
- if (errno == ENOSYS)
+ if (errno == EOPNOTSUPP)
ksft_exit_skip("MSG_COPY not supported\n");
ksft_test_result_fail("Failed to copy IPC message: %m (%d)\n", errno);
diff --git a/tools/testing/selftests/kselftest_harness.h b/tools/testing/selftests/kselftest_harness.h
index 261e4df94d9d..29a19bc87084 100644
--- a/tools/testing/selftests/kselftest_harness.h
+++ b/tools/testing/selftests/kselftest_harness.h
@@ -467,7 +467,7 @@ static inline void __kselftest_memset_safe(void *s, int c, size_t n)
!__atomic_test_and_set(_metadata->no_teardown, __ATOMIC_RELAXED)) \
fixture_name##_teardown(_metadata, self, variant); \
} \
- static struct __test_metadata *_##fixture_name##_##test_name##_object; \
+ static struct __test_metadata *_##fixture_name##_##test_name##_object __maybe_unused; \
static void __attribute__((constructor(KSELFTEST_PRIO_TEST))) \
_register_##fixture_name##_##test_name(void) \
{ \
diff --git a/tools/testing/selftests/kvm/Makefile.kvm b/tools/testing/selftests/kvm/Makefile.kvm
index 6fc34e9bf8e1..96bab7002d39 100644
--- a/tools/testing/selftests/kvm/Makefile.kvm
+++ b/tools/testing/selftests/kvm/Makefile.kvm
@@ -11,6 +11,7 @@ LIBKVM += lib/kvm_util.c
LIBKVM += lib/lru_gen_util.c
LIBKVM += lib/memstress.c
LIBKVM += lib/guest_sprintf.c
+LIBKVM += lib/proc_util.c
LIBKVM += lib/rbtree.c
LIBKVM += lib/sparsebit.c
LIBKVM += lib/test_util.c
@@ -66,6 +67,7 @@ TEST_GEN_PROGS_COMMON += kvm_page_table_test
TEST_GEN_PROGS_COMMON += set_memory_region_test
TEST_GEN_PROGS_COMMON += memslot_modification_stress_test
TEST_GEN_PROGS_COMMON += memslot_perf_test
+TEST_GEN_PROGS_COMMON += vm_types_test
# Compiled test targets
TEST_GEN_PROGS_x86 = $(TEST_GEN_PROGS_COMMON)
@@ -106,6 +108,7 @@ TEST_GEN_PROGS_x86 += x86/pmu_counters_test
TEST_GEN_PROGS_x86 += x86/pmu_event_filter_test
TEST_GEN_PROGS_x86 += x86/private_mem_conversions_test
TEST_GEN_PROGS_x86 += x86/private_mem_kvm_exits_test
+TEST_GEN_PROGS_x86 += x86/save_restore_pf_stress_test
TEST_GEN_PROGS_x86 += x86/set_boot_cpu_id
TEST_GEN_PROGS_x86 += x86/set_sregs_test
TEST_GEN_PROGS_x86 += x86/smaller_maxphyaddr_emulation_test
@@ -120,6 +123,7 @@ TEST_GEN_PROGS_x86 += x86/svm_nested_soft_inject_test
TEST_GEN_PROGS_x86 += x86/svm_nested_vmcb12_gpa
TEST_GEN_PROGS_x86 += x86/svm_nested_pat_test
TEST_GEN_PROGS_x86 += x86/svm_lbr_nested_state
+TEST_GEN_PROGS_x86 += x86/svm_pmu_host_guest_test
TEST_GEN_PROGS_x86 += x86/tsc_scaling_sync
TEST_GEN_PROGS_x86 += x86/sync_regs_test
TEST_GEN_PROGS_x86 += x86/ucna_injection_test
@@ -156,6 +160,7 @@ TEST_GEN_PROGS_x86 += coalesced_io_test
TEST_GEN_PROGS_x86 += dirty_log_perf_test
TEST_GEN_PROGS_x86 += guest_memfd_test
TEST_GEN_PROGS_x86 += hardware_disable_test
+TEST_GEN_PROGS_x86 += irq_test
TEST_GEN_PROGS_x86 += mmu_stress_test
TEST_GEN_PROGS_x86 += rseq_test
TEST_GEN_PROGS_x86 += steal_time
@@ -180,6 +185,7 @@ TEST_GEN_PROGS_arm64 += arm64/psci_test
TEST_GEN_PROGS_arm64 += arm64/sea_to_user
TEST_GEN_PROGS_arm64 += arm64/set_id_regs
TEST_GEN_PROGS_arm64 += arm64/smccc_filter
+TEST_GEN_PROGS_arm64 += arm64/stage2_block_transitions
TEST_GEN_PROGS_arm64 += arm64/vcpu_width_config
TEST_GEN_PROGS_arm64 += arm64/vgic_init
TEST_GEN_PROGS_arm64 += arm64/vgic_irq
@@ -228,7 +234,8 @@ TEST_GEN_PROGS_riscv += mmu_stress_test
TEST_GEN_PROGS_riscv += rseq_test
TEST_GEN_PROGS_riscv += steal_time
-TEST_GEN_PROGS_loongarch = loongarch/pmu_test
+TEST_GEN_PROGS_loongarch = loongarch/fpu_test
+TEST_GEN_PROGS_loongarch += loongarch/pmu_test
TEST_GEN_PROGS_loongarch += arch_timer
TEST_GEN_PROGS_loongarch += coalesced_io_test
TEST_GEN_PROGS_loongarch += demand_paging_test
@@ -260,6 +267,10 @@ OVERRIDE_TARGETS = 1
include ../lib.mk
include ../cgroup/lib/libcgroup.mk
+ifeq ($(ARCH),x86)
+include ../vfio/lib/libvfio.mk
+endif
+
INSTALL_HDR_PATH = $(top_srcdir)/usr
LINUX_HDR_PATH = $(INSTALL_HDR_PATH)/include/
LINUX_TOOL_INCLUDE = $(top_srcdir)/tools/include
@@ -313,7 +324,11 @@ LIBKVM_S := $(filter %.S,$(LIBKVM))
LIBKVM_C_OBJ := $(patsubst %.c, $(OUTPUT)/%.o, $(LIBKVM_C))
LIBKVM_S_OBJ := $(patsubst %.S, $(OUTPUT)/%.o, $(LIBKVM_S))
LIBKVM_STRING_OBJ := $(patsubst %.c, $(OUTPUT)/%.o, $(LIBKVM_STRING))
-LIBKVM_OBJS = $(LIBKVM_C_OBJ) $(LIBKVM_S_OBJ) $(LIBKVM_STRING_OBJ) $(LIBCGROUP_O)
+LIBKVM_OBJS = $(LIBKVM_C_OBJ) $(LIBKVM_S_OBJ) $(LIBKVM_STRING_OBJ)
+LIBKVM_OBJS += $(LIBCGROUP_O)
+ifeq ($(ARCH),x86)
+LIBKVM_OBJS += $(LIBVFIO_O)
+endif
SPLIT_TEST_GEN_PROGS := $(patsubst %, $(OUTPUT)/%, $(SPLIT_TESTS))
SPLIT_TEST_GEN_OBJ := $(patsubst %, $(OUTPUT)/$(ARCH)/%.o, $(SPLIT_TESTS))
diff --git a/tools/testing/selftests/kvm/arch_timer.c b/tools/testing/selftests/kvm/arch_timer.c
index 90c475a61b22..03160ad666f5 100644
--- a/tools/testing/selftests/kvm/arch_timer.c
+++ b/tools/testing/selftests/kvm/arch_timer.c
@@ -85,7 +85,7 @@ static u32 test_get_pcpu(void)
cpu_set_t online_cpuset;
nproc_conf = get_nprocs_conf();
- sched_getaffinity(0, sizeof(cpu_set_t), &online_cpuset);
+ kvm_sched_getaffinity(0, sizeof(cpu_set_t), &online_cpuset);
/* Randomly find an available pCPU to place a vCPU on */
do {
@@ -141,33 +141,27 @@ static void test_run(struct kvm_vm *vm)
{
pthread_t pt_vcpu_migration;
unsigned int i;
- int ret;
pthread_mutex_init(&vcpu_done_map_lock, NULL);
vcpu_done_map = bitmap_zalloc(test_args.nr_vcpus);
TEST_ASSERT(vcpu_done_map, "Failed to allocate vcpu done bitmap");
- for (i = 0; i < (unsigned long)test_args.nr_vcpus; i++) {
- ret = pthread_create(&pt_vcpu_run[i], NULL, test_vcpu_run,
- (void *)(unsigned long)i);
- TEST_ASSERT(!ret, "Failed to create vCPU-%d pthread", i);
- }
+ for (i = 0; i < (unsigned long)test_args.nr_vcpus; i++)
+ kvm_pthread_create(&pt_vcpu_run[i], NULL, test_vcpu_run,
+ (void *)(unsigned long)i);
/* Spawn a thread to control the vCPU migrations */
if (test_args.migration_freq_ms) {
srand(time(NULL));
- ret = pthread_create(&pt_vcpu_migration, NULL,
- test_vcpu_migration, NULL);
- TEST_ASSERT(!ret, "Failed to create the migration pthread");
+ kvm_pthread_create(&pt_vcpu_migration, NULL, test_vcpu_migration, NULL);
}
-
for (i = 0; i < test_args.nr_vcpus; i++)
- pthread_join(pt_vcpu_run[i], NULL);
+ kvm_pthread_join(pt_vcpu_run[i], NULL);
if (test_args.migration_freq_ms)
- pthread_join(pt_vcpu_migration, NULL);
+ kvm_pthread_join(pt_vcpu_migration, NULL);
bitmap_free(vcpu_done_map);
}
diff --git a/tools/testing/selftests/kvm/arm64/arch_timer_edge_cases.c b/tools/testing/selftests/kvm/arm64/arch_timer_edge_cases.c
index f7625eb711d6..d9c9377a6325 100644
--- a/tools/testing/selftests/kvm/arm64/arch_timer_edge_cases.c
+++ b/tools/testing/selftests/kvm/arm64/arch_timer_edge_cases.c
@@ -1039,7 +1039,7 @@ int main(int argc, char *argv[])
if (!parse_args(argc, argv))
exit(KSFT_SKIP);
- sched_getaffinity(0, sizeof(default_cpuset), &default_cpuset);
+ kvm_sched_getaffinity(0, sizeof(default_cpuset), &default_cpuset);
set_counter_defaults();
if (test_args.test_virtual) {
diff --git a/tools/testing/selftests/kvm/arm64/debug-exceptions.c b/tools/testing/selftests/kvm/arm64/debug-exceptions.c
index 3eb4b1b6682d..7dc5f0b4f6ad 100644
--- a/tools/testing/selftests/kvm/arm64/debug-exceptions.c
+++ b/tools/testing/selftests/kvm/arm64/debug-exceptions.c
@@ -527,6 +527,46 @@ void test_single_step_from_userspace(int test_cnt)
kvm_vm_free(vm);
}
+static void guest_code_wp(void)
+{
+ write_data = 'x';
+ GUEST_DONE();
+}
+
+/*
+ * A userspace hardware watchpoint (KVM_GUESTDBG_USE_HW) must fire and report
+ * the accessed address in debug.arch.far, exercising the watchpoint exit path.
+ */
+static void test_watchpoint_from_userspace(void)
+{
+ struct kvm_guest_debug debug = {};
+ struct kvm_vcpu *vcpu;
+ struct kvm_run *run;
+ struct kvm_vm *vm;
+
+ vm = vm_create_with_one_vcpu(&vcpu, guest_code_wp);
+ run = vcpu->run;
+
+ debug.control = KVM_GUESTDBG_ENABLE | KVM_GUESTDBG_USE_HW;
+ debug.arch.dbg_wcr[0] = DBGWCR_LEN8 | DBGWCR_RD | DBGWCR_WR |
+ DBGWCR_EL1 | DBGWCR_E;
+ /*
+ * BAS = 0xff (LEN8) requires a doubleword-aligned DBGWVR; FAR still
+ * reports the exact accessed byte.
+ */
+ debug.arch.dbg_wvr[0] = PC(write_data) & ~7UL;
+ vcpu_guest_debug_set(vcpu, &debug);
+
+ vcpu_run(vcpu);
+ TEST_ASSERT(run->exit_reason == KVM_EXIT_DEBUG,
+ "Expected KVM_EXIT_DEBUG, got %u", run->exit_reason);
+ TEST_ASSERT((u64)run->debug.arch.far == PC(write_data),
+ "Watchpoint FAR 0x%lx != accessed address 0x%lx",
+ (u64)run->debug.arch.far, PC(write_data));
+
+ kvm_vm_free(vm);
+}
+
/*
* Run debug testing using the various breakpoint#, watchpoint# and
* context-aware breakpoint# with the given ID_AA64DFR0_EL1 configuration.
@@ -600,6 +640,7 @@ int main(int argc, char *argv[])
test_guest_debug_exceptions_all(aa64dfr0);
test_single_step_from_userspace(ss_iteration);
+ test_watchpoint_from_userspace();
return 0;
}
diff --git a/tools/testing/selftests/kvm/arm64/get-reg-list.c b/tools/testing/selftests/kvm/arm64/get-reg-list.c
index 0a3a94c4cca1..533994687b5d 100644
--- a/tools/testing/selftests/kvm/arm64/get-reg-list.c
+++ b/tools/testing/selftests/kvm/arm64/get-reg-list.c
@@ -67,6 +67,7 @@ static struct feature_id_reg feat_id_regs[] = {
REG_FEAT(VDISR_EL2, ID_AA64PFR0_EL1, RAS, IMP),
REG_FEAT(VSESR_EL2, ID_AA64PFR0_EL1, RAS, IMP),
REG_FEAT(VNCR_EL2, ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY),
+ REG_FEAT(NVHCR_EL2, ID_AA64MMFR4_EL1, NV_frac, NV3),
REG_FEAT(CNTHV_CTL_EL2, ID_AA64MMFR1_EL1, VH, IMP),
REG_FEAT(CNTHV_CVAL_EL2,ID_AA64MMFR1_EL1, VH, IMP),
REG_FEAT(ZCR_EL2, ID_AA64PFR0_EL1, SVE, IMP),
@@ -532,6 +533,7 @@ static __u64 base_regs[] = {
static __u64 pmu_regs[] = {
ARM64_SYS_REG(3, 0, 9, 14, 1), /* PMINTENSET_EL1 */
ARM64_SYS_REG(3, 0, 9, 14, 2), /* PMINTENCLR_EL1 */
+ ARM64_SYS_REG(3, 0, 9, 14, 6), /* PMMIR_EL1 */
ARM64_SYS_REG(3, 3, 9, 12, 0), /* PMCR_EL0 */
ARM64_SYS_REG(3, 3, 9, 12, 1), /* PMCNTENSET_EL0 */
ARM64_SYS_REG(3, 3, 9, 12, 2), /* PMCNTENCLR_EL0 */
@@ -770,6 +772,7 @@ static __u64 el2_regs[] = {
SYS_REG(SP_EL2),
SYS_REG(VDISR_EL2),
SYS_REG(VSESR_EL2),
+ SYS_REG(NVHCR_EL2),
};
static __u64 el2_e2h0_regs[] = {
diff --git a/tools/testing/selftests/kvm/arm64/stage2_block_transitions.c b/tools/testing/selftests/kvm/arm64/stage2_block_transitions.c
new file mode 100644
index 000000000000..5fd47f4ada1f
--- /dev/null
+++ b/tools/testing/selftests/kvm/arm64/stage2_block_transitions.c
@@ -0,0 +1,226 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Copyright (c) 2026 Google LLC
+ * Author: Fuad Tabba <fuad.tabba@linux.dev>
+ *
+ * stage2_block_transitions - Exercise stage-2 block/page granularity changes
+ * that dirty logging forces at fault time, and assert the guest completes.
+ *
+ * Both scenarios need the fault handler to allocate at fault time (a fresh
+ * mapping and/or page-table pages while holding mmu_lock), so a fault path
+ * that fails to stage that memory manifests as a KVM_RUN error or, worse, a
+ * host crash. The asserted property is host-agnostic: the guest runs the
+ * sequence to completion and every KVM_RUN succeeds. On a pKVM host, where a
+ * non-protected guest's stage-2 faults are serviced by the pkvm_pgtable_*()
+ * backend, the same sequences also guard that backend's fault-time staging.
+ *
+ * Scenario 1 - block collapse on dirty-logging disable:
+ * A write under dirty logging installs a 4K page; GET_DIRTY_LOG
+ * re-write-protects it; logging is disabled; a second write takes a
+ * permission fault that collapses the page into a hugetlb-backed block,
+ * which requires a fresh mapping object under mmu_lock.
+ *
+ * Scenario 2 - block split under dirty logging:
+ * Several hugetlb-backed blocks are faulted in as non-executable blocks,
+ * dirty logging is enabled (write-protect only), then the guest executes
+ * into each block. Each instruction fetch takes an execute permission
+ * fault that must split the block into pages during logging, draining
+ * page-table pages. Skipped on CTR_EL0.DIC hardware, where mappings are
+ * made executable eagerly and the execute fault never occurs.
+ */
+#include <linux/bitfield.h>
+#include <linux/bitmap.h>
+#include <linux/mman.h>
+#include <linux/sizes.h>
+#include <sys/mman.h>
+
+#include <asm/sysreg.h>
+
+#include "kvm_util.h"
+#include "processor.h"
+#include "test_util.h"
+#include "ucall.h"
+
+#define DATA_SLOT 1
+#define TEST_GVA 0xc0000000UL
+#define BLOCK_SIZE SZ_2M
+
+/* AArch64 "ret" (ret x30): a self-contained, returnable executable payload. */
+#define RET_INSN 0xd65f03c0U
+
+/*
+ * A non-protected guest's per-VM stage-2 pool is seeded only with the PGD
+ * donation, which stage-2 init immediately consumes, so the page-table budget
+ * for a fault that does not top up is just the handful (~2x the stage-2 min
+ * pages) of memcache leftovers. Executing into this many distinct blocks
+ * demands far more than that budget: a fault path that tops up on every fault
+ * completes all of them, one that skips non-write faults runs out mid-sequence.
+ */
+#define NR_BLOCKS 16
+
+/* Scenario 2 guest -> host sync stages. */
+#define STAGE_SKIP_DIC 1
+#define STAGE_BLOCKS_READY 2
+
+static void collapse_guest_code(u64 gva)
+{
+ u64 *data = (u64 *)gva;
+
+ /* Under dirty logging: install a 4K writable page. */
+ WRITE_ONCE(*data, 0x1);
+ GUEST_SYNC(1);
+
+ /* Logging disabled: a permission fault collapses the page into a block. */
+ WRITE_ONCE(*data, 0x2);
+ GUEST_SYNC(2);
+
+ GUEST_DONE();
+}
+
+static void test_block_collapse(void)
+{
+ struct kvm_vcpu *vcpu;
+ unsigned long *bmap;
+ struct kvm_vm *vm;
+ struct ucall uc;
+ size_t npages;
+ u64 gpa;
+
+ vm = vm_create_with_one_vcpu(&vcpu, collapse_guest_code);
+ npages = BLOCK_SIZE / vm->page_size;
+
+ gpa = (vm_compute_max_gfn(vm) * vm->page_size) - BLOCK_SIZE;
+ gpa = align_down(gpa, BLOCK_SIZE);
+
+ vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS_HUGETLB_2MB, gpa,
+ DATA_SLOT, npages, KVM_MEM_LOG_DIRTY_PAGES);
+ virt_map(vm, TEST_GVA, gpa, npages);
+ vcpu_args_set(vcpu, 1, TEST_GVA);
+
+ bmap = bitmap_zalloc(BLOCK_SIZE / getpagesize());
+
+ vcpu_run(vcpu);
+ TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC && uc.args[1] == 1,
+ "Expected first sync, got cmd %lu arg %lu", uc.cmd, uc.args[1]);
+
+ /* GET_DIRTY_LOG re-write-protects the dirtied page; then stop logging. */
+ kvm_vm_get_dirty_log(vm, DATA_SLOT, bmap);
+ vm_mem_region_set_flags(vm, DATA_SLOT, 0);
+
+ /* The collapsing permission fault: a broken fault path faults here. */
+ vcpu_run(vcpu);
+ TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC && uc.args[1] == 2,
+ "Expected second sync, got cmd %lu arg %lu", uc.cmd, uc.args[1]);
+
+ vcpu_run(vcpu);
+ TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_DONE,
+ "Expected done, got cmd %lu", uc.cmd);
+
+ free(bmap);
+ kvm_vm_free(vm);
+}
+
+static void guest_sync_insn(u64 va)
+{
+ /* Make the just-written instruction coherent for execution (!DIC). */
+ asm volatile("dc cvau, %0\n"
+ "dsb ish\n"
+ "ic ivau, %0\n"
+ "dsb ish\n"
+ "isb\n"
+ :: "r" (va) : "memory");
+}
+
+static void split_guest_code(u64 base_gva, u64 nblocks)
+{
+ u64 i, va;
+
+ if (FIELD_GET(CTR_EL0_DIC_MASK, read_sysreg(ctr_el0))) {
+ GUEST_SYNC(STAGE_SKIP_DIC);
+ GUEST_DONE();
+ return;
+ }
+
+ /* Fault in each block (non-executable) and stage an executable payload. */
+ for (i = 0; i < nblocks; i++) {
+ va = base_gva + i * BLOCK_SIZE;
+ WRITE_ONCE(*(u32 *)va, RET_INSN);
+ guest_sync_insn(va);
+ }
+ GUEST_SYNC(STAGE_BLOCKS_READY);
+
+ /* Logging is now on: executing into each block splits it into pages. */
+ for (i = 0; i < nblocks; i++) {
+ va = base_gva + i * BLOCK_SIZE;
+ ((void (*)(void))va)();
+ }
+
+ GUEST_DONE();
+}
+
+static void test_exec_split_drain(void)
+{
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ struct ucall uc;
+ size_t npages;
+ u64 gpa;
+
+ vm = vm_create_with_one_vcpu(&vcpu, split_guest_code);
+ npages = NR_BLOCKS * (BLOCK_SIZE / vm->page_size);
+
+ gpa = (vm_compute_max_gfn(vm) * vm->page_size) - NR_BLOCKS * BLOCK_SIZE;
+ gpa = align_down(gpa, BLOCK_SIZE);
+
+ vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS_HUGETLB_2MB, gpa,
+ DATA_SLOT, npages, 0);
+ virt_map(vm, TEST_GVA, gpa, npages);
+ vcpu_args_set(vcpu, 2, TEST_GVA, (u64)NR_BLOCKS);
+
+ vcpu_run(vcpu);
+ TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC,
+ "Expected sync, got cmd %lu", uc.cmd);
+ if (uc.args[1] == STAGE_SKIP_DIC) {
+ ksft_print_msg("SKIP block split: CTR_EL0.DIC == 1\n");
+ kvm_vm_free(vm);
+ return;
+ }
+ TEST_ASSERT(uc.args[1] == STAGE_BLOCKS_READY,
+ "Expected blocks-ready sync, got arg %lu", uc.args[1]);
+
+ /* Write-protect the blocks; the guest then splits them by executing. */
+ vm_mem_region_set_flags(vm, DATA_SLOT, KVM_MEM_LOG_DIRTY_PAGES);
+
+ vcpu_run(vcpu);
+ TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_DONE,
+ "Expected done, got cmd %lu", uc.cmd);
+
+ kvm_vm_free(vm);
+}
+
+/*
+ * The explicit-size hugetlb backing hard-fails region creation if the pages
+ * are not already reserved, so probe here and skip rather than abort. The
+ * peak reservation is scenario 2's; the two scenarios run and free in turn.
+ */
+static void require_hugepages(size_t bytes)
+{
+ void *mem = mmap(NULL, bytes, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB | MAP_HUGE_2MB,
+ -1, 0);
+
+ if (mem == MAP_FAILED)
+ ksft_exit_skip("Need %zu bytes of reserved 2M hugepages\n", bytes);
+ munmap(mem, bytes);
+}
+
+int main(void)
+{
+ require_hugepages(NR_BLOCKS * BLOCK_SIZE);
+
+ test_block_collapse();
+ test_exec_split_drain();
+
+ ksft_print_msg("All ok!\n");
+ return 0;
+}
diff --git a/tools/testing/selftests/kvm/arm64/vgic_irq.c b/tools/testing/selftests/kvm/arm64/vgic_irq.c
index 5e231998617e..3cb85932ee85 100644
--- a/tools/testing/selftests/kvm/arm64/vgic_irq.c
+++ b/tools/testing/selftests/kvm/arm64/vgic_irq.c
@@ -988,7 +988,7 @@ static void test_vgic_two_cpus(void *gcode)
struct test_args args = {};
struct kvm_vm *vm;
gva_t args_gva;
- int gic_fd, ret;
+ int gic_fd;
vm = vm_create_with_vcpus(2, gcode, vcpus);
@@ -1004,15 +1004,11 @@ static void test_vgic_two_cpus(void *gcode)
gic_fd = vgic_v3_setup(vm, 2, 64);
- ret = pthread_create(&thr[0], NULL, test_vcpu_run, vcpus[0]);
- if (ret)
- TEST_FAIL("Can't create thread for vcpu 0 (%d)\n", ret);
- ret = pthread_create(&thr[1], NULL, test_vcpu_run, vcpus[1]);
- if (ret)
- TEST_FAIL("Can't create thread for vcpu 1 (%d)\n", ret);
+ kvm_pthread_create(&thr[0], NULL, test_vcpu_run, vcpus[0]);
+ kvm_pthread_create(&thr[1], NULL, test_vcpu_run, vcpus[1]);
- pthread_join(thr[0], NULL);
- pthread_join(thr[1], NULL);
+ kvm_pthread_join(thr[0], NULL);
+ kvm_pthread_join(thr[1], NULL);
close(gic_fd);
kvm_vm_free(vm);
diff --git a/tools/testing/selftests/kvm/arm64/vgic_lpi_stress.c b/tools/testing/selftests/kvm/arm64/vgic_lpi_stress.c
index d64d434d3f06..9a69c439a0f8 100644
--- a/tools/testing/selftests/kvm/arm64/vgic_lpi_stress.c
+++ b/tools/testing/selftests/kvm/arm64/vgic_lpi_stress.c
@@ -311,23 +311,23 @@ static void run_test(void)
pthread_barrier_init(&test_setup_barrier, NULL, nr_vcpus + nr_devices + 1);
for (i = 0; i < nr_vcpus; i++)
- pthread_create(&vcpu_threads[i], NULL, vcpu_worker_thread, vcpus[i]);
+ kvm_pthread_create(&vcpu_threads[i], NULL, vcpu_worker_thread, vcpus[i]);
for (i = 0; i < nr_devices; i++)
- pthread_create(&lpi_threads[i], NULL, lpi_worker_thread, (void *)i);
+ kvm_pthread_create(&lpi_threads[i], NULL, lpi_worker_thread, (void *)i);
pthread_barrier_wait(&test_setup_barrier);
clock_gettime(CLOCK_MONOTONIC, &start);
for (i = 0; i < nr_devices; i++)
- pthread_join(lpi_threads[i], NULL);
+ kvm_pthread_join(lpi_threads[i], NULL);
delta = timespec_elapsed(start);
write_guest_global(vm, test_data.request_vcpus_stop, true);
for (i = 0; i < nr_vcpus; i++)
- pthread_join(vcpu_threads[i], NULL);
+ kvm_pthread_join(vcpu_threads[i], NULL);
report_stats(delta);
}
diff --git a/tools/testing/selftests/kvm/demand_paging_test.c b/tools/testing/selftests/kvm/demand_paging_test.c
index 302c4923d093..f8b3d0b68830 100644
--- a/tools/testing/selftests/kvm/demand_paging_test.c
+++ b/tools/testing/selftests/kvm/demand_paging_test.c
@@ -57,7 +57,7 @@ static void vcpu_worker(struct memstress_vcpu_args *vcpu_args)
static int handle_uffd_page_request(int uffd_mode, int uffd,
struct uffd_msg *msg)
{
- pid_t tid = syscall(__NR_gettid);
+ pid_t tid = kvm_gettid();
u64 addr = msg->arg.pagefault.address;
struct timespec start;
struct timespec ts_diff;
diff --git a/tools/testing/selftests/kvm/dirty_log_perf_test.c b/tools/testing/selftests/kvm/dirty_log_perf_test.c
index ef779fa91827..7c5abe1ae9e0 100644
--- a/tools/testing/selftests/kvm/dirty_log_perf_test.c
+++ b/tools/testing/selftests/kvm/dirty_log_perf_test.c
@@ -311,7 +311,7 @@ int main(int argc, char *argv[])
int opt;
/* Override the seed to be deterministic by default. */
- guest_random_seed = 1;
+ kvm_random_seed = 1;
dirty_log_manual_caps =
kvm_check_cap(KVM_CAP_MANUAL_DIRTY_LOG_PROTECT2);
@@ -357,7 +357,7 @@ int main(int argc, char *argv[])
p.phys_offset = strtoull(optarg, NULL, 0);
break;
case 'r':
- guest_random_seed = atoi_positive("Random seed", optarg);
+ kvm_random_seed = atoi_positive("Random seed", optarg);
break;
case 's':
p.backing_src = parse_backing_src_type(optarg);
diff --git a/tools/testing/selftests/kvm/dirty_log_test.c b/tools/testing/selftests/kvm/dirty_log_test.c
index 74ca096bf976..af5eb0334a74 100644
--- a/tools/testing/selftests/kvm/dirty_log_test.c
+++ b/tools/testing/selftests/kvm/dirty_log_test.c
@@ -121,7 +121,7 @@ static void guest_code(void)
while (true) {
while (!READ_ONCE(vcpu_stop)) {
addr = guest_test_virt_mem;
- addr += (guest_random_u64(&guest_rng) % guest_num_pages)
+ addr += (kvm_random_u64(&kvm_rng) % guest_num_pages)
* guest_page_size;
addr = align_down(addr, host_page_size);
@@ -691,7 +691,7 @@ static void run_test(enum vm_guest_mode mode, void *arg)
TEST_ASSERT_EQ(vcpu_stop, false);
- pthread_create(&vcpu_thread, NULL, vcpu_worker, vcpu);
+ kvm_pthread_create(&vcpu_thread, NULL, vcpu_worker, vcpu);
for (iteration = 1; iteration <= p->iterations; iteration++) {
unsigned long i, reap_i;
@@ -708,8 +708,7 @@ static void run_test(enum vm_guest_mode mode, void *arg)
sync_global_to_guest(vm, iteration);
- WRITE_ONCE(nr_writes, 0);
- sync_global_to_guest(vm, nr_writes);
+ WRITE_AND_SYNC_TO_GUEST(vm, nr_writes, 0);
dirty_ring_prev_iteration_last_page = dirty_ring_last_page;
WRITE_ONCE(dirty_ring_vcpu_ring_full, false);
@@ -775,16 +774,14 @@ static void run_test(enum vm_guest_mode mode, void *arg)
* writing memory during verification, pages that this thread
* sees as clean may be written with this iteration's value.
*/
- WRITE_ONCE(vcpu_stop, true);
- sync_global_to_guest(vm, vcpu_stop);
+ WRITE_AND_SYNC_TO_GUEST(vm, vcpu_stop, true);
sem_wait(&sem_vcpu_stop);
/*
* Clear vcpu_stop after the vCPU thread has acknowledge the
* stop request and is waiting, i.e. is definitely not running!
*/
- WRITE_ONCE(vcpu_stop, false);
- sync_global_to_guest(vm, vcpu_stop);
+ WRITE_AND_SYNC_TO_GUEST(vm, vcpu_stop, false);
/*
* Sync the number of writes performed before verification, the
@@ -808,7 +805,7 @@ static void run_test(enum vm_guest_mode mode, void *arg)
WRITE_ONCE(host_quit, true);
sem_post(&sem_vcpu_cont);
- pthread_join(vcpu_thread, NULL);
+ kvm_pthread_join(vcpu_thread, NULL);
pr_info("Total bits checked: dirty (%lu), clear (%lu)\n",
host_dirty_count, host_clear_count);
diff --git a/tools/testing/selftests/kvm/hardware_disable_test.c b/tools/testing/selftests/kvm/hardware_disable_test.c
index 3147f5c97e94..43a36ef3ead8 100644
--- a/tools/testing/selftests/kvm/hardware_disable_test.c
+++ b/tools/testing/selftests/kvm/hardware_disable_test.c
@@ -5,7 +5,6 @@
* return notifiers.
*/
#include <fcntl.h>
-#include <pthread.h>
#include <semaphore.h>
#include <stdint.h>
#include <stdlib.h>
@@ -14,20 +13,23 @@
#include <test_util.h>
+#include "kvm_syscalls.h"
#include "kvm_util.h"
+#include "ucall_common.h"
-#define VCPU_NUM 4
-#define SLEEPING_THREAD_NUM (1 << 4)
-#define FORK_NUM (1ULL << 9)
-#define DELAY_US_MAX 2000
+#define NR_VCPUS 4
+#define NR_SLEEPERS_PER_VCPU 16
+#define NR_ITERATIONS 512
+#define DELAY_US_MAX 2000
-sem_t *sem;
+static cpu_set_t threads_cpu_set;
+static sem_t *sem;
static void guest_code(void)
{
for (;;)
; /* Some busy work */
- printf("Should not be reached.\n");
+ GUEST_ASSERT(0);
}
static void *run_vcpu(void *arg)
@@ -35,86 +37,61 @@ static void *run_vcpu(void *arg)
struct kvm_vcpu *vcpu = arg;
struct kvm_run *run = vcpu->run;
+#ifndef _GNU_SOURCE
+ kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set);
+#endif
+
vcpu_run(vcpu);
- TEST_ASSERT(false, "%s: exited with reason %d: %s",
- __func__, run->exit_reason,
- exit_reason_str(run->exit_reason));
- pthread_exit(NULL);
+ TEST_FAIL("vCPU%d exited with reason %d: %s",
+ vcpu->id, run->exit_reason, exit_reason_str(run->exit_reason));
}
static void *sleeping_thread(void *arg)
{
int fd;
- while (true) {
+#ifndef _GNU_SOURCE
+ kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set);
+#endif
+
+ while (1) {
fd = open("/dev/null", O_RDWR);
close(fd);
}
- TEST_ASSERT(false, "%s: exited", __func__);
- pthread_exit(NULL);
-}
-
-static inline void check_create_thread(pthread_t *thread, pthread_attr_t *attr,
- void *(*f)(void *), void *arg)
-{
- int r;
-
- r = pthread_create(thread, attr, f, arg);
- TEST_ASSERT(r == 0, "%s: failed to create thread", __func__);
-}
-
-static inline void check_set_affinity(pthread_t thread, cpu_set_t *cpu_set)
-{
- int r;
-
- r = pthread_setaffinity_np(thread, sizeof(cpu_set_t), cpu_set);
- TEST_ASSERT(r == 0, "%s: failed set affinity", __func__);
-}
-
-static inline void check_join(pthread_t thread, void **retval)
-{
- int r;
-
- r = pthread_join(thread, retval);
- TEST_ASSERT(r == 0, "%s: failed to join thread", __func__);
+ TEST_FAIL("%s: exited", __func__);
}
static void run_test(u32 run)
{
struct kvm_vcpu *vcpu;
+ pthread_attr_t attr;
struct kvm_vm *vm;
- cpu_set_t cpu_set;
- pthread_t threads[VCPU_NUM];
- pthread_t throw_away;
- void *b;
+ pthread_t thread;
u32 i, j;
- CPU_ZERO(&cpu_set);
- for (i = 0; i < VCPU_NUM; i++)
- CPU_SET(i, &cpu_set);
+ TEST_ASSERT_EQ(pthread_attr_init(&attr), 0);
+#ifdef _GNU_SOURCE
+ TEST_ASSERT_EQ(pthread_attr_setaffinity_np(&attr, sizeof(cpu_set_t), &threads_cpu_set), 0);
+#endif
- vm = vm_create(VCPU_NUM);
+ vm = vm_create(NR_VCPUS);
pr_debug("%s: [%d] start vcpus\n", __func__, run);
- for (i = 0; i < VCPU_NUM; ++i) {
+ for (i = 0; i < NR_VCPUS; ++i) {
vcpu = vm_vcpu_add(vm, i, guest_code);
- check_create_thread(&threads[i], NULL, run_vcpu, vcpu);
- check_set_affinity(threads[i], &cpu_set);
+ kvm_pthread_create(&thread, &attr, run_vcpu, vcpu);
- for (j = 0; j < SLEEPING_THREAD_NUM; ++j) {
- check_create_thread(&throw_away, NULL, sleeping_thread,
- (void *)NULL);
- check_set_affinity(throw_away, &cpu_set);
- }
+ for (j = 0; j < NR_SLEEPERS_PER_VCPU; ++j)
+ kvm_pthread_create(&thread, &attr, sleeping_thread, (void *)NULL);
}
pr_debug("%s: [%d] all threads launched\n", __func__, run);
sem_post(sem);
- for (i = 0; i < VCPU_NUM; ++i)
- check_join(threads[i], &b);
- /* Should not be reached */
- TEST_ASSERT(false, "%s: [%d] child escaped the ninja", __func__, run);
+
+ /* Wait for the parent to SIGKILL this child. */
+ while (1)
+ pause();
}
void wait_for_child_setup(pid_t pid)
@@ -149,14 +126,22 @@ void wait_for_child_setup(pid_t pid)
int main(int argc, char **argv)
{
- u32 i;
- int s, r;
+ cpu_set_t allowed_cpu_set;
+ int s, r, cpu, i;
pid_t pid;
+ kvm_sched_getaffinity(0, sizeof(cpu_set_t), &allowed_cpu_set);
+
+ for (i = 0; i < NR_VCPUS && CPU_COUNT(&allowed_cpu_set); i++) {
+ cpu = kvm_pick_random_cpu(&allowed_cpu_set);
+ CPU_CLR(cpu, &allowed_cpu_set);
+ CPU_SET(cpu, &threads_cpu_set);
+ }
+
sem = sem_open("vm_sem", O_CREAT | O_EXCL, 0644, 0);
sem_unlink("vm_sem");
- for (i = 0; i < FORK_NUM; ++i) {
+ for (i = 0; i < NR_ITERATIONS; ++i) {
pid = fork();
TEST_ASSERT(pid >= 0, "%s: unable to fork", __func__);
if (pid == 0)
diff --git a/tools/testing/selftests/kvm/include/kvm_syscalls.h b/tools/testing/selftests/kvm/include/kvm_syscalls.h
index 6cb3bed29b81..dc7461894aef 100644
--- a/tools/testing/selftests/kvm/include/kvm_syscalls.h
+++ b/tools/testing/selftests/kvm/include/kvm_syscalls.h
@@ -12,6 +12,8 @@
#include <sys/mman.h>
#include <sys/syscall.h>
+#include <pthread.h>
+#include <sched.h>
#include <test_util.h>
#define MAP_ARGS0(m,...)
@@ -83,11 +85,64 @@ static inline int kvm_dup(int fd)
return new_fd;
}
+static inline pid_t kvm_gettid(void)
+{
+ return syscall(__NR_gettid);
+}
+
__KVM_SYSCALL_DEFINE(munmap, 2, void *, mem, size_t, size);
__KVM_SYSCALL_DEFINE(close, 1, int, fd);
__KVM_SYSCALL_DEFINE(fallocate, 4, int, fd, int, mode, loff_t, offset, loff_t, len);
__KVM_SYSCALL_DEFINE(ftruncate, 2, unsigned int, fd, off_t, length);
__KVM_SYSCALL_DEFINE(madvise, 3, void *, addr, size_t, length, int, advice);
+__KVM_SYSCALL_DEFINE(sched_getaffinity, 3, pid_t, pid, size_t, cpusetsize, cpu_set_t *, mask);
+__KVM_SYSCALL_DEFINE(sched_setaffinity, 3, pid_t, pid, size_t, cpusetsize, cpu_set_t *, mask);
+
+__KVM_SYSCALL_DEFINE(pthread_getaffinity_np, 3, pthread_t, thread,
+ size_t, cpusetsize, cpu_set_t *, cpuset);
+__KVM_SYSCALL_DEFINE(pthread_setaffinity_np, 3, pthread_t, thread,
+ size_t, cpusetsize, const cpu_set_t *, cpuset);
+
+static inline void kvm_pthread_getaffinity(pthread_t thread, cpu_set_t *cpuset)
+{
+ kvm_pthread_getaffinity_np(thread, sizeof(cpu_set_t), cpuset);
+}
+
+static inline void kvm_pthread_setaffinity(pthread_t thread,
+ const cpu_set_t *cpuset)
+{
+ kvm_pthread_setaffinity_np(thread, sizeof(cpu_set_t), cpuset);
+}
+
+typedef void *(*pthread_fn_t)(void *);
+__KVM_SYSCALL_DEFINE(pthread_create, 4, pthread_t *, thread,
+ const pthread_attr_t *, attr, pthread_fn_t, fn, void *, arg);
+__KVM_SYSCALL_DEFINE(pthread_join, 2, pthread_t, thread, void **, thread_return);
+__KVM_SYSCALL_DEFINE(pthread_cancel, 1, pthread_t, thread);
+
+static inline void __kvm_pthread_cancel_join(pthread_t thread, void **r)
+{
+ kvm_pthread_cancel(thread);
+ kvm_pthread_join(thread, r);
+}
+
+static inline void kvm_pthread_cancel_join(pthread_t thread)
+{
+ __kvm_pthread_cancel_join(thread, NULL);
+}
+
+/*
+ * Cancel+Join a pthread that was configured with PTHREAD_CANCEL_ASYNCHRONOUS
+ * and is expected to exit only in response to cancellation.
+ */
+static inline void kvm_pthread_cancel_join_async(pthread_t thread)
+{
+ void *r;
+
+ __kvm_pthread_cancel_join(thread, &r);
+ TEST_ASSERT(r == PTHREAD_CANCELED,
+ "expected retval=%p, got %p", PTHREAD_CANCELED, r);
+}
#define kvm_free_fd(fd) \
do { \
diff --git a/tools/testing/selftests/kvm/include/kvm_util.h b/tools/testing/selftests/kvm/include/kvm_util.h
index 04a910164a29..5a7a455b5387 100644
--- a/tools/testing/selftests/kvm/include/kvm_util.h
+++ b/tools/testing/selftests/kvm/include/kvm_util.h
@@ -1084,6 +1084,8 @@ struct kvm_vcpu *vm_recreate_with_one_vcpu(struct kvm_vm *vm);
void kvm_set_files_rlimit(u32 nr_vcpus);
+int kvm_pick_random_cpu(cpu_set_t *possible_cpus);
+
int __pin_task_to_cpu(pthread_t task, int cpu);
static inline void pin_task_to_cpu(pthread_t task, int cpu)
@@ -1094,6 +1096,15 @@ static inline void pin_task_to_cpu(pthread_t task, int cpu)
TEST_ASSERT(!r, "Failed to set thread affinity to pCPU '%u'", cpu);
}
+static inline int pin_task_to_random_cpu(pthread_t task, cpu_set_t *possible_cpus)
+{
+ int cpu;
+
+ cpu = kvm_pick_random_cpu(possible_cpus);
+ pin_task_to_cpu(task, cpu);
+ return cpu;
+}
+
static inline int pin_task_to_any_cpu(pthread_t task)
{
int cpu = sched_getcpu();
@@ -1138,6 +1149,16 @@ vm_adjust_num_guest_pages(enum vm_guest_mode mode, unsigned int num_guest_pages)
memcpy(&(g), _p, sizeof(g)); \
})
+#define SYNC_FROM_GUEST_AND_READ(_vm, _variable) ({ \
+ sync_global_from_guest(_vm, _variable); \
+ READ_ONCE(_variable); \
+})
+
+#define WRITE_AND_SYNC_TO_GUEST(_vm, _variable, _value) do { \
+ WRITE_ONCE(_variable, _value); \
+ sync_global_to_guest(_vm, _variable); \
+} while (0)
+
/*
* Write a global value, but only in the VM's (guest's) domain. Primarily used
* for "globals" that hold per-VM values (VMs always duplicate code and global
diff --git a/tools/testing/selftests/kvm/include/loongarch/processor.h b/tools/testing/selftests/kvm/include/loongarch/processor.h
index 93dc1fbd2e79..0e59daa8024e 100644
--- a/tools/testing/selftests/kvm/include/loongarch/processor.h
+++ b/tools/testing/selftests/kvm/include/loongarch/processor.h
@@ -70,6 +70,16 @@
#define PS_64K 0x00000010
#define PS_DEFAULT_SIZE PS_16K
+#define LOONGARCH_CPUCFG2 0x2
+#define CPUCFG2_FP BIT(0)
+#define CPUCFG2_FPSP BIT(1)
+#define CPUCFG2_FPDP BIT(2)
+#define CPUCFG2_FPVERS GENMASK(5, 3)
+#define CPUCFG2_LSX BIT(6)
+#define CPUCFG2_LASX BIT(7)
+#define CPUCFG2_LLFTP BIT(14)
+#define CPUCFG2_LLFTPREV GENMASK(17, 15)
+
/* LoongArch Basic CSR registers */
#define LOONGARCH_CSR_CRMD 0x0 /* Current mode info */
#define CSR_CRMD_PG_SHIFT 4
@@ -82,6 +92,14 @@
#define PLV_MASK 0x3
#define LOONGARCH_CSR_PRMD 0x1
#define LOONGARCH_CSR_EUEN 0x2
+#define CSR_EUEN_LBTEN_SHIFT 3
+#define CSR_EUEN_LBTEN BIT_ULL(CSR_EUEN_LBTEN_SHIFT)
+#define CSR_EUEN_LASXEN_SHIFT 2
+#define CSR_EUEN_LASXEN BIT_ULL(CSR_EUEN_LASXEN_SHIFT)
+#define CSR_EUEN_LSXEN_SHIFT 1
+#define CSR_EUEN_LSXEN BIT_ULL(CSR_EUEN_LSXEN_SHIFT)
+#define CSR_EUEN_FPEN_SHIFT 0
+#define CSR_EUEN_FPEN BIT_ULL(CSR_EUEN_FPEN_SHIFT)
#define LOONGARCH_CSR_ECFG 0x4
#define ECFGB_PMU 10
#define ECFGF_PMU (BIT_ULL(ECFGB_PMU))
diff --git a/tools/testing/selftests/kvm/include/proc_util.h b/tools/testing/selftests/kvm/include/proc_util.h
new file mode 100644
index 000000000000..d1ddc967d11d
--- /dev/null
+++ b/tools/testing/selftests/kvm/include/proc_util.h
@@ -0,0 +1,11 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+#ifndef SELFTEST_KVM_PROC_UTIL_H
+#define SELFTEST_KVM_PROC_UTIL_H
+
+#include <stdint.h>
+
+unsigned int vfio_msix_to_host_irq(const char *vfio_device_bdf, int msix);
+
+void proc_irq_set_smp_affinity(unsigned int irq, int cpu);
+
+#endif /* SELFTEST_KVM_PROC_UTIL_H */
diff --git a/tools/testing/selftests/kvm/include/test_util.h b/tools/testing/selftests/kvm/include/test_util.h
index a56271c237ae..a6a3e1657895 100644
--- a/tools/testing/selftests/kvm/include/test_util.h
+++ b/tools/testing/selftests/kvm/include/test_util.h
@@ -23,6 +23,7 @@
#include <linux/mman.h>
#include <linux/types.h>
+#include <linux/stringify.h>
#define msecs_to_usecs(msec) ((msec) * 1000ULL)
@@ -108,32 +109,35 @@ struct timespec timespec_sub(struct timespec ts1, struct timespec ts2);
struct timespec timespec_elapsed(struct timespec start);
struct timespec timespec_div(struct timespec ts, int divisor);
-struct guest_random_state {
+struct kvm_random_state {
u32 seed;
};
-extern u32 guest_random_seed;
-extern struct guest_random_state guest_rng;
+extern u32 kvm_random_seed;
+extern struct kvm_random_state kvm_rng;
-struct guest_random_state new_guest_random_state(u32 seed);
-u32 guest_random_u32(struct guest_random_state *state);
+struct kvm_random_state new_kvm_random_state(u32 seed);
+u32 kvm_random_u32(struct kvm_random_state *state);
-static inline bool __guest_random_bool(struct guest_random_state *state,
+static inline bool __kvm_random_bool(struct kvm_random_state *state,
u8 percent)
{
- return (guest_random_u32(state) % 100) < percent;
+ return (kvm_random_u32(state) % 100) < percent;
}
-static inline bool guest_random_bool(struct guest_random_state *state)
+static inline bool kvm_random_bool(struct kvm_random_state *state)
{
- return __guest_random_bool(state, 50);
+ return __kvm_random_bool(state, 50);
}
-static inline u64 guest_random_u64(struct guest_random_state *state)
+static inline u64 kvm_random_u64(struct kvm_random_state *state)
{
- return ((u64)guest_random_u32(state) << 32) | guest_random_u32(state);
+ return ((u64)kvm_random_u32(state) << 32) | kvm_random_u32(state);
}
+u32 kvm_random_u32_in_range(struct kvm_random_state *state, u32 min, u32 max);
+u64 kvm_random_u64_in_range(struct kvm_random_state *state, u64 min, u64 max);
+
enum vm_mem_backing_src_type {
VM_MEM_SRC_ANONYMOUS,
VM_MEM_SRC_ANONYMOUS_THP,
diff --git a/tools/testing/selftests/kvm/include/x86/evmcs.h b/tools/testing/selftests/kvm/include/x86/evmcs.h
index be79bda024bf..82a8ea6b661f 100644
--- a/tools/testing/selftests/kvm/include/x86/evmcs.h
+++ b/tools/testing/selftests/kvm/include/x86/evmcs.h
@@ -1207,30 +1207,23 @@ static inline int evmcs_vmlaunch(void)
current_evmcs->hv_clean_fields = 0;
- __asm__ __volatile__("push %%rbp;"
- "push %%rcx;"
- "push %%rdx;"
- "push %%rsi;"
- "push %%rdi;"
- "push $0;"
+ __asm__ __volatile__("push $0;"
"mov %%rsp, (%[host_rsp]);"
"lea 1f(%%rip), %%rax;"
"mov %%rax, (%[host_rip]);"
+ VMX_SWITCH_GPRS_ASM
"vmlaunch;"
"incq (%%rsp);"
- "1: pop %%rax;"
- "pop %%rdi;"
- "pop %%rsi;"
- "pop %%rdx;"
- "pop %%rcx;"
- "pop %%rbp;"
+ "1: ;"
+ VMX_SWITCH_GPRS_ASM
+ "pop %%rax;"
: [ret]"=&a"(ret)
: [host_rsp]"r"
((u64)&current_evmcs->host_rsp),
[host_rip]"r"
- ((u64)&current_evmcs->host_rip)
- : "memory", "cc", "rbx", "r8", "r9", "r10",
- "r11", "r12", "r13", "r14", "r15");
+ ((u64)&current_evmcs->host_rip),
+ GUEST_REGS_OFFSETS
+ : "memory", "cc");
return ret;
}
@@ -1246,30 +1239,23 @@ static inline int evmcs_vmresume(void)
/* HOST_RSP */
current_evmcs->hv_clean_fields &= ~HV_VMX_ENLIGHTENED_CLEAN_FIELD_HOST_POINTER;
- __asm__ __volatile__("push %%rbp;"
- "push %%rcx;"
- "push %%rdx;"
- "push %%rsi;"
- "push %%rdi;"
- "push $0;"
+ __asm__ __volatile__("push $0;"
"mov %%rsp, (%[host_rsp]);"
"lea 1f(%%rip), %%rax;"
"mov %%rax, (%[host_rip]);"
+ VMX_SWITCH_GPRS_ASM
"vmresume;"
"incq (%%rsp);"
- "1: pop %%rax;"
- "pop %%rdi;"
- "pop %%rsi;"
- "pop %%rdx;"
- "pop %%rcx;"
- "pop %%rbp;"
+ "1: ;"
+ VMX_SWITCH_GPRS_ASM
+ "pop %%rax;"
: [ret]"=&a"(ret)
: [host_rsp]"r"
((u64)&current_evmcs->host_rsp),
[host_rip]"r"
- ((u64)&current_evmcs->host_rip)
- : "memory", "cc", "rbx", "r8", "r9", "r10",
- "r11", "r12", "r13", "r14", "r15");
+ ((u64)&current_evmcs->host_rip),
+ GUEST_REGS_OFFSETS
+ : "memory", "cc");
return ret;
}
diff --git a/tools/testing/selftests/kvm/include/x86/kvm_util_arch.h b/tools/testing/selftests/kvm/include/x86/kvm_util_arch.h
index c33ab6e04171..6904dbda79f9 100644
--- a/tools/testing/selftests/kvm/include/x86/kvm_util_arch.h
+++ b/tools/testing/selftests/kvm/include/x86/kvm_util_arch.h
@@ -55,9 +55,9 @@ static inline bool __vm_arch_has_protected_memory(struct kvm_vm_arch *arch)
do { \
const typeof(mem) val = (__val); \
\
- if (!is_forced_emulation_enabled || guest_random_bool(&guest_rng)) { \
+ if (!is_forced_emulation_enabled || kvm_random_bool(&kvm_rng)) { \
(mem) = val; \
- } else if (guest_random_bool(&guest_rng)) { \
+ } else if (kvm_random_bool(&kvm_rng)) { \
__asm__ __volatile__(KVM_FEP "mov %1, %0" \
: "+m" (mem) \
: "r" (val) : "memory"); \
diff --git a/tools/testing/selftests/kvm/include/x86/pmu.h b/tools/testing/selftests/kvm/include/x86/pmu.h
index 98537cc8840d..608ed83d7c6a 100644
--- a/tools/testing/selftests/kvm/include/x86/pmu.h
+++ b/tools/testing/selftests/kvm/include/x86/pmu.h
@@ -38,6 +38,12 @@
#define ARCH_PERFMON_EVENTSEL_INV BIT_ULL(23)
#define ARCH_PERFMON_EVENTSEL_CMASK GENMASK_ULL(31, 24)
+/*
+ * These are AMD-specific bits.
+ */
+#define AMD64_EVENTSEL_GUESTONLY BIT_ULL(40)
+#define AMD64_EVENTSEL_HOSTONLY BIT_ULL(41)
+
/* RDPMC control flags, Intel only. */
#define INTEL_RDPMC_METRICS BIT_ULL(29)
#define INTEL_RDPMC_FIXED BIT_ULL(30)
diff --git a/tools/testing/selftests/kvm/include/x86/processor.h b/tools/testing/selftests/kvm/include/x86/processor.h
index 513e4a1075fa..6e6f70035508 100644
--- a/tools/testing/selftests/kvm/include/x86/processor.h
+++ b/tools/testing/selftests/kvm/include/x86/processor.h
@@ -24,6 +24,7 @@ extern bool host_cpu_is_amd;
extern bool host_cpu_is_hygon;
extern bool host_cpu_is_amd_compatible;
extern u64 guest_tsc_khz;
+extern struct kvm_mmu guest_mmu;
#ifndef MAX_NR_CPUID_ENTRIES
#define MAX_NR_CPUID_ENTRIES 100
@@ -208,6 +209,7 @@ struct kvm_x86_cpu_feature {
#define X86_FEATURE_PERFCTR_NB KVM_X86_CPU_FEATURE(0x80000001, 0, ECX, 24)
#define X86_FEATURE_PERFCTR_LLC KVM_X86_CPU_FEATURE(0x80000001, 0, ECX, 28)
#define X86_FEATURE_NX KVM_X86_CPU_FEATURE(0x80000001, 0, EDX, 20)
+#define X86_FEATURE_FXSR_OPT KVM_X86_CPU_FEATURE(0x80000001, 0, EDX, 25)
#define X86_FEATURE_GBPAGES KVM_X86_CPU_FEATURE(0x80000001, 0, EDX, 26)
#define X86_FEATURE_RDTSCP KVM_X86_CPU_FEATURE(0x80000001, 0, EDX, 27)
#define X86_FEATURE_LM KVM_X86_CPU_FEATURE(0x80000001, 0, EDX, 29)
@@ -226,6 +228,7 @@ struct kvm_x86_cpu_feature {
#define X86_FEATURE_SEV KVM_X86_CPU_FEATURE(0x8000001F, 0, EAX, 1)
#define X86_FEATURE_SEV_ES KVM_X86_CPU_FEATURE(0x8000001F, 0, EAX, 3)
#define X86_FEATURE_SEV_SNP KVM_X86_CPU_FEATURE(0x8000001F, 0, EAX, 4)
+#define X86_FEATURE_AUTOIBRS KVM_X86_CPU_FEATURE(0x80000021, 0, EAX, 8)
#define X86_FEATURE_GP_ON_USER_CPUID KVM_X86_CPU_FEATURE(0x80000021, 0, EAX, 17)
#define X86_FEATURE_PERFMON_V2 KVM_X86_CPU_FEATURE(0x80000022, 0, EAX, 0)
#define X86_FEATURE_LBR_PMC_FREEZE KVM_X86_CPU_FEATURE(0x80000022, 0, EAX, 2)
@@ -396,8 +399,7 @@ static inline unsigned int x86_model(unsigned int eax)
#define PTE_GET_PA(pte) ((pte) & PHYSICAL_PAGE_MASK)
#define PTE_GET_PFN(pte) (PTE_GET_PA(pte) >> PAGE_SHIFT)
-/* General Registers in 64-Bit Mode */
-struct gpr64_regs {
+struct guest_regs {
u64 rax;
u64 rcx;
u64 rdx;
@@ -414,8 +416,38 @@ struct gpr64_regs {
u64 r13;
u64 r14;
u64 r15;
+ u64 rflags;
};
+extern struct guest_regs guest_regs;
+
+#define GUEST_REG_OFFSET(name) \
+ [off_##name] "i" (offsetof(struct guest_regs, name))
+
+#define GUEST_REGS_OFFSETS \
+ GUEST_REG_OFFSET(rax), \
+ GUEST_REG_OFFSET(rcx), \
+ GUEST_REG_OFFSET(rdx), \
+ GUEST_REG_OFFSET(rbx), \
+ GUEST_REG_OFFSET(rsp), \
+ GUEST_REG_OFFSET(rbp), \
+ GUEST_REG_OFFSET(rsi), \
+ GUEST_REG_OFFSET(rdi), \
+ GUEST_REG_OFFSET(r8), \
+ GUEST_REG_OFFSET(r9), \
+ GUEST_REG_OFFSET(r10), \
+ GUEST_REG_OFFSET(r11), \
+ GUEST_REG_OFFSET(r12), \
+ GUEST_REG_OFFSET(r13), \
+ GUEST_REG_OFFSET(r14), \
+ GUEST_REG_OFFSET(r15), \
+ GUEST_REG_OFFSET(rflags)
+
+#define GUEST_REG(name) "guest_regs + %c[off_" #name "]"
+
+#define GUEST_SWITCH_GPR_ASM(name) \
+ "xchg %%" #name ", " GUEST_REG(name) "\n\t"
+
struct desc64 {
u16 limit0;
u16 base0;
@@ -582,6 +614,14 @@ static inline void set_cr0(u64 val)
__asm__ __volatile__("mov %0, %%cr0" : : "r" (val) : "memory");
}
+static inline u64 get_cr2(void)
+{
+ u64 cr2;
+
+ __asm__ __volatile__("mov %%cr2, %[cr2]" : [cr2]"=r"(cr2));
+ return cr2;
+}
+
static inline u64 get_cr3(void)
{
u64 cr3;
@@ -877,6 +917,11 @@ static inline void write_sse_reg(int reg, const sse128_t *data)
}
}
+static inline void invlpg(u64 addr)
+{
+ __asm__ __volatile__("invlpg (%0)" : : "r"(addr) : "memory");
+}
+
static inline void cpu_relax(void)
{
asm volatile("rep; nop" ::: "memory");
@@ -913,6 +958,11 @@ struct kvm_x86_state *vcpu_save_state(struct kvm_vcpu *vcpu);
void vcpu_load_state(struct kvm_vcpu *vcpu, struct kvm_x86_state *state);
void kvm_x86_state_cleanup(struct kvm_x86_state *state);
+static inline bool kvm_x86_state_is_guest_mode(struct kvm_x86_state *state)
+{
+ return state->nested.size && (state->nested.flags & KVM_STATE_NESTED_GUEST_MODE);
+}
+
const struct kvm_msr_list *kvm_get_msr_index_list(void);
const struct kvm_msr_list *kvm_get_feature_msr_index_list(void);
bool kvm_msr_is_in_save_restore_list(u32 msr_index);
@@ -1226,6 +1276,8 @@ struct idt_entry {
void vm_install_exception_handler(struct kvm_vm *vm, int vector,
void (*handler)(struct ex_regs *));
+gva_t vm_alloc_stack(struct kvm_vm *vm, int nr_pages);
+
/*
* Exception fixup morphs #DE to an arbitrary magic vector so that '0' can be
* used to signal "no expcetion".
@@ -1392,6 +1444,14 @@ static inline bool kvm_is_pmu_enabled(void)
return get_kvm_param_bool("enable_pmu");
}
+static inline bool kvm_is_mediated_pmu_enabled(void)
+{
+ if (host_cpu_is_intel)
+ return get_kvm_intel_param_bool("enable_mediated_pmu");
+
+ return get_kvm_amd_param_bool("enable_mediated_pmu");
+}
+
static inline bool kvm_is_forced_emulation_enabled(void)
{
return !!get_kvm_param_integer("force_emulation_prefix");
diff --git a/tools/testing/selftests/kvm/include/x86/svm_util.h b/tools/testing/selftests/kvm/include/x86/svm_util.h
index 6c013eb838be..c201c30485e7 100644
--- a/tools/testing/selftests/kvm/include/x86/svm_util.h
+++ b/tools/testing/selftests/kvm/include/x86/svm_util.h
@@ -28,6 +28,9 @@ struct svm_test_data {
void *msr_hva;
u64 msr_gpa;
+ /* Stack */
+ void *stack; /* gva */
+
/* NPT */
u64 ncr3_gpa;
};
@@ -57,7 +60,7 @@ static inline void vmmcall(void)
)
struct svm_test_data *vcpu_alloc_svm(struct kvm_vm *vm, gva_t *p_svm_gva);
-void generic_svm_setup(struct svm_test_data *svm, void *guest_rip, void *guest_rsp);
+void generic_svm_setup(struct svm_test_data *svm, void *guest_rip);
void run_guest(struct vmcb *vmcb, u64 vmcb_gpa);
static inline bool kvm_cpu_has_npt(void)
diff --git a/tools/testing/selftests/kvm/include/x86/vmx.h b/tools/testing/selftests/kvm/include/x86/vmx.h
index 90fffaf91595..04f5e34dea3a 100644
--- a/tools/testing/selftests/kvm/include/x86/vmx.h
+++ b/tools/testing/selftests/kvm/include/x86/vmx.h
@@ -290,6 +290,23 @@ struct vmx_msr_entry {
u64 value;
} __attribute__ ((aligned(16)));
+#define VMX_SWITCH_GPRS_ASM \
+ GUEST_SWITCH_GPR_ASM(rax) \
+ GUEST_SWITCH_GPR_ASM(rbx) \
+ GUEST_SWITCH_GPR_ASM(rcx) \
+ GUEST_SWITCH_GPR_ASM(rdx) \
+ GUEST_SWITCH_GPR_ASM(rbp) \
+ GUEST_SWITCH_GPR_ASM(rsi) \
+ GUEST_SWITCH_GPR_ASM(rdi) \
+ GUEST_SWITCH_GPR_ASM(r8) \
+ GUEST_SWITCH_GPR_ASM(r9) \
+ GUEST_SWITCH_GPR_ASM(r10) \
+ GUEST_SWITCH_GPR_ASM(r11) \
+ GUEST_SWITCH_GPR_ASM(r12) \
+ GUEST_SWITCH_GPR_ASM(r13) \
+ GUEST_SWITCH_GPR_ASM(r14) \
+ GUEST_SWITCH_GPR_ASM(r15)
+
#include "evmcs.h"
static inline int vmxon(u64 phys)
@@ -363,9 +380,6 @@ static inline u64 vmptrstz(void)
return value;
}
-/*
- * No guest state (e.g. GPRs) is established by this vmlaunch.
- */
static inline int vmlaunch(void)
{
int ret;
@@ -373,34 +387,24 @@ static inline int vmlaunch(void)
if (enable_evmcs)
return evmcs_vmlaunch();
- __asm__ __volatile__("push %%rbp;"
- "push %%rcx;"
- "push %%rdx;"
- "push %%rsi;"
- "push %%rdi;"
- "push $0;"
+ __asm__ __volatile__("push $0;"
"vmwrite %%rsp, %[host_rsp];"
"lea 1f(%%rip), %%rax;"
"vmwrite %%rax, %[host_rip];"
+ VMX_SWITCH_GPRS_ASM
"vmlaunch;"
"incq (%%rsp);"
- "1: pop %%rax;"
- "pop %%rdi;"
- "pop %%rsi;"
- "pop %%rdx;"
- "pop %%rcx;"
- "pop %%rbp;"
+ "1: ;"
+ VMX_SWITCH_GPRS_ASM
+ "pop %%rax;"
: [ret]"=&a"(ret)
: [host_rsp]"r"((u64)HOST_RSP),
- [host_rip]"r"((u64)HOST_RIP)
- : "memory", "cc", "rbx", "r8", "r9", "r10",
- "r11", "r12", "r13", "r14", "r15");
+ [host_rip]"r"((u64)HOST_RIP),
+ GUEST_REGS_OFFSETS
+ : "memory", "cc");
return ret;
}
-/*
- * No guest state (e.g. GPRs) is established by this vmresume.
- */
static inline int vmresume(void)
{
int ret;
@@ -408,28 +412,21 @@ static inline int vmresume(void)
if (enable_evmcs)
return evmcs_vmresume();
- __asm__ __volatile__("push %%rbp;"
- "push %%rcx;"
- "push %%rdx;"
- "push %%rsi;"
- "push %%rdi;"
- "push $0;"
+ __asm__ __volatile__("push $0;"
"vmwrite %%rsp, %[host_rsp];"
"lea 1f(%%rip), %%rax;"
"vmwrite %%rax, %[host_rip];"
+ VMX_SWITCH_GPRS_ASM
"vmresume;"
"incq (%%rsp);"
- "1: pop %%rax;"
- "pop %%rdi;"
- "pop %%rsi;"
- "pop %%rdx;"
- "pop %%rcx;"
- "pop %%rbp;"
+ "1: ;"
+ VMX_SWITCH_GPRS_ASM
+ "pop %%rax;"
: [ret]"=&a"(ret)
: [host_rsp]"r"((u64)HOST_RSP),
- [host_rip]"r"((u64)HOST_RIP)
- : "memory", "cc", "rbx", "r8", "r9", "r10",
- "r11", "r12", "r13", "r14", "r15");
+ [host_rip]"r"((u64)HOST_RIP),
+ GUEST_REGS_OFFSETS
+ : "memory", "cc");
return ret;
}
@@ -524,6 +521,8 @@ struct vmx_pages {
u64 apic_access_gpa;
void *apic_access;
+ void *stack;
+
u64 eptp_gpa;
};
@@ -552,7 +551,7 @@ union vmx_ctrl_msr {
struct vmx_pages *vcpu_alloc_vmx(struct kvm_vm *vm, gva_t *p_vmx_gva);
bool prepare_for_vmx_operation(struct vmx_pages *vmx);
-void prepare_vmcs(struct vmx_pages *vmx, void *guest_rip, void *guest_rsp);
+void prepare_vmcs(struct vmx_pages *vmx, void *guest_rip);
bool load_vmcs(struct vmx_pages *vmx);
bool ept_1g_pages_supported(void);
diff --git a/tools/testing/selftests/kvm/irq_test.c b/tools/testing/selftests/kvm/irq_test.c
new file mode 100644
index 000000000000..168c52734fef
--- /dev/null
+++ b/tools/testing/selftests/kvm/irq_test.c
@@ -0,0 +1,362 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "kvm_util.h"
+#include "test_util.h"
+#include "apic.h"
+#include "processor.h"
+#include "proc_util.h"
+
+#include <libvfio.h>
+#include <linux/sizes.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <unistd.h>
+#include <pthread.h>
+#include <sys/eventfd.h>
+#include <sys/sysinfo.h>
+
+static u64 timeout_ns = 2ULL * 1000 * 1000 * 1000;
+static bool guest_ready_for_irqs[KVM_MAX_VCPUS];
+static bool guest_received_irq[KVM_MAX_VCPUS];
+static bool guest_received_nmi[KVM_MAX_VCPUS];
+static bool x2apic = true;
+static bool irq_affinity;
+static bool done;
+
+#define GUEST_RECEIVED_IRQ(__vcpu) \
+ SYNC_FROM_GUEST_AND_READ((__vcpu)->vm, guest_received_irq[(__vcpu)->id])
+#define GUEST_RECEIVED_NMI(__vcpu) \
+ SYNC_FROM_GUEST_AND_READ((__vcpu)->vm, guest_received_nmi[(__vcpu)->id])
+
+#define GUEST_RECEIVED_INTERRUPT(__vcpu, __nmi) \
+ ((__nmi) ? GUEST_RECEIVED_NMI(__vcpu) : GUEST_RECEIVED_IRQ(__vcpu))
+
+static u32 guest_get_vcpu_id(void)
+{
+ if (x2apic)
+ return x2apic_read_reg(APIC_ID);
+ else
+ return xapic_read_reg(APIC_ID) >> 24;
+}
+
+static void guest_irq_handler(struct ex_regs *regs)
+{
+ WRITE_ONCE(guest_received_irq[guest_get_vcpu_id()], true);
+
+ if (x2apic)
+ x2apic_write_reg(APIC_EOI, 0);
+ else
+ xapic_write_reg(APIC_EOI, 0);
+}
+
+static void guest_nmi_handler(struct ex_regs *regs)
+{
+ WRITE_ONCE(guest_received_nmi[guest_get_vcpu_id()], true);
+}
+
+static void guest_code(void)
+{
+ if (x2apic)
+ x2apic_enable();
+ else
+ xapic_enable();
+
+ sti_nop();
+
+ WRITE_ONCE(guest_ready_for_irqs[guest_get_vcpu_id()], true);
+
+ while (!READ_ONCE(done))
+ cpu_relax();
+
+ GUEST_DONE();
+}
+
+static void *vcpu_thread_main(void *arg)
+{
+ struct kvm_vcpu *vcpu = arg;
+ struct ucall uc;
+
+ vcpu_run(vcpu);
+ TEST_ASSERT_EQ(UCALL_DONE, get_ucall(vcpu, &uc));
+
+ return NULL;
+}
+
+static int vfio_setup_msi(struct vfio_pci_device *device)
+{
+ const int flags = MAP_SHARED | MAP_ANONYMOUS;
+ const int prot = PROT_READ | PROT_WRITE;
+ struct iova_allocator *allocator;
+ struct dma_region *region;
+
+ /* Sanity check that the device+driver can actually send MSIs. */
+ TEST_REQUIRE(device->driver.ops);
+ TEST_REQUIRE(device->driver.ops->send_msi);
+
+ /*
+ * Set up a DMA-able region for the driver to use. Very few devices
+ * provide a way to arbitrarily send interrupts (MSIs), e.g. by writing
+ * an MMIO register. Instead, most devices send MSIs when an action is
+ * completed, and practically all actions involve DMA of some form.
+ */
+ allocator = iova_allocator_init(device->iommu);
+
+ region = &device->driver.region;
+ region->size = SZ_2M;
+ region->iova = iova_allocator_alloc(allocator, region->size);
+ region->vaddr = kvm_mmap(region->size, prot, flags, -1);
+ TEST_ASSERT(region->vaddr != MAP_FAILED, "mmap() failed\n");
+ iommu_map(device->iommu, region);
+
+ iova_allocator_cleanup(allocator);
+
+ vfio_pci_driver_init(device);
+
+ return device->driver.msi;
+}
+
+static void trigger_interrupt(struct vfio_pci_device *device, int eventfd)
+{
+ if (device)
+ vfio_pci_driver_send_msi(device);
+ else
+ eventfd_write(eventfd, 1);
+}
+
+
+static void kvm_route_msi(struct kvm_vm *vm, u32 gsi, struct kvm_vcpu *vcpu,
+ u8 vector, bool use_nmi)
+{
+ struct {
+ struct kvm_irq_routing header;
+ struct kvm_irq_routing_entry entry;
+ } routing = {
+ .header.nr = 1,
+ .entry = {
+ .gsi = gsi,
+ .type = KVM_IRQ_ROUTING_MSI,
+ .u.msi.address_lo = 0xFEE00000 | (vcpu->id & GENMASK(7, 0)) << 12,
+ .u.msi.address_hi = vcpu->id & GENMASK(31, 8),
+ .u.msi.data = use_nmi ? NMI_VECTOR | (4 << 8) : vector,
+ },
+ };
+
+ vm_ioctl(vm, KVM_SET_GSI_ROUTING, &routing.header);
+}
+
+static void kvm_set_empty_gsi_routing(struct kvm_vm *vm)
+{
+ struct kvm_irq_routing routing = {};
+
+ vm_ioctl(vm, KVM_SET_GSI_ROUTING, &routing);
+}
+
+static const char *probe_iommu_type(void)
+{
+ int io_fd;
+
+ io_fd = open("/dev/iommu", O_RDONLY);
+ if (io_fd >= 0) {
+ close(io_fd);
+ return MODE_IOMMUFD;
+ }
+
+ io_fd = __open_path_or_exit("/dev/vfio/vfio", O_RDONLY,
+ "Is VFIO (or IOMMUFD) loaded and enabled?");
+ close(io_fd);
+ return MODE_VFIO_TYPE1_IOMMU;
+}
+
+static void help(const char *name)
+{
+ printf("Usage: %s [-a] [-d <segment:bus:device.function>] [-e] [-h] [-i nr_irqs] [-m] [-n] [-t iommu_type] [-v nr_vcpus] [-x]\n", name);
+ printf("\n");
+ printf("Tests KVM interrupt routing and delivery via irqfd.\n");
+ printf("-a Affine the device's host IRQ to a random physical CPU\n");
+ printf("-d Use a VFIO device to send MSI-X interrupts instead of manually signaling the eventfd\n");
+ printf("-e Set empty GSI routing in-between some interrupts\n");
+ printf("-i The number of IRQs to generate during the test\n");
+ printf("-m Pin target vCPU to random physical CPU before triggering interrupt\n");
+ printf("-n Deliver 50 percent of IRQs as non-maskable interrupts\n");
+ printf("-t Override the IOMMU type to use (vfio_type1_iommu or iommufd)\n");
+ printf("-v Number of vCPUS to run\n");
+ printf("-x Use xAPIC mode instead of x2APIC mode in the guest\n");
+ printf("\n");
+ exit(KSFT_FAIL);
+}
+
+int main(int argc, char **argv)
+{
+ /*
+ * Pick a random vector and a random GSI to use for device IRQ.
+ *
+ * Pick an IRQ vector in range [32, UINT8_MAX]. Min value is 32 because
+ * Linux/x86 reserves vectors 0-31 for exceptions and architecture
+ * defined NMIs and interrupts.
+ *
+ * Pick a GSI in range [24, KVM_MAX_IRQ_ROUTES - 1]. The min value is 24
+ * because KVM reserves GSIs 0-15 for legacy ISA IRQs and 16-23 only go
+ * to the IOAPIC. The max is KVM_MAX_IRQ_ROUTES - 1, because
+ * KVM_MAX_IRQ_ROUTES is exclusive.
+ */
+ u32 gsi = kvm_random_u64_in_range(&kvm_rng, 24, KVM_MAX_IRQ_ROUTES - 1);
+ u8 vector = kvm_random_u64_in_range(&kvm_rng, 32, UINT8_MAX);
+
+ pthread_t vcpu_threads[KVM_MAX_VCPUS];
+ struct kvm_vcpu *vcpus[KVM_MAX_VCPUS];
+ struct vfio_pci_device *device = NULL;
+ int nr_irqs = 1000, nr_vcpus = 1;
+ bool set_empty_routing = false;
+ const char *device_bdf = NULL;
+ const char *iommu_type = NULL;
+ int i, j, c, msix, eventfd;
+ bool migrate_vcpus = false;
+ cpu_set_t available_cpus;
+ bool use_nmi = false;
+ struct iommu *iommu;
+ struct kvm_vm *vm;
+ int irq, irq_cpu;
+
+ while ((c = getopt(argc, argv, "ad:ehi:mnt:v:x")) != -1) {
+ switch (c) {
+ case 'a':
+ irq_affinity = true;
+ break;
+ case 'd':
+ device_bdf = optarg;
+ break;
+ case 'e':
+ set_empty_routing = true;
+ break;
+ case 'i':
+ nr_irqs = atoi_positive("Number of IRQs", optarg);
+ break;
+ case 'm':
+ migrate_vcpus = true;
+ break;
+ case 'n':
+ use_nmi = true;
+ break;
+ case 't':
+ iommu_type = optarg;
+ break;
+ case 'v':
+ nr_vcpus = atoi_positive("Number of vCPUS", optarg);
+ TEST_ASSERT(nr_vcpus <= KVM_MAX_VCPUS,
+ "KVM selftests support at most %u vCPUs", KVM_MAX_VCPUS);
+ break;
+ case 'x':
+ x2apic = false;
+ break;
+ case 'h':
+ default:
+ help(argv[0]);
+ }
+ }
+
+ TEST_REQUIRE(kvm_arch_has_default_irqchip());
+
+ vm = vm_create_with_vcpus(nr_vcpus, guest_code, vcpus);
+ vm_enable_cap(vm, KVM_CAP_X2APIC_API, KVM_X2APIC_API_USE_32BIT_IDS |
+ KVM_X2APIC_API_DISABLE_BROADCAST_QUIRK);
+
+ vm_install_exception_handler(vm, vector, guest_irq_handler);
+ vm_install_exception_handler(vm, NMI_VECTOR, guest_nmi_handler);
+
+ if (!x2apic) {
+ TEST_ASSERT(nr_vcpus < 256, "xAPIC can only target IDs [0-254] (255 vCPUs)");
+ virt_pg_map(vm, APIC_DEFAULT_GPA, APIC_DEFAULT_GPA);
+ }
+
+ if (device_bdf) {
+ if (!iommu_type)
+ iommu_type = probe_iommu_type();
+ iommu = iommu_init(iommu_type);
+ device = vfio_pci_device_init(device_bdf, iommu);
+ msix = vfio_setup_msi(device);
+ irq = vfio_msix_to_host_irq(device_bdf, msix);
+ eventfd = device->msi_eventfds[msix];
+ printf("Using device %s MSI-X[%d] (IRQ-%u)\n", device_bdf, msix,
+ irq);
+ } else {
+ TEST_ASSERT(!irq_affinity,
+ "Setting IRQ affinity (-a) requires a backing device (-d)");
+
+ eventfd = kvm_new_eventfd();
+ irq = -1;
+ }
+
+ pr_info("Injecting interrupts for GSI %d (guest vector 0x%x) %d times\n",
+ gsi, vector, nr_irqs);
+
+ kvm_assign_irqfd(vm, gsi, eventfd);
+
+ sync_global_to_guest(vm, x2apic);
+
+ if (migrate_vcpus)
+ kvm_sched_getaffinity(0, sizeof(available_cpus), &available_cpus);
+
+ for (i = 0; i < nr_vcpus; i++)
+ kvm_pthread_create(&vcpu_threads[i], NULL, vcpu_thread_main, vcpus[i]);
+
+ for (i = 0; i < nr_vcpus; i++) {
+ struct kvm_vcpu *vcpu = vcpus[i];
+
+ while (!SYNC_FROM_GUEST_AND_READ(vm, guest_ready_for_irqs[vcpu->id]))
+ continue;
+ }
+
+ irq_cpu = -1;
+
+ for (i = 0; i < nr_irqs; i++) {
+ const bool do_set_empty_routing = set_empty_routing && (i & BIT(3));
+ const bool do_use_nmi = use_nmi && (i & BIT(2));
+ struct kvm_vcpu *vcpu = vcpus[i % nr_vcpus];
+ struct timespec start;
+
+ if (do_set_empty_routing)
+ kvm_set_empty_gsi_routing(vm);
+
+ kvm_route_msi(vm, gsi, vcpu, vector, do_use_nmi);
+
+ if (irq_affinity) {
+ irq_cpu = kvm_random_u64(&kvm_rng) % get_nprocs();
+ proc_irq_set_smp_affinity(irq, irq_cpu);
+ }
+
+ if (migrate_vcpus)
+ pin_task_to_random_cpu(vcpu_threads[i % nr_vcpus], &available_cpus);
+
+ for (j = 0; j < nr_vcpus; j++) {
+ TEST_ASSERT(!GUEST_RECEIVED_IRQ(vcpus[j]),
+ "IRQ flag for vCPU %d not clear prior to test",
+ vcpus[j]->id);
+ TEST_ASSERT(!GUEST_RECEIVED_NMI(vcpus[j]),
+ "NMI flag for vCPU %d not clear prior to test",
+ vcpus[j]->id);
+ }
+
+ trigger_interrupt(device, eventfd);
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+ while (!GUEST_RECEIVED_INTERRUPT(vcpu, do_use_nmi) &&
+ timespec_to_ns(timespec_elapsed(start)) <= timeout_ns)
+ cpu_relax();
+
+ TEST_ASSERT(GUEST_RECEIVED_INTERRUPT(vcpu, do_use_nmi),
+ "vCPU %d timed out waiting for %s (vector 0x%x) from GSI %d (via CPU %d)\n",
+ vcpu->id, do_use_nmi ? "NMI" : "IRQ",
+ do_use_nmi ? NMI_VECTOR : vector, gsi, irq_cpu);
+
+ if (do_use_nmi)
+ WRITE_AND_SYNC_TO_GUEST(vm, guest_received_nmi[vcpu->id], false);
+ else
+ WRITE_AND_SYNC_TO_GUEST(vm, guest_received_irq[vcpu->id], false);
+ }
+
+ WRITE_AND_SYNC_TO_GUEST(vm, done, true);
+
+ for (i = 0; i < nr_vcpus; i++)
+ kvm_pthread_join(vcpu_threads[i], NULL);
+
+ return 0;
+}
diff --git a/tools/testing/selftests/kvm/irqfd_test.c b/tools/testing/selftests/kvm/irqfd_test.c
index 5d7590d01868..ebb2478878de 100644
--- a/tools/testing/selftests/kvm/irqfd_test.c
+++ b/tools/testing/selftests/kvm/irqfd_test.c
@@ -128,7 +128,7 @@ int main(int argc, char *argv[])
close(__eventfd);
- pthread_create(&racing_thread, NULL, secondary_irqfd_juggler, vm2);
+ kvm_pthread_create(&racing_thread, NULL, secondary_irqfd_juggler, vm2);
for (i = 0; i < 10000; i++) {
WRITE_ONCE(__eventfd, kvm_new_eventfd());
@@ -139,5 +139,5 @@ int main(int argc, char *argv[])
}
WRITE_ONCE(done, true);
- pthread_join(racing_thread, NULL);
+ kvm_pthread_join(racing_thread, NULL);
}
diff --git a/tools/testing/selftests/kvm/kvm_page_table_test.c b/tools/testing/selftests/kvm/kvm_page_table_test.c
index a910e3abb8c7..974813ea8953 100644
--- a/tools/testing/selftests/kvm/kvm_page_table_test.c
+++ b/tools/testing/selftests/kvm/kvm_page_table_test.c
@@ -364,8 +364,8 @@ static void run_test(enum vm_guest_mode mode, void *arg)
*current_stage = KVM_BEFORE_MAPPINGS;
for (i = 0; i < nr_vcpus; i++)
- pthread_create(&vcpu_threads[i], NULL, vcpu_worker,
- test_args.vcpus[i]);
+ kvm_pthread_create(&vcpu_threads[i], NULL, vcpu_worker,
+ test_args.vcpus[i]);
vcpus_complete_new_stage(*current_stage);
pr_info("Started all vCPUs successfully\n");
@@ -413,7 +413,7 @@ static void run_test(enum vm_guest_mode mode, void *arg)
}
for (i = 0; i < nr_vcpus; i++)
- pthread_join(vcpu_threads[i], NULL);
+ kvm_pthread_join(vcpu_threads[i], NULL);
ret = sem_destroy(&test_stage_updated);
TEST_ASSERT(ret == 0, "Error in sem_destroy");
diff --git a/tools/testing/selftests/kvm/lib/assert.c b/tools/testing/selftests/kvm/lib/assert.c
index 8be0d09ecf0f..3e353ac39eeb 100644
--- a/tools/testing/selftests/kvm/lib/assert.c
+++ b/tools/testing/selftests/kvm/lib/assert.c
@@ -10,6 +10,7 @@
#include <sys/syscall.h>
#include "kselftest.h"
+#include "kvm_syscalls.h"
#ifdef __GLIBC__
#include <execinfo.h>
@@ -64,11 +65,6 @@ static void test_dump_stack(void)
static void test_dump_stack(void) {}
#endif
-static pid_t _gettid(void)
-{
- return syscall(SYS_gettid);
-}
-
void __attribute__((noinline))
test_assert(bool exp, const char *exp_str,
const char *file, unsigned int line, const char *fmt, ...)
@@ -78,10 +74,10 @@ test_assert(bool exp, const char *exp_str,
if (!(exp)) {
va_start(ap, fmt);
- fprintf(stderr, "==== Test Assertion Failure ====\n"
+ fprintf(stderr, "\n==== Test Assertion Failure ====\n"
" %s:%u: %s\n"
" pid=%d tid=%d errno=%d - %s\n",
- file, line, exp_str, getpid(), _gettid(),
+ file, line, exp_str, getpid(), kvm_gettid(),
errno, strerror(errno));
test_dump_stack();
if (fmt) {
diff --git a/tools/testing/selftests/kvm/lib/kvm_util.c b/tools/testing/selftests/kvm/lib/kvm_util.c
index 195f3fdae1e3..9ddc047d5c27 100644
--- a/tools/testing/selftests/kvm/lib/kvm_util.c
+++ b/tools/testing/selftests/kvm/lib/kvm_util.c
@@ -15,14 +15,22 @@
#include <sys/resource.h>
#include <sys/types.h>
#include <sys/stat.h>
+#include <time.h>
#include <unistd.h>
#include <linux/kernel.h>
#define KVM_UTIL_MIN_PFN 2
-u32 guest_random_seed;
-struct guest_random_state guest_rng;
-static u32 last_guest_seed;
+u32 kvm_random_seed;
+struct kvm_random_state kvm_rng;
+static u32 last_kvm_seed;
+
+static void kvm_seed_rng(u32 seed)
+{
+ kvm_random_seed = last_kvm_seed = seed;
+ pr_info("Random seed: 0x%x\n", kvm_random_seed);
+ kvm_rng = new_kvm_random_state(kvm_random_seed);
+}
static size_t vcpu_mmap_sz(void);
@@ -515,12 +523,10 @@ struct kvm_vm *__vm_create(struct vm_shape shape, u32 nr_runnable_vcpus,
slot0 = memslot2region(vm, 0);
ucall_init(vm, slot0->region.guest_phys_addr + slot0->region.memory_size);
- if (guest_random_seed != last_guest_seed) {
- pr_info("Random seed: 0x%x\n", guest_random_seed);
- last_guest_seed = guest_random_seed;
- }
- guest_rng = new_guest_random_state(guest_random_seed);
- sync_global_to_guest(vm, guest_rng);
+ if (kvm_random_seed != last_kvm_seed)
+ kvm_seed_rng(kvm_random_seed);
+
+ sync_global_to_guest(vm, kvm_rng);
kvm_arch_vm_post_create(vm, nr_runnable_vcpus);
@@ -662,19 +668,37 @@ void kvm_print_vcpu_pinning_help(void)
" (default: no pinning)\n", name, name);
}
+int kvm_pick_random_cpu(cpu_set_t *possible_cpus)
+{
+ int target_idx;
+ int nr_cpus;
+ int cpu;
+
+ nr_cpus = CPU_COUNT(possible_cpus);
+ TEST_ASSERT(nr_cpus > 0, "No CPUs available in possible_cpus");
+
+ target_idx = kvm_random_u64(&kvm_rng) % nr_cpus;
+
+ for (cpu = 0; cpu < CPU_SETSIZE; cpu++) {
+ if (CPU_ISSET(cpu, possible_cpus) && target_idx-- == 0)
+ return cpu;
+ }
+ TEST_FAIL("Failed to find random CPU in possible_cpus");
+ return -1;
+}
+
void kvm_parse_vcpu_pinning(const char *pcpus_string, u32 vcpu_to_pcpu[],
int nr_vcpus)
{
cpu_set_t allowed_mask;
char *cpu, *cpu_list;
char delim[2] = ",";
- int i, r;
+ int i;
cpu_list = strdup(pcpus_string);
TEST_ASSERT(cpu_list, "strdup() allocation failed.");
- r = sched_getaffinity(0, sizeof(allowed_mask), &allowed_mask);
- TEST_ASSERT(!r, "sched_getaffinity() failed");
+ kvm_sched_getaffinity(0, sizeof(allowed_mask), &allowed_mask);
cpu = strtok(cpu_list, delim);
@@ -2279,8 +2303,8 @@ void __attribute((constructor)) kvm_selftest_init(void)
sigaction(SIGILL, &sig_sa, NULL);
sigaction(SIGFPE, &sig_sa, NULL);
- guest_random_seed = last_guest_seed = random();
- pr_info("Random seed: 0x%x\n", guest_random_seed);
+ srandom(time(0));
+ kvm_seed_rng(random());
kvm_selftest_arch_init();
}
diff --git a/tools/testing/selftests/kvm/lib/loongarch/processor.c b/tools/testing/selftests/kvm/lib/loongarch/processor.c
index 64d91fb76522..47bbde3e205a 100644
--- a/tools/testing/selftests/kvm/lib/loongarch/processor.c
+++ b/tools/testing/selftests/kvm/lib/loongarch/processor.c
@@ -278,7 +278,7 @@ static void loongarch_set_csr(struct kvm_vcpu *vcpu, u64 id, u64 val)
void loongarch_vcpu_setup(struct kvm_vcpu *vcpu)
{
- int width;
+ int ret, width;
unsigned int cfg;
unsigned long val;
struct kvm_vm *vm = vcpu->vm;
@@ -292,6 +292,15 @@ void loongarch_vcpu_setup(struct kvm_vcpu *vcpu)
TEST_FAIL("Unknown guest mode, mode: 0x%x", vm->mode);
}
+ /* Enable LSX/LASX feature by detecting host capability */
+ cfg = CPUCFG2_FP | CPUCFG2_FPSP | CPUCFG2_FPDP | CPUCFG2_LLFTP;
+ ret = __kvm_has_device_attr(vm->fd, KVM_LOONGARCH_VM_FEAT_CTRL, KVM_LOONGARCH_VM_FEAT_LSX);
+ if (!ret)
+ cfg |= CPUCFG2_LSX;
+ ret = __kvm_has_device_attr(vm->fd, KVM_LOONGARCH_VM_FEAT_CTRL, KVM_LOONGARCH_VM_FEAT_LASX);
+ if (!ret)
+ cfg |= CPUCFG2_LASX;
+ loongarch_set_cpucfg(vcpu, LOONGARCH_CPUCFG2, cfg);
cfg = read_cpucfg(LOONGARCH_CPUCFG6);
loongarch_set_cpucfg(vcpu, LOONGARCH_CPUCFG6, cfg);
diff --git a/tools/testing/selftests/kvm/lib/memstress.c b/tools/testing/selftests/kvm/lib/memstress.c
index 6dcd15910a06..2864562f57b4 100644
--- a/tools/testing/selftests/kvm/lib/memstress.c
+++ b/tools/testing/selftests/kvm/lib/memstress.c
@@ -48,14 +48,14 @@ void memstress_guest_code(u32 vcpu_idx)
{
struct memstress_args *args = &memstress_args;
struct memstress_vcpu_args *vcpu_args = &args->vcpu_args[vcpu_idx];
- struct guest_random_state rand_state;
+ struct kvm_random_state rand_state;
gva_t gva;
u64 pages;
u64 addr;
u64 page;
int i;
- rand_state = new_guest_random_state(guest_random_seed + vcpu_idx);
+ rand_state = new_kvm_random_state(kvm_random_seed + vcpu_idx);
gva = vcpu_args->gva;
pages = vcpu_args->pages;
@@ -69,13 +69,13 @@ void memstress_guest_code(u32 vcpu_idx)
for (i = 0; i < pages; i++) {
if (args->random_access)
- page = guest_random_u32(&rand_state) % pages;
+ page = kvm_random_u32(&rand_state) % pages;
else
page = i;
addr = gva + (page * args->guest_page_size);
- if (__guest_random_bool(&rand_state, args->write_percent))
+ if (__kvm_random_bool(&rand_state, args->write_percent))
*(u64 *)addr = 0x0123456789ABCDEF;
else
READ_ONCE(*(u64 *)addr);
@@ -294,7 +294,7 @@ void memstress_start_vcpu_threads(int nr_vcpus,
vcpu->vcpu_idx = i;
WRITE_ONCE(vcpu->running, false);
- pthread_create(&vcpu->thread, NULL, vcpu_thread_main, vcpu);
+ kvm_pthread_create(&vcpu->thread, NULL, vcpu_thread_main, vcpu);
}
for (i = 0; i < nr_vcpus; i++) {
@@ -312,7 +312,7 @@ void memstress_join_vcpu_threads(int nr_vcpus)
WRITE_ONCE(memstress_args.stop_vcpus, true);
for (i = 0; i < nr_vcpus; i++)
- pthread_join(vcpu_threads[i].thread, NULL);
+ kvm_pthread_join(vcpu_threads[i].thread, NULL);
}
static void toggle_dirty_logging(struct kvm_vm *vm, int slots, bool enable)
diff --git a/tools/testing/selftests/kvm/lib/proc_util.c b/tools/testing/selftests/kvm/lib/proc_util.c
new file mode 100644
index 000000000000..3960b3841d63
--- /dev/null
+++ b/tools/testing/selftests/kvm/lib/proc_util.c
@@ -0,0 +1,54 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "kvm_util.h"
+#include "test_util.h"
+#include "proc_util.h"
+
+static FILE *open_proc_interrupts(void)
+{
+ FILE *fp;
+
+ fp = fopen("/proc/interrupts", "r");
+ TEST_ASSERT(fp, "fopen(/proc/interrupts) failed");
+
+ return fp;
+}
+
+unsigned int vfio_msix_to_host_irq(const char *device_bdf, int msix)
+{
+ char search_string[64];
+ char line[4096];
+ int irq = -1;
+ FILE *fp;
+
+ fp = open_proc_interrupts();
+
+ snprintf(search_string, sizeof(search_string), "vfio-msix[%d]", msix);
+
+ while (fgets(line, sizeof(line), fp)) {
+ if (strstr(line, device_bdf) && strstr(line, search_string)) {
+ TEST_ASSERT_EQ(1, sscanf(line, "%d:", &irq));
+ break;
+ }
+ }
+
+ fclose(fp);
+
+ TEST_ASSERT(irq != -1, "Failed to locate IRQ for %s %s", device_bdf,
+ search_string);
+ return (unsigned int)irq;
+}
+
+void proc_irq_set_smp_affinity(unsigned int irq, int cpu)
+{
+ char path[PATH_MAX];
+ int r, fd;
+
+ snprintf(path, sizeof(path), "/proc/irq/%u/smp_affinity_list", irq);
+ fd = open(path, O_RDWR);
+ TEST_ASSERT(fd >= 0, "Failed to open %s", path);
+
+ r = dprintf(fd, "%d\n", cpu);
+ TEST_ASSERT(r > 0, "Failed to affinitize IRQ-%u to CPU %d", irq, cpu);
+
+ kvm_close(fd);
+}
diff --git a/tools/testing/selftests/kvm/lib/test_util.c b/tools/testing/selftests/kvm/lib/test_util.c
index bab1bd2b775b..4dff72f6bd34 100644
--- a/tools/testing/selftests/kvm/lib/test_util.c
+++ b/tools/testing/selftests/kvm/lib/test_util.c
@@ -17,6 +17,7 @@
#include "linux/kernel.h"
#include "test_util.h"
+#include "kvm_syscalls.h"
sigjmp_buf expect_sigbus_jmpbuf;
@@ -30,18 +31,50 @@ void __attribute__((used)) expect_sigbus_handler(int signum)
* Park-Miller LCG using standard constants.
*/
-struct guest_random_state new_guest_random_state(u32 seed)
+struct kvm_random_state new_kvm_random_state(u32 seed)
{
- struct guest_random_state s = {.seed = seed};
+ struct kvm_random_state s = {.seed = seed};
return s;
}
-u32 guest_random_u32(struct guest_random_state *state)
+u32 kvm_random_u32(struct kvm_random_state *state)
{
state->seed = (u64)state->seed * 48271 % ((u32)(1 << 31) - 1);
return state->seed;
}
+/* Returns a random u32 in the inclusive range [min, max] */
+u32 kvm_random_u32_in_range(struct kvm_random_state *state, u32 min, u32 max)
+{
+ u32 value, range;
+
+ TEST_ASSERT(min <= max, "PEBKAC, min = 0x%x, max = 0x%x", min, max);
+
+ value = kvm_random_u32(state);
+
+ range = max - min;
+ if (range == UINT_MAX)
+ return value;
+
+ return min + (value % (range + 1));
+}
+
+/* Returns a random u64 in the inclusive range [min, max] */
+u64 kvm_random_u64_in_range(struct kvm_random_state *state, u64 min, u64 max)
+{
+ u64 value, range;
+
+ TEST_ASSERT(min <= max, "PEBKAC, min = 0x%lx, max = 0x%lx", min, max);
+
+ value = kvm_random_u64(state);
+
+ range = max - min;
+ if (range == ULLONG_MAX)
+ return value;
+
+ return min + (value % (range + 1));
+}
+
/*
* Parses "[0-9]+[kmgt]?".
*/
@@ -377,7 +410,7 @@ long get_run_delay(void)
long val[2];
FILE *fp;
- sprintf(path, "/proc/%ld/schedstat", syscall(SYS_gettid));
+ sprintf(path, "/proc/%ld/schedstat", (long)kvm_gettid());
fp = fopen(path, "r");
/* Return MIN_RUN_DELAY_NS upon failure just to be safe */
if (fscanf(fp, "%ld %ld ", &val[0], &val[1]) < 2)
diff --git a/tools/testing/selftests/kvm/lib/userfaultfd_util.c b/tools/testing/selftests/kvm/lib/userfaultfd_util.c
index ef8d76f71f83..f7ce5a6ddcc2 100644
--- a/tools/testing/selftests/kvm/lib/userfaultfd_util.c
+++ b/tools/testing/selftests/kvm/lib/userfaultfd_util.c
@@ -167,8 +167,8 @@ struct uffd_desc *uffd_setup_demand_paging(int uffd_mode, useconds_t delay,
uffd_desc->reader_args[i].handler = handler;
uffd_desc->reader_args[i].pipe = pipes[0];
- pthread_create(&uffd_desc->readers[i], NULL, uffd_handler_thread_fn,
- &uffd_desc->reader_args[i]);
+ kvm_pthread_create(&uffd_desc->readers[i], NULL, uffd_handler_thread_fn,
+ &uffd_desc->reader_args[i]);
PER_VCPU_DEBUG("Created uffd thread %i for HVA range [%p, %p)\n",
i, hva, hva + len);
@@ -187,8 +187,7 @@ void uffd_stop_demand_paging(struct uffd_desc *uffd)
"Unable to write to pipefd %i for uffd_desc %p", i, uffd);
for (i = 0; i < uffd->num_readers; ++i)
- TEST_ASSERT(!pthread_join(uffd->readers[i], NULL),
- "Pthread_join failed on reader %i for uffd_desc %p", i, uffd);
+ kvm_pthread_join(uffd->readers[i], NULL);
close(uffd->uffd);
diff --git a/tools/testing/selftests/kvm/lib/x86/memstress.c b/tools/testing/selftests/kvm/lib/x86/memstress.c
index 61cf952cd2dc..e19e8b5a09c5 100644
--- a/tools/testing/selftests/kvm/lib/x86/memstress.c
+++ b/tools/testing/selftests/kvm/lib/x86/memstress.c
@@ -30,21 +30,15 @@ __asm__(
" ud2;"
);
-#define L2_GUEST_STACK_SIZE 64
-
static void l1_vmx_code(struct vmx_pages *vmx, u64 vcpu_id)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
- unsigned long *rsp;
-
GUEST_ASSERT(vmx->vmcs_gpa);
GUEST_ASSERT(prepare_for_vmx_operation(vmx));
GUEST_ASSERT(load_vmcs(vmx));
GUEST_ASSERT(ept_1g_pages_supported());
- rsp = &l2_guest_stack[L2_GUEST_STACK_SIZE - 1];
- *rsp = vcpu_id;
- prepare_vmcs(vmx, memstress_l2_guest_entry, rsp);
+ *(u64 *)vmx->stack = vcpu_id;
+ prepare_vmcs(vmx, memstress_l2_guest_entry);
GUEST_ASSERT(!vmlaunch());
GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_VMCALL);
@@ -53,13 +47,8 @@ static void l1_vmx_code(struct vmx_pages *vmx, u64 vcpu_id)
static void l1_svm_code(struct svm_test_data *svm, u64 vcpu_id)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
- unsigned long *rsp;
-
-
- rsp = &l2_guest_stack[L2_GUEST_STACK_SIZE - 1];
- *rsp = vcpu_id;
- generic_svm_setup(svm, memstress_l2_guest_entry, rsp);
+ *(u64 *)svm->stack = vcpu_id;
+ generic_svm_setup(svm, memstress_l2_guest_entry);
run_guest(svm->vmcb, svm->vmcb_gpa);
GUEST_ASSERT_EQ(svm->vmcb->control.exit_code, SVM_EXIT_VMMCALL);
diff --git a/tools/testing/selftests/kvm/lib/x86/processor.c b/tools/testing/selftests/kvm/lib/x86/processor.c
index 4ca48de7a926..d31fa81ea075 100644
--- a/tools/testing/selftests/kvm/lib/x86/processor.c
+++ b/tools/testing/selftests/kvm/lib/x86/processor.c
@@ -28,6 +28,9 @@ bool host_cpu_is_hygon;
bool host_cpu_is_amd_compatible;
bool is_forced_emulation_enabled;
u64 guest_tsc_khz;
+struct kvm_mmu guest_mmu;
+
+struct guest_regs guest_regs;
const char *ex_str(int vector)
{
@@ -778,6 +781,30 @@ void assert_on_unhandled_exception(struct kvm_vcpu *vcpu)
REPORT_GUEST_ASSERT(uc);
}
+gva_t vm_alloc_stack(struct kvm_vm *vm, int nr_pages)
+{
+ int size = nr_pages * getpagesize();
+ gva_t stack_gva;
+
+ stack_gva = __vm_alloc(vm, size, DEFAULT_GUEST_STACK_VADDR_MIN, MEM_REGION_DATA);
+ stack_gva += size;
+
+ /*
+ * Align stack to match calling sequence requirements in section "The
+ * Stack Frame" of the System V ABI AMD64 Architecture Processor
+ * Supplement, which requires the value (%rsp + 8) to be a multiple of
+ * 16 when control is transferred to the function entry point.
+ *
+ * If this code is ever used to launch a vCPU with 32-bit entry point it
+ * may need to subtract 4 bytes instead of 8 bytes.
+ */
+ TEST_ASSERT(IS_ALIGNED(stack_gva, PAGE_SIZE),
+ "__vm_alloc() did not provide a page-aligned address");
+ stack_gva -= 8;
+
+ return stack_gva;
+}
+
void kvm_arch_vm_post_create(struct kvm_vm *vm, unsigned int nr_vcpus)
{
int r;
@@ -805,6 +832,17 @@ void kvm_arch_vm_post_create(struct kvm_vm *vm, unsigned int nr_vcpus)
TEST_ASSERT(r > 0, "KVM_GET_TSC_KHZ did not provide a valid TSC frequency.");
guest_tsc_khz = r;
sync_global_to_guest(vm, guest_tsc_khz);
+
+ /*
+ * The guest MMU is just a placeholder to provide access to PTE masks
+ * (for now). The guest does not have mappings for its own page tables
+ * by default, so any meaningful use of guest page tables requires
+ * explicit setup by the test. Zero the PGD to make it obvious the guest
+ * page tables are not immediately usable by guest code.
+ */
+ guest_mmu = vm->mmu;
+ guest_mmu.pgd = 0;
+ sync_global_to_guest(vm, guest_mmu);
}
void vcpu_arch_set_entry_point(struct kvm_vcpu *vcpu, void *guest_code)
@@ -820,27 +858,8 @@ struct kvm_vcpu *vm_arch_vcpu_add(struct kvm_vm *vm, u32 vcpu_id)
{
struct kvm_mp_state mp_state;
struct kvm_regs regs;
- gva_t stack_gva;
struct kvm_vcpu *vcpu;
- stack_gva = __vm_alloc(vm, DEFAULT_STACK_PGS * getpagesize(),
- DEFAULT_GUEST_STACK_VADDR_MIN, MEM_REGION_DATA);
-
- stack_gva += DEFAULT_STACK_PGS * getpagesize();
-
- /*
- * Align stack to match calling sequence requirements in section "The
- * Stack Frame" of the System V ABI AMD64 Architecture Processor
- * Supplement, which requires the value (%rsp + 8) to be a multiple of
- * 16 when control is transferred to the function entry point.
- *
- * If this code is ever used to launch a vCPU with 32-bit entry point it
- * may need to subtract 4 bytes instead of 8 bytes.
- */
- TEST_ASSERT(IS_ALIGNED(stack_gva, PAGE_SIZE),
- "__vm_alloc() did not provide a page-aligned address");
- stack_gva -= 8;
-
vcpu = __vm_vcpu_add(vm, vcpu_id);
vcpu_init_cpuid(vcpu, kvm_get_supported_cpuid());
vcpu_init_sregs(vm, vcpu);
@@ -849,7 +868,7 @@ struct kvm_vcpu *vm_arch_vcpu_add(struct kvm_vm *vm, u32 vcpu_id)
/* Setup guest general purpose registers */
vcpu_regs_get(vcpu, &regs);
regs.rflags = regs.rflags | X86_EFLAGS_FIXED;
- regs.rsp = stack_gva;
+ regs.rsp = vm_alloc_stack(vm, DEFAULT_STACK_PGS);
vcpu_regs_set(vcpu, &regs);
/* Setup the MP state */
diff --git a/tools/testing/selftests/kvm/lib/x86/svm.c b/tools/testing/selftests/kvm/lib/x86/svm.c
index 3b01605ab016..b05be50f075d 100644
--- a/tools/testing/selftests/kvm/lib/x86/svm.c
+++ b/tools/testing/selftests/kvm/lib/x86/svm.c
@@ -13,9 +13,6 @@
#define SEV_DEV_PATH "/dev/sev"
-struct gpr64_regs guest_regs;
-u64 rflags;
-
/* Allocate memory regions for nested SVM tests.
*
* Input Args:
@@ -46,6 +43,8 @@ vcpu_alloc_svm(struct kvm_vm *vm, gva_t *p_svm_gva)
svm->msr_gpa = addr_gva2gpa(vm, (uintptr_t)svm->msr);
memset(svm->msr_hva, 0, getpagesize());
+ svm->stack = (void *)vm_alloc_stack(vm, 1);
+
if (vm->stage2_mmu.pgd_created)
svm->ncr3_gpa = vm->stage2_mmu.pgd;
@@ -81,7 +80,7 @@ void vm_enable_npt(struct kvm_vm *vm)
tdp_mmu_init(vm, vm->mmu.pgtable_levels, &pte_masks);
}
-void generic_svm_setup(struct svm_test_data *svm, void *guest_rip, void *guest_rsp)
+void generic_svm_setup(struct svm_test_data *svm, void *guest_rip)
{
struct vmcb *vmcb = svm->vmcb;
u64 vmcb_gpa = svm->vmcb_gpa;
@@ -122,7 +121,7 @@ void generic_svm_setup(struct svm_test_data *svm, void *guest_rip, void *guest_r
ctrl->msrpm_base_pa = svm->msr_gpa;
vmcb->save.rip = (u64)guest_rip;
- vmcb->save.rsp = (u64)guest_rsp;
+ vmcb->save.rsp = (u64)svm->stack;
guest_regs.rdi = (u64)svm;
if (svm->ncr3_gpa) {
@@ -135,46 +134,47 @@ void generic_svm_setup(struct svm_test_data *svm, void *guest_rip, void *guest_r
* save/restore 64-bit general registers except rax, rip, rsp
* which are directly handed through the VMCB guest processor state
*/
-#define SAVE_GPR_C \
- "xchg %%rbx, guest_regs+0x20\n\t" \
- "xchg %%rcx, guest_regs+0x10\n\t" \
- "xchg %%rdx, guest_regs+0x18\n\t" \
- "xchg %%rbp, guest_regs+0x30\n\t" \
- "xchg %%rsi, guest_regs+0x38\n\t" \
- "xchg %%rdi, guest_regs+0x40\n\t" \
- "xchg %%r8, guest_regs+0x48\n\t" \
- "xchg %%r9, guest_regs+0x50\n\t" \
- "xchg %%r10, guest_regs+0x58\n\t" \
- "xchg %%r11, guest_regs+0x60\n\t" \
- "xchg %%r12, guest_regs+0x68\n\t" \
- "xchg %%r13, guest_regs+0x70\n\t" \
- "xchg %%r14, guest_regs+0x78\n\t" \
- "xchg %%r15, guest_regs+0x80\n\t"
-
-#define LOAD_GPR_C SAVE_GPR_C
+#define SVM_SWITCH_GPRS_ASM \
+ GUEST_SWITCH_GPR_ASM(rbx) \
+ GUEST_SWITCH_GPR_ASM(rcx) \
+ GUEST_SWITCH_GPR_ASM(rdx) \
+ GUEST_SWITCH_GPR_ASM(rbp) \
+ GUEST_SWITCH_GPR_ASM(rsi) \
+ GUEST_SWITCH_GPR_ASM(rdi) \
+ GUEST_SWITCH_GPR_ASM(r8) \
+ GUEST_SWITCH_GPR_ASM(r9) \
+ GUEST_SWITCH_GPR_ASM(r10) \
+ GUEST_SWITCH_GPR_ASM(r11) \
+ GUEST_SWITCH_GPR_ASM(r12) \
+ GUEST_SWITCH_GPR_ASM(r13) \
+ GUEST_SWITCH_GPR_ASM(r14) \
+ GUEST_SWITCH_GPR_ASM(r15)
/*
* selftests do not use interrupts so we dropped clgi/sti/cli/stgi
- * for now. registers involved in LOAD/SAVE_GPR_C are eventually
+ * for now. Registers involved in SVM_SWITCH_GPRS_ASM are eventually
* unmodified so they do not need to be in the clobber list.
*/
void run_guest(struct vmcb *vmcb, u64 vmcb_gpa)
{
asm volatile (
"vmload %[vmcb_gpa]\n\t"
- "mov rflags, %%r15\n\t" // rflags
- "mov %%r15, 0x170(%[vmcb])\n\t"
- "mov guest_regs, %%r15\n\t" // rax
- "mov %%r15, 0x1f8(%[vmcb])\n\t"
- LOAD_GPR_C
+ "mov " GUEST_REG(rflags) ", %%r15\n\t"
+ "mov %%r15, %[vmcb_rflags]\n\t"
+ "mov " GUEST_REG(rax) ", %%r15\n\t"
+ "mov %%r15, %[vmcb_rax]\n\t"
+ SVM_SWITCH_GPRS_ASM
"vmrun %[vmcb_gpa]\n\t"
- SAVE_GPR_C
- "mov 0x170(%[vmcb]), %%r15\n\t" // rflags
- "mov %%r15, rflags\n\t"
- "mov 0x1f8(%[vmcb]), %%r15\n\t" // rax
- "mov %%r15, guest_regs\n\t"
+ SVM_SWITCH_GPRS_ASM
+ "mov %[vmcb_rflags], %%r15\n\t"
+ "mov %%r15, " GUEST_REG(rflags) "\n\t"
+ "mov %[vmcb_rax], %%r15\n\t" // rax
+ "mov %%r15, " GUEST_REG(rax) "\n\t"
"vmsave %[vmcb_gpa]\n\t"
- : : [vmcb] "r" (vmcb), [vmcb_gpa] "a" (vmcb_gpa)
+ : [vmcb_rflags] "+m" (vmcb->save.rflags),
+ [vmcb_rax] "+m" (vmcb->save.rax)
+ : [vmcb_gpa] "a" (vmcb_gpa),
+ GUEST_REGS_OFFSETS
: "r15", "memory");
}
diff --git a/tools/testing/selftests/kvm/lib/x86/ucall.c b/tools/testing/selftests/kvm/lib/x86/ucall.c
index e7dd5791959b..38050c60a067 100644
--- a/tools/testing/selftests/kvm/lib/x86/ucall.c
+++ b/tools/testing/selftests/kvm/lib/x86/ucall.c
@@ -10,36 +10,8 @@
void ucall_arch_do_ucall(gva_t uc)
{
- /*
- * FIXME: Revert this hack (the entire commit that added it) once nVMX
- * preserves L2 GPRs across a nested VM-Exit. If a ucall from L2, e.g.
- * to do a GUEST_SYNC(), lands the vCPU in L1, any and all GPRs can be
- * clobbered by L1. Save and restore non-volatile GPRs (clobbering RBP
- * in particular is problematic) along with RDX and RDI (which are
- * inputs), and clobber volatile GPRs. *sigh*
- */
-#define HORRIFIC_L2_UCALL_CLOBBER_HACK \
- "rcx", "rsi", "r8", "r9", "r10", "r11"
-
- asm volatile("push %%rbp\n\t"
- "push %%r15\n\t"
- "push %%r14\n\t"
- "push %%r13\n\t"
- "push %%r12\n\t"
- "push %%rbx\n\t"
- "push %%rdx\n\t"
- "push %%rdi\n\t"
- "in %[port], %%al\n\t"
- "pop %%rdi\n\t"
- "pop %%rdx\n\t"
- "pop %%rbx\n\t"
- "pop %%r12\n\t"
- "pop %%r13\n\t"
- "pop %%r14\n\t"
- "pop %%r15\n\t"
- "pop %%rbp\n\t"
- : : [port] "d" (UCALL_PIO_PORT), "D" (uc) : "rax", "memory",
- HORRIFIC_L2_UCALL_CLOBBER_HACK);
+ asm volatile("in %[port], %%al"
+ : : [port] "d" (UCALL_PIO_PORT), "D" (uc) : "rax", "memory");
}
void *ucall_arch_get_ucall(struct kvm_vcpu *vcpu)
diff --git a/tools/testing/selftests/kvm/lib/x86/vmx.c b/tools/testing/selftests/kvm/lib/x86/vmx.c
index 7c10ba6e6fb4..089e1a8af53f 100644
--- a/tools/testing/selftests/kvm/lib/x86/vmx.c
+++ b/tools/testing/selftests/kvm/lib/x86/vmx.c
@@ -116,6 +116,8 @@ vcpu_alloc_vmx(struct kvm_vm *vm, gva_t *p_vmx_gva)
vmx->vmwrite_gpa = addr_gva2gpa(vm, (uintptr_t)vmx->vmwrite);
memset(vmx->vmwrite_hva, 0, getpagesize());
+ vmx->stack = (void *)vm_alloc_stack(vm, 1);
+
if (vm->stage2_mmu.pgd_created)
vmx->eptp_gpa = vm->stage2_mmu.pgd;
@@ -230,7 +232,7 @@ static inline void init_vmcs_control_fields(struct vmx_pages *vmx)
vmwrite(EXCEPTION_BITMAP, 0);
vmwrite(PAGE_FAULT_ERROR_CODE_MASK, 0);
- vmwrite(PAGE_FAULT_ERROR_CODE_MATCH, -1); /* Never match */
+ vmwrite(PAGE_FAULT_ERROR_CODE_MATCH, 0);
vmwrite(CR3_TARGET_COUNT, 0);
vmwrite(VM_EXIT_CONTROLS, rdmsr(MSR_IA32_VMX_EXIT_CTLS) |
VM_EXIT_HOST_ADDR_SPACE_SIZE); /* 64-bit host */
@@ -366,11 +368,11 @@ static inline void init_vmcs_guest_state(void *rip, void *rsp)
vmwrite(GUEST_SYSENTER_EIP, vmreadz(HOST_IA32_SYSENTER_EIP));
}
-void prepare_vmcs(struct vmx_pages *vmx, void *guest_rip, void *guest_rsp)
+void prepare_vmcs(struct vmx_pages *vmx, void *guest_rip)
{
init_vmcs_control_fields(vmx);
init_vmcs_host_state();
- init_vmcs_guest_state(guest_rip, guest_rsp);
+ init_vmcs_guest_state(guest_rip, vmx->stack);
}
bool kvm_cpu_has_ept(void)
diff --git a/tools/testing/selftests/kvm/loongarch/fpu_test.c b/tools/testing/selftests/kvm/loongarch/fpu_test.c
new file mode 100644
index 000000000000..c54ec35883dc
--- /dev/null
+++ b/tools/testing/selftests/kvm/loongarch/fpu_test.c
@@ -0,0 +1,137 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <stdio.h>
+#include <string.h>
+#include "kvm_util.h"
+#include "processor.h"
+#include "loongarch/processor.h"
+
+struct kvm_fpureg __aligned(64) vector = {{1, 2, 3, 4 }};
+
+static void guest_code(void)
+{
+ unsigned long val;
+ struct kvm_fpureg *fp = &vector;
+
+ val = csr_read(LOONGARCH_CSR_EUEN);
+ val |= CSR_EUEN_FPEN | CSR_EUEN_LSXEN | CSR_EUEN_LASXEN;
+ csr_write(val, LOONGARCH_CSR_EUEN);
+
+ __asm__ __volatile__("fld.d $f0, %0, 0\n" : : "r"(fp) : "$f0");
+ GUEST_SYNC(0);
+
+ __asm__ __volatile__("vld $vr0, %0, 0\n" : : "r"(fp) : "$vr0");
+ GUEST_SYNC(1);
+
+ __asm__ __volatile__("xvld $xr0, %0, 0\n" : : "r"(fp) : "$xr0");
+ GUEST_SYNC(2);
+
+ __asm__ __volatile__("fst.d $f0, %0, 0\n" : : "r"(fp) : "memory");
+ GUEST_SYNC(3);
+
+ __asm__ __volatile__("vst $vr0, %0, 0\n" : : "r"(fp) : "memory");
+ GUEST_SYNC(4);
+
+ __asm__ __volatile__("xvst $xr0, %0, 0\n" : : "r"(fp) : "memory");
+ GUEST_SYNC(5);
+
+ GUEST_DONE();
+}
+
+static void run_vcpu(struct kvm_vcpu *vcpu)
+{
+ int cont = 1;
+ struct ucall uc;
+
+ while (cont) {
+ vcpu_run(vcpu);
+
+ switch (get_ucall(vcpu, &uc)) {
+ case UCALL_PRINTF:
+ printf("%s", (const char *)uc.buffer);
+ break;
+ case UCALL_DONE:
+ printf("FPU test PASSED\n");
+ fallthrough;
+ case UCALL_SYNC:
+ cont = 0;
+ break;
+ case UCALL_ABORT:
+ REPORT_GUEST_ASSERT(uc);
+ default:
+ TEST_ASSERT(false, "Unexpected exit: %s",
+ exit_reason_str(vcpu->run->exit_reason));
+ }
+ }
+}
+
+int main(void)
+{
+ int ret;
+ struct kvm_vm *vm;
+ struct kvm_vcpu *vcpu;
+ struct kvm_fpu fpu;
+ struct kvm_fpureg *fp = &vector;
+
+ vm = vm_create_with_one_vcpu(&vcpu, guest_code);
+
+ ret = __kvm_has_device_attr(vm->fd, KVM_LOONGARCH_VM_FEAT_CTRL, KVM_LOONGARCH_VM_FEAT_LSX);
+ __TEST_REQUIRE(!ret, "LSX not available, skipping test\n");
+
+ ret = __kvm_has_device_attr(vm->fd, KVM_LOONGARCH_VM_FEAT_CTRL, KVM_LOONGARCH_VM_FEAT_LASX);
+ __TEST_REQUIRE(!ret, "LASX not available, skipping test\n");
+
+ run_vcpu(vcpu);
+ vcpu_fpu_get(vcpu, &fpu);
+ TEST_ASSERT(!memcmp(fpu.fpr, fp, 8), "Wanted 0x%llx from f0, got 0x%llx",
+ fp->val64[0], fpu.fpr[0].val64[0]);
+
+ run_vcpu(vcpu);
+ vcpu_fpu_get(vcpu, &fpu);
+ TEST_ASSERT(!memcmp(fpu.fpr, fp, 16), "Wanted 0x%llx %llx from vr0, got 0x%llx %llx",
+ fp->val64[0], fp->val64[1],
+ fpu.fpr[0].val64[0], fpu.fpr[0].val64[1]);
+
+ run_vcpu(vcpu);
+ vcpu_fpu_get(vcpu, &fpu);
+ TEST_ASSERT(!memcmp(fpu.fpr, fp, 32),
+ "Wanted 0x%llx %llx %llx %llx from xr0, got 0x%llx %llx %llx %llx",
+ fp->val64[0], fp->val64[1], fp->val64[2], fp->val64[3],
+ fpu.fpr[0].val64[0], fpu.fpr[0].val64[1], fpu.fpr[0].val64[2], fpu.fpr[0].val64[3]);
+
+ fpu.fpr[0].val64[0] += random();
+ vcpu_fpu_set(vcpu, &fpu);
+ run_vcpu(vcpu);
+ vcpu_fpu_get(vcpu, &fpu);
+ sync_global_from_guest(vm, *fp);
+ TEST_ASSERT(!memcmp(fpu.fpr, fp, 8), "Wanted 0x%llx from f0, got 0x%llx",
+ fp->val64[0], fpu.fpr[0].val64[0]);
+
+ fpu.fpr[0].val64[0] += random();
+ fpu.fpr[0].val64[1] += random();
+ vcpu_fpu_set(vcpu, &fpu);
+ run_vcpu(vcpu);
+ vcpu_fpu_get(vcpu, &fpu);
+ sync_global_from_guest(vm, *fp);
+ TEST_ASSERT(!memcmp(fpu.fpr, fp, 16), "Wanted 0x%llx %llx from vr0, got 0x%llx %llx",
+ fp->val64[0], fp->val64[1],
+ fpu.fpr[0].val64[0], fpu.fpr[0].val64[1]);
+
+ fpu.fpr[0].val64[0] += random();
+ fpu.fpr[0].val64[1] += random();
+ fpu.fpr[0].val64[2] += random();
+ fpu.fpr[0].val64[3] += random();
+ vcpu_fpu_set(vcpu, &fpu);
+ run_vcpu(vcpu);
+ vcpu_fpu_get(vcpu, &fpu);
+ sync_global_from_guest(vm, *fp);
+ TEST_ASSERT(!memcmp(fpu.fpr, fp, 32),
+ "Wanted 0x%llx %llx %llx %llx from xr0, got 0x%llx %llx %llx %llx",
+ fp->val64[0], fp->val64[1], fp->val64[2], fp->val64[3],
+ fpu.fpr[0].val64[0], fpu.fpr[0].val64[1], fpu.fpr[0].val64[2], fpu.fpr[0].val64[3]);
+
+ run_vcpu(vcpu);
+ kvm_vm_free(vm);
+
+ return 0;
+}
diff --git a/tools/testing/selftests/kvm/memslot_perf_test.c b/tools/testing/selftests/kvm/memslot_perf_test.c
index 4d9ad6104a6e..e1299611be32 100644
--- a/tools/testing/selftests/kvm/memslot_perf_test.c
+++ b/tools/testing/selftests/kvm/memslot_perf_test.c
@@ -366,7 +366,7 @@ static void launch_vm(struct vm_data *data)
{
pr_info_v("Launching the test VM\n");
- pthread_create(&data->vcpu_thread, NULL, vcpu_worker, data);
+ kvm_pthread_create(&data->vcpu_thread, NULL, vcpu_worker, data);
/* Ensure the guest thread is spun up. */
wait_for_vcpu();
@@ -381,7 +381,7 @@ static void free_vm(struct vm_data *data)
static void wait_guest_exit(struct vm_data *data)
{
- pthread_join(data->vcpu_thread, NULL);
+ kvm_pthread_join(data->vcpu_thread, NULL);
}
static void let_guest_run(struct sync_area *sync)
diff --git a/tools/testing/selftests/kvm/mmu_stress_test.c b/tools/testing/selftests/kvm/mmu_stress_test.c
index 54d281419d31..741e343962ed 100644
--- a/tools/testing/selftests/kvm/mmu_stress_test.c
+++ b/tools/testing/selftests/kvm/mmu_stress_test.c
@@ -155,10 +155,8 @@ static void *vcpu_worker(void *data)
"Expected EFAULT on write to RO memory, got r = %d, errno = %d", r, errno);
atomic_inc(&nr_ro_faults);
- if (atomic_read(&nr_ro_faults) == nr_vcpus) {
- WRITE_ONCE(all_vcpus_hit_ro_fault, true);
- sync_global_to_guest(vm, all_vcpus_hit_ro_fault);
- }
+ if (atomic_read(&nr_ro_faults) == nr_vcpus)
+ WRITE_AND_SYNC_TO_GUEST(vm, all_vcpus_hit_ro_fault, true);
#if defined(__x86_64__) || defined(__aarch64__)
/*
@@ -224,7 +222,7 @@ static pthread_t *spawn_workers(struct kvm_vm *vm, struct kvm_vcpu **vcpus,
info[i].vcpu = vcpus[i];
info[i].start_gpa = gpa;
info[i].end_gpa = gpa + nr_bytes;
- pthread_create(&threads[i], NULL, vcpu_worker, &info[i]);
+ kvm_pthread_create(&threads[i], NULL, vcpu_worker, &info[i]);
}
return threads;
}
@@ -257,11 +255,7 @@ static void rendezvous_with_vcpus(struct timespec *time, const char *name)
static void calc_default_nr_vcpus(void)
{
cpu_set_t possible_mask;
- int r;
-
- r = sched_getaffinity(0, sizeof(possible_mask), &possible_mask);
- TEST_ASSERT(!r, "sched_getaffinity failed, errno = %d (%s)",
- errno, strerror(errno));
+ kvm_sched_getaffinity(0, sizeof(possible_mask), &possible_mask);
nr_vcpus = CPU_COUNT(&possible_mask);
TEST_ASSERT(nr_vcpus > 0, "Uh, no CPUs?");
@@ -383,8 +377,7 @@ int main(int argc, char *argv[])
rendezvous_with_vcpus(&time_run2, "run 2");
mprotect(mem, slot_size, PROT_READ);
- mprotect_ro_done = true;
- sync_global_to_guest(vm, mprotect_ro_done);
+ WRITE_AND_SYNC_TO_GUEST(vm, mprotect_ro_done, true);
rendezvous_with_vcpus(&time_ro, "mprotect RO");
mprotect(mem, slot_size, PROT_READ | PROT_WRITE);
@@ -416,7 +409,7 @@ int main(int argc, char *argv[])
/* Sanity check that the vCPUs actually ran. */
for (i = 0; i < nr_vcpus; i++)
- pthread_join(threads[i], NULL);
+ kvm_pthread_join(threads[i], NULL);
/*
* Deliberately exit without deleting the remaining memslots or closing
diff --git a/tools/testing/selftests/kvm/pre_fault_memory_test.c b/tools/testing/selftests/kvm/pre_fault_memory_test.c
index a0fcae3cb7a8..c57631aab3d3 100644
--- a/tools/testing/selftests/kvm/pre_fault_memory_test.c
+++ b/tools/testing/selftests/kvm/pre_fault_memory_test.c
@@ -84,7 +84,7 @@ static void pre_fault_memory(struct kvm_vcpu *vcpu, u64 base_gpa, u64 offset,
* Concurrently delete (and recreate) the slot to test KVM's handling
* of a racing memslot deletion with prefaulting.
*/
- pthread_create(&slot_worker, NULL, delete_slot_worker, &data);
+ kvm_pthread_create(&slot_worker, NULL, delete_slot_worker, &data);
while (!READ_ONCE(data.worker_ready))
cpu_relax();
@@ -116,7 +116,7 @@ static void pre_fault_memory(struct kvm_vcpu *vcpu, u64 base_gpa, u64 offset,
*/
if (!slot_recreated) {
WRITE_ONCE(data.recreate_slot, true);
- pthread_join(slot_worker, NULL);
+ kvm_pthread_join(slot_worker, NULL);
slot_recreated = true;
/*
diff --git a/tools/testing/selftests/kvm/riscv/get-reg-list.c b/tools/testing/selftests/kvm/riscv/get-reg-list.c
index cb86cb6b3635..bf3980e9db12 100644
--- a/tools/testing/selftests/kvm/riscv/get-reg-list.c
+++ b/tools/testing/selftests/kvm/riscv/get-reg-list.c
@@ -90,6 +90,8 @@ bool filter_reg(__u64 reg)
case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICBOP:
case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICBOZ:
case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICCRSE:
+ case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICFILP:
+ case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICFISS:
case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICNTR:
case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICOND:
case KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICSR:
@@ -358,6 +360,8 @@ static const char *core_id_to_str(const char *prefix, __u64 id)
"KVM_REG_RISCV_CSR_AIA | KVM_REG_RISCV_CSR_REG(" #csr ")"
#define RISCV_CSR_SMSTATEEN(csr) \
"KVM_REG_RISCV_CSR_SMSTATEEN | KVM_REG_RISCV_CSR_REG(" #csr ")"
+#define RISCV_CSR_ZICFISS(csr) \
+ "KVM_REG_RISCV_CSR_ZICFISS | KVM_REG_RISCV_CSR_ZICFISS_REG(" #csr ")"
static const char *general_csr_id_to_str(__u64 reg_off)
{
@@ -425,6 +429,18 @@ static const char *smstateen_csr_id_to_str(__u64 reg_off)
return NULL;
}
+static const char *zicfiss_csr_id_to_str(__u64 reg_off)
+{
+ /* reg_off is the offset into struct kvm_riscv_zicfiss_csr */
+ switch (reg_off) {
+ case KVM_REG_RISCV_CSR_ZICFISS_REG(ssp):
+ return RISCV_CSR_ZICFISS(ssp);
+ }
+
+ TEST_FAIL("Unknown zicfiss csr reg: 0x%llx", reg_off);
+ return NULL;
+}
+
static const char *csr_id_to_str(const char *prefix, __u64 id)
{
__u64 reg_off = id & ~(REG_MASK | KVM_REG_RISCV_CSR);
@@ -441,6 +457,8 @@ static const char *csr_id_to_str(const char *prefix, __u64 id)
return aia_csr_id_to_str(reg_off);
case KVM_REG_RISCV_CSR_SMSTATEEN:
return smstateen_csr_id_to_str(reg_off);
+ case KVM_REG_RISCV_CSR_ZICFISS:
+ return zicfiss_csr_id_to_str(reg_off);
}
return strdup_printf("%lld | %lld /* UNKNOWN */", reg_subtype, reg_off);
@@ -585,6 +603,8 @@ static const char *isa_ext_single_id_to_str(__u64 reg_off)
KVM_ISA_EXT_ARR(ZICBOP),
KVM_ISA_EXT_ARR(ZICBOZ),
KVM_ISA_EXT_ARR(ZICCRSE),
+ KVM_ISA_EXT_ARR(ZICFILP),
+ KVM_ISA_EXT_ARR(ZICFISS),
KVM_ISA_EXT_ARR(ZICNTR),
KVM_ISA_EXT_ARR(ZICOND),
KVM_ISA_EXT_ARR(ZICSR),
@@ -745,6 +765,15 @@ static const char *sbi_fwft_id_to_str(__u64 reg_off)
case 3: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pointer_masking.enable)";
case 4: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pointer_masking.flags)";
case 5: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pointer_masking.value)";
+ case 6: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pte_ad_hw_updating.enable)";
+ case 7: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pte_ad_hw_updating.flags)";
+ case 8: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pte_ad_hw_updating.value)";
+ case 9: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(landing_pad.enable)";
+ case 10: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(landing_pad.flags)";
+ case 11: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(landing_pad.value)";
+ case 12: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(shadow_stack.enable)";
+ case 13: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(shadow_stack.flags)";
+ case 14: return "KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(shadow_stack.value)";
}
return strdup_printf("KVM_REG_RISCV_SBI_FWFT | %lld /* UNKNOWN */", reg_off);
}
@@ -944,6 +973,27 @@ static __u64 sbi_fwft_pointer_masking_regs[] = {
KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pointer_masking.value),
};
+static __u64 sbi_fwft_pte_ad_hw_updating_regs[] = {
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_EXT | KVM_REG_RISCV_SBI_SINGLE | KVM_RISCV_SBI_EXT_FWFT,
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pte_ad_hw_updating.enable),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pte_ad_hw_updating.flags),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(pte_ad_hw_updating.value),
+};
+
+static __u64 sbi_fwft_landing_pad_regs[] = {
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_EXT | KVM_REG_RISCV_SBI_SINGLE | KVM_RISCV_SBI_EXT_FWFT,
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(landing_pad.enable),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(landing_pad.flags),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(landing_pad.value),
+};
+
+static __u64 sbi_fwft_shadow_stack_regs[] = {
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_EXT | KVM_REG_RISCV_SBI_SINGLE | KVM_RISCV_SBI_EXT_FWFT,
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(shadow_stack.enable),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(shadow_stack.flags),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_SBI_STATE | KVM_REG_RISCV_SBI_FWFT | KVM_REG_RISCV_SBI_FWFT_REG(shadow_stack.value),
+};
+
static __u64 zicbom_regs[] = {
KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_CONFIG | KVM_REG_RISCV_CONFIG_REG(zicbom_block_size),
KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICBOM,
@@ -959,6 +1009,11 @@ static __u64 zicboz_regs[] = {
KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICBOZ,
};
+static __u64 zicfiss_regs[] = {
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_CSR | KVM_REG_RISCV_CSR_ZICFISS | KVM_REG_RISCV_CSR_ZICFISS_REG(ssp),
+ KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_ISA_EXT | KVM_REG_RISCV_ISA_SINGLE | KVM_RISCV_ISA_EXT_ZICFISS,
+};
+
static __u64 aia_regs[] = {
KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_CSR | KVM_REG_RISCV_CSR_AIA | KVM_REG_RISCV_CSR_AIA_REG(siselect),
KVM_REG_RISCV | KVM_REG_SIZE_ULONG | KVM_REG_RISCV_CSR | KVM_REG_RISCV_CSR_AIA | KVM_REG_RISCV_CSR_AIA_REG(iprio1),
@@ -1210,6 +1265,8 @@ KVM_ISA_EXT_SUBLIST_CONFIG(zicbom, ZICBOM);
KVM_ISA_EXT_SUBLIST_CONFIG(zicbop, ZICBOP);
KVM_ISA_EXT_SUBLIST_CONFIG(zicboz, ZICBOZ);
KVM_ISA_EXT_SIMPLE_CONFIG(ziccrse, ZICCRSE);
+KVM_ISA_EXT_SIMPLE_CONFIG(zicfilp, ZICFILP);
+KVM_ISA_EXT_SUBLIST_CONFIG(zicfiss, ZICFISS);
KVM_ISA_EXT_SIMPLE_CONFIG(zicntr, ZICNTR);
KVM_ISA_EXT_SIMPLE_CONFIG(zicond, ZICOND);
KVM_ISA_EXT_SIMPLE_CONFIG(zicsr, ZICSR);
@@ -1259,6 +1316,34 @@ static struct vcpu_reg_list config_sbi_fwft_pointer_masking = {
},
};
+static struct vcpu_reg_list config_sbi_fwft_pte_ad_hw_updating = {
+ .sublists = {
+ SUBLIST_BASE,
+ SUBLIST_ISA(svade, SVADE),
+ SUBLIST_ISA(svadu, SVADU),
+ SUBLIST_SBI(fwft_pte_ad_hw_updating, FWFT),
+ {0},
+ },
+};
+
+static struct vcpu_reg_list config_sbi_fwft_landing_pad = {
+ .sublists = {
+ SUBLIST_BASE,
+ SUBLIST_ISA(zicfilp, ZICFILP),
+ SUBLIST_SBI(fwft_landing_pad, FWFT),
+ {0},
+ },
+};
+
+static struct vcpu_reg_list config_sbi_fwft_shadow_stack = {
+ .sublists = {
+ SUBLIST_BASE,
+ SUBLIST_ISA(zicfiss, ZICFISS),
+ SUBLIST_SBI(fwft_shadow_stack, FWFT),
+ {0},
+ },
+};
+
struct vcpu_reg_list *vcpu_configs[] = {
&config_sbi_base,
&config_sbi_sta,
@@ -1268,6 +1353,9 @@ struct vcpu_reg_list *vcpu_configs[] = {
&config_sbi_mpxy,
&config_sbi_fwft_misaligned_deleg,
&config_sbi_fwft_pointer_masking,
+ &config_sbi_fwft_pte_ad_hw_updating,
+ &config_sbi_fwft_landing_pad,
+ &config_sbi_fwft_shadow_stack,
&config_aia,
&config_fp_f,
&config_fp_d,
@@ -1311,6 +1399,8 @@ struct vcpu_reg_list *vcpu_configs[] = {
&config_zicbop,
&config_zicboz,
&config_ziccrse,
+ &config_zicfilp,
+ &config_zicfiss,
&config_zicntr,
&config_zicond,
&config_zicsr,
diff --git a/tools/testing/selftests/kvm/riscv/sbi_pmu_test.c b/tools/testing/selftests/kvm/riscv/sbi_pmu_test.c
index e56a3dd6a51e..20388f0b959d 100644
--- a/tools/testing/selftests/kvm/riscv/sbi_pmu_test.c
+++ b/tools/testing/selftests/kvm/riscv/sbi_pmu_test.c
@@ -492,7 +492,7 @@ static void test_pmu_events_snaphost(void)
struct riscv_pmu_snapshot_data *snapshot_data = snapshot_gva;
int i;
- /* Verify presence of SBI PMU and minimum requrired SBI version */
+ /* Verify presence of SBI PMU and minimum required SBI version */
verify_sbi_requirement_assert();
snapshot_set_shmem(snapshot_gpa, 0);
@@ -518,7 +518,7 @@ static void test_pmu_events_overflow(void)
{
int num_counters = 0, i = 0;
- /* Verify presence of SBI PMU and minimum requrired SBI version */
+ /* Verify presence of SBI PMU and minimum required SBI version */
verify_sbi_requirement_assert();
snapshot_set_shmem(snapshot_gpa, 0);
diff --git a/tools/testing/selftests/kvm/rseq_test.c b/tools/testing/selftests/kvm/rseq_test.c
index f80ad6b47d16..b386aece68fd 100644
--- a/tools/testing/selftests/kvm/rseq_test.c
+++ b/tools/testing/selftests/kvm/rseq_test.c
@@ -75,7 +75,7 @@ static void *migration_worker(void *__rseq_tid)
{
pid_t rseq_tid = (pid_t)(unsigned long)__rseq_tid;
cpu_set_t allowed_mask;
- int r, i, cpu;
+ int i, cpu;
CPU_ZERO(&allowed_mask);
@@ -96,9 +96,7 @@ static void *migration_worker(void *__rseq_tid)
* stable, i.e. while changing affinity is in-progress.
*/
smp_wmb();
- r = sched_setaffinity(rseq_tid, sizeof(allowed_mask), &allowed_mask);
- TEST_ASSERT(!r, "sched_setaffinity failed, errno = %d (%s)",
- errno, strerror(errno));
+ kvm_sched_setaffinity(rseq_tid, sizeof(allowed_mask), &allowed_mask);
smp_wmb();
atomic_inc(&seq_cnt);
@@ -226,9 +224,7 @@ int main(int argc, char *argv[])
}
}
- r = sched_getaffinity(0, sizeof(possible_mask), &possible_mask);
- TEST_ASSERT(!r, "sched_getaffinity failed, errno = %d (%s)", errno,
- strerror(errno));
+ kvm_sched_getaffinity(0, sizeof(possible_mask), &possible_mask);
calc_min_max_cpu();
@@ -243,8 +239,8 @@ int main(int argc, char *argv[])
*/
vm = vm_create_with_one_vcpu(&vcpu, guest_code);
- pthread_create(&migration_thread, NULL, migration_worker,
- (void *)(unsigned long)syscall(SYS_gettid));
+ kvm_pthread_create(&migration_thread, NULL, migration_worker,
+ (void *)(unsigned long)kvm_gettid());
if (latency >= 0) {
/*
@@ -316,7 +312,7 @@ int main(int argc, char *argv[])
" e.g. via cpuidle.off=1 or via -l <latency>, or run with -u to\n"
" disable this sanity check.", i);
- pthread_join(migration_thread, NULL);
+ kvm_pthread_join(migration_thread, NULL);
kvm_vm_free(vm);
diff --git a/tools/testing/selftests/kvm/s390/memop.c b/tools/testing/selftests/kvm/s390/memop.c
index 0244848621b3..ce8db96ccf7e 100644
--- a/tools/testing/selftests/kvm/s390/memop.c
+++ b/tools/testing/selftests/kvm/s390/memop.c
@@ -678,7 +678,7 @@ static void test_cmpxchg_key_concurrent(void)
HOST_SYNC(t.vcpu, STAGE_SKEYS_SET);
prepare_mem12();
MOP(t.vcpu, LOGICAL, WRITE, mem1, max_block, GADDR_V(mem2));
- pthread_create(&thread, NULL, run_guest, &t.vcpu);
+ kvm_pthread_create(&thread, NULL, run_guest, &t.vcpu);
for (int i = 0; i < cmpxchg_iter_outer; i++) {
do {
@@ -701,7 +701,7 @@ static void test_cmpxchg_key_concurrent(void)
}
}
- pthread_join(thread, NULL);
+ kvm_pthread_join(thread, NULL);
MOP(t.vcpu, LOGICAL, READ, mem2, max_block, GADDR_V(mem2));
TEST_ASSERT(popcount_eq(*(__uint128_t *)mem1, *(__uint128_t *)mem2),
diff --git a/tools/testing/selftests/kvm/set_memory_region_test.c b/tools/testing/selftests/kvm/set_memory_region_test.c
index a152ab65c657..dffe9654f45b 100644
--- a/tools/testing/selftests/kvm/set_memory_region_test.c
+++ b/tools/testing/selftests/kvm/set_memory_region_test.c
@@ -133,7 +133,7 @@ static struct kvm_vm *spawn_vm(struct kvm_vcpu **vcpu, pthread_t *vcpu_thread,
hva = addr_gpa2hva(vm, MEM_REGION_GPA);
memset(hva, 0, 2 * 4096);
- pthread_create(vcpu_thread, NULL, vcpu_worker, *vcpu);
+ kvm_pthread_create(vcpu_thread, NULL, vcpu_worker, *vcpu);
/* Ensure the guest thread is spun up. */
wait_for_vcpu();
@@ -216,7 +216,7 @@ static void test_move_memory_region(bool disable_slot_zap_quirk)
/* Defered sync from when the memslot was misaligned (above). */
wait_for_vcpu();
- pthread_join(vcpu_thread, NULL);
+ kvm_pthread_join(vcpu_thread, NULL);
kvm_vm_free(vm);
}
@@ -302,7 +302,7 @@ static void test_delete_memory_region(bool disable_slot_zap_quirk)
*/
vm_mem_region_delete(vm, 0);
- pthread_join(vcpu_thread, NULL);
+ kvm_pthread_join(vcpu_thread, NULL);
run = vcpu->run;
diff --git a/tools/testing/selftests/kvm/steal_time.c b/tools/testing/selftests/kvm/steal_time.c
index 76fcdd1fd3cb..bc3c62b72c58 100644
--- a/tools/testing/selftests/kvm/steal_time.c
+++ b/tools/testing/selftests/kvm/steal_time.c
@@ -70,8 +70,8 @@ static bool is_steal_time_supported(struct kvm_vcpu *vcpu)
static void steal_time_init(struct kvm_vcpu *vcpu, u32 i)
{
/* ST_GPA_BASE is identity mapped */
- st_gva[i] = (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE);
- sync_global_to_guest(vcpu->vm, st_gva[i]);
+ WRITE_AND_SYNC_TO_GUEST(vcpu->vm, st_gva[i],
+ (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE));
vcpu_set_msr(vcpu, MSR_KVM_STEAL_TIME, (ulong)st_gva[i] | KVM_MSR_ENABLED);
}
@@ -187,8 +187,7 @@ static void steal_time_init(struct kvm_vcpu *vcpu, u32 i)
};
/* ST_GPA_BASE is identity mapped */
- st_gva[i] = (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE);
- sync_global_to_guest(vm, st_gva[i]);
+ WRITE_AND_SYNC_TO_GUEST(vm, st_gva[i], (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE));
st_ipa = (ulong)st_gva[i];
vcpu_ioctl(vcpu, KVM_SET_DEVICE_ATTR, &dev);
@@ -310,10 +309,8 @@ static bool is_steal_time_supported(struct kvm_vcpu *vcpu)
static void steal_time_init(struct kvm_vcpu *vcpu, u32 i)
{
/* ST_GPA_BASE is identity mapped */
- st_gva[i] = (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE);
- st_gpa[i] = addr_gva2gpa(vcpu->vm, (gva_t)st_gva[i]);
- sync_global_to_guest(vcpu->vm, st_gva[i]);
- sync_global_to_guest(vcpu->vm, st_gpa[i]);
+ WRITE_AND_SYNC_TO_GUEST(vcpu->vm, st_gva[i], (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE));
+ WRITE_AND_SYNC_TO_GUEST(vcpu->vm, st_gpa[i], addr_gva2gpa(vcpu->vm, (gva_t)st_gva[i]));
}
static void steal_time_dump(struct kvm_vm *vm, u32 vcpu_idx)
@@ -442,8 +439,7 @@ static void steal_time_init(struct kvm_vcpu *vcpu, u32 i)
};
/* ST_GPA_BASE is identity mapped */
- st_gva[i] = (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE);
- sync_global_to_guest(vm, st_gva[i]);
+ WRITE_AND_SYNC_TO_GUEST(vm, st_gva[i], (void *)(ST_GPA_BASE + i * STEAL_TIME_SIZE));
err = __vcpu_ioctl(vcpu, KVM_HAS_DEVICE_ATTR, &attr);
TEST_ASSERT(err == 0, "No PV stealtime Feature");
@@ -508,23 +504,18 @@ int main(int ac, char **av)
{
struct kvm_vcpu *vcpus[NR_VCPUS];
struct kvm_vm *vm;
- pthread_attr_t attr;
pthread_t thread;
cpu_set_t cpuset;
unsigned int gpages;
long stolen_time;
long run_delay;
bool verbose;
- int i;
+ int i, cpu;
verbose = ac > 1 && (!strncmp(av[1], "-v", 3) || !strncmp(av[1], "--verbose", 10));
/* Set CPU affinity so we can force preemption of the VCPU */
- CPU_ZERO(&cpuset);
- CPU_SET(0, &cpuset);
- pthread_attr_init(&attr);
- pthread_attr_setaffinity_np(&attr, sizeof(cpu_set_t), &cpuset);
- pthread_setaffinity_np(pthread_self(), sizeof(cpu_set_t), &cpuset);
+ cpu = pin_self_to_any_cpu();
/* Create a VM and an identity mapped memslot for the steal time structure */
vm = vm_create_with_vcpus(NR_VCPUS, guest_code, vcpus);
@@ -549,8 +540,7 @@ int main(int ac, char **av)
/* Second VCPU run, expect guest stolen time to be <= run_delay */
run_vcpu(vcpus[i]);
- sync_global_from_guest(vm, guest_stolen_time[i]);
- stolen_time = guest_stolen_time[i];
+ stolen_time = SYNC_FROM_GUEST_AND_READ(vm, guest_stolen_time[i]);
run_delay = get_run_delay();
TEST_ASSERT(stolen_time <= run_delay,
"Expected stolen time <= %ld, got %ld",
@@ -558,11 +548,15 @@ int main(int ac, char **av)
/* Steal time from the VCPU. The steal time thread has the same CPU affinity as the VCPUs. */
run_delay = get_run_delay();
- pthread_create(&thread, &attr, do_steal_time, NULL);
+ kvm_pthread_create(&thread, NULL, do_steal_time, NULL);
+ kvm_pthread_getaffinity(thread, &cpuset);
+ TEST_ASSERT(CPU_COUNT(&cpuset) == 1 && CPU_ISSET(cpu, &cpuset),
+ "Worker failed to inherit parent's CPU affinity");
+
do
sched_yield();
while (get_run_delay() - run_delay < MIN_RUN_DELAY_NS);
- pthread_join(thread, NULL);
+ kvm_pthread_join(thread, NULL);
run_delay = get_run_delay() - run_delay;
TEST_ASSERT(run_delay >= MIN_RUN_DELAY_NS,
"Expected run_delay >= %ld, got %ld",
@@ -570,8 +564,7 @@ int main(int ac, char **av)
/* Run VCPU again to confirm stolen time is consistent with run_delay */
run_vcpu(vcpus[i]);
- sync_global_from_guest(vm, guest_stolen_time[i]);
- stolen_time = guest_stolen_time[i] - stolen_time;
+ stolen_time = SYNC_FROM_GUEST_AND_READ(vm, guest_stolen_time[i]) - stolen_time;
TEST_ASSERT(stolen_time >= run_delay,
"Expected stolen time >= %ld, got %ld",
run_delay, stolen_time);
diff --git a/tools/testing/selftests/kvm/vm_types_test.c b/tools/testing/selftests/kvm/vm_types_test.c
new file mode 100644
index 000000000000..6c421871e74b
--- /dev/null
+++ b/tools/testing/selftests/kvm/vm_types_test.c
@@ -0,0 +1,48 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Verify that KVM_CREATE_VM accepts exactly the VM types enumerated by
+ * KVM_CAP_VM_TYPES, and rejects every other type with -EINVAL.
+ */
+#include <errno.h>
+#include <stdbool.h>
+#include <unistd.h>
+
+#include <linux/kvm.h>
+
+#include "kvm_util.h"
+#include "test_util.h"
+
+int main(void)
+{
+ unsigned long type, supported_types;
+ int kvm_fd;
+
+ TEST_REQUIRE(kvm_has_cap(KVM_CAP_VM_TYPES));
+
+ kvm_fd = open_kvm_dev_path_or_exit();
+ supported_types = kvm_check_cap(KVM_CAP_VM_TYPES);
+ pr_info("Supported VM types: 0x%lx\n", supported_types);
+
+ /*
+ * For compatibility with 32-bit kernels, KVM_CHECK_EXTENSION restricts
+ * its return to 32-bit values, i.e. only types 0..31 can be advertised.
+ * Walk past that range as well to confirm that any out-of-range type is
+ * rejected rather than silently accepted (or truncated).
+ */
+ for (type = 0; type < BITS_PER_TYPE(supported_types); type++) {
+ int fd = __kvm_ioctl(kvm_fd, KVM_CREATE_VM, (void *)type);
+
+ if (supported_types & BIT(type)) {
+ TEST_ASSERT(fd >= 0,
+ "KVM_CREATE_VM(%lu) should succeed, supported types = 0x%lx",
+ type, supported_types);
+ kvm_close(fd);
+ } else {
+ TEST_ASSERT(fd < 0 && errno == EINVAL,
+ "KVM_CREATE_VM(%lu) should fail with EINVAL, supported types = 0x%lx",
+ type, supported_types);
+ }
+ }
+
+ return 0;
+}
diff --git a/tools/testing/selftests/kvm/x86/aperfmperf_test.c b/tools/testing/selftests/kvm/x86/aperfmperf_test.c
index c91660103137..845cb685f174 100644
--- a/tools/testing/selftests/kvm/x86/aperfmperf_test.c
+++ b/tools/testing/selftests/kvm/x86/aperfmperf_test.c
@@ -54,8 +54,6 @@ static void guest_read_aperf_mperf(void)
GUEST_SYNC2(rdmsr(MSR_IA32_APERF), rdmsr(MSR_IA32_MPERF));
}
-#define L2_GUEST_STACK_SIZE 64
-
static void l2_guest_code(void)
{
guest_read_aperf_mperf();
@@ -64,21 +62,18 @@ static void l2_guest_code(void)
static void l1_svm_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
- generic_svm_setup(svm, l2_guest_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
run_guest(vmcb, svm->vmcb_gpa);
}
static void l1_vmx_code(struct vmx_pages *vmx)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
GUEST_ASSERT_EQ(prepare_for_vmx_operation(vmx), true);
GUEST_ASSERT_EQ(load_vmcs(vmx), true);
- prepare_vmcs(vmx, NULL, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx, NULL);
/*
* Enable MSR bitmaps (the bitmap itself is allocated, zeroed, and set
diff --git a/tools/testing/selftests/kvm/x86/evmcs_smm_controls_test.c b/tools/testing/selftests/kvm/x86/evmcs_smm_controls_test.c
index 5b3aef109cfc..aa7f3b405fd3 100644
--- a/tools/testing/selftests/kvm/x86/evmcs_smm_controls_test.c
+++ b/tools/testing/selftests/kvm/x86/evmcs_smm_controls_test.c
@@ -22,9 +22,6 @@
#define SYNC_PORT 0xe
-#define STR(x) #x
-#define XSTR(s) STR(s)
-
/*
* SMI handler: runs in real-address mode.
* Reports SMRAM_STAGE via port IO, then does RSM.
@@ -37,7 +34,7 @@ static u8 smi_handler[] = {
static inline void sync_with_host(u64 phase)
{
- asm volatile("in $" XSTR(SYNC_PORT) ", %%al \n"
+ asm volatile("in $" __stringify(SYNC_PORT) ", %%al \n"
: "+a" (phase));
}
@@ -52,8 +49,6 @@ static void l2_guest_code(void)
static void guest_code(struct vmx_pages *vmx_pages,
struct hyperv_test_pages *hv_pages)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
/* Set up Hyper-V enlightenments and eVMCS */
wrmsr(HV_X64_MSR_GUEST_OS_ID, HYPERV_LINUX_OS_ID);
@@ -62,8 +57,7 @@ static void guest_code(struct vmx_pages *vmx_pages,
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_evmcs(hv_pages));
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
GUEST_ASSERT(!vmlaunch());
diff --git a/tools/testing/selftests/kvm/x86/fix_hypercall_test.c b/tools/testing/selftests/kvm/x86/fix_hypercall_test.c
index 753a0e730ea8..4931ec22768e 100644
--- a/tools/testing/selftests/kvm/x86/fix_hypercall_test.c
+++ b/tools/testing/selftests/kvm/x86/fix_hypercall_test.c
@@ -6,7 +6,6 @@
*/
#include <asm/kvm_para.h>
#include <linux/kvm_para.h>
-#include <linux/stringify.h>
#include <stdint.h>
#include "kvm_test_harness.h"
diff --git a/tools/testing/selftests/kvm/x86/hyperv_clock.c b/tools/testing/selftests/kvm/x86/hyperv_clock.c
index c083cea546dc..d5d779623cc6 100644
--- a/tools/testing/selftests/kvm/x86/hyperv_clock.c
+++ b/tools/testing/selftests/kvm/x86/hyperv_clock.c
@@ -56,7 +56,7 @@ static inline void check_tsc_msr_rdtsc(void)
tsc_freq = rdmsr(HV_X64_MSR_TSC_FREQUENCY);
GUEST_ASSERT(tsc_freq > 0);
- /* For increased accuracy, take mean rdtsc() before and afrer rdmsr() */
+ /* For increased accuracy, take mean rdtsc() before and after rdmsr() */
r1 = rdtsc();
t1 = rdmsr(HV_X64_MSR_TIME_REF_COUNT);
r1 = (r1 + rdtsc()) / 2;
@@ -181,7 +181,7 @@ static void host_check_tsc_msr_rdtsc(struct kvm_vcpu *vcpu)
tsc_freq = vcpu_get_msr(vcpu, HV_X64_MSR_TSC_FREQUENCY);
TEST_ASSERT(tsc_freq > 0, "TSC frequency must be nonzero");
- /* For increased accuracy, take mean rdtsc() before and afrer ioctl */
+ /* For increased accuracy, take mean rdtsc() before and after ioctl */
r1 = rdtsc();
t1 = vcpu_get_msr(vcpu, HV_X64_MSR_TIME_REF_COUNT);
r1 = (r1 + rdtsc()) / 2;
diff --git a/tools/testing/selftests/kvm/x86/hyperv_evmcs.c b/tools/testing/selftests/kvm/x86/hyperv_evmcs.c
index c7fa114aee20..63ea1533e4ea 100644
--- a/tools/testing/selftests/kvm/x86/hyperv_evmcs.c
+++ b/tools/testing/selftests/kvm/x86/hyperv_evmcs.c
@@ -78,9 +78,6 @@ void l2_guest_code(void)
void guest_code(struct vmx_pages *vmx_pages, struct hyperv_test_pages *hv_pages,
gpa_t hv_hcall_page_gpa)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
wrmsr(HV_X64_MSR_GUEST_OS_ID, HYPERV_LINUX_OS_ID);
wrmsr(HV_X64_MSR_HYPERCALL, hv_hcall_page_gpa);
@@ -100,8 +97,7 @@ void guest_code(struct vmx_pages *vmx_pages, struct hyperv_test_pages *hv_pages,
GUEST_SYNC(4);
GUEST_ASSERT(vmptrstz() == hv_pages->enlightened_vmcs_gpa);
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
GUEST_SYNC(5);
GUEST_ASSERT(vmptrstz() == hv_pages->enlightened_vmcs_gpa);
@@ -129,7 +125,7 @@ void guest_code(struct vmx_pages *vmx_pages, struct hyperv_test_pages *hv_pages,
/*
* NMI forces L2->L1 exit, resuming L2 and hope that EVMCS is
* up-to-date (RIP points where it should and not at the beginning
- * of l2_guest_code(). GUEST_SYNC(9) checkes that.
+ * of l2_guest_code(). GUEST_SYNC(9) checks that.
*/
GUEST_ASSERT(!vmresume());
diff --git a/tools/testing/selftests/kvm/x86/hyperv_ipi.c b/tools/testing/selftests/kvm/x86/hyperv_ipi.c
index 771535f9aad3..a2fc0bca8011 100644
--- a/tools/testing/selftests/kvm/x86/hyperv_ipi.c
+++ b/tools/testing/selftests/kvm/x86/hyperv_ipi.c
@@ -222,30 +222,13 @@ static void *vcpu_thread(void *arg)
return NULL;
}
-static void cancel_join_vcpu_thread(pthread_t thread, struct kvm_vcpu *vcpu)
-{
- void *retval;
- int r;
-
- r = pthread_cancel(thread);
- TEST_ASSERT(!r, "pthread_cancel on vcpu_id=%d failed with errno=%d",
- vcpu->id, r);
-
- r = pthread_join(thread, &retval);
- TEST_ASSERT(!r, "pthread_join on vcpu_id=%d failed with errno=%d",
- vcpu->id, r);
- TEST_ASSERT(retval == PTHREAD_CANCELED,
- "expected retval=%p, got %p", PTHREAD_CANCELED,
- retval);
-}
-
int main(int argc, char *argv[])
{
struct kvm_vm *vm;
struct kvm_vcpu *vcpu[3];
gva_t hcall_page;
pthread_t threads[2];
- int stage = 1, r;
+ int stage = 1;
struct ucall uc;
TEST_REQUIRE(kvm_has_cap(KVM_CAP_HYPERV_SEND_IPI));
@@ -272,11 +255,8 @@ int main(int argc, char *argv[])
vcpu_args_set(vcpu[0], 2, hcall_page, addr_gva2gpa(vm, hcall_page));
vcpu_set_hv_cpuid(vcpu[0]);
- r = pthread_create(&threads[0], NULL, vcpu_thread, vcpu[1]);
- TEST_ASSERT(!r, "pthread_create failed errno=%d", r);
-
- r = pthread_create(&threads[1], NULL, vcpu_thread, vcpu[2]);
- TEST_ASSERT(!r, "pthread_create failed errno=%d", errno);
+ kvm_pthread_create(&threads[0], NULL, vcpu_thread, vcpu[1]);
+ kvm_pthread_create(&threads[1], NULL, vcpu_thread, vcpu[2]);
while (true) {
vcpu_run(vcpu[0]);
@@ -302,9 +282,9 @@ int main(int argc, char *argv[])
}
done:
- cancel_join_vcpu_thread(threads[0], vcpu[1]);
- cancel_join_vcpu_thread(threads[1], vcpu[2]);
+ kvm_pthread_cancel_join_async(threads[0]);
+ kvm_pthread_cancel_join_async(threads[1]);
kvm_vm_free(vm);
- return r;
+ return 0;
}
diff --git a/tools/testing/selftests/kvm/x86/hyperv_svm_test.c b/tools/testing/selftests/kvm/x86/hyperv_svm_test.c
index 7a62f6a9d606..1f74b0fa9b83 100644
--- a/tools/testing/selftests/kvm/x86/hyperv_svm_test.c
+++ b/tools/testing/selftests/kvm/x86/hyperv_svm_test.c
@@ -18,8 +18,6 @@
#include "svm_util.h"
#include "hyperv.h"
-#define L2_GUEST_STACK_SIZE 256
-
/* Exit to L1 from L2 with RDMSR instruction */
static inline void rdmsr_from_l2(u32 msr)
{
@@ -69,7 +67,6 @@ static void __attribute__((__flatten__)) guest_code(struct svm_test_data *svm,
struct hyperv_test_pages *hv_pages,
gpa_t pgs_gpa)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
struct hv_vmcb_enlightenments *hve = &vmcb->control.hv_enlightenments;
@@ -81,8 +78,7 @@ static void __attribute__((__flatten__)) guest_code(struct svm_test_data *svm,
GUEST_ASSERT(svm->vmcb_gpa);
/* Prepare for L2 execution. */
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
/* L2 TLB flush setup */
hve->partition_assist_page = hv_pages->partition_assist_gpa;
diff --git a/tools/testing/selftests/kvm/x86/hyperv_tlb_flush.c b/tools/testing/selftests/kvm/x86/hyperv_tlb_flush.c
index b4be9a175379..dc9d40830a7f 100644
--- a/tools/testing/selftests/kvm/x86/hyperv_tlb_flush.c
+++ b/tools/testing/selftests/kvm/x86/hyperv_tlb_flush.c
@@ -548,23 +548,6 @@ static void *vcpu_thread(void *arg)
return NULL;
}
-static void cancel_join_vcpu_thread(pthread_t thread, struct kvm_vcpu *vcpu)
-{
- void *retval;
- int r;
-
- r = pthread_cancel(thread);
- TEST_ASSERT(!r, "pthread_cancel on vcpu_id=%d failed with errno=%d",
- vcpu->id, r);
-
- r = pthread_join(thread, &retval);
- TEST_ASSERT(!r, "pthread_join on vcpu_id=%d failed with errno=%d",
- vcpu->id, r);
- TEST_ASSERT(retval == PTHREAD_CANCELED,
- "expected retval=%p, got %p", PTHREAD_CANCELED,
- retval);
-}
-
int main(int argc, char *argv[])
{
struct kvm_vm *vm;
@@ -575,7 +558,7 @@ int main(int argc, char *argv[])
u64 *pte;
struct test_data *data;
struct ucall uc;
- int stage = 1, r, i;
+ int stage = 1, i;
TEST_REQUIRE(kvm_has_cap(KVM_CAP_HYPERV_TLBFLUSH));
@@ -632,11 +615,8 @@ int main(int argc, char *argv[])
vcpu_set_msr(vcpu[2], HV_X64_MSR_VP_INDEX, WORKER_VCPU_ID_2);
vcpu_set_hv_cpuid(vcpu[2]);
- r = pthread_create(&threads[0], NULL, vcpu_thread, vcpu[1]);
- TEST_ASSERT(!r, "pthread_create() failed");
-
- r = pthread_create(&threads[1], NULL, vcpu_thread, vcpu[2]);
- TEST_ASSERT(!r, "pthread_create() failed");
+ kvm_pthread_create(&threads[0], NULL, vcpu_thread, vcpu[1]);
+ kvm_pthread_create(&threads[1], NULL, vcpu_thread, vcpu[2]);
while (true) {
vcpu_run(vcpu[0]);
@@ -661,8 +641,8 @@ int main(int argc, char *argv[])
}
done:
- cancel_join_vcpu_thread(threads[0], vcpu[1]);
- cancel_join_vcpu_thread(threads[1], vcpu[2]);
+ kvm_pthread_cancel_join_async(threads[0]);
+ kvm_pthread_cancel_join_async(threads[1]);
kvm_vm_free(vm);
return 0;
diff --git a/tools/testing/selftests/kvm/x86/kvm_buslock_test.c b/tools/testing/selftests/kvm/x86/kvm_buslock_test.c
index 52014a3210c8..25a182be00a9 100644
--- a/tools/testing/selftests/kvm/x86/kvm_buslock_test.c
+++ b/tools/testing/selftests/kvm/x86/kvm_buslock_test.c
@@ -26,8 +26,6 @@ static void guest_generate_buslocks(void)
atomic_inc(val);
}
-#define L2_GUEST_STACK_SIZE 64
-
static void l2_guest_code(void)
{
guest_generate_buslocks();
@@ -36,21 +34,18 @@ static void l2_guest_code(void)
static void l1_svm_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
- generic_svm_setup(svm, l2_guest_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
run_guest(vmcb, svm->vmcb_gpa);
}
static void l1_vmx_code(struct vmx_pages *vmx)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
GUEST_ASSERT_EQ(prepare_for_vmx_operation(vmx), true);
GUEST_ASSERT_EQ(load_vmcs(vmx), true);
- prepare_vmcs(vmx, NULL, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx, NULL);
GUEST_ASSERT(!vmwrite(GUEST_RIP, (u64)l2_guest_code));
GUEST_ASSERT(!vmlaunch());
diff --git a/tools/testing/selftests/kvm/x86/nested_close_kvm_test.c b/tools/testing/selftests/kvm/x86/nested_close_kvm_test.c
index 761fec293408..b974cfb347d6 100644
--- a/tools/testing/selftests/kvm/x86/nested_close_kvm_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_close_kvm_test.c
@@ -21,8 +21,6 @@ enum {
PORT_L0_EXIT = 0x2000,
};
-#define L2_GUEST_STACK_SIZE 64
-
static void l2_guest_code(void)
{
/* Exit to L0 */
@@ -32,14 +30,11 @@ static void l2_guest_code(void)
static void l1_vmx_code(struct vmx_pages *vmx_pages)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
/* Prepare the VMCS for L2 execution. */
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
GUEST_ASSERT(!vmlaunch());
GUEST_ASSERT(0);
@@ -47,11 +42,8 @@ static void l1_vmx_code(struct vmx_pages *vmx_pages)
static void l1_svm_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
/* Prepare the VMCB for L2 execution. */
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
run_guest(svm->vmcb, svm->vmcb_gpa);
GUEST_ASSERT(0);
diff --git a/tools/testing/selftests/kvm/x86/nested_dirty_log_test.c b/tools/testing/selftests/kvm/x86/nested_dirty_log_test.c
index 0e67cce83570..26b474bf1353 100644
--- a/tools/testing/selftests/kvm/x86/nested_dirty_log_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_dirty_log_test.c
@@ -40,8 +40,6 @@
#define TEST_HVA(vm, idx) addr_gpa2hva(vm, TEST_GPA(idx))
-#define L2_GUEST_STACK_SIZE 64
-
/* Use the page offset bits to communicate the access+fault type. */
#define TEST_SYNC_READ_FAULT BIT(0)
#define TEST_SYNC_WRITE_FAULT BIT(1)
@@ -92,7 +90,6 @@ static void l2_guest_code_tdp_disabled(void)
void l1_vmx_code(struct vmx_pages *vmx)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
void *l2_rip;
GUEST_ASSERT(vmx->vmcs_gpa);
@@ -104,7 +101,7 @@ void l1_vmx_code(struct vmx_pages *vmx)
else
l2_rip = l2_guest_code_tdp_disabled;
- prepare_vmcs(vmx, l2_rip, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx, l2_rip);
GUEST_SYNC(TEST_SYNC_NO_FAULT);
GUEST_ASSERT(!vmlaunch());
@@ -115,7 +112,6 @@ void l1_vmx_code(struct vmx_pages *vmx)
static void l1_svm_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
void *l2_rip;
if (svm->ncr3_gpa)
@@ -123,7 +119,7 @@ static void l1_svm_code(struct svm_test_data *svm)
else
l2_rip = l2_guest_code_tdp_disabled;
- generic_svm_setup(svm, l2_rip, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_rip);
GUEST_SYNC(TEST_SYNC_NO_FAULT);
run_guest(svm->vmcb, svm->vmcb_gpa);
diff --git a/tools/testing/selftests/kvm/x86/nested_emulation_test.c b/tools/testing/selftests/kvm/x86/nested_emulation_test.c
index fb7dcbe53ac7..e08c6b0697e5 100644
--- a/tools/testing/selftests/kvm/x86/nested_emulation_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_emulation_test.c
@@ -57,7 +57,7 @@ static void guest_code(void *test_data)
struct svm_test_data *svm = test_data;
struct vmcb *vmcb = svm->vmcb;
- generic_svm_setup(svm, NULL, NULL);
+ generic_svm_setup(svm, NULL);
vmcb->save.idtr.limit = 0;
vmcb->save.rip = (u64)l2_guest_code;
@@ -69,7 +69,7 @@ static void guest_code(void *test_data)
GUEST_ASSERT(prepare_for_vmx_operation(test_data));
GUEST_ASSERT(load_vmcs(test_data));
- prepare_vmcs(test_data, NULL, NULL);
+ prepare_vmcs(test_data, NULL);
GUEST_ASSERT(!vmwrite(GUEST_IDTR_LIMIT, 0));
GUEST_ASSERT(!vmwrite(GUEST_RIP, (u64)l2_guest_code));
GUEST_ASSERT(!vmwrite(EXCEPTION_BITMAP, 0));
diff --git a/tools/testing/selftests/kvm/x86/nested_exceptions_test.c b/tools/testing/selftests/kvm/x86/nested_exceptions_test.c
index 186e980aa8ee..aeec3121c8e8 100644
--- a/tools/testing/selftests/kvm/x86/nested_exceptions_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_exceptions_test.c
@@ -5,8 +5,6 @@
#include "vmx.h"
#include "svm_util.h"
-#define L2_GUEST_STACK_SIZE 256
-
/*
* Arbitrary, never shoved into KVM/hardware, just need to avoid conflict with
* the "real" exceptions used, #SS/#GP/#DF (12/13/8).
@@ -91,9 +89,8 @@ static void svm_run_l2(struct svm_test_data *svm, void *l2_code, int vector,
static void l1_svm_code(struct svm_test_data *svm)
{
struct vmcb_control_area *ctrl = &svm->vmcb->control;
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
- generic_svm_setup(svm, NULL, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, NULL);
svm->vmcb->save.idtr.limit = 0;
ctrl->intercept |= BIT_ULL(INTERCEPT_SHUTDOWN);
@@ -128,13 +125,11 @@ static void vmx_run_l2(void *l2_code, int vector, u32 error_code)
static void l1_vmx_code(struct vmx_pages *vmx)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
GUEST_ASSERT_EQ(prepare_for_vmx_operation(vmx), true);
GUEST_ASSERT_EQ(load_vmcs(vmx), true);
- prepare_vmcs(vmx, NULL, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx, NULL);
GUEST_ASSERT_EQ(vmwrite(GUEST_IDTR_LIMIT, 0), 0);
/*
diff --git a/tools/testing/selftests/kvm/x86/nested_invalid_cr3_test.c b/tools/testing/selftests/kvm/x86/nested_invalid_cr3_test.c
index 11fd2467d823..8c2ba9674558 100644
--- a/tools/testing/selftests/kvm/x86/nested_invalid_cr3_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_invalid_cr3_test.c
@@ -11,8 +11,6 @@
#include "kselftest.h"
-#define L2_GUEST_STACK_SIZE 64
-
static void l2_guest_code(void)
{
vmcall();
@@ -20,11 +18,9 @@ static void l2_guest_code(void)
static void l1_svm_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
uintptr_t save_cr3;
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
/* Try to run L2 with invalid CR3 and make sure it fails */
save_cr3 = svm->vmcb->save.cr3;
@@ -42,14 +38,12 @@ static void l1_svm_code(struct svm_test_data *svm)
static void l1_vmx_code(struct vmx_pages *vmx_pages)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
uintptr_t save_cr3;
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
/* Try to run L2 with invalid CR3 and make sure it fails */
save_cr3 = vmreadz(GUEST_CR3);
diff --git a/tools/testing/selftests/kvm/x86/nested_tdp_fault_test.c b/tools/testing/selftests/kvm/x86/nested_tdp_fault_test.c
index fa95568f55ff..2e04563790ff 100644
--- a/tools/testing/selftests/kvm/x86/nested_tdp_fault_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_tdp_fault_test.c
@@ -9,8 +9,6 @@
#include "svm_util.h"
#include "vmx.h"
-#define L2_GUEST_STACK_SIZE 64
-
enum test_type {
TEST_FINAL_PAGE_UNMAPPED, /* Final data page not present */
TEST_PT_PAGE_UNMAPPED, /* Page table page not present */
@@ -54,14 +52,13 @@ static void l2_guest_code_ins(void)
static void l1_vmx_code(struct vmx_pages *vmx, u64 expected_fault_gpa,
u64 test_type)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u64 exit_qual;
GUEST_ASSERT(vmx->vmcs_gpa);
GUEST_ASSERT(prepare_for_vmx_operation(vmx));
GUEST_ASSERT(load_vmcs(vmx));
- prepare_vmcs(vmx, l2_entry, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx, l2_entry);
GUEST_ASSERT(!vmlaunch());
@@ -120,12 +117,10 @@ static void l1_vmx_code(struct vmx_pages *vmx, u64 expected_fault_gpa,
static void l1_svm_code(struct svm_test_data *svm, u64 expected_fault_gpa,
u64 test_type)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
u64 exit_info_1;
- generic_svm_setup(svm, l2_entry,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_entry);
run_guest(vmcb, svm->vmcb_gpa);
diff --git a/tools/testing/selftests/kvm/x86/nested_tsc_adjust_test.c b/tools/testing/selftests/kvm/x86/nested_tsc_adjust_test.c
index f0e4adac4751..cb79d7b9619c 100644
--- a/tools/testing/selftests/kvm/x86/nested_tsc_adjust_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_tsc_adjust_test.c
@@ -34,8 +34,6 @@
#define TSC_ADJUST_VALUE (1ll << 32)
#define TSC_OFFSET_VALUE -(1ll << 48)
-#define L2_GUEST_STACK_SIZE 64
-
enum {
PORT_ABORT = 0x1000,
PORT_REPORT,
@@ -75,8 +73,6 @@ static void l2_guest_code(void)
static void l1_guest_code(void *data)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
/* Set TSC from L1 and make sure TSC_ADJUST is updated correctly */
GUEST_ASSERT(rdtsc() < TSC_ADJUST_VALUE);
wrmsr(MSR_IA32_TSC, rdtsc() - TSC_ADJUST_VALUE);
@@ -93,8 +89,7 @@ static void l1_guest_code(void *data)
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
control = vmreadz(CPU_BASED_VM_EXEC_CONTROL);
control |= CPU_BASED_USE_MSR_BITMAPS | CPU_BASED_USE_TSC_OFFSETTING;
vmwrite(CPU_BASED_VM_EXEC_CONTROL, control);
@@ -105,8 +100,7 @@ static void l1_guest_code(void *data)
} else {
struct svm_test_data *svm = data;
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
svm->vmcb->control.tsc_offset = TSC_OFFSET_VALUE;
run_guest(svm->vmcb, svm->vmcb_gpa);
diff --git a/tools/testing/selftests/kvm/x86/nested_tsc_scaling_test.c b/tools/testing/selftests/kvm/x86/nested_tsc_scaling_test.c
index 190e93af20a1..18f765835bf4 100644
--- a/tools/testing/selftests/kvm/x86/nested_tsc_scaling_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_tsc_scaling_test.c
@@ -22,8 +22,6 @@
#define TSC_OFFSET_L2 ((u64)-33125236320908)
#define TSC_MULTIPLIER_L2 (L2_SCALE_FACTOR << 48)
-#define L2_GUEST_STACK_SIZE 64
-
enum { USLEEP, UCHECK_L1, UCHECK_L2 };
#define GUEST_SLEEP(sec) ucall(UCALL_SYNC, 2, USLEEP, sec)
#define GUEST_CHECK(level, freq) ucall(UCALL_SYNC, 2, level, freq)
@@ -82,13 +80,10 @@ static void l2_guest_code(void)
static void l1_svm_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
/* check that L1's frequency looks alright before launching L2 */
check_tsc_freq(UCHECK_L1);
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
/* enable TSC scaling for L2 */
wrmsr(MSR_AMD64_TSC_RATIO, L2_SCALE_FACTOR << 32);
@@ -105,7 +100,6 @@ static void l1_svm_code(struct svm_test_data *svm)
static void l1_vmx_code(struct vmx_pages *vmx_pages)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u32 control;
/* check that L1's frequency looks alright before launching L2 */
@@ -115,7 +109,7 @@ static void l1_vmx_code(struct vmx_pages *vmx_pages)
GUEST_ASSERT(load_vmcs(vmx_pages));
/* prepare the VMCS for L2 execution */
- prepare_vmcs(vmx_pages, l2_guest_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
/* enable TSC offsetting and TSC scaling for L2 */
control = vmreadz(CPU_BASED_VM_EXEC_CONTROL);
diff --git a/tools/testing/selftests/kvm/x86/nested_vmsave_vmload_test.c b/tools/testing/selftests/kvm/x86/nested_vmsave_vmload_test.c
index 85d3f4cc76f3..a130759f39a1 100644
--- a/tools/testing/selftests/kvm/x86/nested_vmsave_vmload_test.c
+++ b/tools/testing/selftests/kvm/x86/nested_vmsave_vmload_test.c
@@ -28,8 +28,6 @@
#define TEST_VMCB_L2_GPA TEST_VMCB_L1_GPA(0)
-#define L2_GUEST_STACK_SIZE 64
-
static void l2_guest_code_vmsave(void)
{
asm volatile("vmsave %0" : : "a"(TEST_VMCB_L2_GPA) : "memory");
@@ -70,10 +68,8 @@ static void l2_guest_code_vmcb1(void)
static void l1_guest_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
/* Each test case initializes the guest RIP below */
- generic_svm_setup(svm, NULL, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, NULL);
/* Set VMSAVE/VMLOAD intercepts and make sure they work with.. */
svm->vmcb->control.intercept |= (BIT_ULL(INTERCEPT_VMSAVE) |
diff --git a/tools/testing/selftests/kvm/x86/pmu_counters_test.c b/tools/testing/selftests/kvm/x86/pmu_counters_test.c
index dc6afac3aa91..c3e784e16348 100644
--- a/tools/testing/selftests/kvm/x86/pmu_counters_test.c
+++ b/tools/testing/selftests/kvm/x86/pmu_counters_test.c
@@ -87,14 +87,18 @@ static struct kvm_intel_pmu_event intel_event_to_feature(u8 idx)
return __intel_event_to_feature[idx];
}
-static struct kvm_vm *pmu_vm_create_with_one_vcpu(struct kvm_vcpu **vcpu,
- void *guest_code,
- u8 pmu_version,
- u64 perf_capabilities)
+static struct kvm_vm *pmu_vm_create_with_vcpus(u32 nr_vcpus, void *guest_code,
+ u8 pmu_version,
+ u64 perf_capabilities,
+ struct kvm_vcpu **__vcpus[])
{
+ struct kvm_vcpu **vcpus = calloc(nr_vcpus, sizeof(*vcpus));
struct kvm_vm *vm;
+ int i;
- vm = vm_create_with_one_vcpu(vcpu, guest_code);
+ *__vcpus = vcpus;
+
+ vm = vm_create_with_vcpus(nr_vcpus, guest_code, vcpus);
sync_global_to_guest(vm, kvm_pmu_version);
sync_global_to_guest(vm, hardware_pmu_arch_events);
@@ -102,13 +106,22 @@ static struct kvm_vm *pmu_vm_create_with_one_vcpu(struct kvm_vcpu **vcpu,
* Set PERF_CAPABILITIES before PMU version as KVM disallows enabling
* features via PERF_CAPABILITIES if the guest doesn't have a vPMU.
*/
- if (kvm_has_perf_caps)
- vcpu_set_msr(*vcpu, MSR_IA32_PERF_CAPABILITIES, perf_capabilities);
+ for (i = 0; i < nr_vcpus; i++) {
+ if (kvm_has_perf_caps)
+ vcpu_set_msr(vcpus[i], MSR_IA32_PERF_CAPABILITIES, perf_capabilities);
+
+ vcpu_set_cpuid_property(vcpus[i], X86_PROPERTY_PMU_VERSION, pmu_version);
+ }
- vcpu_set_cpuid_property(*vcpu, X86_PROPERTY_PMU_VERSION, pmu_version);
return vm;
}
+static void pmu_vm_free(struct kvm_vm *vm, struct kvm_vcpu **vcpus)
+{
+ kvm_vm_free(vm);
+ free(vcpus);
+}
+
static void run_vcpu(struct kvm_vcpu *vcpu)
{
struct ucall uc;
@@ -146,9 +159,9 @@ static u8 guest_get_pmu_version(void)
/*
* If an architectural event is supported and guaranteed to generate at least
- * one "hit, assert that its count is non-zero. If an event isn't supported or
- * the test can't guarantee the associated action will occur, then all bets are
- * off regarding the count, i.e. no checks can be done.
+ * one "hit", assert that its count is non-zero. If an event isn't supported
+ * or the test can't guarantee the associated action will occur, then all bets
+ * are off regarding the count, i.e. no checks can be done.
*
* Sanity check that in all cases, the event doesn't count when it's disabled,
* and that KVM correctly emulates the write of an arbitrary value.
@@ -273,6 +286,7 @@ static void guest_test_arch_event(u8 idx)
struct kvm_x86_pmu_feature gp_event, fixed_event;
u32 base_pmc_msr;
unsigned int i;
+ u64 eventsel;
/* The host side shouldn't invoke this without a guest PMU. */
GUEST_ASSERT(pmu_version);
@@ -287,19 +301,16 @@ static void guest_test_arch_event(u8 idx)
GUEST_ASSERT_EQ(idx, gp_event.f.bit);
GUEST_ASSERT(nr_gp_counters);
+ i = kvm_random_u32_in_range(&kvm_rng, 0, nr_gp_counters - 1);
- for (i = 0; i < nr_gp_counters; i++) {
- u64 eventsel = ARCH_PERFMON_EVENTSEL_OS |
- ARCH_PERFMON_EVENTSEL_ENABLE |
- intel_pmu_arch_events[idx];
+ eventsel = ARCH_PERFMON_EVENTSEL_OS | ARCH_PERFMON_EVENTSEL_ENABLE |
+ intel_pmu_arch_events[idx];
- wrmsr(MSR_P6_EVNTSEL0 + i, 0);
- if (guest_has_perf_global_ctrl)
- wrmsr(MSR_CORE_PERF_GLOBAL_CTRL, BIT_ULL(i));
+ wrmsr(MSR_P6_EVNTSEL0 + i, 0);
+ if (guest_has_perf_global_ctrl)
+ wrmsr(MSR_CORE_PERF_GLOBAL_CTRL, BIT_ULL(i));
- __guest_test_arch_event(idx, i, base_pmc_msr + i,
- MSR_P6_EVNTSEL0 + i, eventsel);
- }
+ __guest_test_arch_event(idx, i, base_pmc_msr + i, MSR_P6_EVNTSEL0 + i, eventsel);
if (!guest_has_perf_global_ctrl)
return;
@@ -328,30 +339,70 @@ static void guest_test_arch_events(void)
GUEST_DONE();
}
-static void test_arch_events(u8 pmu_version, u64 perf_capabilities,
- u8 length, u32 unavailable_mask)
+static void __test_arch_events(struct kvm_vcpu *vcpu, u8 length, u32 unavailable_mask)
{
- struct kvm_vcpu *vcpu;
- struct kvm_vm *vm;
-
- /* Testing arch events requires a vPMU (there are no negative tests). */
- if (!pmu_version)
- return;
-
unavailable_mask &= GENMASK(X86_PROPERTY_PMU_EVENTS_MASK.hi_bit,
X86_PROPERTY_PMU_EVENTS_MASK.lo_bit);
- vm = pmu_vm_create_with_one_vcpu(&vcpu, guest_test_arch_events,
- pmu_version, perf_capabilities);
-
vcpu_set_cpuid_property(vcpu, X86_PROPERTY_PMU_EBX_BIT_VECTOR_LENGTH,
length);
vcpu_set_cpuid_property(vcpu, X86_PROPERTY_PMU_EVENTS_MASK,
unavailable_mask);
run_vcpu(vcpu);
+}
- kvm_vm_free(vm);
+static void test_arch_events(u8 pmu_version, u64 perf_capabilities)
+{
+ struct kvm_vcpu **vcpus;
+ struct kvm_vm *vm;
+ int i = 0;
+ u32 k;
+ u8 j;
+
+ /*
+ * To keep the total runtime reasonable, test only a handful of select,
+ * semi-arbitrary values for the mask of unavailable PMU events. Test
+ * 0 (all events available) and all ones (no events available) as well
+ * as alternating bit sequences, e.g. to detect if KVM is checking the
+ * wrong bit(s).
+ */
+ const u32 unavailable_masks[] = {
+ 0x0,
+ 0xffffffffu,
+ 0xaaaaaaaau,
+ 0x55555555u,
+ 0xf0f0f0f0u,
+ 0x0f0f0f0fu,
+ 0xa0a0a0a0u,
+ 0x0a0a0a0au,
+ 0x50505050u,
+ 0x05050505u,
+ };
+
+ pr_info("Testing arch events, PMU version %u, perf_caps = %lx\n",
+ pmu_version, perf_capabilities);
+
+ /* Testing arch events requires a vPMU (there are no negative tests). */
+ if (!pmu_version)
+ return;
+
+ vm = pmu_vm_create_with_vcpus((NR_INTEL_ARCH_EVENTS + 2) * (ARRAY_SIZE(unavailable_masks) - 1),
+ guest_test_arch_events, pmu_version,
+ perf_capabilities, &vcpus);
+
+ /*
+ * Test single bits for all PMU version and lengths up the number of
+ * events +1 (to verify KVM doesn't do weird things if the guest length
+ * is greater than the host length). Explicitly test a mask of '0' and
+ * all ones i.e. all events being available and unavailable.
+ */
+ for (j = 0; j <= NR_INTEL_ARCH_EVENTS + 1; j++) {
+ for (k = 1; k < ARRAY_SIZE(unavailable_masks); k++)
+ __test_arch_events(vcpus[i++], j, unavailable_masks[k]);
+ }
+
+ pmu_vm_free(vm, vcpus);
}
/*
@@ -495,21 +546,26 @@ static void guest_test_gp_counters(void)
GUEST_DONE();
}
-static void test_gp_counters(u8 pmu_version, u64 perf_capabilities,
- u8 nr_gp_counters)
+static void test_gp_counters(u8 pmu_version, u64 perf_capabilities)
{
- struct kvm_vcpu *vcpu;
+ u8 nr_gp_counters = kvm_cpu_property(X86_PROPERTY_PMU_NR_GP_COUNTERS);
+ struct kvm_vcpu **vcpus;
struct kvm_vm *vm;
+ u8 j;
- vm = pmu_vm_create_with_one_vcpu(&vcpu, guest_test_gp_counters,
- pmu_version, perf_capabilities);
+ pr_info("Testing %u GP counters, PMU version %u, perf_caps = %lx\n",
+ nr_gp_counters, pmu_version, perf_capabilities);
- vcpu_set_cpuid_property(vcpu, X86_PROPERTY_PMU_NR_GP_COUNTERS,
- nr_gp_counters);
+ vm = pmu_vm_create_with_vcpus(nr_gp_counters + 1, guest_test_gp_counters,
+ pmu_version, perf_capabilities, &vcpus);
- run_vcpu(vcpu);
+ for (j = 0; j <= nr_gp_counters; j++) {
+ vcpu_set_cpuid_property(vcpus[j], X86_PROPERTY_PMU_NR_GP_COUNTERS, j);
- kvm_vm_free(vm);
+ run_vcpu(vcpus[j]);
+ }
+
+ pmu_vm_free(vm, vcpus);
}
static void guest_test_fixed_counters(void)
@@ -561,33 +617,47 @@ static void guest_test_fixed_counters(void)
GUEST_DONE();
}
-static void test_fixed_counters(u8 pmu_version, u64 perf_capabilities,
- u8 nr_fixed_counters, u32 supported_bitmask)
+static void __test_fixed_counters(struct kvm_vcpu *vcpu, u8 nr_fixed_counters,
+ u32 supported_bitmask)
{
- struct kvm_vcpu *vcpu;
- struct kvm_vm *vm;
-
- vm = pmu_vm_create_with_one_vcpu(&vcpu, guest_test_fixed_counters,
- pmu_version, perf_capabilities);
-
vcpu_set_cpuid_property(vcpu, X86_PROPERTY_PMU_FIXED_COUNTERS_BITMASK,
supported_bitmask);
vcpu_set_cpuid_property(vcpu, X86_PROPERTY_PMU_NR_FIXED_COUNTERS,
nr_fixed_counters);
run_vcpu(vcpu);
+}
- kvm_vm_free(vm);
+static void test_fixed_counters(u8 pmu_version, u64 perf_capabilities)
+{
+ u8 nr_fixed_counters = kvm_cpu_property(X86_PROPERTY_PMU_NR_FIXED_COUNTERS);
+ struct kvm_vcpu **vcpus;
+ struct kvm_vm *vm;
+ int i = 0;
+ u32 k;
+ u8 j;
+
+ pr_info("Testing %u fixed counters, PMU version %u, perf_caps = %lx\n",
+ nr_fixed_counters, pmu_version, perf_capabilities);
+
+
+ vm = pmu_vm_create_with_vcpus((nr_fixed_counters + 1) * BIT(nr_fixed_counters),
+ guest_test_fixed_counters,
+ pmu_version, perf_capabilities, &vcpus);
+
+ for (j = 0; j <= nr_fixed_counters; j++) {
+ for (k = 0; k <= (BIT(nr_fixed_counters) - 1); k++)
+ __test_fixed_counters(vcpus[i++], j, k);
+ }
+
+ pmu_vm_free(vm, vcpus);
}
static void test_intel_counters(void)
{
- u8 nr_fixed_counters = kvm_cpu_property(X86_PROPERTY_PMU_NR_FIXED_COUNTERS);
- u8 nr_gp_counters = kvm_cpu_property(X86_PROPERTY_PMU_NR_GP_COUNTERS);
u8 pmu_version = kvm_cpu_property(X86_PROPERTY_PMU_VERSION);
unsigned int i;
- u8 v, j;
- u32 k;
+ u8 v;
const u64 perf_caps[] = {
0,
@@ -595,26 +665,6 @@ static void test_intel_counters(void)
};
/*
- * To keep the total runtime reasonable, test only a handful of select,
- * semi-arbitrary values for the mask of unavailable PMU events. Test
- * 0 (all events available) and all ones (no events available) as well
- * as alternating bit sequencues, e.g. to detect if KVM is checking the
- * wrong bit(s).
- */
- const u32 unavailable_masks[] = {
- 0x0,
- 0xffffffffu,
- 0xaaaaaaaau,
- 0x55555555u,
- 0xf0f0f0f0u,
- 0x0f0f0f0fu,
- 0xa0a0a0a0u,
- 0x0a0a0a0au,
- 0x50505050u,
- 0x05050505u,
- };
-
- /*
* Test up to PMU v5, which is the current maximum version defined by
* Intel, i.e. is the last version that is guaranteed to be backwards
* compatible with KVM's existing behavior.
@@ -649,32 +699,9 @@ static void test_intel_counters(void)
if (!kvm_has_perf_caps && perf_caps[i])
continue;
- pr_info("Testing arch events, PMU version %u, perf_caps = %lx\n",
- v, perf_caps[i]);
-
- /*
- * Test single bits for all PMU version and lengths up
- * the number of events +1 (to verify KVM doesn't do
- * weird things if the guest length is greater than the
- * host length). Explicitly test a mask of '0' and all
- * ones i.e. all events being available and unavailable.
- */
- for (j = 0; j <= NR_INTEL_ARCH_EVENTS + 1; j++) {
- for (k = 1; k < ARRAY_SIZE(unavailable_masks); k++)
- test_arch_events(v, perf_caps[i], j, unavailable_masks[k]);
- }
-
- pr_info("Testing GP counters, PMU version %u, perf_caps = %lx\n",
- v, perf_caps[i]);
- for (j = 0; j <= nr_gp_counters; j++)
- test_gp_counters(v, perf_caps[i], j);
-
- pr_info("Testing fixed counters, PMU version %u, perf_caps = %lx\n",
- v, perf_caps[i]);
- for (j = 0; j <= nr_fixed_counters; j++) {
- for (k = 0; k <= (BIT(nr_fixed_counters) - 1); k++)
- test_fixed_counters(v, perf_caps[i], j, k);
- }
+ test_arch_events(v, perf_caps[i]);
+ test_gp_counters(v, perf_caps[i]);
+ test_fixed_counters(v, perf_caps[i]);
}
}
}
diff --git a/tools/testing/selftests/kvm/x86/pmu_event_filter_test.c b/tools/testing/selftests/kvm/x86/pmu_event_filter_test.c
index 84e4c6ca67a3..2d401ff09112 100644
--- a/tools/testing/selftests/kvm/x86/pmu_event_filter_test.c
+++ b/tools/testing/selftests/kvm/x86/pmu_event_filter_test.c
@@ -62,7 +62,7 @@ struct {
/*
* If we encounter a #GP during the guest PMU sanity check, then the guest
- * PMU is not functional. Inform the hypervisor via GUEST_SYNC(0).
+ * PMU is not functional. Inform the hypervisor via GUEST_SYNC(-EFAULT).
*/
static void guest_gp_handler(struct ex_regs *regs)
{
@@ -73,7 +73,7 @@ static void guest_gp_handler(struct ex_regs *regs)
* Check that we can write a new value to the given MSR and read it back.
* The caller should provide a non-empty set of bits that are safe to flip.
*
- * Return on success. GUEST_SYNC(0) on error.
+ * Return on success, GUEST_SYNC(-EIO) on error.
*/
static void check_msr(u32 msr, u64 bits_to_flip)
{
diff --git a/tools/testing/selftests/kvm/x86/private_mem_conversions_test.c b/tools/testing/selftests/kvm/x86/private_mem_conversions_test.c
index 1d2f5d4fd45d..3aa0673f79f3 100644
--- a/tools/testing/selftests/kvm/x86/private_mem_conversions_test.c
+++ b/tools/testing/selftests/kvm/x86/private_mem_conversions_test.c
@@ -412,13 +412,13 @@ static void test_mem_conversions(enum vm_mem_backing_src_type src_type, u32 nr_v
*/
virt_map(vm, gpa, gpa, PER_CPU_DATA_SIZE / vm->page_size);
- pthread_create(&threads[i], NULL, __test_mem_conversions, vcpus[i]);
+ kvm_pthread_create(&threads[i], NULL, __test_mem_conversions, vcpus[i]);
}
WRITE_ONCE(run_vcpus, true);
for (i = 0; i < nr_vcpus; i++)
- pthread_join(threads[i], NULL);
+ kvm_pthread_join(threads[i], NULL);
kvm_vm_free(vm);
diff --git a/tools/testing/selftests/kvm/x86/private_mem_kvm_exits_test.c b/tools/testing/selftests/kvm/x86/private_mem_kvm_exits_test.c
index 10db9fe6d906..e6d16e9d6120 100644
--- a/tools/testing/selftests/kvm/x86/private_mem_kvm_exits_test.c
+++ b/tools/testing/selftests/kvm/x86/private_mem_kvm_exits_test.c
@@ -65,13 +65,12 @@ static void test_private_access_memslot_deleted(void)
/* Request to access page privately */
vm_mem_set_private(vm, EXITS_TEST_GPA, EXITS_TEST_SIZE);
- pthread_create(&vm_thread, NULL,
- (void *(*)(void *))run_vcpu_get_exit_reason,
- (void *)vcpu);
+ kvm_pthread_create(&vm_thread, NULL,
+ (pthread_fn_t)run_vcpu_get_exit_reason, (void *)vcpu);
vm_mem_region_delete(vm, EXITS_TEST_SLOT);
- pthread_join(vm_thread, &thread_return);
+ kvm_pthread_join(vm_thread, &thread_return);
exit_reason = (u32)(u64)thread_return;
TEST_ASSERT_EQ(exit_reason, KVM_EXIT_MEMORY_FAULT);
diff --git a/tools/testing/selftests/kvm/x86/recalc_apic_map_test.c b/tools/testing/selftests/kvm/x86/recalc_apic_map_test.c
index cbc92a862ea9..6252a96320ee 100644
--- a/tools/testing/selftests/kvm/x86/recalc_apic_map_test.c
+++ b/tools/testing/selftests/kvm/x86/recalc_apic_map_test.c
@@ -57,7 +57,7 @@ int main(void)
for (i = 0; i < KVM_MAX_VCPUS; i++)
vcpu_set_msr(vcpus[i], MSR_IA32_APICBASE, LAPIC_X2APIC);
- TEST_ASSERT_EQ(pthread_create(&thread, NULL, race, vcpus[0]), 0);
+ kvm_pthread_create(&thread, NULL, race, vcpus[0]);
vcpuN = vcpus[KVM_MAX_VCPUS - 1];
for (t = time(NULL) + TIMEOUT; time(NULL) < t;) {
@@ -65,8 +65,7 @@ int main(void)
vcpu_set_msr(vcpuN, MSR_IA32_APICBASE, LAPIC_DISABLED);
}
- TEST_ASSERT_EQ(pthread_cancel(thread), 0);
- TEST_ASSERT_EQ(pthread_join(thread, NULL), 0);
+ kvm_pthread_cancel_join(thread);
kvm_vm_free(vm);
diff --git a/tools/testing/selftests/kvm/x86/save_restore_pf_stress_test.c b/tools/testing/selftests/kvm/x86/save_restore_pf_stress_test.c
new file mode 100644
index 000000000000..507391ab2c93
--- /dev/null
+++ b/tools/testing/selftests/kvm/x86/save_restore_pf_stress_test.c
@@ -0,0 +1,288 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <errno.h>
+#include <sys/types.h>
+#include <time.h>
+#include <pthread.h>
+#include <signal.h>
+#include <unistd.h>
+#include <getopt.h>
+
+#include "test_util.h"
+#include "kvm_util.h"
+#include "processor.h"
+#include "svm_util.h"
+#include "vmx.h"
+
+#define NR_ITERATIONS 500
+
+#define PTRS_PER_PTE 512
+#define PXD_INDEX(vaddr, level) (((vaddr) >> PG_LEVEL_SHIFT(level)) & (PTRS_PER_PTE - 1))
+
+#define TEST_MEM_BASE_GVA 0xc0000000ULL
+#define TEST_PGTABLE_GVA_OFFSET 0xd0000000ULL
+#define PATTERN 0xabcdefabcdefabcdULL
+
+static u64 expected_vaddr;
+static u64 guest_faults;
+
+static u64 *guest_get_pte(u64 vaddr)
+{
+ u64 pgtable_pa, pte;
+ u64 *pgtable;
+ int level;
+
+ level = (get_cr4() & X86_CR4_LA57) ? PG_LEVEL_256T : PG_LEVEL_512G;
+
+ pgtable_pa = get_cr3() & PHYSICAL_PAGE_MASK;
+ for (; level > PG_LEVEL_4K; level--) {
+ pgtable = (u64 *)(pgtable_pa + TEST_PGTABLE_GVA_OFFSET);
+ pte = pgtable[PXD_INDEX(vaddr, level)];
+ GUEST_ASSERT(pte & PTE_PRESENT_MASK(&guest_mmu));
+ GUEST_ASSERT(!(pte & PTE_HUGE_MASK(&guest_mmu)));
+ pgtable_pa = PTE_GET_PA(pte);
+ }
+
+ pgtable = (u64 *)(pgtable_pa + TEST_PGTABLE_GVA_OFFSET);
+ return &pgtable[PXD_INDEX(vaddr, PG_LEVEL_4K)];
+}
+
+static void guest_pf_handler(struct ex_regs *regs)
+{
+ u64 fault_addr;
+ u64 *ptep;
+
+ fault_addr = get_cr2();
+ GUEST_ASSERT_EQ(fault_addr, READ_ONCE(expected_vaddr));
+
+ ptep = guest_get_pte(fault_addr);
+ GUEST_ASSERT(ptep);
+ GUEST_ASSERT(!(*ptep & PTE_PRESENT_MASK(&guest_mmu)));
+
+ *ptep |= PTE_PRESENT_MASK(&guest_mmu);
+ guest_faults++;
+}
+
+static void guest_access_memory(void *arg)
+{
+ u64 vaddr, val;
+ int i;
+
+ for (i = 0; ; i++) {
+ vaddr = TEST_MEM_BASE_GVA + (i % PTRS_PER_PTE) * PAGE_SIZE;
+ WRITE_ONCE(expected_vaddr, vaddr);
+
+ /* Read to trigger #PF */
+ val = READ_ONCE(*(u64 *)vaddr);
+ GUEST_ASSERT_EQ(val, PATTERN);
+
+ /* Clear the present bit again so it faults next time */
+ *guest_get_pte(vaddr) &= ~PTE_PRESENT_MASK(&guest_mmu);
+ invlpg(vaddr);
+ }
+}
+
+static void l1_svm_code(struct svm_test_data *svm)
+{
+ generic_svm_setup(svm, guest_access_memory);
+ svm->vmcb->control.intercept_exceptions |= BIT(UD_VECTOR);
+
+ while (1) {
+ run_guest(svm->vmcb, svm->vmcb_gpa);
+ GUEST_ASSERT_EQ(svm->vmcb->control.exit_code,
+ (SVM_EXIT_EXCP_BASE + UD_VECTOR));
+ }
+}
+
+static void l1_vmx_code(struct vmx_pages *vmx)
+{
+ GUEST_ASSERT(prepare_for_vmx_operation(vmx));
+ GUEST_ASSERT(load_vmcs(vmx));
+ prepare_vmcs(vmx, guest_access_memory);
+
+ GUEST_ASSERT(!vmwrite(EXCEPTION_BITMAP, BIT(UD_VECTOR)));
+
+ GUEST_ASSERT(!vmlaunch());
+ while (1) {
+ GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_EXCEPTION_NMI);
+ GUEST_ASSERT_EQ(vmreadz(VM_EXIT_INTR_INFO) & 0xff, UD_VECTOR);
+ GUEST_ASSERT(!vmresume());
+ }
+}
+
+static void l1_guest_code(void *test_data)
+{
+ if (this_cpu_has(X86_FEATURE_SVM))
+ l1_svm_code(test_data);
+ else
+ l1_vmx_code(test_data);
+}
+
+static void *sigusr_thread_fn(void *arg)
+{
+ pthread_t vcpu_thread = (pthread_t)arg;
+
+ for (;;) {
+ pthread_testcancel();
+ pthread_kill(vcpu_thread, SIGUSR1);
+ usleep(msecs_to_usecs(1));
+ }
+ return NULL;
+}
+
+static void dummy_signal_handler(int signo) {}
+static struct sigaction sa;
+
+static void vcpu_sigusr_listen(void)
+{
+ sa.sa_handler = dummy_signal_handler;
+ sigaction(SIGUSR1, &sa, NULL);
+}
+
+static void vcpu_sigusr_ignore(void)
+{
+ sa.sa_handler = SIG_IGN;
+ sigaction(SIGUSR1, &sa, NULL);
+}
+
+static void kvm_x86_state_queue_ud(struct kvm_x86_state *state)
+{
+ if (state->events.exception.pending || state->events.exception.injected)
+ return;
+
+ state->events.flags |= KVM_VCPUEVENT_VALID_PAYLOAD;
+ state->events.exception.pending = true;
+ state->events.exception.injected = false;
+ state->events.exception.nr = UD_VECTOR;
+ state->events.exception.has_error_code = false;
+ state->events.exception_has_payload = false;
+}
+
+static void run_test(bool nested)
+{
+ struct kvm_x86_state *state;
+ int r, i, level;
+ pthread_t sigusr_thread;
+ gpa_t gpa, pgtable_gpa;
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ struct ucall uc;
+ u64 *pgtable;
+ gva_t gva;
+ u64 pte;
+
+ vm = vm_create_with_one_vcpu(&vcpu, nested ? l1_guest_code : guest_access_memory);
+ vm_install_exception_handler(vm, PF_VECTOR, guest_pf_handler);
+
+ if (nested) {
+ vm_enable_cap(vm, KVM_CAP_EXCEPTION_PAYLOAD, -2ul);
+ if (kvm_cpu_has(X86_FEATURE_SVM))
+ vcpu_alloc_svm(vm, &gva);
+ else
+ vcpu_alloc_vmx(vm, &gva);
+ vcpu_args_set(vcpu, 1, gva);
+ }
+
+ /* Allocate a page and write the pattern to it */
+ gva = vm_alloc_page(vm);
+ *(u64 *)addr_gva2hva(vm, gva) = PATTERN;
+ gpa = addr_gva2gpa(vm, gva);
+
+ /*
+ * Map all virtual addresses to the pattern page and clear the present
+ * bit such that guest accesses will cause a #PF.
+ */
+ for (i = 0; i < PTRS_PER_PTE; i++) {
+ gva = TEST_MEM_BASE_GVA + i * getpagesize();
+ virt_pg_map(vm, gva, gpa);
+ *vm_get_pte(vm, gva) &= ~PTE_PRESENT_MASK(&vm->mmu);
+ }
+
+ /*
+ * Now create mappings for the page tables created above so that the
+ * guest #PF handler can walk them. All PTEs for test virtual addresses
+ * should lie on the same PTE page, so one page is mapped for each page
+ * table level.
+ *
+ * Use an offset for the GVA instead of creating identity mappings to
+ * avoid collision with existing mappings at low GVAs (e.g. ELF).
+ */
+ pgtable_gpa = vm->mmu.pgd;
+ for (level = vm->mmu.pgtable_levels; level >= PG_LEVEL_4K; level--) {
+ virt_map(vm, pgtable_gpa + TEST_PGTABLE_GVA_OFFSET, pgtable_gpa, 1);
+ pgtable = addr_gpa2hva(vm, pgtable_gpa);
+ pte = pgtable[PXD_INDEX(TEST_MEM_BASE_GVA, level)];
+ pgtable_gpa = PTE_GET_PA(pte);
+ }
+
+ /* Initialize the thread sending SIGUSR and install the handler */
+ vcpu_sigusr_ignore();
+ r = pthread_create(&sigusr_thread, NULL, sigusr_thread_fn,
+ (void *)pthread_self());
+ TEST_ASSERT(!r, "pthread_create() failed: %d", r);
+
+ for (i = 1; i <= NR_ITERATIONS; i++) {
+ /*
+ * Only handle SIGUSR while the vCPU is running, otherwise
+ * ignore it to avoid interrupting other ioctls/syscalls.
+ */
+ vcpu_sigusr_listen();
+ r = __vcpu_run(vcpu);
+ TEST_ASSERT(!r || errno == EINTR, "Expected success or SIGUSR1");
+ vcpu_sigusr_ignore();
+
+ /* The guest only exits due to a signal or failed assertion */
+ if (!r) {
+ TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_IO);
+ TEST_ASSERT_EQ(get_ucall(vcpu, &uc), UCALL_ABORT);
+ REPORT_GUEST_ASSERT(uc);
+ break;
+ }
+
+ state = vcpu_save_state(vcpu);
+
+ /*
+ * If the vCPU is in guest mode, inject a #UD to trigger an
+ * L2->L1 VM-Exit every other iteration.
+ */
+ if (kvm_x86_state_is_guest_mode(state) && i % 2 == 0)
+ kvm_x86_state_queue_ud(state);
+
+ kvm_vm_release(vm);
+ vcpu = vm_recreate_with_one_vcpu(vm);
+ if (nested)
+ vm_enable_cap(vm, KVM_CAP_EXCEPTION_PAYLOAD, -2ul);
+ vcpu_load_state(vcpu, state);
+ kvm_x86_state_cleanup(state);
+
+ pr_info("\rSave+restore iterations: %d", i);
+ }
+ pr_info("\n");
+
+ sync_global_from_guest(vm, guest_faults);
+ TEST_ASSERT(guest_faults, "No guest page faults triggered");
+ pr_info("Guest page faults%s: %lu\n", nested ? " (in L2)" : "", guest_faults);
+
+ pthread_cancel(sigusr_thread);
+ pthread_join(sigusr_thread, NULL);
+ kvm_vm_free(vm);
+}
+
+int main(int argc, char *argv[])
+{
+ pr_info("Running save+restore stress test...\n");
+ run_test(/*nested=*/false);
+
+ if (!kvm_has_cap(KVM_CAP_EXCEPTION_PAYLOAD) ||
+ !kvm_has_cap(KVM_CAP_NESTED_STATE) ||
+ (!kvm_cpu_has(X86_FEATURE_SVM) && !kvm_cpu_has(X86_FEATURE_VMX))) {
+ pr_info("Nested virtualization not supported, skipping nested test\n");
+ return 0;
+ }
+
+ pr_info("Running save+restore stress test with a nested guest...\n");
+ run_test(/*nested=*/true);
+ return 0;
+}
diff --git a/tools/testing/selftests/kvm/x86/set_sregs_test.c b/tools/testing/selftests/kvm/x86/set_sregs_test.c
index 8e654cc9ab16..603226ffe437 100644
--- a/tools/testing/selftests/kvm/x86/set_sregs_test.c
+++ b/tools/testing/selftests/kvm/x86/set_sregs_test.c
@@ -21,20 +21,20 @@
#include "kvm_util.h"
#include "processor.h"
-#define TEST_INVALID_CR_BIT(vcpu, cr, orig, bit) \
+#define TEST_INVALID_SREG_BIT(vcpu, reg, orig, bit) \
do { \
struct kvm_sregs new; \
int rc; \
\
/* Skip the sub-test, the feature/bit is supported. */ \
- if (orig.cr & bit) \
+ if (orig.reg & bit) \
break; \
\
- memcpy(&new, &orig, sizeof(sregs)); \
- new.cr |= bit; \
+ memcpy(&new, &orig, sizeof(new)); \
+ new.reg |= bit; \
\
rc = _vcpu_sregs_set(vcpu, &new); \
- TEST_ASSERT(rc, "KVM allowed invalid " #cr " bit (0x%lx)", bit); \
+ TEST_ASSERT(rc, "KVM allowed invalid " #reg " bit (0x%lx)", (u64)bit); \
\
/* Sanity check that KVM didn't change anything. */ \
vcpu_sregs_get(vcpu, &new); \
@@ -46,6 +46,8 @@ do { \
X86_CR4_MCE | X86_CR4_PGE | X86_CR4_PCE | \
X86_CR4_OSFXSR | X86_CR4_OSXMMEXCPT)
+#define KVM_ALWAYS_ALLOWED_EFER EFER_SCE
+
static u64 calc_supported_cr4_feature_bits(void)
{
u64 cr4 = KVM_ALWAYS_ALLOWED_CR4;
@@ -74,6 +76,24 @@ static u64 calc_supported_cr4_feature_bits(void)
return cr4;
}
+static u64 calc_supported_efer_feature_bits(void)
+{
+ u64 efer = KVM_ALWAYS_ALLOWED_EFER;
+
+ if (kvm_cpu_has(X86_FEATURE_LM))
+ efer |= (EFER_LME | EFER_LMA);
+ if (kvm_cpu_has(X86_FEATURE_NX))
+ efer |= EFER_NX;
+ if (kvm_cpu_has(X86_FEATURE_SVM))
+ efer |= EFER_SVME;
+ if (kvm_cpu_has(X86_FEATURE_FXSR_OPT))
+ efer |= EFER_FFXSR;
+ if (kvm_cpu_has(X86_FEATURE_AUTOIBRS))
+ efer |= EFER_AUTOIBRS;
+
+ return efer;
+}
+
static void test_cr_bits(struct kvm_vcpu *vcpu, u64 cr4)
{
struct kvm_sregs sregs;
@@ -96,26 +116,45 @@ static void test_cr_bits(struct kvm_vcpu *vcpu, u64 cr4)
(sregs.cr4 & X86_CR4_PKE) ? "set" : "clear");
vcpu_sregs_get(vcpu, &sregs);
- TEST_ASSERT(sregs.cr4 == cr4, "sregs.CR4 (0x%llx) != CR4 (0x%lx)",
- sregs.cr4, cr4);
-
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_UMIP);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_LA57);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_VMXE);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_SMXE);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_FSGSBASE);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_PCIDE);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_OSXSAVE);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_SMEP);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_SMAP);
- TEST_INVALID_CR_BIT(vcpu, cr4, sregs, X86_CR4_PKE);
+ TEST_ASSERT_EQ(sregs.cr4, cr4);
+
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_UMIP);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_LA57);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_VMXE);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_SMXE);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_FSGSBASE);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_PCIDE);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_OSXSAVE);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_SMEP);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_SMAP);
+ TEST_INVALID_SREG_BIT(vcpu, cr4, sregs, X86_CR4_PKE);
for (i = 32; i < 64; i++)
- TEST_INVALID_CR_BIT(vcpu, cr0, sregs, BIT(i));
+ TEST_INVALID_SREG_BIT(vcpu, cr0, sregs, BIT(i));
/* NW without CD is illegal, as is PG without PE. */
- TEST_INVALID_CR_BIT(vcpu, cr0, sregs, X86_CR0_NW);
- TEST_INVALID_CR_BIT(vcpu, cr0, sregs, X86_CR0_PG);
+ TEST_INVALID_SREG_BIT(vcpu, cr0, sregs, X86_CR0_NW);
+ TEST_INVALID_SREG_BIT(vcpu, cr0, sregs, X86_CR0_PG);
+}
+
+static void test_efer_bits(struct kvm_vcpu *vcpu, u64 efer)
+{
+ struct kvm_sregs sregs;
+ int rc;
+
+ vcpu_sregs_get(vcpu, &sregs);
+ sregs.efer |= efer;
+ rc = _vcpu_sregs_set(vcpu, &sregs);
+ TEST_ASSERT(!rc, "Failed to set supported EFER bits (0x%llx)", sregs.efer);
+
+ vcpu_sregs_get(vcpu, &sregs);
+ TEST_ASSERT_EQ(sregs.efer, efer);
+
+ TEST_INVALID_SREG_BIT(vcpu, efer, sregs, EFER_LME);
+ TEST_INVALID_SREG_BIT(vcpu, efer, sregs, EFER_NX);
+ TEST_INVALID_SREG_BIT(vcpu, efer, sregs, EFER_SVME);
+ TEST_INVALID_SREG_BIT(vcpu, efer, sregs, EFER_FFXSR);
+ TEST_INVALID_SREG_BIT(vcpu, efer, sregs, EFER_AUTOIBRS);
}
int main(int argc, char *argv[])
@@ -132,6 +171,7 @@ int main(int argc, char *argv[])
*/
vm = vm_create_barebones();
vcpu = __vm_vcpu_add(vm, 0);
+ test_efer_bits(vcpu, KVM_ALWAYS_ALLOWED_EFER);
test_cr_bits(vcpu, KVM_ALWAYS_ALLOWED_CR4);
kvm_vm_free(vm);
@@ -151,6 +191,7 @@ int main(int argc, char *argv[])
sregs.apic_base);
test_cr_bits(vcpu, calc_supported_cr4_feature_bits());
+ test_efer_bits(vcpu, calc_supported_efer_feature_bits());
kvm_vm_free(vm);
diff --git a/tools/testing/selftests/kvm/x86/sev_dbg_test.c b/tools/testing/selftests/kvm/x86/sev_dbg_test.c
index a9d8e4c059f9..eaa8201b937d 100644
--- a/tools/testing/selftests/kvm/x86/sev_dbg_test.c
+++ b/tools/testing/selftests/kvm/x86/sev_dbg_test.c
@@ -34,7 +34,7 @@ static void validate_buffers(void)
static void ____test_sev_dbg(struct kvm_vm *vm, int i, int j, int nr_bytes)
{
- u8 pattern = guest_random_u32(&guest_rng);
+ u8 pattern = kvm_random_u32(&kvm_rng);
if (i + nr_bytes > BUFFER_SIZE || j + nr_bytes > BUFFER_SIZE)
return;
diff --git a/tools/testing/selftests/kvm/x86/sev_init2_tests.c b/tools/testing/selftests/kvm/x86/sev_init2_tests.c
index 689390c10f7c..87bff4fbb7ed 100644
--- a/tools/testing/selftests/kvm/x86/sev_init2_tests.c
+++ b/tools/testing/selftests/kvm/x86/sev_init2_tests.c
@@ -77,10 +77,6 @@ void test_vm_types(void)
{
test_init2(KVM_X86_SEV_VM, &(struct kvm_sev_init){});
- /*
- * TODO: check that unsupported types cannot be created. Probably
- * a separate selftest.
- */
if (have_sev_es)
test_init2(KVM_X86_SEV_ES_VM, &(struct kvm_sev_init){});
diff --git a/tools/testing/selftests/kvm/x86/sev_migrate_tests.c b/tools/testing/selftests/kvm/x86/sev_migrate_tests.c
index d59abb198d86..de006e1c861c 100644
--- a/tools/testing/selftests/kvm/x86/sev_migrate_tests.c
+++ b/tools/testing/selftests/kvm/x86/sev_migrate_tests.c
@@ -128,10 +128,10 @@ static void test_sev_migrate_locking(void)
sizeof(input[i].source_vms));
for (i = 0; i < NR_LOCK_TESTING_THREADS; ++i)
- pthread_create(&pt[i], NULL, locking_test_thread, &input[i]);
+ kvm_pthread_create(&pt[i], NULL, locking_test_thread, &input[i]);
for (i = 0; i < NR_LOCK_TESTING_THREADS; ++i)
- pthread_join(pt[i], NULL);
+ kvm_pthread_join(pt[i], NULL);
for (i = 0; i < NR_LOCK_TESTING_THREADS; ++i)
kvm_vm_free(input[i].vm);
}
diff --git a/tools/testing/selftests/kvm/x86/smm_test.c b/tools/testing/selftests/kvm/x86/smm_test.c
index 740051167dbd..d1edafd5af75 100644
--- a/tools/testing/selftests/kvm/x86/smm_test.c
+++ b/tools/testing/selftests/kvm/x86/smm_test.c
@@ -22,9 +22,6 @@
#define SMRAM_GPA 0x1000000
#define SMRAM_STAGE 0xfe
-#define STR(x) #x
-#define XSTR(s) STR(s)
-
#define SYNC_PORT 0xe
#define DONE 0xff
@@ -42,7 +39,7 @@ u8 smi_handler[] = {
static inline void sync_with_host(u64 phase)
{
- asm volatile("in $" XSTR(SYNC_PORT)", %%al \n"
+ asm volatile("in $" __stringify(SYNC_PORT)", %%al \n"
: "+a" (phase));
}
@@ -63,8 +60,6 @@ static void l2_guest_code(void)
static void guest_code(void *arg)
{
- #define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u64 apicbase = rdmsr(MSR_IA32_APICBASE);
struct svm_test_data *svm = arg;
struct vmx_pages *vmx_pages = arg;
@@ -81,13 +76,11 @@ static void guest_code(void *arg)
if (arg) {
if (this_cpu_has(X86_FEATURE_SVM)) {
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
} else {
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
}
sync_with_host(5);
diff --git a/tools/testing/selftests/kvm/x86/state_test.c b/tools/testing/selftests/kvm/x86/state_test.c
index 409c6cc9f921..4a1056a6cb8d 100644
--- a/tools/testing/selftests/kvm/x86/state_test.c
+++ b/tools/testing/selftests/kvm/x86/state_test.c
@@ -19,8 +19,6 @@
#include "vmx.h"
#include "svm_util.h"
-#define L2_GUEST_STACK_SIZE 256
-
void svm_l2_guest_code(void)
{
GUEST_SYNC(4);
@@ -35,13 +33,11 @@ void svm_l2_guest_code(void)
static void svm_l1_guest_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
GUEST_ASSERT(svm->vmcb_gpa);
/* Prepare for L2 execution. */
- generic_svm_setup(svm, svm_l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, svm_l2_guest_code);
vmcb->control.int_ctl |= (V_GIF_ENABLE_MASK | V_GIF_MASK);
@@ -78,8 +74,6 @@ void vmx_l2_guest_code(void)
static void vmx_l1_guest_code(struct vmx_pages *vmx_pages)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
GUEST_ASSERT(vmx_pages->vmcs_gpa);
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_SYNC(3);
@@ -89,8 +83,7 @@ static void vmx_l1_guest_code(struct vmx_pages *vmx_pages)
GUEST_SYNC(4);
GUEST_ASSERT(vmptrstz() == vmx_pages->vmcs_gpa);
- prepare_vmcs(vmx_pages, vmx_l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, vmx_l2_guest_code);
GUEST_SYNC(5);
GUEST_ASSERT(vmptrstz() == vmx_pages->vmcs_gpa);
diff --git a/tools/testing/selftests/kvm/x86/svm_int_ctl_test.c b/tools/testing/selftests/kvm/x86/svm_int_ctl_test.c
index d3cc5e4f7883..7b1f4a4818bd 100644
--- a/tools/testing/selftests/kvm/x86/svm_int_ctl_test.c
+++ b/tools/testing/selftests/kvm/x86/svm_int_ctl_test.c
@@ -54,15 +54,12 @@ static void l2_guest_code(struct svm_test_data *svm)
static void l1_guest_code(struct svm_test_data *svm)
{
- #define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
x2apic_enable();
/* Prepare for L2 execution. */
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
/* No virtual interrupt masking */
vmcb->control.int_ctl &= ~V_INTR_MASKING_MASK;
diff --git a/tools/testing/selftests/kvm/x86/svm_lbr_nested_state.c b/tools/testing/selftests/kvm/x86/svm_lbr_nested_state.c
index 7fbfaa054c95..77c6ce9f4507 100644
--- a/tools/testing/selftests/kvm/x86/svm_lbr_nested_state.c
+++ b/tools/testing/selftests/kvm/x86/svm_lbr_nested_state.c
@@ -9,8 +9,6 @@
#include "svm_util.h"
-#define L2_GUEST_STACK_SIZE 64
-
#define DO_BRANCH() do { asm volatile("jmp 1f\n 1: nop"); } while (0)
struct lbr_branch {
@@ -55,7 +53,6 @@ static void l2_guest_code(struct svm_test_data *svm)
static void l1_guest_code(struct svm_test_data *svm, bool nested_lbrv)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
struct lbr_branch l1_branch;
@@ -65,8 +62,7 @@ static void l1_guest_code(struct svm_test_data *svm, bool nested_lbrv)
CHECK_BRANCH_MSRS(&l1_branch);
/* Run L2, which will also do the same */
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
if (nested_lbrv)
vmcb->control.misc_ctl2 = SVM_MISC2_ENABLE_V_LBR;
diff --git a/tools/testing/selftests/kvm/x86/svm_nested_clear_efer_svme.c b/tools/testing/selftests/kvm/x86/svm_nested_clear_efer_svme.c
index 6a89eaffc657..6bc301207cbc 100644
--- a/tools/testing/selftests/kvm/x86/svm_nested_clear_efer_svme.c
+++ b/tools/testing/selftests/kvm/x86/svm_nested_clear_efer_svme.c
@@ -8,8 +8,6 @@
#include "kselftest.h"
-#define L2_GUEST_STACK_SIZE 64
-
static void l2_guest_code(void)
{
unsigned long efer = rdmsr(MSR_EFER);
@@ -24,10 +22,7 @@ static void l2_guest_code(void)
static void l1_guest_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
run_guest(svm->vmcb, svm->vmcb_gpa);
/* Unreachable, L1 should be shutdown */
diff --git a/tools/testing/selftests/kvm/x86/svm_nested_pat_test.c b/tools/testing/selftests/kvm/x86/svm_nested_pat_test.c
index 92da8ff34da1..14ec9d6ad195 100644
--- a/tools/testing/selftests/kvm/x86/svm_nested_pat_test.c
+++ b/tools/testing/selftests/kvm/x86/svm_nested_pat_test.c
@@ -25,8 +25,6 @@
#include "processor.h"
#include "svm_util.h"
-#define L2_GUEST_STACK_SIZE 256
-
#define PAT_DEFAULT 0x0007040600070406ULL
#define L1_PAT_VALUE 0x0007040600070404ULL /* Change PA0 to WT */
#define L2_VMCB12_PAT 0x0606060606060606ULL /* All WB */
@@ -59,14 +57,13 @@ static void l2_guest_code(void)
static void l1_guest_code(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
int i;
wrmsr(MSR_IA32_CR_PAT, L1_PAT_VALUE);
GUEST_ASSERT_EQ(rdmsr(MSR_IA32_CR_PAT), L1_PAT_VALUE);
- generic_svm_setup(svm, l2_guest_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
vmcb->save.g_pat = L2_VMCB12_PAT;
vmcb->control.intercept &= ~(1ULL << INTERCEPT_MSR_PROT);
@@ -94,11 +91,10 @@ static void l1_guest_code(struct svm_test_data *svm)
static void l1_guest_code_invalid_gpat(struct svm_test_data *svm)
{
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
/* VMRUN should fail without running L2 */
- generic_svm_setup(svm, NULL, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, NULL);
vmcb->save.g_pat = INVALID_PAT_VALUE;
run_guest(vmcb, svm->vmcb_gpa);
diff --git a/tools/testing/selftests/kvm/x86/svm_nested_shutdown_test.c b/tools/testing/selftests/kvm/x86/svm_nested_shutdown_test.c
index c6ea3d609a62..2a4a216954bb 100644
--- a/tools/testing/selftests/kvm/x86/svm_nested_shutdown_test.c
+++ b/tools/testing/selftests/kvm/x86/svm_nested_shutdown_test.c
@@ -19,12 +19,9 @@ static void l2_guest_code(struct svm_test_data *svm)
static void l1_guest_code(struct svm_test_data *svm, struct idt_entry *idt)
{
- #define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
vmcb->control.intercept &= ~(BIT(INTERCEPT_SHUTDOWN));
diff --git a/tools/testing/selftests/kvm/x86/svm_nested_soft_inject_test.c b/tools/testing/selftests/kvm/x86/svm_nested_soft_inject_test.c
index f72f11d4c4f8..0b640d09d194 100644
--- a/tools/testing/selftests/kvm/x86/svm_nested_soft_inject_test.c
+++ b/tools/testing/selftests/kvm/x86/svm_nested_soft_inject_test.c
@@ -78,17 +78,13 @@ static void l2_guest_code_nmi(void)
static void l1_guest_code(struct svm_test_data *svm, u64 is_nmi, u64 idt_alt)
{
- #define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
if (is_nmi)
x2apic_enable();
/* Prepare for L2 execution. */
- generic_svm_setup(svm,
- is_nmi ? l2_guest_code_nmi : l2_guest_code_int,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, is_nmi ? l2_guest_code_nmi : l2_guest_code_int);
vmcb->control.intercept_exceptions |= BIT(PF_VECTOR) | BIT(UD_VECTOR);
vmcb->control.intercept |= BIT(INTERCEPT_NMI) | BIT(INTERCEPT_HLT);
diff --git a/tools/testing/selftests/kvm/x86/svm_nested_vmcb12_gpa.c b/tools/testing/selftests/kvm/x86/svm_nested_vmcb12_gpa.c
index a4935ce2fb99..b3f45035745f 100644
--- a/tools/testing/selftests/kvm/x86/svm_nested_vmcb12_gpa.c
+++ b/tools/testing/selftests/kvm/x86/svm_nested_vmcb12_gpa.c
@@ -9,14 +9,9 @@
#include "kvm_test_harness.h"
#include "test_util.h"
-
-#define L2_GUEST_STACK_SIZE 64
-
#define SYNC_GP 101
#define SYNC_L2_STARTED 102
-static unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
static void guest_gp_handler(struct ex_regs *regs)
{
GUEST_SYNC(SYNC_GP);
@@ -30,28 +25,28 @@ static void l2_code(void)
static void l1_vmrun(struct svm_test_data *svm, gpa_t gpa)
{
- generic_svm_setup(svm, l2_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_code);
asm volatile ("vmrun %[gpa]" : : [gpa] "a" (gpa) : "memory");
}
static void l1_vmload(struct svm_test_data *svm, gpa_t gpa)
{
- generic_svm_setup(svm, l2_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_code);
asm volatile ("vmload %[gpa]" : : [gpa] "a" (gpa) : "memory");
}
static void l1_vmsave(struct svm_test_data *svm, gpa_t gpa)
{
- generic_svm_setup(svm, l2_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_code);
asm volatile ("vmsave %[gpa]" : : [gpa] "a" (gpa) : "memory");
}
static void l1_vmexit(struct svm_test_data *svm, gpa_t gpa)
{
- generic_svm_setup(svm, l2_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_code);
run_guest(svm->vmcb, svm->vmcb_gpa);
GUEST_ASSERT(svm->vmcb->control.exit_code == SVM_EXIT_VMMCALL);
diff --git a/tools/testing/selftests/kvm/x86/svm_pmu_host_guest_test.c b/tools/testing/selftests/kvm/x86/svm_pmu_host_guest_test.c
new file mode 100644
index 000000000000..c5b5cd788d93
--- /dev/null
+++ b/tools/testing/selftests/kvm/x86/svm_pmu_host_guest_test.c
@@ -0,0 +1,215 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * KVM nested SVM PMU Host-Only/Guest-Only test
+ *
+ * Copyright (C) 2026, Google LLC.
+ *
+ * Test that KVM correctly virtualizes the AMD PMU Host-Only (bit 41) and
+ * Guest-Only (bit 40) event selector bits across all SVM state
+ * transitions.
+ *
+ * Programs 4 PMCs simultaneously with all combinations of Host-Only and
+ * Guest-Only bits, then verifies correct counting behavior with different
+ * combinations of EFER.SVME and host/guest mode -- as well as event filtering.
+ */
+#include <fcntl.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
+#include "test_util.h"
+#include "kvm_util.h"
+#include "processor.h"
+#include "svm_util.h"
+#include "pmu.h"
+
+#define EVENTSEL_RETIRED_INSNS (ARCH_PERFMON_EVENTSEL_OS | \
+ ARCH_PERFMON_EVENTSEL_USR | \
+ ARCH_PERFMON_EVENTSEL_ENABLE | \
+ AMD_ZEN_INSTRUCTIONS_RETIRED)
+
+/* PMC configurations: index corresponds to Host-Only | Guest-Only bits */
+#define PMC_NONE 0 /* Neither bit set */
+#define PMC_G 1 /* Guest-Only bit set */
+#define PMC_H 2 /* Host-Only bit set */
+#define PMC_HG 3 /* Both bits set */
+#define NR_PMCS 4
+
+#define LOOP_INSNS 1000
+
+static __always_inline void run_instruction_loop(void)
+{
+ unsigned int i;
+
+ for (i = 0; i < LOOP_INSNS; i++)
+ __asm__ __volatile__("nop");
+}
+
+static __always_inline void read_counters(uint64_t *counts)
+{
+ int i;
+
+ for (i = 0; i < NR_PMCS; i++)
+ counts[i] = rdmsr(MSR_F15H_PERF_CTR + 2 * i);
+}
+
+static __always_inline void run_and_measure(uint64_t *deltas)
+{
+ uint64_t before[NR_PMCS], after[NR_PMCS];
+ int i;
+
+ read_counters(before);
+ run_instruction_loop();
+ read_counters(after);
+
+ for (i = 0; i < NR_PMCS; i++)
+ deltas[i] = after[i] - before[i];
+}
+
+static void assert_pmc_counts(uint64_t *deltas, unsigned int expected_counting)
+{
+ int i;
+
+ for (i = 0; i < NR_PMCS; i++) {
+ if (expected_counting & BIT(i))
+ GUEST_ASSERT_NE(deltas[i], 0);
+ else
+ GUEST_ASSERT_EQ(deltas[i], 0);
+ }
+}
+
+static uint64_t l2_deltas[NR_PMCS];
+
+static void l2_guest_code(void)
+{
+ run_and_measure(l2_deltas);
+ vmmcall();
+}
+
+static void l1_guest_code(struct svm_test_data *svm)
+{
+ struct vmcb *vmcb = svm->vmcb;
+ uint64_t deltas[NR_PMCS];
+ uint64_t eventsel;
+ int i;
+
+ /* Program 4 PMCs with all combinations of Host-Only/Guest-Only bits */
+ for (i = 0; i < NR_PMCS; i++) {
+ eventsel = EVENTSEL_RETIRED_INSNS;
+ if (i & PMC_G)
+ eventsel |= AMD64_EVENTSEL_GUESTONLY;
+ if (i & PMC_H)
+ eventsel |= AMD64_EVENTSEL_HOSTONLY;
+ wrmsr(MSR_F15H_PERF_CTL + 2 * i, eventsel);
+ wrmsr(MSR_F15H_PERF_CTR + 2 * i, 0);
+ }
+
+ /* Step 1: SVME=0 - Only the counter with neither bits set counts */
+ wrmsr(MSR_EFER, rdmsr(MSR_EFER) & ~EFER_SVME);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE));
+
+ /* Step 2: Set SVME=1 - In L1 "host mode"; Guest-Only stops */
+ wrmsr(MSR_EFER, rdmsr(MSR_EFER) | EFER_SVME);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE) | BIT(PMC_H) | BIT(PMC_HG));
+
+ /* Step 3: VMRUN to L2 - In "guest mode"; Host-Only stops */
+ generic_svm_setup(svm, l2_guest_code);
+ vmcb->control.intercept &= ~(1ULL << INTERCEPT_MSR_PROT);
+
+ run_guest(vmcb, svm->vmcb_gpa);
+
+ GUEST_ASSERT_EQ(vmcb->control.exit_code, SVM_EXIT_VMMCALL);
+ assert_pmc_counts(l2_deltas, BIT(PMC_NONE) | BIT(PMC_G) | BIT(PMC_HG));
+
+ /* Step 4: After VMEXIT to L1 - Back in "host mode"; Guest-Only stops */
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE) | BIT(PMC_H) | BIT(PMC_HG));
+
+ /* Step 5: Set KVM_PMU_EVENT_DENY - all counters stop */
+ GUEST_SYNC(KVM_PMU_EVENT_DENY);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, 0);
+
+ /* Step 6: Set KVM_PMU_EVENT_ALLOW - back to all except Guest-only */
+ GUEST_SYNC(KVM_PMU_EVENT_ALLOW);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE) | BIT(PMC_H) | BIT(PMC_HG));
+
+ /* Step 7: Clear Host-Only for PMC_HG - counter stops in "host mode" */
+ eventsel = rdmsr(MSR_F15H_PERF_CTL + 2 * PMC_HG);
+ wrmsr(MSR_F15H_PERF_CTL + 2 * PMC_HG, eventsel & ~AMD64_EVENTSEL_HOSTONLY);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE) | BIT(PMC_H));
+
+ /* Step 8: Restore Host-Only for PMC_HG - counter counts again */
+ wrmsr(MSR_F15H_PERF_CTL + 2 * PMC_HG, eventsel);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE) | BIT(PMC_H) | BIT(PMC_HG));
+
+ /* Step 9: Clear SVME - Only the counter with neither bits set counts */
+ wrmsr(MSR_EFER, rdmsr(MSR_EFER) & ~EFER_SVME);
+ run_and_measure(deltas);
+ assert_pmc_counts(deltas, BIT(PMC_NONE));
+
+ GUEST_DONE();
+}
+
+static struct kvm_pmu_event_filter *alloc_event_filter(u64 event)
+{
+ struct kvm_pmu_event_filter *filter;
+
+ filter = malloc(sizeof(*filter) + sizeof(event));
+ TEST_ASSERT(filter != NULL, "Filter allocation failed");
+
+ memset(filter, 0, sizeof(*filter));
+ memcpy(filter->events, &event, sizeof(event));
+ filter->nevents = 1;
+ filter->action = KVM_PMU_EVENT_ALLOW;
+
+ return filter;
+}
+
+int main(int argc, char *argv[])
+{
+ struct kvm_pmu_event_filter *filter;
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ struct ucall uc;
+ gva_t svm_gva;
+
+ TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_SVM));
+ TEST_REQUIRE(kvm_is_pmu_enabled());
+ TEST_REQUIRE(kvm_is_mediated_pmu_enabled());
+
+ vm = vm_create_with_one_vcpu(&vcpu, l1_guest_code);
+
+ vcpu_alloc_svm(vm, &svm_gva);
+ vcpu_args_set(vcpu, 1, svm_gva);
+
+ filter = alloc_event_filter(AMD_ZEN_INSTRUCTIONS_RETIRED);
+
+ for (;;) {
+ vcpu_run(vcpu);
+ TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_IO);
+
+ switch (get_ucall(vcpu, &uc)) {
+ case UCALL_ABORT:
+ REPORT_GUEST_ASSERT(uc);
+ goto done;
+ case UCALL_DONE:
+ goto done;
+ case UCALL_SYNC:
+ filter->action = uc.args[1];
+ vm_ioctl(vm, KVM_SET_PMU_EVENT_FILTER, filter);
+ break;
+ default:
+ TEST_FAIL("Unknown ucall %lu", uc.cmd);
+ goto done;
+ }
+ }
+done:
+ kvm_vm_free(vm);
+ return 0;
+}
diff --git a/tools/testing/selftests/kvm/x86/svm_vmcall_test.c b/tools/testing/selftests/kvm/x86/svm_vmcall_test.c
index b1887242f3b8..7c57fb7e6422 100644
--- a/tools/testing/selftests/kvm/x86/svm_vmcall_test.c
+++ b/tools/testing/selftests/kvm/x86/svm_vmcall_test.c
@@ -19,13 +19,10 @@ static void l2_guest_code(struct svm_test_data *svm)
static void l1_guest_code(struct svm_test_data *svm)
{
- #define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
struct vmcb *vmcb = svm->vmcb;
/* Prepare for L2 execution. */
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
run_guest(vmcb, svm->vmcb_gpa);
diff --git a/tools/testing/selftests/kvm/x86/sync_regs_test.c b/tools/testing/selftests/kvm/x86/sync_regs_test.c
index 5b0c2359bbb4..ed0c21b427c9 100644
--- a/tools/testing/selftests/kvm/x86/sync_regs_test.c
+++ b/tools/testing/selftests/kvm/x86/sync_regs_test.c
@@ -181,7 +181,7 @@ static void race_sync_regs(struct kvm_vcpu *vcpu, void *racer)
!!(run->s.regs.sregs.cr4 & X86_CR4_PAE),
!!(run->s.regs.sregs.efer & EFER_LME));
- TEST_ASSERT_EQ(pthread_create(&thread, NULL, racer, (void *)run), 0);
+ kvm_pthread_create(&thread, NULL, racer, (void *)run);
for (t = time(NULL) + TIMEOUT; time(NULL) < t;) {
/*
@@ -199,8 +199,7 @@ static void race_sync_regs(struct kvm_vcpu *vcpu, void *racer)
}
}
- TEST_ASSERT_EQ(pthread_cancel(thread), 0);
- TEST_ASSERT_EQ(pthread_join(thread, NULL), 0);
+ kvm_pthread_cancel_join(thread);
kvm_x86_state_cleanup(state);
}
diff --git a/tools/testing/selftests/kvm/x86/triple_fault_event_test.c b/tools/testing/selftests/kvm/x86/triple_fault_event_test.c
index f1c488e0d497..0d83516f4bd0 100644
--- a/tools/testing/selftests/kvm/x86/triple_fault_event_test.c
+++ b/tools/testing/selftests/kvm/x86/triple_fault_event_test.c
@@ -21,9 +21,6 @@ static void l2_guest_code(void)
: : [port] "d" (ARBITRARY_IO_PORT) : "rax");
}
-#define L2_GUEST_STACK_SIZE 64
-unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
void l1_guest_code_vmx(struct vmx_pages *vmx)
{
@@ -31,8 +28,7 @@ void l1_guest_code_vmx(struct vmx_pages *vmx)
GUEST_ASSERT(prepare_for_vmx_operation(vmx));
GUEST_ASSERT(load_vmcs(vmx));
- prepare_vmcs(vmx, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx, l2_guest_code);
GUEST_ASSERT(!vmlaunch());
/* L2 should triple fault after a triple fault event injected. */
@@ -44,8 +40,7 @@ void l1_guest_code_svm(struct svm_test_data *svm)
{
struct vmcb *vmcb = svm->vmcb;
- generic_svm_setup(svm, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ generic_svm_setup(svm, l2_guest_code);
/* don't intercept shutdown to test the case of SVM allowing to do so */
vmcb->control.intercept &= ~(BIT(INTERCEPT_SHUTDOWN));
diff --git a/tools/testing/selftests/kvm/x86/tsc_scaling_sync.c b/tools/testing/selftests/kvm/x86/tsc_scaling_sync.c
index 59da8d4da607..2a3a9bd8b338 100644
--- a/tools/testing/selftests/kvm/x86/tsc_scaling_sync.c
+++ b/tools/testing/selftests/kvm/x86/tsc_scaling_sync.c
@@ -94,12 +94,12 @@ int main(int argc, char *argv[])
pthread_t cpu_threads[NR_TEST_VCPUS];
unsigned long cpu;
for (cpu = 0; cpu < NR_TEST_VCPUS; cpu++)
- pthread_create(&cpu_threads[cpu], NULL, run_vcpu, (void *)cpu);
+ kvm_pthread_create(&cpu_threads[cpu], NULL, run_vcpu, (void *)cpu);
unsigned long failures = 0;
for (cpu = 0; cpu < NR_TEST_VCPUS; cpu++) {
void *this_cpu_failures;
- pthread_join(cpu_threads[cpu], &this_cpu_failures);
+ kvm_pthread_join(cpu_threads[cpu], &this_cpu_failures);
failures += (unsigned long)this_cpu_failures;
}
diff --git a/tools/testing/selftests/kvm/x86/vmx_apic_access_test.c b/tools/testing/selftests/kvm/x86/vmx_apic_access_test.c
index 1720113eae79..463f73aa9159 100644
--- a/tools/testing/selftests/kvm/x86/vmx_apic_access_test.c
+++ b/tools/testing/selftests/kvm/x86/vmx_apic_access_test.c
@@ -36,16 +36,13 @@ static void l2_guest_code(void)
static void l1_guest_code(struct vmx_pages *vmx_pages, unsigned long high_gpa)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u32 control;
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
/* Prepare the VMCS for L2 execution. */
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
control = vmreadz(CPU_BASED_VM_EXEC_CONTROL);
control |= CPU_BASED_ACTIVATE_SECONDARY_CONTROLS;
vmwrite(CPU_BASED_VM_EXEC_CONTROL, control);
diff --git a/tools/testing/selftests/kvm/x86/vmx_apicv_updates_test.c b/tools/testing/selftests/kvm/x86/vmx_apicv_updates_test.c
index 80a4fd1e5bbb..f9b88a6f6113 100644
--- a/tools/testing/selftests/kvm/x86/vmx_apicv_updates_test.c
+++ b/tools/testing/selftests/kvm/x86/vmx_apicv_updates_test.c
@@ -31,15 +31,13 @@ static void l2_guest_code(void)
static void l1_guest_code(struct vmx_pages *vmx_pages)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u32 control;
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
/* Prepare the VMCS for L2 execution. */
- prepare_vmcs(vmx_pages, l2_guest_code, &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
control = vmreadz(CPU_BASED_VM_EXEC_CONTROL);
control |= CPU_BASED_USE_MSR_BITMAPS;
vmwrite(CPU_BASED_VM_EXEC_CONTROL, control);
diff --git a/tools/testing/selftests/kvm/x86/vmx_invalid_nested_guest_state.c b/tools/testing/selftests/kvm/x86/vmx_invalid_nested_guest_state.c
index a2eaceed9ad5..c8379124b317 100644
--- a/tools/testing/selftests/kvm/x86/vmx_invalid_nested_guest_state.c
+++ b/tools/testing/selftests/kvm/x86/vmx_invalid_nested_guest_state.c
@@ -2,6 +2,7 @@
#include "test_util.h"
#include "kvm_util.h"
#include "processor.h"
+#include "smm.h"
#include "vmx.h"
#include <string.h>
@@ -9,9 +10,23 @@
#include "kselftest.h"
-#define ARBITRARY_IO_PORT 0x2000
+#define ARBITRARY_IO_PORT 0x80
-static struct kvm_vm *vm;
+/*
+ * The 64-bit SMRAM state-save area starts at SMBASE + 0xfe00. TR starts at
+ * offset 0xfe90, and attributes is the second 16-bit field in the descriptor.
+ */
+#define SMRAM64_TR_ATTRIBUTES_OFFSET 0xfe92
+#define SMRAM_GPA 0x1000000
+
+/*
+ * SMI handler that runs in 16-bit Real Mode. Syncs with L0 via port I/O, then
+ * executes RSM to trigger the consumption of invalid guest state.
+ */
+static u8 smi_handler[] = {
+ 0xe4, ARBITRARY_IO_PORT, /* IN $ARBITRARY_IO_PORT, %al */
+ 0x0f, 0xaa, /* RSM */
+};
static void l2_guest_code(void)
{
@@ -19,21 +34,18 @@ static void l2_guest_code(void)
* Generate an exit to L0 userspace, i.e. main(), via I/O to an
* arbitrary port.
*/
- asm volatile("inb %%dx, %%al"
- : : [port] "d" (ARBITRARY_IO_PORT) : "rax");
+ asm volatile("inb $" __stringify(ARBITRARY_IO_PORT) ", %%al"
+ ::: "rax");
+ GUEST_FAIL("L2 resumed after stuffing invalid guest state");
}
static void l1_guest_code(struct vmx_pages *vmx_pages)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
-
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
/* Prepare the VMCS for L2 execution. */
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
/*
* L2 must be run without unrestricted guest, verify that the selftests
@@ -50,38 +62,53 @@ static void l1_guest_code(struct vmx_pages *vmx_pages)
GUEST_DONE();
}
-int main(int argc, char *argv[])
+static void vcpu_run_to_io(struct kvm_vcpu *vcpu, bool want_l2)
{
- gva_t vmx_pages_gva;
- struct kvm_sregs sregs;
- struct kvm_vcpu *vcpu;
- struct kvm_run *run;
- struct ucall uc;
+ struct kvm_run *run = vcpu->run;
- TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_VMX));
+ vcpu_run(vcpu);
- vm = vm_create_with_one_vcpu(&vcpu, l1_guest_code);
+ TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_IO);
- /* Allocate VMX pages and shared descriptors (vmx_pages). */
- vcpu_alloc_vmx(vm, &vmx_pages_gva);
- vcpu_args_set(vcpu, 1, vmx_pages_gva);
+ TEST_ASSERT(run->io.port == ARBITRARY_IO_PORT &&
+ (!!(run->flags & KVM_RUN_X86_GUEST_MODE) == want_l2 ||
+ !kvm_has_cap(KVM_CAP_X86_GUEST_MODE)),
+ "Expected IN from port 0x%x from L%u, got port 0x%x from L%u",
+ ARBITRARY_IO_PORT, 1 + want_l2, run->io.port,
+ 1 + !!(run->flags & KVM_RUN_X86_GUEST_MODE));
+}
- vcpu_run(vcpu);
+static struct kvm_vm *vm_create_and_run_l2(struct kvm_vcpu **vcpu)
+{
+ gva_t vmx_pages_gva;
+ struct kvm_vm *vm;
- run = vcpu->run;
+ vm = vm_create_with_one_vcpu(vcpu, l1_guest_code);
+
+ /* Allocate VMX pages and shared descriptors (vmx_pages). */
+ vcpu_alloc_vmx(vm, &vmx_pages_gva);
+ vcpu_args_set(*vcpu, 1, vmx_pages_gva);
/*
* The first exit to L0 userspace should be an I/O access from L2.
* Running L1 should launch L2 without triggering an exit to userspace.
*/
- TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_IO);
+ vcpu_run_to_io(*vcpu, true);
+
+ return vm;
+}
+
+static void test_invalid_l2_guest_state(void)
+{
+ struct kvm_sregs sregs;
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ struct ucall uc;
- TEST_ASSERT(run->io.port == ARBITRARY_IO_PORT,
- "Expected IN from port %d from L2, got port %d",
- ARBITRARY_IO_PORT, run->io.port);
+ vm = vm_create_and_run_l2(&vcpu);
/*
- * Stuff invalid guest state for L2 by making TR unusuable. The next
+ * Stuff invalid guest state for L2 by making TR unusable. The next
* KVM_RUN should induce a TRIPLE_FAULT in L2 as KVM doesn't support
* emulating invalid guest state for L2.
*/
@@ -100,4 +127,49 @@ int main(int argc, char *argv[])
default:
TEST_FAIL("Unexpected ucall: %lu", uc.cmd);
}
+
+ kvm_vm_free(vm);
+}
+
+static void test_invalid_l2_guest_state_rsm(void)
+{
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ u16 *tr_attrs;
+
+ if (!kvm_has_cap(KVM_CAP_X86_SMM))
+ return;
+
+ vm = vm_create_and_run_l2(&vcpu);
+
+ /*
+ * Inject SMI while L2 is active, run the vCPU to get I/O exit from L1,
+ * then stuff TR in the SMRAM state-save area so that RSM restores
+ * invalid L2 state.
+ */
+ setup_smram(vm, vcpu, SMRAM_GPA, smi_handler, sizeof(smi_handler));
+ inject_smi(vcpu);
+
+ vcpu_run_to_io(vcpu, false);
+
+ /* Clear the present bit in SMRAM to make TR unusable. */
+ tr_attrs = addr_gpa2hva(vm, SMRAM_GPA + SMRAM64_TR_ATTRIBUTES_OFFSET);
+ *tr_attrs &= ~BIT(7);
+
+ vcpu_run(vcpu);
+
+ /*
+ * For RSM, L1 gets the SHUTDOWN because RSM is architecturally defined
+ * to result in shutdown if the CPU detects invalid state in SMRAM.
+ */
+ TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_SHUTDOWN);
+ kvm_vm_free(vm);
+}
+
+int main(int argc, char *argv[])
+{
+ TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_VMX));
+
+ test_invalid_l2_guest_state();
+ test_invalid_l2_guest_state_rsm();
}
diff --git a/tools/testing/selftests/kvm/x86/vmx_nested_la57_state_test.c b/tools/testing/selftests/kvm/x86/vmx_nested_la57_state_test.c
index f13dee317383..75073efa926d 100644
--- a/tools/testing/selftests/kvm/x86/vmx_nested_la57_state_test.c
+++ b/tools/testing/selftests/kvm/x86/vmx_nested_la57_state_test.c
@@ -27,8 +27,6 @@ static void l2_guest_code(void)
static void l1_guest_code(struct vmx_pages *vmx_pages)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u64 guest_cr4;
gpa_t pml5_pa, pml4_pa;
u64 *pml5;
@@ -42,8 +40,7 @@ static void l1_guest_code(struct vmx_pages *vmx_pages)
GUEST_ASSERT(prepare_for_vmx_operation(vmx_pages));
GUEST_ASSERT(load_vmcs(vmx_pages));
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
/*
* Set up L2 with a 4-level page table by pointing its CR3 to
diff --git a/tools/testing/selftests/kvm/x86/vmx_preemption_timer_test.c b/tools/testing/selftests/kvm/x86/vmx_preemption_timer_test.c
index 1b7b6ba23de7..eb8021c33cd4 100644
--- a/tools/testing/selftests/kvm/x86/vmx_preemption_timer_test.c
+++ b/tools/testing/selftests/kvm/x86/vmx_preemption_timer_test.c
@@ -66,8 +66,6 @@ void l2_guest_code(void)
void l1_guest_code(struct vmx_pages *vmx_pages)
{
-#define L2_GUEST_STACK_SIZE 64
- unsigned long l2_guest_stack[L2_GUEST_STACK_SIZE];
u64 l1_vmx_pt_start;
u64 l1_vmx_pt_finish;
u64 l1_tsc_deadline, l2_tsc_deadline;
@@ -77,8 +75,7 @@ void l1_guest_code(struct vmx_pages *vmx_pages)
GUEST_ASSERT(load_vmcs(vmx_pages));
GUEST_ASSERT(vmptrstz() == vmx_pages->vmcs_gpa);
- prepare_vmcs(vmx_pages, l2_guest_code,
- &l2_guest_stack[L2_GUEST_STACK_SIZE]);
+ prepare_vmcs(vmx_pages, l2_guest_code);
/*
* Check for Preemption timer support
diff --git a/tools/testing/selftests/kvm/x86/xapic_ipi_test.c b/tools/testing/selftests/kvm/x86/xapic_ipi_test.c
index 39ce9a9369f5..469e3ab16460 100644
--- a/tools/testing/selftests/kvm/x86/xapic_ipi_test.c
+++ b/tools/testing/selftests/kvm/x86/xapic_ipi_test.c
@@ -17,7 +17,7 @@
* amongst the available numa nodes on the machine.
*
* Migration is a command line option. When used on non-numa machines will
- * exit with error. Test is still usefull on non-numa for testing IPIs.
+ * exit with error. Test is still useful on non-numa for testing IPIs.
*/
#include <getopt.h>
#include <pthread.h>
@@ -228,25 +228,6 @@ static void *vcpu_thread(void *arg)
return NULL;
}
-static void cancel_join_vcpu_thread(pthread_t thread, struct kvm_vcpu *vcpu)
-{
- void *retval;
- int r;
-
- r = pthread_cancel(thread);
- TEST_ASSERT(r == 0,
- "pthread_cancel on vcpu_id=%d failed with errno=%d",
- vcpu->id, r);
-
- r = pthread_join(thread, &retval);
- TEST_ASSERT(r == 0,
- "pthread_join on vcpu_id=%d failed with errno=%d",
- vcpu->id, r);
- TEST_ASSERT(retval == PTHREAD_CANCELED,
- "expected retval=%p, got %p", PTHREAD_CANCELED,
- retval);
-}
-
void do_migrations(struct test_data_page *data, int run_secs, int delay_usecs,
u64 *pipis_rcvd)
{
@@ -387,7 +368,6 @@ void get_cmdline_args(int argc, char *argv[], int *run_secs,
int main(int argc, char *argv[])
{
- int r;
int wait_secs;
const int max_halter_wait = 10;
int run_secs = 0;
@@ -428,9 +408,7 @@ int main(int argc, char *argv[])
params[1].pipis_rcvd = pipis_rcvd;
/* Start halter vCPU thread and wait for it to execute first HLT. */
- r = pthread_create(&threads[0], NULL, vcpu_thread, &params[0]);
- TEST_ASSERT(r == 0,
- "pthread_create halter failed errno=%d", errno);
+ kvm_pthread_create(&threads[0], NULL, vcpu_thread, &params[0]);
fprintf(stderr, "Halter vCPU thread started\n");
wait_secs = 0;
@@ -447,8 +425,7 @@ int main(int argc, char *argv[])
"Halter vCPU thread reported its APIC ID: %u after %d seconds.\n",
data->halter_apic_id, wait_secs);
- r = pthread_create(&threads[1], NULL, vcpu_thread, &params[1]);
- TEST_ASSERT(r == 0, "pthread_create sender failed errno=%d", errno);
+ kvm_pthread_create(&threads[1], NULL, vcpu_thread, &params[1]);
fprintf(stderr,
"IPI sender vCPU thread started. Letting vCPUs run for %d seconds.\n",
@@ -462,8 +439,8 @@ int main(int argc, char *argv[])
/*
* Cancel threads and wait for them to stop.
*/
- cancel_join_vcpu_thread(threads[0], params[0].vcpu);
- cancel_join_vcpu_thread(threads[1], params[1].vcpu);
+ kvm_pthread_cancel_join_async(threads[0]);
+ kvm_pthread_cancel_join_async(threads[1]);
/*
* If the host support Idle HLT, i.e. KVM *might* be using Idle HLT,
diff --git a/tools/testing/selftests/kvm/x86/xen_shinfo_test.c b/tools/testing/selftests/kvm/x86/xen_shinfo_test.c
index 5076f6a75455..aaa414c677ad 100644
--- a/tools/testing/selftests/kvm/x86/xen_shinfo_test.c
+++ b/tools/testing/selftests/kvm/x86/xen_shinfo_test.c
@@ -437,7 +437,6 @@ int main(int argc, char *argv[])
struct kvm_vm *vm;
pthread_t thread;
bool verbose;
- int ret;
verbose = argc > 1 && (!strncmp(argv[1], "-v", 3) ||
!strncmp(argv[1], "--verbose", 10));
@@ -894,8 +893,7 @@ int main(int argc, char *argv[])
if (verbose)
printf("Testing shinfo lock corruption (KVM_XEN_HVM_EVTCHN_SEND)\n");
- ret = pthread_create(&thread, NULL, &juggle_shinfo_state, (void *)vm);
- TEST_ASSERT(ret == 0, "pthread_create() failed: %s", strerror(ret));
+ kvm_pthread_create(&thread, NULL, &juggle_shinfo_state, (void *)vm);
struct kvm_irq_routing_xen_evtchn uxe = {
.port = 1,
@@ -949,11 +947,7 @@ int main(int argc, char *argv[])
TEST_ASSERT(!evtchn_irq_expected,
"Expected event channel IRQ but it didn't happen");
- ret = pthread_cancel(thread);
- TEST_ASSERT(ret == 0, "pthread_cancel() failed: %s", strerror(ret));
-
- ret = pthread_join(thread, 0);
- TEST_ASSERT(ret == 0, "pthread_join() failed: %s", strerror(ret));
+ kvm_pthread_cancel_join(thread);
goto done;
case TEST_GUEST_SAW_IRQ:
diff --git a/tools/testing/selftests/landlock/audit.h b/tools/testing/selftests/landlock/audit.h
index f45fdef35681..d428ce802f49 100644
--- a/tools/testing/selftests/landlock/audit.h
+++ b/tools/testing/selftests/landlock/audit.h
@@ -214,41 +214,6 @@ static int audit_set_status(int fd, __u32 key, __u32 val)
return audit_request(fd, &msg, NULL);
}
-/* Returns a pointer to the last filled character of @dst, which is `\0`. */
-static __maybe_unused char *regex_escape(const char *const src, char *dst,
- size_t dst_size)
-{
- char *d = dst;
-
- for (const char *s = src; *s; s++) {
- switch (*s) {
- case '$':
- case '*':
- case '.':
- case '[':
- case '\\':
- case ']':
- case '^':
- if (d >= dst + dst_size - 2)
- return (char *)-ENOMEM;
-
- *d++ = '\\';
- *d++ = *s;
- break;
- default:
- if (d >= dst + dst_size - 1)
- return (char *)-ENOMEM;
-
- *d++ = *s;
- }
- }
- if (d >= dst + dst_size - 1)
- return (char *)-ENOMEM;
-
- *d = '\0';
- return d;
-}
-
/*
* @domain_id: The domain ID extracted from the audit message (if the first part
* of @pattern is REGEX_LANDLOCK_PREFIX). It is set to 0 if the domain ID is
diff --git a/tools/testing/selftests/landlock/base_test.c b/tools/testing/selftests/landlock/base_test.c
index cbd3c1669951..d20ab8f0862c 100644
--- a/tools/testing/selftests/landlock/base_test.c
+++ b/tools/testing/selftests/landlock/base_test.c
@@ -76,7 +76,7 @@ TEST(abi_version)
const struct landlock_ruleset_attr ruleset_attr = {
.handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE,
};
- ASSERT_EQ(10, landlock_create_ruleset(NULL, 0,
+ ASSERT_EQ(11, landlock_create_ruleset(NULL, 0,
LANDLOCK_CREATE_RULESET_VERSION));
ASSERT_EQ(-1, landlock_create_ruleset(&ruleset_attr, 0,
@@ -255,12 +255,24 @@ TEST(restrict_self_checks_ordering)
/* Checks unprivileged enforcement without no_new_privs. */
drop_caps(_metadata);
+ /*
+ * The flags validity is checked before the no_new_privs /
+ * CAP_SYS_ADMIN requirement.
+ */
ASSERT_EQ(-1, landlock_restrict_self(-1, -1));
- ASSERT_EQ(EPERM, errno);
+ ASSERT_EQ(EINVAL, errno);
ASSERT_EQ(-1, landlock_restrict_self(-1, 0));
ASSERT_EQ(EPERM, errno);
ASSERT_EQ(-1, landlock_restrict_self(ruleset_fd, 0));
ASSERT_EQ(EPERM, errno);
+ /*
+ * LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS fulfills the no_new_privs /
+ * CAP_SYS_ADMIN requirement but requires a ruleset, so the FD is
+ * checked next.
+ */
+ ASSERT_EQ(-1, landlock_restrict_self(
+ -1, LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ ASSERT_EQ(EBADF, errno);
ASSERT_EQ(0, prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0));
@@ -277,6 +289,41 @@ TEST(restrict_self_checks_ordering)
ASSERT_EQ(0, close(ruleset_fd));
}
+TEST(restrict_self_max_layers)
+{
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_EXECUTE,
+ };
+ struct landlock_path_beneath_attr path_beneath_attr = {
+ .allowed_access = LANDLOCK_ACCESS_FS_EXECUTE,
+ .parent_fd = -1,
+ };
+ const int ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ path_beneath_attr.parent_fd =
+ open("/tmp", O_PATH | O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_LE(0, path_beneath_attr.parent_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath_attr, 0));
+ ASSERT_EQ(0, close(path_beneath_attr.parent_fd));
+
+ /* Enforces the maximum number of allowed layers. */
+ for (int i = 0; i < LANDLOCK_MAX_NUM_LAYERS; i++)
+ ASSERT_EQ(0, landlock_restrict_self(ruleset_fd, 0));
+
+ /* Enforces one too many rulesets. */
+ drop_caps(_metadata);
+ ASSERT_EQ(-1, landlock_restrict_self(
+ ruleset_fd, LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ ASSERT_EQ(E2BIG, errno);
+
+ /* Checks that the failed call did not set no_new_privs. */
+ ASSERT_EQ(0, prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0));
+ ASSERT_EQ(0, close(ruleset_fd));
+}
+
TEST(restrict_self_fd)
{
int fd;
@@ -288,7 +335,7 @@ TEST(restrict_self_fd)
EXPECT_EQ(EBADFD, errno);
}
-TEST(restrict_self_fd_logging_flags)
+TEST(restrict_self_fd_flags)
{
int fd;
@@ -302,11 +349,16 @@ TEST(restrict_self_fd_logging_flags)
EXPECT_EQ(-1, landlock_restrict_self(
fd, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF));
EXPECT_EQ(EBADFD, errno);
+
+ /* LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS requires a ruleset FD. */
+ EXPECT_EQ(-1, landlock_restrict_self(
+ fd, LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ EXPECT_EQ(EBADFD, errno);
}
-TEST(restrict_self_logging_flags)
+TEST(restrict_self_flags)
{
- const __u32 last_flag = LANDLOCK_RESTRICT_SELF_TSYNC;
+ const __u32 last_flag = LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS;
/* Tests invalid flag combinations. */
@@ -349,6 +401,17 @@ TEST(restrict_self_logging_flags)
LANDLOCK_RESTRICT_SELF_LOG_NEW_EXEC_ON));
EXPECT_EQ(EBADF, errno);
+ /* LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS requires a ruleset FD. */
+
+ EXPECT_EQ(-1, landlock_restrict_self(
+ -1, LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ EXPECT_EQ(EBADF, errno);
+
+ EXPECT_EQ(-1, landlock_restrict_self(
+ -1, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF |
+ LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ EXPECT_EQ(EBADF, errno);
+
/* Tests with an invalid ruleset_fd. */
EXPECT_EQ(-1, landlock_restrict_self(
@@ -359,6 +422,37 @@ TEST(restrict_self_logging_flags)
-1, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF));
}
+TEST(restrict_self_no_new_privs)
+{
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+ const int ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+
+ ASSERT_LE(0, ruleset_fd);
+
+ /*
+ * The calling thread does not need CAP_SYS_ADMIN nor an explicit
+ * prctl(2) PR_SET_NO_NEW_PRIVS call.
+ */
+ drop_caps(_metadata);
+ ASSERT_EQ(0, prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0));
+
+ /* Checks that a failed call does not set no_new_privs. */
+ EXPECT_EQ(-1, landlock_restrict_self(
+ -1, LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ EXPECT_EQ(EBADF, errno);
+ EXPECT_EQ(0, prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0));
+
+ /* Checks that a successful call sets no_new_privs. */
+ ASSERT_EQ(0, landlock_restrict_self(
+ ruleset_fd, LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS));
+ EXPECT_EQ(1, prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0));
+
+ EXPECT_EQ(0, close(ruleset_fd));
+}
+
TEST(ruleset_fd_io)
{
struct landlock_ruleset_attr ruleset_attr = {
diff --git a/tools/testing/selftests/landlock/common.h b/tools/testing/selftests/landlock/common.h
index 7206d5105d66..c5124de68a51 100644
--- a/tools/testing/selftests/landlock/common.h
+++ b/tools/testing/selftests/landlock/common.h
@@ -253,3 +253,50 @@ static void __maybe_unused set_unix_address(struct service_fixture *const srv,
srv->unix_addr_len = SUN_LEN(&srv->unix_addr);
srv->unix_addr.sun_path[0] = '\0';
}
+
+/**
+ * regex_escape - Escape BRE metacharacters in a string
+ *
+ * @src: Source string to escape.
+ * @dst: Destination buffer for the escaped string.
+ * @dst_size: Size of the destination buffer.
+ *
+ * Escapes characters that have special meaning in POSIX Basic Regular
+ * Expressions: $ * . [ \ ] ^
+ *
+ * Returns a pointer to the NUL terminator in @dst (cursor-style API for
+ * chaining), or (char *)-ENOMEM if the buffer is too small.
+ */
+static __maybe_unused char *regex_escape(const char *const src, char *dst,
+ size_t dst_size)
+{
+ char *d = dst;
+
+ for (const char *s = src; *s; s++) {
+ switch (*s) {
+ case '$':
+ case '*':
+ case '.':
+ case '[':
+ case '\\':
+ case ']':
+ case '^':
+ if (d >= dst + dst_size - 2)
+ return (char *)-ENOMEM;
+
+ *d++ = '\\';
+ *d++ = *s;
+ break;
+ default:
+ if (d >= dst + dst_size - 1)
+ return (char *)-ENOMEM;
+
+ *d++ = *s;
+ }
+ }
+ if (d >= dst + dst_size - 1)
+ return (char *)-ENOMEM;
+
+ *d = '\0';
+ return d;
+}
diff --git a/tools/testing/selftests/landlock/config b/tools/testing/selftests/landlock/config
index 8fe9b461b1fd..d86321936fd8 100644
--- a/tools/testing/selftests/landlock/config
+++ b/tools/testing/selftests/landlock/config
@@ -2,6 +2,8 @@ CONFIG_AF_UNIX_OOB=y
CONFIG_AUDIT=y
CONFIG_CGROUPS=y
CONFIG_CGROUP_SCHED=y
+CONFIG_ENABLE_DEFAULT_TRACERS=y
+CONFIG_FTRACE=y
CONFIG_INET=y
CONFIG_IPV6=y
CONFIG_KEYS=y
diff --git a/tools/testing/selftests/landlock/fs_test.c b/tools/testing/selftests/landlock/fs_test.c
index 86e08aa6e0a7..6e979cef884d 100644
--- a/tools/testing/selftests/landlock/fs_test.c
+++ b/tools/testing/selftests/landlock/fs_test.c
@@ -44,6 +44,9 @@
#include "audit.h"
#include "common.h"
+#include "trace.h"
+
+#define TRACE_TASK "fs_test"
#ifndef renameat2
int renameat2(int olddirfd, const char *oldpath, int newdirfd,
@@ -2247,6 +2250,177 @@ TEST_F_FORK(layout1, rename_file)
RENAME_EXCHANGE));
}
+TEST_F_FORK(layout1, rename_whiteout_denied)
+{
+ /* The affected file is a FIFO. */
+ ASSERT_EQ(0, unlink(file1_s3d3));
+ ASSERT_EQ(0, mknod(file1_s3d3, S_IFIFO | 0600, 0));
+
+ /* Deny MAKE_REG, but allow MAKE_FIFO. */
+ enforce_fs(_metadata, LANDLOCK_ACCESS_FS_MAKE_REG, NULL);
+
+ /*
+ * Try to rename a file with RENAME_WHITEOUT.
+ * file1_s3d3 is in dir_s3d2 (tmpfs), so it supports RENAME_WHITEOUT.
+ * Denied, because whiteout creation is guarded with MAKE_REG.
+ */
+ EXPECT_EQ(-1, renameat2(AT_FDCWD, file1_s3d3, AT_FDCWD,
+ TMP_DIR "/s3d1/s3d2/s3d3/f2", RENAME_WHITEOUT));
+ EXPECT_EQ(EACCES, errno);
+}
+
+static bool is_whiteout(const char *const path)
+{
+ struct stat st;
+
+ if (stat(path, &st) == -1)
+ return false;
+
+ return S_ISCHR(st.st_mode) && st.st_rdev == makedev(0, 0);
+}
+
+static bool is_fifo(const char *const path)
+{
+ struct stat st;
+
+ return stat(path, &st) == 0 && S_ISFIFO(st.st_mode);
+}
+
+static bool is_missing(const char *const path)
+{
+ struct stat st;
+
+ return stat(path, &st) == -1 && errno == ENOENT;
+}
+
+TEST_F_FORK(layout1, rename_whiteout_allowed)
+{
+ const struct rule rules[] = {
+ {
+ .path = dir_s3d3,
+ .access = LANDLOCK_ACCESS_FS_MAKE_REG,
+ },
+ {},
+ };
+
+ /* The affected file is a FIFO. */
+ ASSERT_EQ(0, unlink(file1_s3d3));
+ ASSERT_EQ(0, mknod(file1_s3d3, S_IFIFO | 0600, 0));
+
+ /* Allow MAKE_REG below dir_s3d3. */
+ enforce_fs(_metadata, LANDLOCK_ACCESS_FS_MAKE_REG, rules);
+
+ /*
+ * Rename a file with RENAME_WHITEOUT within the same directory.
+ * Allowed, because MAKE_REG is granted for the whiteout object which
+ * gets created in the source location.
+ */
+ EXPECT_EQ(0, renameat2(AT_FDCWD, file1_s3d3, AT_FDCWD,
+ TMP_DIR "/s3d1/s3d2/s3d3/f2", RENAME_WHITEOUT));
+
+ /* A whiteout object took the place of the moved FIFO. */
+ EXPECT_TRUE(is_whiteout(file1_s3d3));
+ EXPECT_TRUE(is_fifo(TMP_DIR "/s3d1/s3d2/s3d3/f2"));
+}
+
+TEST_F_FORK(layout1, rename_whiteout_reparenting)
+{
+ const struct rule rules[] = {
+ {
+ .path = dir_s3d2,
+ .access = LANDLOCK_ACCESS_FS_REFER,
+ },
+ {
+ .path = dir_s3d3,
+ .access = LANDLOCK_ACCESS_FS_MAKE_REG,
+ },
+ {},
+ };
+
+ /* The moved files are FIFOs. */
+ ASSERT_EQ(0, unlink(file1_s3d3));
+ ASSERT_EQ(0, mknod(file1_s3d3, S_IFIFO | 0600, 0));
+ ASSERT_EQ(0, unlink(file1_s3d4));
+ ASSERT_EQ(0, mknod(file1_s3d4, S_IFIFO | 0600, 0));
+
+ /* Allow REFER below dir_s3d2, but MAKE_REG only below dir_s3d3. */
+ enforce_fs(_metadata,
+ LANDLOCK_ACCESS_FS_MAKE_REG | LANDLOCK_ACCESS_FS_REFER,
+ rules);
+
+ /*
+ * The whiteout object is created in the source directory: Moving the
+ * FIFO out of dir_s3d4 is denied because MAKE_REG is not granted
+ * there, even though it is granted in the destination directory
+ * dir_s3d3.
+ */
+ EXPECT_EQ(-1, renameat2(AT_FDCWD, file1_s3d4, AT_FDCWD,
+ TMP_DIR "/s3d1/s3d2/s3d3/f2", RENAME_WHITEOUT));
+ EXPECT_EQ(EACCES, errno);
+
+ /*
+ * Moving the FIFO out of dir_s3d3 is allowed, because MAKE_REG is
+ * granted there for the created whiteout object.
+ */
+ EXPECT_EQ(0, renameat2(AT_FDCWD, file1_s3d3, AT_FDCWD,
+ TMP_DIR "/s3d1/s3d2/s3d4/f2", RENAME_WHITEOUT));
+
+ /* A whiteout object took the place of the moved FIFO. */
+ EXPECT_TRUE(is_whiteout(file1_s3d3));
+ EXPECT_TRUE(is_fifo(TMP_DIR "/s3d1/s3d2/s3d4/f2"));
+}
+
+TEST_F_FORK(layout1, rename_whiteout_exchange)
+{
+ const char *const whiteout_s3d3 = TMP_DIR "/s3d1/s3d2/s3d3/f2";
+ const struct rule rules[] = {
+ {
+ .path = dir_s3d2,
+ .access = LANDLOCK_ACCESS_FS_REFER,
+ },
+ {
+ .path = dir_s3d3,
+ .access = LANDLOCK_ACCESS_FS_MAKE_REG,
+ },
+ {},
+ };
+
+ /* The exchanged files are FIFOs and an existing whiteout object. */
+ ASSERT_EQ(0, unlink(file1_s3d3));
+ ASSERT_EQ(0, mknod(file1_s3d3, S_IFIFO | 0600, 0));
+ ASSERT_EQ(0, mknod(whiteout_s3d3, S_IFCHR | 0600, makedev(0, 0)));
+ ASSERT_EQ(0, unlink(file1_s3d4));
+ ASSERT_EQ(0, mknod(file1_s3d4, S_IFIFO | 0600, 0));
+
+ /* Allow REFER below dir_s3d2, but MAKE_REG only below dir_s3d3. */
+ enforce_fs(_metadata,
+ LANDLOCK_ACCESS_FS_MAKE_REG | LANDLOCK_ACCESS_FS_REFER,
+ rules);
+
+ /*
+ * With RENAME_EXCHANGE, the whiteout object moves into the source
+ * directory of the rename: Exchanging the FIFO in dir_s3d4 with the
+ * whiteout object is denied because MAKE_REG is not granted in
+ * dir_s3d4, even though it is granted in the whiteout object's own
+ * directory dir_s3d3.
+ */
+ EXPECT_EQ(-1, renameat2(AT_FDCWD, file1_s3d4, AT_FDCWD, whiteout_s3d3,
+ RENAME_EXCHANGE));
+ EXPECT_EQ(EACCES, errno);
+
+ /*
+ * Exchanging the FIFO in dir_s3d3 with the whiteout object is
+ * allowed, because MAKE_REG is granted in the directory into which
+ * the whiteout object moves.
+ */
+ EXPECT_EQ(0, renameat2(AT_FDCWD, file1_s3d3, AT_FDCWD, whiteout_s3d3,
+ RENAME_EXCHANGE));
+
+ /* The FIFO and the whiteout object swapped places. */
+ EXPECT_TRUE(is_whiteout(file1_s3d3));
+ EXPECT_TRUE(is_fifo(whiteout_s3d3));
+}
+
TEST_F_FORK(layout1, rename_dir)
{
const struct rule rules[] = {
@@ -3270,6 +3444,18 @@ TEST_F_FORK(layout1, make_char)
makedev(1, 3));
}
+TEST_F_FORK(layout1, make_whiteout)
+{
+ /*
+ * Creates a whiteout object (creation guarded by MAKE_REG).
+ *
+ * Contrary to the other character devices, this does not require
+ * CAP_MKNOD, cf. vfs_mknod().
+ */
+ test_make_file(_metadata, LANDLOCK_ACCESS_FS_MAKE_REG, S_IFCHR,
+ makedev(0, 0));
+}
+
TEST_F_FORK(layout1, make_block)
{
/* Creates a /dev/loop0 device. */
@@ -6459,6 +6645,8 @@ static const char lower_fo1[] = LOWER_DATA "/fo1";
static const char lower_do1[] = LOWER_DATA "/do1";
static const char lower_do1_fo2[] = LOWER_DATA "/do1/fo2";
static const char lower_do1_fl3[] = LOWER_DATA "/do1/fl3";
+/* lower_pl1 is a FIFO and is deliberately not in the lists below. */
+static const char lower_pl1[] = LOWER_DATA "/pl1";
static const char (*lower_base_files[])[] = {
&lower_fl1,
@@ -6508,6 +6696,8 @@ static const char (*upper_sub_files[])[] = {
#define MERGE_BASE TMP_DIR "/merge"
#define MERGE_DATA MERGE_BASE "/data"
static const char merge_fl1[] = MERGE_DATA "/fl1";
+/* merge_pl1 is a FIFO and is deliberately not in the lists below. */
+static const char merge_pl1[] = MERGE_DATA "/pl1";
static const char merge_dl1[] = MERGE_DATA "/dl1";
static const char merge_dl1_fl2[] = MERGE_DATA "/dl1/fl2";
static const char merge_fu1[] = MERGE_DATA "/fu1";
@@ -6548,7 +6738,8 @@ static const char (*merge_sub_files[])[] = {
* │   │   ├── fl3
* │   │   └── fo2
* │   ├── fl1
- * │   └── fo1
+ * │   ├── fo1
+ * │   └── pl1 [FIFO]
* ├── merge
* │   └── data
* │   ├── dl1
@@ -6561,7 +6752,8 @@ static const char (*merge_sub_files[])[] = {
* │   │   └── fu2
* │   ├── fl1
* │   ├── fo1
- * │   └── fu1
+ * │   ├── fu1
+ * │   └── pl1 [FIFO]
* └── upper
* ├── data
* │   ├── do1
@@ -6599,6 +6791,7 @@ FIXTURE_SETUP(layout2_overlay)
create_file(_metadata, lower_fo1);
create_file(_metadata, lower_do1_fo2);
create_file(_metadata, lower_do1_fl3);
+ ASSERT_EQ(0, mknod(lower_pl1, S_IFIFO | 0600, 0));
create_directory(_metadata, UPPER_BASE);
set_cap(_metadata, CAP_SYS_ADMIN);
@@ -6631,6 +6824,7 @@ FIXTURE_TEARDOWN_PARENT(layout2_overlay)
EXPECT_EQ(0, remove_path(lower_fl1));
EXPECT_EQ(0, remove_path(lower_do1_fo2));
EXPECT_EQ(0, remove_path(lower_fo1));
+ EXPECT_EQ(0, remove_path(lower_pl1));
/* umount(LOWER_BASE)) is handled by namespace lifetime. */
EXPECT_EQ(0, remove_path(LOWER_BASE));
@@ -6927,7 +7121,7 @@ TEST_F_FORK(layout2_overlay, same_content_different_file)
ASSERT_EQ(0, test_open(path_entry, O_RDWR));
}
- /* Only allowes access to the merge hierarchy. */
+ /* Only allows access to the merge hierarchy. */
enforce_fs(_metadata, ACCESS_RW, layer5_merge_only);
/* Checks new accesses on lower layer. */
@@ -6951,6 +7145,43 @@ TEST_F_FORK(layout2_overlay, same_content_different_file)
}
}
+TEST_F_FORK(layout2_overlay, rename_in_overlay_without_make_reg)
+{
+ const char *const merge_pl1_renamed = MERGE_DATA "/pl1_renamed";
+
+ if (self->skip_test)
+ SKIP(return, "overlayfs is not supported (test)");
+
+ /*
+ * merge_pl1 is a FIFO which only exists in the lower layer. Before
+ * the rename, the upper layer has no entry under this name.
+ */
+ ASSERT_TRUE(is_fifo(merge_pl1));
+ ASSERT_TRUE(is_missing(UPPER_DATA "/pl1"));
+
+ /* MAKE_REG is restricted, but MAKE_FIFO is not. */
+ enforce_fs(_metadata, LANDLOCK_ACCESS_FS_MAKE_REG, NULL);
+
+ /*
+ * Rename the FIFO through OverlayFS. merge_pl1 originates from the
+ * lower layer, so this triggers a copy-up and creates the whiteout in
+ * the upper layer to hide the lower layer FIFO file. Even though
+ * MAKE_REG is restricted, the rename on the OverlayFS works.
+ */
+ EXPECT_EQ(0, rename(merge_pl1, merge_pl1_renamed));
+
+ /* Check that the rename worked. */
+ EXPECT_TRUE(is_fifo(merge_pl1_renamed));
+ EXPECT_TRUE(is_missing(merge_pl1));
+
+ /*
+ * Check that the whiteout object was created on the underlying "upper"
+ * filesystem during the rename. This is OK because the whiteout object
+ * was created by OverlayFS, not by the calling task.
+ */
+ EXPECT_TRUE(is_whiteout(UPPER_DATA "/pl1"));
+}
+
FIXTURE(layout3_fs)
{
bool has_created_dir;
@@ -7436,7 +7667,7 @@ TEST_F(audit_layout1, make_char)
enforce_fs(_metadata, ACCESS_ALL, NULL);
- EXPECT_EQ(-1, mknod(file1_s1d3, S_IFCHR | 0644, 0));
+ EXPECT_EQ(-1, mknod(file1_s1d3, S_IFCHR | 0644, makedev(7, 0)));
EXPECT_EQ(EACCES, errno);
EXPECT_EQ(0, matches_log_fs(_metadata, self->audit_fd, "fs\\.make_char",
dir_s1d3));
@@ -7446,6 +7677,25 @@ TEST_F(audit_layout1, make_char)
EXPECT_EQ(1, records.domain);
}
+TEST_F(audit_layout1, make_whiteout)
+{
+ struct audit_records records;
+
+ EXPECT_EQ(0, unlink(file1_s1d3));
+
+ enforce_fs(_metadata, ACCESS_ALL, NULL);
+
+ /* Whiteout creation is denied and logged as fs.make_reg. */
+ EXPECT_EQ(-1, mknod(file1_s1d3, S_IFCHR | 0644, makedev(0, 0)));
+ EXPECT_EQ(EACCES, errno);
+ EXPECT_EQ(0, matches_log_fs(_metadata, self->audit_fd, "fs\\.make_reg",
+ dir_s1d3));
+
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(1, records.domain);
+}
+
TEST_F(audit_layout1, make_dir)
{
struct audit_records records;
@@ -10189,4 +10439,484 @@ TEST_F(audit_quiet_rename, quiet_behind_mountpoint_disconnected)
ASSERT_EQ(0, records.access);
}
+/* clang-format off */
+FIXTURE(trace_layout1) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_layout1)
+{
+ struct stat st;
+
+ /*
+ * Check tracefs availability before creating the layout, following the
+ * layout3_fs pattern: skip before any layout creation to avoid leaving
+ * stale TMP_DIR on skip.
+ */
+ if (stat(TRACEFS_LANDLOCK_DIR, &st)) {
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ /* Isolate tracefs state (PID filter, event enables). */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ prepare_layout(_metadata);
+ create_layout1(_metadata);
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_EQ(0, tracefs_fixture_setup());
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ ASSERT_EQ(0, tracefs_set_pid_filter(getpid()));
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+}
+
+FIXTURE_TEARDOWN_PARENT(trace_layout1)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false);
+ tracefs_clear_pid_filter();
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+
+ remove_layout1(_metadata);
+ cleanup_layout(_metadata);
+}
+
+/*
+ * Verifies that check_rule_inode events include correct field values: domain,
+ * dev, ino, access_request, and grants. All values are verified against stat()
+ * of the rule path on a deterministic tmpfs layout.
+ */
+TEST_F(trace_layout1, check_rule_fs_fields)
+{
+ struct stat dir_stat;
+ char expected_dev[32];
+ char expected_ino[32];
+ char *buf;
+ char field[64];
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ ASSERT_EQ(0, stat(dir_s1d1, &dir_stat));
+ snprintf(expected_dev, sizeof(expected_dev), "%u:%u",
+ major(dir_stat.st_dev), minor(dir_stat.st_dev));
+ snprintf(expected_ino, sizeof(expected_ino), "%lu", dir_stat.st_ino);
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+
+ sandbox_child_fs_access(_metadata, dir_s1d1,
+ LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, dir_s1d1);
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK)))
+ {
+ TH_LOG("Expected 1 check_rule_inode event\n%s", buf);
+ }
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "dev", field, sizeof(field)));
+ EXPECT_STREQ(expected_dev, field)
+ {
+ TH_LOG("Expected dev=%s, got %s", expected_dev, field);
+ }
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "ino", field, sizeof(field)));
+ EXPECT_STREQ(expected_ino, field)
+ {
+ TH_LOG("Expected ino=%s, got %s", expected_ino, field);
+ }
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "access_request", field,
+ sizeof(field)));
+ EXPECT_STREQ("read_dir", field)
+ {
+ TH_LOG("Expected access_request=read_dir, got %s", field);
+ }
+
+ /*
+ * The domain handles only READ_DIR, so the rule carries the
+ * unhandled-rights padding; intersecting with the request leaves just
+ * the requested read_dir (no padding, no hex).
+ */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "grants", field, sizeof(field)));
+ EXPECT_STREQ("{read_dir}", field)
+ {
+ TH_LOG("Expected grants={read_dir}, got %s", field);
+ }
+
+ free(buf);
+}
+
+/*
+ * Verifies check_rule_inode behavior with multiple rules. With rules at s1d1
+ * and s1d2 (a child of s1d1), accessing s1d2 produces only 1 event because the
+ * pathwalk short-circuits after the first rule fully unmasks the single layer.
+ */
+TEST_F(trace_layout1, check_rule_fs_multiple_rules)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ int count;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd, fd;
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open(dir_s1d1, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0))
+ _exit(1);
+ close(path_beneath.parent_fd);
+
+ path_beneath.parent_fd =
+ open(dir_s1d2, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0))
+ _exit(1);
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ fd = open(dir_s1d2, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_NE(NULL, buf);
+
+ /*
+ * Only one check_rule_inode event: the rule on dir_s1d2 fully unmasks
+ * the single layer, so the pathwalk short-circuits before reaching the
+ * dir_s1d1 rule.
+ */
+ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 check_rule_inode event, got %d\n%s", count,
+ buf);
+ }
+
+ free(buf);
+}
+
+/*
+ * Verifies the grants array is intersected with the request: a handled,
+ * granted, but unrequested right (execute) is filtered out, leaving only the
+ * requested read_dir.
+ */
+TEST_F(trace_layout1, check_rule_fs_request_subset)
+{
+ char *buf;
+ char field[64];
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+
+ /*
+ * Handle and grant READ_DIR|EXECUTE; the open only requests read_dir.
+ */
+ sandbox_child_fs_access(
+ _metadata, dir_s1d1,
+ LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_EXECUTE,
+ LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_EXECUTE,
+ dir_s1d1);
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_NE(NULL, buf);
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "access_request", field,
+ sizeof(field)));
+ EXPECT_STREQ("read_dir", field);
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "grants", field, sizeof(field)));
+ EXPECT_STREQ("{read_dir}", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that the optional TRUNCATE access right, which hook_file_open()
+ * speculatively evaluates on every open, appears in the access_request= and
+ * grants= fields. Opening file1_s1d1 read-only needs only read_file, but the
+ * open hook also evaluates truncate; the domain handles and the rule grants
+ * both, so the event reports access_request=read_file|truncate and
+ * grants={read_file|truncate}, and the open is allowed.
+ */
+TEST_F(trace_layout1, check_rule_fs_optional_access)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ char field[64];
+ int count;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE |
+ LANDLOCK_ACCESS_FS_TRUNCATE,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_FILE |
+ LANDLOCK_ACCESS_FS_TRUNCATE,
+ };
+ int ruleset_fd, fd;
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open(dir_s1d1, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0))
+ _exit(1);
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ /* Read-only open needs only read_file; truncate is optional. */
+ fd = open(file1_s1d1, O_RDONLY | O_CLOEXEC);
+ if (fd < 0)
+ _exit(1);
+ close(fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ /* The open is allowed: the required read_file is granted. */
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_NE(NULL, buf);
+
+ /* The rule at dir_s1d1 matches when opening file1_s1d1. */
+ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 check_rule_inode event, got %d\n%s", count,
+ buf);
+ }
+
+ /* The open hook adds the optional truncate to the request. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "access_request", field,
+ sizeof(field)));
+ EXPECT_STREQ("read_file|truncate", field);
+
+ /* The rule grants both, so truncate appears in the grants array. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "grants", field, sizeof(field)));
+ EXPECT_STREQ("{read_file|truncate}", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that check_rule_inode fires for a rule matching the inode even when
+ * it grants none of the requested rights, so the grants set is empty. Landlock
+ * cannot know a rule ignores the request before reading it, so the event is
+ * still emitted (grants={}), which lets a tracer see that the rule matched.
+ * The domain handles READ_DIR|EXECUTE, dir_s1d2 grants only EXECUTE and its
+ * parent dir_s1d1 grants only READ_DIR. Reading dir_s1d2 (requesting read_dir)
+ * first matches the dir_s1d2 rule, which grants nothing requested (grants={});
+ * walking up to dir_s1d1 then grants read_dir (grants={read_dir}) and allows
+ * the access.
+ */
+TEST_F(trace_layout1, check_rule_fs_empty_grant)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ int count;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR |
+ LANDLOCK_ACCESS_FS_EXECUTE,
+ };
+ struct landlock_path_beneath_attr path_beneath = {};
+ int ruleset_fd, fd;
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ /* Parent dir_s1d1 grants only READ_DIR. */
+ path_beneath.allowed_access = LANDLOCK_ACCESS_FS_READ_DIR;
+ path_beneath.parent_fd =
+ open(dir_s1d1, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0))
+ _exit(1);
+ close(path_beneath.parent_fd);
+
+ /* Child dir_s1d2 grants only EXECUTE. */
+ path_beneath.allowed_access = LANDLOCK_ACCESS_FS_EXECUTE;
+ path_beneath.parent_fd =
+ open(dir_s1d2, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0))
+ _exit(1);
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ fd = open(dir_s1d2, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd < 0)
+ _exit(1);
+ close(fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ set_cap(_metadata, CAP_DAC_OVERRIDE);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_DAC_OVERRIDE);
+ ASSERT_NE(NULL, buf);
+
+ /*
+ * dir_s1d2 (grants nothing requested) then dir_s1d1 (grants read_dir).
+ */
+ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_EQ(2, count)
+ {
+ TH_LOG("Expected 2 check_rule_inode events, got %d\n%s", count,
+ buf);
+ }
+
+ /* The dir_s1d2 rule matches the inode but grants none of read_dir. */
+ EXPECT_EQ(
+ 1,
+ tracefs_count_matches(
+ buf,
+ TRACE_PREFIX(
+ TRACE_TASK) "landlock_check_rule_inode: domain=[0-9a-f]\\+ "
+ "access_request=read_dir "
+ "dev=[0-9]\\+:[0-9]\\+ ino=[0-9]\\+ "
+ "grants={}$"))
+ {
+ TH_LOG("Expected a grants={} event\n%s", buf);
+ }
+
+ /* Walking up to dir_s1d1 grants the requested read_dir. */
+ EXPECT_EQ(
+ 1,
+ tracefs_count_matches(
+ buf,
+ TRACE_PREFIX(
+ TRACE_TASK) "landlock_check_rule_inode: domain=[0-9a-f]\\+ "
+ "access_request=read_dir "
+ "dev=[0-9]\\+:[0-9]\\+ ino=[0-9]\\+ "
+ "grants={read_dir}$"))
+ {
+ TH_LOG("Expected a grants={read_dir} event\n%s", buf);
+ }
+
+ free(buf);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/net_test.c b/tools/testing/selftests/landlock/net_test.c
index be2eb88092fb..16afbfdf06bb 100644
--- a/tools/testing/selftests/landlock/net_test.c
+++ b/tools/testing/selftests/landlock/net_test.c
@@ -10,11 +10,12 @@
#include <arpa/inet.h>
#include <errno.h>
#include <fcntl.h>
-#include <linux/landlock.h>
#include <linux/in.h>
+#include <linux/landlock.h>
#include <sched.h>
#include <stdint.h>
#include <string.h>
+#include <sys/mount.h>
#include <sys/prctl.h>
#include <sys/socket.h>
#include <sys/syscall.h>
@@ -22,6 +23,9 @@
#include "audit.h"
#include "common.h"
+#include "trace.h"
+
+#define TRACE_TASK "net_test"
const short sock_port_start = (1 << 10);
@@ -3285,4 +3289,609 @@ TEST_F(audit, sendmsg)
EXPECT_EQ(0, close(sock_fd));
}
+/* Trace tests */
+
+/* clang-format off */
+FIXTURE(trace_net) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_net)
+{
+ int ret;
+
+ /* Isolate the network namespace so the bound port cannot collide. */
+ setup_loopback(_metadata);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_net)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * Baseline: verifies that without Landlock, the bind succeeds and no
+ * deny_access_net trace event fires.
+ */
+/* clang-format off */
+FIXTURE_VARIANT(trace_net)
+{
+ /* clang-format on */
+ bool sandbox;
+ int bind_port_offset; /* 0 = allowed port, 1 = denied port */
+ int expect_denied;
+};
+
+/* Unsandboxed: no Landlock, bind should succeed with no events. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net, unsandboxed) {
+ /* clang-format on */
+ .sandbox = false,
+ .bind_port_offset = 0,
+ .expect_denied = 0,
+};
+
+/* Denied: sandboxed, bind to port not in ruleset. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net, bind_denied) {
+ /* clang-format on */
+ .sandbox = true,
+ .bind_port_offset = 1,
+ .expect_denied = 1,
+};
+
+/* Allowed: sandboxed, bind to port in ruleset. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net, bind_allowed) {
+ /* clang-format on */
+ .sandbox = true,
+ .bind_port_offset = 0,
+ .expect_denied = 0,
+};
+
+TEST_F(trace_net, deny_access_net_bind)
+{
+ char *buf;
+ int count, status;
+ pid_t child;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ struct sockaddr_in addr = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ };
+ int sock_fd;
+
+ if (variant->sandbox) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_net =
+ LANDLOCK_ACCESS_NET_BIND_TCP,
+ };
+ struct landlock_net_port_attr port_attr = {
+ .allowed_access = LANDLOCK_ACCESS_NET_BIND_TCP,
+ .port = sock_port_start,
+ };
+ int ruleset_fd;
+
+ ruleset_fd = landlock_create_ruleset(
+ &ruleset_attr, sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ if (landlock_add_rule(ruleset_fd,
+ LANDLOCK_RULE_NET_PORT,
+ &port_attr, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+ }
+
+ sock_fd = socket(AF_INET, SOCK_STREAM | SOCK_CLOEXEC, 0);
+ if (sock_fd < 0)
+ _exit(1);
+
+ addr.sin_port =
+ htons(sock_port_start + variant->bind_port_offset);
+ if (variant->expect_denied) {
+ /* Bind should be denied. */
+ if (bind(sock_fd, (struct sockaddr *)&addr,
+ sizeof(addr)) == 0) {
+ close(sock_fd);
+ _exit(2);
+ }
+ if (errno != EACCES) {
+ close(sock_fd);
+ _exit(3);
+ }
+ } else {
+ /* Bind should succeed. */
+ if (bind(sock_fd, (struct sockaddr *)&addr,
+ sizeof(addr))) {
+ close(sock_fd);
+ _exit(2);
+ }
+ }
+ close(sock_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK));
+ if (variant->expect_denied) {
+ EXPECT_EQ(variant->expect_denied, count)
+ {
+ TH_LOG("Expected deny_access_net event, got %d\n%s",
+ count, buf);
+ }
+ } else {
+ EXPECT_EQ(0, count)
+ {
+ TH_LOG("Expected 0 deny_access_net events, "
+ "got %d\n%s",
+ count, buf);
+ }
+ }
+
+ free(buf);
+}
+
+/*
+ * Anchors the denial fields shared by every deny_access_net event so a port
+ * test also proves the denying domain, execution status, logging verdict, and
+ * exact blocked access.
+ */
+static void
+expect_net_deny_common_fields(struct __test_metadata *const _metadata,
+ const char *const buf,
+ const char *const expected_blockers)
+{
+ char field[64];
+
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK),
+ "domain", field, sizeof(field)));
+ EXPECT_STRNE("0", field);
+
+ /* Same exec that restricted itself, no exec in between. */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK),
+ "same_exec", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+
+ /* Default flags, same exec: audit would log this denial. */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK),
+ "blockers", field, sizeof(field)));
+ EXPECT_STREQ(expected_blockers, field);
+}
+
+enum trace_net_operation {
+ TRACE_NET_BIND,
+ TRACE_NET_SEND,
+};
+
+/* clang-format off */
+FIXTURE(trace_net_address) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_net_address)
+{
+ int ret;
+
+ /* Isolate the network namespace so the bound port cannot collide. */
+ setup_loopback(_metadata);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_net_address)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_net_address) {
+ /* clang-format on */
+ int socket_family;
+ int socket_type;
+ enum trace_net_operation operation;
+ int address_family;
+ socklen_t addrlen;
+ __u64 handled_access;
+ const char *expected_blockers;
+ bool address_port_zero;
+ bool expected_address_port;
+ int expected_port;
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net_address, ipv4_tcp_bind) {
+ /* clang-format on */
+ .socket_family = AF_INET,
+ .socket_type = SOCK_STREAM,
+ .operation = TRACE_NET_BIND,
+ .address_family = AF_INET,
+ .addrlen = sizeof(struct sockaddr_in),
+ .handled_access = LANDLOCK_ACCESS_NET_BIND_TCP,
+ .expected_blockers = "bind_tcp",
+ .expected_address_port = true,
+};
+
+/* Explicit bind(0) has a checked zero port. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net_address, ipv4_udp_bind_zero) {
+ /* clang-format on */
+ .socket_family = AF_INET,
+ .socket_type = SOCK_DGRAM,
+ .operation = TRACE_NET_BIND,
+ .address_family = AF_INET,
+ .addrlen = sizeof(struct sockaddr_in),
+ .handled_access = LANDLOCK_ACCESS_NET_BIND_UDP,
+ .expected_blockers = "bind_udp",
+ .address_port_zero = true,
+ .expected_port = 0,
+};
+
+/* A UDP send can deny its synthetic unspecified bind endpoint. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net_address, ipv6_udp_autobind) {
+ /* clang-format on */
+ .socket_family = AF_INET6,
+ .socket_type = SOCK_DGRAM,
+ .operation = TRACE_NET_SEND,
+ .address_family = AF_INET6,
+ .addrlen = sizeof(struct sockaddr_in6),
+ .handled_access = LANDLOCK_ACCESS_NET_BIND_UDP,
+ .expected_blockers = "bind_udp",
+ .expected_port = 0,
+};
+
+/* A family-only address has no checked port. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_net_address, ipv6_unspec_udp_send_min) {
+ /* clang-format on */
+ .socket_family = AF_INET6,
+ .socket_type = SOCK_DGRAM,
+ .operation = TRACE_NET_SEND,
+ .address_family = AF_UNSPEC,
+ .addrlen = sizeof(sa_family_t),
+ .handled_access = LANDLOCK_ACCESS_NET_CONNECT_SEND_UDP,
+ .expected_blockers = "connect_send_udp",
+ .expected_port = -1,
+};
+
+static void set_trace_net_address(struct sockaddr_storage *const storage,
+ const int socket_family,
+ const int address_family,
+ const unsigned short port)
+{
+ memset(storage, 0, sizeof(*storage));
+
+ if (socket_family == AF_INET) {
+ struct sockaddr_in *const addr4 = (struct sockaddr_in *)storage;
+
+ addr4->sin_family = address_family;
+ addr4->sin_port = htons(port);
+ addr4->sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ } else {
+ struct sockaddr_in6 *const addr6 =
+ (struct sockaddr_in6 *)storage;
+
+ addr6->sin6_family = address_family;
+ addr6->sin6_port = htons(port);
+ addr6->sin6_addr = in6addr_loopback;
+ }
+}
+
+/* Verifies the actionable signed port for representative checked shapes. */
+TEST_F(trace_net_address, deny_access_net)
+{
+ const char *const event_regex = REGEX_DENY_ACCESS_NET(TRACE_TASK);
+ const unsigned short address_port =
+ variant->address_port_zero ? 0 : sock_port_start + 1;
+ const int expected_port = variant->expected_address_port ?
+ address_port :
+ variant->expected_port;
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_net = variant->handled_access,
+ };
+ struct sockaddr_storage address;
+ char field[64], expected[16];
+ char *buf;
+ int count, ret, ruleset_fd, socket_fd, status;
+ pid_t child;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ set_trace_net_address(&address, variant->socket_family,
+ variant->address_family, address_port);
+ socket_fd = socket(variant->socket_family,
+ variant->socket_type | SOCK_CLOEXEC, 0);
+ ASSERT_LE(0, socket_fd);
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ child = fork();
+ ASSERT_LE(0, child);
+ if (child == 0) {
+ if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0))
+ _exit(1);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(2);
+ close(ruleset_fd);
+
+ switch (variant->operation) {
+ case TRACE_NET_BIND:
+ ret = bind(socket_fd, (const struct sockaddr *)&address,
+ variant->addrlen);
+ break;
+ case TRACE_NET_SEND:
+ ret = sendto(socket_fd, "A", 1, MSG_NOSIGNAL,
+ (const struct sockaddr *)&address,
+ variant->addrlen);
+ break;
+ default:
+ _exit(3);
+ }
+ if (ret >= 0 || errno != EACCES)
+ _exit(4);
+ close(socket_fd);
+
+ _exit(0);
+ }
+ close(ruleset_fd);
+ close(socket_fd);
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+ count = tracefs_count_matches(buf, event_regex);
+ if (count != 1)
+ TH_LOG("Expected 1 denial event, got %d\n%s", count, buf);
+ ASSERT_EQ(1, count);
+ expect_net_deny_common_fields(_metadata, buf,
+ variant->expected_blockers);
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, event_regex, "port", field,
+ sizeof(field)));
+ snprintf(expected, sizeof(expected), "%d", expected_port);
+ EXPECT_STREQ(expected, field);
+
+ free(buf);
+}
+
+/* Field verification for the check_rule_net_port event on an allowed access. */
+
+/* clang-format off */
+FIXTURE(trace_net_check_rule) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_net_check_rule)
+{
+ int ret;
+
+ /* Isolate the network namespace so the bound port cannot collide. */
+ setup_loopback(_metadata);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_net_check_rule)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * Verifies that an allowed bind matching a net-port rule emits exactly one
+ * landlock_check_rule_net_port event with the enforcing domain, the requested
+ * access, the checked port (host endianness), and the per-layer grants. The
+ * whole event is anchored to exact values so removing the check_rule_net_port
+ * emission or introducing a byte-order or field-plumbing regression fails the
+ * test.
+ */
+TEST_F(trace_net_check_rule, check_rule_net_fields)
+{
+ pid_t child;
+ int status;
+ char *buf;
+ char field[64], expected[16];
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_net = LANDLOCK_ACCESS_NET_BIND_TCP,
+ };
+ struct landlock_net_port_attr port_attr = {
+ .allowed_access = LANDLOCK_ACCESS_NET_BIND_TCP,
+ .port = sock_port_start,
+ };
+ struct sockaddr_in addr = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ };
+ int ruleset_fd, sock_fd;
+
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NET_PORT,
+ &port_attr, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+
+ /* Bind to the allowed port: succeeds and matches the rule. */
+ sock_fd = socket(AF_INET, SOCK_STREAM | SOCK_CLOEXEC, 0);
+ if (sock_fd < 0)
+ _exit(1);
+
+ addr.sin_port = htons(sock_port_start);
+ if (bind(sock_fd, (struct sockaddr *)&addr, sizeof(addr))) {
+ close(sock_fd);
+ _exit(2);
+ }
+ close(sock_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* A single-layer domain matching one port rule emits one event. */
+ EXPECT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CHECK_RULE_NET(TRACE_TASK)))
+ {
+ TH_LOG("Expected 1 check_rule_net_port event\n%s", buf);
+ }
+
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CHECK_RULE_NET(TRACE_TASK),
+ "domain", field, sizeof(field)));
+ EXPECT_STRNE("0", field);
+
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_CHECK_RULE_NET(TRACE_TASK),
+ "access_request", field, sizeof(field)));
+ EXPECT_STREQ("bind_tcp", field);
+
+ /*
+ * The port is reported in host endianness (UAPI convention), so on
+ * little-endian htons(sock_port_start) would print a different value:
+ * the exact match also catches byte-order regressions.
+ */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CHECK_RULE_NET(TRACE_TASK),
+ "port", field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "%llu",
+ (unsigned long long)sock_port_start);
+ EXPECT_STREQ(expected, field);
+
+ /* One layer that fully grants the request: grants={bind_tcp}. */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CHECK_RULE_NET(TRACE_TASK),
+ "grants", field, sizeof(field)));
+ EXPECT_STREQ("{bind_tcp}", field);
+
+ free(buf);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/ptrace_test.c b/tools/testing/selftests/landlock/ptrace_test.c
index 4f64c90583cd..2644445a9d02 100644
--- a/tools/testing/selftests/landlock/ptrace_test.c
+++ b/tools/testing/selftests/landlock/ptrace_test.c
@@ -11,7 +11,9 @@
#include <errno.h>
#include <fcntl.h>
#include <linux/landlock.h>
+#include <sched.h>
#include <signal.h>
+#include <sys/mount.h>
#include <sys/prctl.h>
#include <sys/ptrace.h>
#include <sys/types.h>
@@ -20,6 +22,7 @@
#include "audit.h"
#include "common.h"
+#include "trace.h"
/* Copied from security/yama/yama_lsm.c */
#define YAMA_SCOPE_DISABLED 0
@@ -430,4 +433,403 @@ TEST_F(audit, trace)
EXPECT_EQ(0, records.domain);
}
+/* Trace tests */
+
+/* clang-format off */
+FIXTURE(trace_ptrace) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_ptrace)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_PTRACE_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_ptrace)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_PTRACE_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_ptrace)
+{
+ /* clang-format on */
+ bool sandbox;
+ bool sandbox_target;
+ int expect_denied;
+};
+
+/* Denied: sandboxed child ptraces unsandboxed parent (tracee_domain=0). */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_ptrace, denied) {
+ /* clang-format on */
+ .sandbox = true,
+ .sandbox_target = false,
+ .expect_denied = 1,
+};
+
+/*
+ * Denied: sandboxed child ptraces a sandboxed parent, so the tracee is in a
+ * domain and tracee_domain= is non-zero.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_ptrace, denied_scoped_target) {
+ /* clang-format on */
+ .sandbox = true,
+ .sandbox_target = true,
+ .expect_denied = 1,
+};
+
+/* Allowed: unsandboxed child uses PTRACE_TRACEME. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_ptrace, allowed) {
+ /* clang-format on */
+ .sandbox = false,
+ .sandbox_target = false,
+ .expect_denied = 0,
+};
+
+TEST_F(trace_ptrace, deny_ptrace)
+{
+ char *buf, field[64], expected_pid[16];
+ int count, status;
+ pid_t child, parent;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ parent = getpid();
+
+ /*
+ * Set a known comm so the denied variant can verify both the trace line
+ * task name and the tracee_comm= field.
+ */
+ prctl(PR_SET_NAME, "ll_trace_test");
+
+ /*
+ * For the non-zero tracee_domain case, sandbox the parent (the tracee)
+ * before forking. The child inherits that domain and adds its own
+ * layer, so the child (tracer) is not an ancestor of the tracee and the
+ * ptrace is still denied, with tracee_domain= naming the parent's
+ * domain.
+ */
+ if (variant->sandbox_target)
+ create_domain(_metadata);
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ if (variant->sandbox) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .scoped = LANDLOCK_SCOPE_SIGNAL,
+ };
+ int ruleset_fd;
+
+ /*
+ * Any scope creates a domain. Ptrace denial checks
+ * domain ancestry, not specific flags.
+ */
+ ruleset_fd = landlock_create_ruleset(
+ &ruleset_attr, sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+
+ /* PTRACE_ATTACH on unsandboxed parent: denied. */
+ if (ptrace(PTRACE_ATTACH, parent, NULL, NULL) == 0) {
+ ptrace(PTRACE_DETACH, parent, NULL, NULL);
+ _exit(2);
+ }
+ if (errno != EPERM)
+ _exit(3);
+ } else {
+ /* No sandbox: ptrace should succeed. */
+ if (ptrace(PTRACE_TRACEME) != 0)
+ _exit(1);
+ }
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_PTRACE("ll_trace_test"));
+ if (variant->expect_denied) {
+ EXPECT_EQ(variant->expect_denied, count)
+ {
+ TH_LOG("Expected deny_ptrace event, got %d\n%s", count,
+ buf);
+ }
+
+ /* Verify tracee_pid is the parent's TGID. */
+ snprintf(expected_pid, sizeof(expected_pid), "%d", parent);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PTRACE("ll_trace_test"),
+ "tracee_pid", field, sizeof(field)));
+ EXPECT_STREQ(expected_pid, field);
+
+ /* Verify tracee_comm matches prctl(PR_SET_NAME). */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PTRACE("ll_trace_test"),
+ "tracee_comm", field, sizeof(field)));
+ EXPECT_STREQ("ll_trace_test", field);
+
+ /*
+ * Verify tracee_domain: 0 when the tracee is unsandboxed,
+ * non-zero when the tracee is in a domain.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PTRACE("ll_trace_test"),
+ "tracee_domain", field, sizeof(field)));
+ EXPECT_EQ(variant->sandbox_target, strcmp("0", field) != 0)
+ {
+ TH_LOG("Unexpected tracee_domain=%s", field);
+ }
+ } else {
+ EXPECT_EQ(0, count)
+ {
+ TH_LOG("Expected 0 deny_ptrace events, got %d\n%s",
+ count, buf);
+ }
+ }
+
+ free(buf);
+}
+
+/* clang-format off */
+FIXTURE(trace_ptrace_traceme) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_ptrace_traceme)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_PTRACE_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_ptrace_traceme)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_PTRACE_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_ptrace_traceme)
+{
+ /* clang-format on */
+ bool sandbox_tracer;
+ bool sandbox_tracee;
+ int expect_denied;
+};
+
+/*
+ * Denied: a sandboxed tracer cannot trace the unsandboxed child that asked to
+ * be traced with PTRACE_TRACEME (tracee_domain=0).
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_ptrace_traceme, denied) {
+ /* clang-format on */
+ .sandbox_tracer = true,
+ .sandbox_tracee = false,
+ .expect_denied = 1,
+};
+
+/*
+ * Denied: a sandboxed child in its own domain asks to be traced by a tracer in
+ * an unrelated domain, so the tracee is in a domain and tracee_domain= is
+ * non-zero.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_ptrace_traceme, denied_scoped_tracee) {
+ /* clang-format on */
+ .sandbox_tracer = true,
+ .sandbox_tracee = true,
+ .expect_denied = 1,
+};
+
+/* Allowed: unsandboxed child uses PTRACE_TRACEME with an unsandboxed tracer. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_ptrace_traceme, allowed) {
+ /* clang-format on */
+ .sandbox_tracer = false,
+ .sandbox_tracee = false,
+ .expect_denied = 0,
+};
+
+TEST_F(trace_ptrace_traceme, deny_ptrace)
+{
+ char *buf, field[64], expected_pid[16];
+ int count, status, sync_pipe[2];
+ pid_t child;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ /*
+ * Set a known comm so the denied variant can verify both the trace line
+ * task name and the tracee_comm= field. The tracee is the current
+ * (child) task for PTRACE_TRACEME, so the child inherits this name.
+ */
+ prctl(PR_SET_NAME, "ll_trace_test");
+
+ ASSERT_EQ(0, pipe2(sync_pipe, O_CLOEXEC));
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ char c;
+
+ close(sync_pipe[1]);
+
+ /*
+ * The tracee is the current task; for the non-zero
+ * tracee_domain case it sandboxes itself in its own domain,
+ * unrelated to the tracer's domain, so PTRACE_TRACEME is still
+ * denied and tracee_domain= names the child's own domain.
+ */
+ if (variant->sandbox_tracee)
+ create_domain(_metadata);
+
+ /* Waits for the tracer (parent) to enter its domain, if any. */
+ if (read(sync_pipe[0], &c, 1) != 1)
+ _exit(1);
+ close(sync_pipe[0]);
+
+ if (variant->expect_denied) {
+ if (ptrace(PTRACE_TRACEME) == 0)
+ _exit(2);
+ if (errno != EPERM)
+ _exit(3);
+ } else {
+ if (ptrace(PTRACE_TRACEME) != 0)
+ _exit(4);
+ /* Lets the tracer reap the trace-stop and detach. */
+ raise(SIGSTOP);
+ }
+
+ _exit(0);
+ }
+
+ close(sync_pipe[0]);
+
+ /*
+ * For a denial, the proposed tracer must be in a domain that is not an
+ * ancestor of the tracee's domain. Sandboxing the parent after the
+ * fork gives it a domain unrelated to the child.
+ */
+ if (variant->sandbox_tracer)
+ create_domain(_metadata);
+
+ /* Signals the child that the tracer is in its domain, if any. */
+ ASSERT_EQ(1, write(sync_pipe[1], ".", 1));
+ close(sync_pipe[1]);
+
+ if (!variant->expect_denied) {
+ /* PTRACE_TRACEME succeeded: reap the SIGSTOP and detach. */
+ ASSERT_EQ(child, waitpid(child, &status, WUNTRACED));
+ ASSERT_TRUE(WIFSTOPPED(status));
+ ASSERT_EQ(0, ptrace(PTRACE_DETACH, child, NULL, 0));
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_PTRACE("ll_trace_test"));
+ if (variant->expect_denied) {
+ EXPECT_EQ(variant->expect_denied, count)
+ {
+ TH_LOG("Expected deny_ptrace event, got %d\n%s", count,
+ buf);
+ }
+
+ /* Verify tracee_pid is the child's TGID (the traced task). */
+ snprintf(expected_pid, sizeof(expected_pid), "%d", child);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PTRACE("ll_trace_test"),
+ "tracee_pid", field, sizeof(field)));
+ EXPECT_STREQ(expected_pid, field);
+
+ /*
+ * Verify tracee_domain: 0 when the tracee is unsandboxed,
+ * non-zero when the tracee is in a domain.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PTRACE("ll_trace_test"),
+ "tracee_domain", field, sizeof(field)));
+ EXPECT_EQ(variant->sandbox_tracee, strcmp("0", field) != 0)
+ {
+ TH_LOG("Unexpected tracee_domain=%s", field);
+ }
+ } else {
+ EXPECT_EQ(0, count)
+ {
+ TH_LOG("Expected 0 deny_ptrace events, got %d\n%s",
+ count, buf);
+ }
+ }
+
+ free(buf);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/scoped_abstract_unix_test.c b/tools/testing/selftests/landlock/scoped_abstract_unix_test.c
index 40fc82fbf01d..5dc0debacb2a 100644
--- a/tools/testing/selftests/landlock/scoped_abstract_unix_test.c
+++ b/tools/testing/selftests/landlock/scoped_abstract_unix_test.c
@@ -12,6 +12,7 @@
#include <sched.h>
#include <signal.h>
#include <stddef.h>
+#include <sys/mount.h>
#include <sys/prctl.h>
#include <sys/socket.h>
#include <sys/stat.h>
@@ -23,6 +24,9 @@
#include "audit.h"
#include "common.h"
#include "scoped_common.h"
+#include "trace.h"
+
+#define TRACE_TASK "scoped_abstract"
/* Number of pending connections queue to be hold. */
const short backlog = 10;
@@ -1205,4 +1209,296 @@ TEST(self_connect)
_metadata->exit_code = KSFT_FAIL;
}
+/* Trace tests */
+
+/* clang-format off */
+FIXTURE(trace_unix) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_unix)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS | CLONE_NEWNET));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(
+ TRACEFS_DENY_SCOPE_ABSTRACT_UNIX_SOCKET_ENABLE,
+ true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_unix)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_SCOPE_ABSTRACT_UNIX_SOCKET_ENABLE,
+ false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+static const char
+ trace_unix_max_name[sizeof(((struct sockaddr_un *)0)->sun_path)] = {
+ [0 ... sizeof(trace_unix_max_name) - 2] = 'x',
+ };
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_unix) {
+ /* clang-format on */
+ int sock_type; /* SOCK_STREAM (connect) or SOCK_DGRAM (sendto). */
+ bool sandbox;
+ bool sandbox_target; /* Peer owned by a domain: peer_domain != 0. */
+ int expect_denied;
+ const char *name; /* NULL generates a PID-based binary name. */
+ size_t name_len;
+};
+
+/* clang-format off */
+
+/* Stream: sandboxed client connect() to an unsandboxed peer (peer_domain=0). */
+FIXTURE_VARIANT_ADD(trace_unix, stream_denied) {
+ .sock_type = SOCK_STREAM, .sandbox = true,
+ .sandbox_target = false, .expect_denied = 1,
+};
+
+/* Stream: peer socket owned by a domain, so peer_domain != 0. */
+FIXTURE_VARIANT_ADD(trace_unix, stream_denied_scoped_peer) {
+ .sock_type = SOCK_STREAM, .sandbox = true,
+ .sandbox_target = true, .expect_denied = 1,
+};
+
+/* Stream: unsandboxed client, connect() succeeds, no event. */
+FIXTURE_VARIANT_ADD(trace_unix, stream_allowed) {
+ .sock_type = SOCK_STREAM, .sandbox = false,
+ .sandbox_target = false, .expect_denied = 0,
+};
+
+/* Stream: lower abstract-name length boundary. */
+FIXTURE_VARIANT_ADD(trace_unix, stream_denied_empty_name) {
+ .sock_type = SOCK_STREAM,
+ .sandbox = true,
+ .sandbox_target = false,
+ .expect_denied = 1,
+ .name = "",
+ .name_len = 0,
+};
+
+/* Stream: upper abstract-name length boundary. */
+FIXTURE_VARIANT_ADD(trace_unix, stream_denied_max_name) {
+ .sock_type = SOCK_STREAM,
+ .sandbox = true,
+ .sandbox_target = false,
+ .expect_denied = 1,
+ .name = trace_unix_max_name,
+ .name_len = sizeof(trace_unix_max_name) - 1,
+};
+
+/* Datagram: sandboxed client sendto() an unsandboxed peer (peer_domain=0). */
+FIXTURE_VARIANT_ADD(trace_unix, dgram_denied) {
+ .sock_type = SOCK_DGRAM, .sandbox = true,
+ .sandbox_target = false, .expect_denied = 1,
+};
+
+/* Datagram: peer socket owned by a domain, so peer_domain != 0. */
+FIXTURE_VARIANT_ADD(trace_unix, dgram_denied_scoped_peer) {
+ .sock_type = SOCK_DGRAM, .sandbox = true,
+ .sandbox_target = true, .expect_denied = 1,
+};
+
+/* Datagram: unsandboxed client, sendto() succeeds, no event. */
+FIXTURE_VARIANT_ADD(trace_unix, dgram_allowed) {
+ .sock_type = SOCK_DGRAM, .sandbox = false,
+ .sandbox_target = false, .expect_denied = 0,
+};
+
+/* clang-format on */
+
+/*
+ * A sandboxed thread reaching an abstract unix socket peer through connect(2)
+ * (stream) or sendto(2) (datagram) is denied and emits
+ * landlock_deny_scope_abstract_unix_socket. The default abstract name has a
+ * space and an embedded NUL followed by an "END" marker to check escaping and
+ * binary length handling. Additional stream variants cover the minimum and
+ * maximum abstract-name lengths. peer_pid is only meaningful for a stream peer
+ * (a datagram peer has no SO_PEERCRED), so it is asserted only there.
+ */
+TEST_F(trace_unix, deny_scope_unix)
+{
+ struct sockaddr_un addr = {
+ .sun_family = AF_UNIX,
+ };
+ char *buf, field[128], expected_pid[16];
+ int server_fd, count, status, name_len, addr_len;
+ pid_t child;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ /*
+ * For the non-zero peer_domain case, sandbox the parent before it
+ * creates the server socket, so the socket carries the parent's domain
+ * and peer_domain= is non-zero.
+ */
+ if (variant->sandbox_target)
+ create_scoped_domain(_metadata,
+ LANDLOCK_SCOPE_ABSTRACT_UNIX_SOCKET);
+
+ server_fd = socket(AF_UNIX, variant->sock_type | SOCK_CLOEXEC, 0);
+ ASSERT_LE(0, server_fd);
+
+ addr.sun_path[0] = '\0';
+ if (variant->name) {
+ ASSERT_LE(variant->name_len, sizeof(addr.sun_path) - 1);
+ memcpy(addr.sun_path + 1, variant->name, variant->name_len);
+ name_len = variant->name_len;
+ } else {
+ name_len = snprintf(addr.sun_path + 1,
+ sizeof(addr.sun_path) - 1,
+ "landlock_trace_test_%d ", getpid());
+ addr.sun_path[1 + name_len] = '\0';
+ memcpy(addr.sun_path + 1 + name_len + 1, "END", 3);
+ name_len += 1 + 3;
+ }
+ addr_len = offsetof(struct sockaddr_un, sun_path) + 1 + name_len;
+
+ ASSERT_EQ(0, bind(server_fd, (struct sockaddr *)&addr, addr_len));
+ if (variant->sock_type == SOCK_STREAM)
+ ASSERT_EQ(0, listen(server_fd, 1));
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ int client_fd, ret;
+
+ if (variant->sandbox) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .scoped = LANDLOCK_SCOPE_ABSTRACT_UNIX_SOCKET,
+ };
+ int ruleset_fd;
+
+ ruleset_fd = landlock_create_ruleset(
+ &ruleset_attr, sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+ }
+
+ client_fd =
+ socket(AF_UNIX, variant->sock_type | SOCK_CLOEXEC, 0);
+ if (client_fd < 0)
+ _exit(1);
+
+ if (variant->sock_type == SOCK_STREAM)
+ ret = connect(client_fd, (struct sockaddr *)&addr,
+ addr_len);
+ else
+ ret = sendto(client_fd, ".", 1, 0,
+ (struct sockaddr *)&addr, addr_len);
+
+ if (variant->sandbox) {
+ /* Reaching the peer should be denied. */
+ if (ret != -1 || errno != EPERM) {
+ close(client_fd);
+ _exit(2);
+ }
+ } else {
+ /* No sandbox: stream connect() == 0, sendto() == 1. */
+ int ok = variant->sock_type == SOCK_STREAM ? 0 : 1;
+
+ if (ret != ok) {
+ close(client_fd);
+ _exit(2);
+ }
+ }
+ close(client_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+ close(server_fd);
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(
+ buf, REGEX_DENY_SCOPE_ABSTRACT_UNIX_SOCKET(TRACE_TASK));
+ if (!variant->expect_denied) {
+ EXPECT_EQ(0, count)
+ {
+ TH_LOG("Expected 0 deny_scope events, got %d\n%s",
+ count, buf);
+ }
+ free(buf);
+ return;
+ }
+
+ EXPECT_EQ(variant->expect_denied, count)
+ {
+ TH_LOG("Expected deny_scope_abstract_unix_socket event, "
+ "got %d\n%s",
+ count, buf);
+ }
+
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf,
+ REGEX_DENY_SCOPE_ABSTRACT_UNIX_SOCKET(TRACE_TASK),
+ "sun_path", field, sizeof(field)));
+ if (variant->name) {
+ EXPECT_STREQ(variant->name, field);
+ } else {
+ /* An embedded NUL must not truncate the following marker. */
+ EXPECT_NE(NULL, strstr(field, "END"))
+ {
+ TH_LOG("sun_path truncated or unescaped: %s", field);
+ }
+ }
+
+ /* peer_pid is the parent's PID for a stream peer (0 for datagram). */
+ if (variant->sock_type == SOCK_STREAM) {
+ snprintf(expected_pid, sizeof(expected_pid), "%d", getpid());
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf,
+ REGEX_DENY_SCOPE_ABSTRACT_UNIX_SOCKET(
+ TRACE_TASK),
+ "peer_pid", field, sizeof(field)));
+ EXPECT_STREQ(expected_pid, field);
+ }
+
+ /* peer_domain: 0 when the peer is unsandboxed, non-zero otherwise. */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf,
+ REGEX_DENY_SCOPE_ABSTRACT_UNIX_SOCKET(TRACE_TASK),
+ "peer_domain", field, sizeof(field)));
+ EXPECT_EQ(variant->sandbox_target, strcmp("0", field) != 0)
+ {
+ TH_LOG("Unexpected peer_domain=%s", field);
+ }
+
+ free(buf);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/scoped_signal_test.c b/tools/testing/selftests/landlock/scoped_signal_test.c
index 2d37d0c06c06..259cdcc8aa5c 100644
--- a/tools/testing/selftests/landlock/scoped_signal_test.c
+++ b/tools/testing/selftests/landlock/scoped_signal_test.c
@@ -10,7 +10,9 @@
#include <fcntl.h>
#include <linux/landlock.h>
#include <pthread.h>
+#include <sched.h>
#include <signal.h>
+#include <sys/mount.h>
#include <sys/prctl.h>
#include <sys/types.h>
#include <sys/wait.h>
@@ -18,6 +20,9 @@
#include "common.h"
#include "scoped_common.h"
+#include "trace.h"
+
+#define TRACE_TASK "scoped_signal_t"
/* This variable is used for handling several signals. */
static volatile sig_atomic_t is_signaled;
@@ -762,4 +767,403 @@ TEST(sigio_to_pgid_self)
EXPECT_EQ(0, close(trigger[1]));
}
+/* Trace tests */
+
+/* clang-format off */
+FIXTURE(trace_signal) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_signal)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_SCOPE_SIGNAL_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_signal)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_SCOPE_SIGNAL_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_signal)
+{
+ /* clang-format on */
+ bool sandbox;
+ bool sandbox_target;
+ int expect_denied;
+};
+
+/* Denied: sandboxed child signals unsandboxed parent (target_domain=0). */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_signal, denied) {
+ /* clang-format on */
+ .sandbox = true,
+ .sandbox_target = false,
+ .expect_denied = 1,
+};
+
+/*
+ * Denied: sandboxed child signals a sandboxed parent, so the target is in a
+ * domain and target_domain= is non-zero.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_signal, denied_scoped_target) {
+ /* clang-format on */
+ .sandbox = true,
+ .sandbox_target = true,
+ .expect_denied = 1,
+};
+
+/* Allowed: unsandboxed child signals unsandboxed parent. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_signal, allowed) {
+ /* clang-format on */
+ .sandbox = false,
+ .sandbox_target = false,
+ .expect_denied = 0,
+};
+
+TEST_F(trace_signal, deny_scope_signal)
+{
+ char *buf, field[64], expected_pid[16];
+ int count, status;
+ pid_t child;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ /*
+ * For the non-zero target_domain case, sandbox the parent (the signal
+ * target) before forking. The child inherits that domain and adds its
+ * own scoped layer, so the signal is still denied and target_domain=
+ * names the parent's domain.
+ */
+ if (variant->sandbox_target)
+ create_scoped_domain(_metadata, LANDLOCK_SCOPE_SIGNAL);
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ if (variant->sandbox) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .scoped = LANDLOCK_SCOPE_SIGNAL,
+ };
+ int ruleset_fd;
+
+ ruleset_fd = landlock_create_ruleset(
+ &ruleset_attr, sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+ }
+
+ if (variant->sandbox) {
+ /* Signal to unsandboxed parent should be denied. */
+ if (kill(getppid(), 0) == 0)
+ _exit(2);
+ if (errno != EPERM)
+ _exit(3);
+ } else {
+ /* No sandbox: kill should succeed. */
+ if (kill(getppid(), 0) != 0)
+ _exit(1);
+ }
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_SCOPE_SIGNAL(TRACE_TASK));
+ if (variant->expect_denied) {
+ EXPECT_EQ(variant->expect_denied, count)
+ {
+ TH_LOG("Expected deny_scope_signal event, got %d\n%s",
+ count, buf);
+ }
+
+ /* Verify target_pid is the parent's PID. */
+ snprintf(expected_pid, sizeof(expected_pid), "%d", getpid());
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_SCOPE_SIGNAL(TRACE_TASK),
+ "target_pid", field, sizeof(field)));
+ EXPECT_STREQ(expected_pid, field);
+
+ /*
+ * Verify target_domain: 0 when the target is unsandboxed,
+ * non-zero when the target is in a domain.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_SCOPE_SIGNAL(TRACE_TASK),
+ "target_domain", field, sizeof(field)));
+ EXPECT_EQ(variant->sandbox_target, strcmp("0", field) != 0)
+ {
+ TH_LOG("Unexpected target_domain=%s", field);
+ }
+ } else {
+ EXPECT_EQ(0, count)
+ {
+ TH_LOG("Expected 0 deny_scope_signal events, "
+ "got %d\n%s",
+ count, buf);
+ }
+ }
+
+ free(buf);
+}
+
+/*
+ * Trace test for the asynchronous SIGIO/SIGURG delivery path
+ * (hook_file_send_sigiotask), which reaches the same landlock_deny_scope_signal
+ * tracepoint as a synchronous kill(2) but through fcntl(F_SETOWN).
+ */
+
+/* clang-format off */
+FIXTURE(trace_fown) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_fown)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_SCOPE_SIGNAL_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_fown)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_DENY_SCOPE_SIGNAL_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_fown)
+{
+ /* clang-format on */
+ bool sandbox;
+ bool sandbox_target;
+ int expect_denied;
+};
+
+/*
+ * Denied: a sandboxed file owner's SIGURG is delivered to an unsandboxed target
+ * process (target_domain=0).
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_fown, denied) {
+ /* clang-format on */
+ .sandbox = true,
+ .sandbox_target = false,
+ .expect_denied = 1,
+};
+
+/*
+ * Denied: the SIGURG target sandboxes itself in its own domain, so the target
+ * is in a domain and target_domain= is non-zero.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_fown, denied_scoped_target) {
+ /* clang-format on */
+ .sandbox = true,
+ .sandbox_target = true,
+ .expect_denied = 1,
+};
+
+/* Allowed: an unsandboxed file owner delivers SIGURG. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(trace_fown, allowed) {
+ /* clang-format on */
+ .sandbox = false,
+ .sandbox_target = false,
+ .expect_denied = 0,
+};
+
+TEST_F(trace_fown, deny_scope_fown)
+{
+ int server_socket, recv_socket;
+ struct service_fixture server_address;
+ char buffer_parent, field[64], *buf;
+ int status, count;
+ int pipe_parent[2], pipe_child[2];
+ pid_t child;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ memset(&server_address, 0, sizeof(server_address));
+ set_unix_address(&server_address, 0);
+
+ ASSERT_EQ(0, pipe2(pipe_parent, O_CLOEXEC));
+ ASSERT_EQ(0, pipe2(pipe_child, O_CLOEXEC));
+
+ child = fork();
+ ASSERT_LE(0, child);
+ if (child == 0) {
+ int client_socket;
+ char buffer_child;
+
+ EXPECT_EQ(0, close(pipe_parent[1]));
+ EXPECT_EQ(0, close(pipe_child[0]));
+
+ ASSERT_EQ(0, setup_signal_handler(SIGURG));
+ client_socket = socket(AF_UNIX, SOCK_STREAM, 0);
+ ASSERT_LE(0, client_socket);
+
+ /*
+ * The SIGURG target is this child; for the non-zero
+ * target_domain case it sandboxes itself in its own domain,
+ * unrelated to the file owner's domain.
+ */
+ if (variant->sandbox_target)
+ create_scoped_domain(_metadata, LANDLOCK_SCOPE_SIGNAL);
+
+ /* Waits for the parent to listen. */
+ ASSERT_EQ(1, read(pipe_parent[0], &buffer_child, 1));
+ ASSERT_EQ(0, connect(client_socket, &server_address.unix_addr,
+ server_address.unix_addr_len));
+
+ /*
+ * Waits for the parent to accept the connection, sandbox
+ * itself, and call fcntl(F_SETOWN).
+ */
+ ASSERT_EQ(1, read(pipe_parent[0], &buffer_child, 1));
+ /* Triggers the asynchronous SIGURG to this file owner. */
+ ASSERT_EQ(1, send(client_socket, ".", 1, MSG_OOB));
+ EXPECT_EQ(0, close(client_socket));
+ ASSERT_EQ(1, write(pipe_child[1], ".", 1));
+ EXPECT_EQ(0, close(pipe_child[1]));
+
+ _exit(0);
+ return;
+ }
+ EXPECT_EQ(0, close(pipe_parent[0]));
+ EXPECT_EQ(0, close(pipe_child[1]));
+
+ server_socket = socket(AF_UNIX, SOCK_STREAM, 0);
+ ASSERT_LE(0, server_socket);
+ ASSERT_EQ(0, bind(server_socket, &server_address.unix_addr,
+ server_address.unix_addr_len));
+ ASSERT_EQ(0, listen(server_socket, backlog));
+ ASSERT_EQ(1, write(pipe_parent[1], ".", 1));
+
+ recv_socket = accept(server_socket, NULL, NULL);
+ ASSERT_LE(0, recv_socket);
+
+ /*
+ * The file owner is the denying subject; its domain is captured at
+ * fcntl(F_SETOWN) time, so sandbox it before setting the owner.
+ */
+ if (variant->sandbox)
+ create_scoped_domain(_metadata, LANDLOCK_SCOPE_SIGNAL);
+
+ /*
+ * Sets the child to receive SIGURG for MSG_OOB. This uncommon use is a
+ * valid attack scenario which also simplifies this test.
+ */
+ ASSERT_EQ(0, fcntl(recv_socket, F_SETOWN, child));
+
+ ASSERT_EQ(1, write(pipe_parent[1], ".", 1));
+
+ /* Waits for the child to send MSG_OOB. */
+ ASSERT_EQ(1, read(pipe_child[0], &buffer_parent, 1));
+ EXPECT_EQ(0, close(pipe_child[0]));
+ ASSERT_EQ(1, recv(recv_socket, &buffer_parent, 1, MSG_OOB));
+ EXPECT_EQ(0, close(recv_socket));
+ EXPECT_EQ(0, close(server_socket));
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_SCOPE_SIGNAL(TRACE_TASK));
+ if (variant->expect_denied) {
+ EXPECT_EQ(variant->expect_denied, count)
+ {
+ TH_LOG("Expected deny_scope_signal event, got %d\n%s",
+ count, buf);
+ }
+
+ /*
+ * Verify target_domain: 0 when the target is unsandboxed,
+ * non-zero when the target is in a domain.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_SCOPE_SIGNAL(TRACE_TASK),
+ "target_domain", field, sizeof(field)));
+ EXPECT_EQ(variant->sandbox_target, strcmp("0", field) != 0)
+ {
+ TH_LOG("Unexpected target_domain=%s", field);
+ }
+ } else {
+ EXPECT_EQ(0, count)
+ {
+ TH_LOG("Expected 0 deny_scope_signal events, "
+ "got %d\n%s",
+ count, buf);
+ }
+ }
+
+ free(buf);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/trace.h b/tools/testing/selftests/landlock/trace.h
new file mode 100644
index 000000000000..2ec863362173
--- /dev/null
+++ b/tools/testing/selftests/landlock/trace.h
@@ -0,0 +1,638 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Landlock trace test helpers
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <regex.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "kselftest_harness.h"
+
+#define TRACEFS_ROOT "/sys/kernel/tracing"
+#define TRACEFS_LANDLOCK_DIR TRACEFS_ROOT "/events/landlock"
+#define TRACEFS_CREATE_RULESET_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_create_ruleset/enable"
+#define TRACEFS_CREATE_DOMAIN_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_create_domain/enable"
+#define TRACEFS_ENFORCE_DOMAIN_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_enforce_domain/enable"
+#define TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_add_rule_path_beneath/enable"
+#define TRACEFS_ADD_RULE_NET_PORT_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_add_rule_net_port/enable"
+#define TRACEFS_CHECK_RULE_FS_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_check_rule_inode/enable"
+#define TRACEFS_CHECK_RULE_NET_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_check_rule_net_port/enable"
+#define TRACEFS_DENY_ACCESS_FS_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_deny_access_fs/enable"
+#define TRACEFS_DENY_ACCESS_NET_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_deny_access_net/enable"
+#define TRACEFS_DENY_PTRACE_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_deny_ptrace/enable"
+#define TRACEFS_DENY_SCOPE_SIGNAL_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_deny_scope_signal/enable"
+#define TRACEFS_DENY_SCOPE_ABSTRACT_UNIX_SOCKET_ENABLE \
+ TRACEFS_LANDLOCK_DIR \
+ "/landlock_deny_scope_abstract_unix_socket/enable"
+#define TRACEFS_FREE_DOMAIN_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_free_domain/enable"
+#define TRACEFS_FREE_RULESET_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_free_ruleset/enable"
+#define TRACEFS_TRACE TRACEFS_ROOT "/trace"
+#define TRACEFS_SET_EVENT_PID TRACEFS_ROOT "/set_event_pid"
+#define TRACEFS_OPTIONS_EVENT_FORK TRACEFS_ROOT "/options/event-fork"
+
+#define TRACE_BUFFER_SIZE (64 * 1024)
+
+/*
+ * Trace line prefix: matches the ftrace "trace" file format. Format: "
+ * <task>-<pid> [<cpu>] <flags> <timestamp>: "
+ *
+ * The task parameter must be a string literal truncated to 15 chars
+ * (TASK_COMM_LEN - 1), matching what the kernel stores in task->comm. The
+ * pattern accepts either the expected task name or "<...>" because the ftrace
+ * comm cache may evict short-lived processes (e.g., forked children that exit
+ * before the trace buffer is read).
+ *
+ * No unescaped '.' in any REGEX macro; literal dots use '\\.'.
+ */
+#define TRACE_PREFIX(task) \
+ "^ *\\(<\\.\\.\\.>" \
+ "\\|" task "\\)" \
+ "-[0-9]\\+ *\\[[0-9]\\+\\] [^ ]\\+ \\+[0-9]\\+\\.[0-9]\\+: "
+
+/*
+ * Task name for events emitted by kworker threads (e.g., free_domain fires from
+ * a work queue, not from the test process).
+ */
+#define KWORKER_TASK "kworker/[0-9]\\+:[0-9]\\+"
+
+#define REGEX_ADD_RULE_PATH_BENEATH(task) \
+ TRACE_PREFIX(task) \
+ "landlock_add_rule_path_beneath: " \
+ "ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \
+ "access_rights=[a-z_|]* " \
+ "dev=[0-9]\\+:[0-9]\\+ " \
+ "ino=[0-9]\\+ " \
+ "path=[^ ]\\+$"
+
+#define REGEX_ADD_RULE_NET_PORT(task) \
+ TRACE_PREFIX(task) \
+ "landlock_add_rule_net_port: " \
+ "ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \
+ "access_rights=[a-z_|]* " \
+ "port=[0-9]\\+$"
+
+#define REGEX_CREATE_RULESET(task) \
+ TRACE_PREFIX(task) \
+ "landlock_create_ruleset: " \
+ "ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \
+ "handled_fs=[a-z_|]* " \
+ "handled_net=[a-z_|]* " \
+ "scoped=[a-z_|]*$"
+
+#define REGEX_CREATE_DOMAIN(task) \
+ TRACE_PREFIX(task) \
+ "landlock_create_domain: " \
+ "domain=[0-9a-f]\\+ " \
+ "parent=[0-9a-f]\\+ " \
+ "ruleset=[0-9a-f]\\+\\.[0-9]\\+$"
+
+#define REGEX_CHECK_RULE_FS(task) \
+ TRACE_PREFIX(task) \
+ "landlock_check_rule_inode: " \
+ "domain=[0-9a-f]\\+ " \
+ "access_request=[a-z_|]* " \
+ "dev=[0-9]\\+:[0-9]\\+ " \
+ "ino=[0-9]\\+ " \
+ "grants={[a-z_|,]*}$"
+
+#define REGEX_CHECK_RULE_NET(task) \
+ TRACE_PREFIX(task) \
+ "landlock_check_rule_net_port: " \
+ "domain=[0-9a-f]\\+ " \
+ "access_request=[a-z_|]* " \
+ "port=[0-9]\\+ " \
+ "grants={[a-z_|,]*}$"
+
+#define REGEX_DENY_ACCESS_FS(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_access_fs: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "blockers=[a-z_|]* " \
+ "dev=[0-9]\\+:[0-9]\\+ " \
+ "ino=[0-9]\\+ " \
+ "path=[^ ]*$"
+
+#define REGEX_DENY_ACCESS_NET(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_access_net: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "blockers=[a-z_|]* " \
+ "port=-\\?[0-9]\\+$"
+
+#define REGEX_DENY_PTRACE(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_ptrace: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "tracee_domain=[0-9a-f]\\+ " \
+ "tracee_pid=[0-9]\\+ " \
+ "tracee_comm=[^ ]*$"
+
+#define REGEX_DENY_SCOPE_SIGNAL(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_scope_signal: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "target_domain=[0-9a-f]\\+ " \
+ "target_pid=[0-9]\\+ " \
+ "target_comm=[^ ]*$"
+
+#define REGEX_DENY_SCOPE_ABSTRACT_UNIX_SOCKET(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_scope_abstract_unix_socket: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "peer_domain=[0-9a-f]\\+ " \
+ "peer_pid=[0-9]\\+ " \
+ "sun_path=[^ ]*$"
+
+#define REGEX_FREE_DOMAIN(task) \
+ TRACE_PREFIX(task) \
+ "landlock_free_domain: " \
+ "domain=[0-9a-f]\\+ " \
+ "denials=[0-9]\\+$"
+
+#define REGEX_FREE_RULESET(task) \
+ TRACE_PREFIX(task) \
+ "landlock_free_ruleset: " \
+ "ruleset=[0-9a-f]\\+\\.[0-9]\\+$"
+
+static int __maybe_unused tracefs_write(const char *path, const char *value)
+{
+ int fd;
+ ssize_t ret;
+ size_t len = strlen(value);
+
+ fd = open(path, O_WRONLY | O_TRUNC | O_CLOEXEC);
+ if (fd < 0)
+ return -errno;
+
+ ret = write(fd, value, len);
+ close(fd);
+ if (ret < 0)
+ return -errno;
+ if ((size_t)ret != len)
+ return -EIO;
+
+ return 0;
+}
+
+static int __maybe_unused tracefs_write_int(const char *path, int value)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%d", value);
+ return tracefs_write(path, buf);
+}
+
+static int __maybe_unused tracefs_setup(void)
+{
+ struct stat st;
+
+ /* Mount tracefs if not already mounted. */
+ if (stat(TRACEFS_ROOT, &st) != 0) {
+ int ret = mount("tracefs", TRACEFS_ROOT, "tracefs", 0, NULL);
+
+ if (ret)
+ return -errno;
+ }
+
+ /* Verify landlock events are available. */
+ if (stat(TRACEFS_LANDLOCK_DIR, &st) != 0)
+ return -ENOENT;
+
+ return 0;
+}
+
+/*
+ * Set up PID-based event filtering so only events from the current process and
+ * its children are recorded. This is analogous to audit's AUDIT_EXE filter: it
+ * prevents events from unrelated processes from polluting the trace buffer.
+ */
+static int __maybe_unused tracefs_set_pid_filter(pid_t pid)
+{
+ int ret;
+
+ /* Enable event-fork so children inherit the PID filter. */
+ ret = tracefs_write(TRACEFS_OPTIONS_EVENT_FORK, "1");
+ if (ret)
+ return ret;
+
+ return tracefs_write_int(TRACEFS_SET_EVENT_PID, pid);
+}
+
+/* Clear the PID filter to stop filtering by PID. */
+static int __maybe_unused tracefs_clear_pid_filter(void)
+{
+ return tracefs_write(TRACEFS_SET_EVENT_PID, "");
+}
+
+static int __maybe_unused tracefs_enable_event(const char *enable_path,
+ bool enable)
+{
+ return tracefs_write(enable_path, enable ? "1" : "0");
+}
+
+static int __maybe_unused tracefs_clear(void)
+{
+ return tracefs_write(TRACEFS_TRACE, "");
+}
+
+/*
+ * Reads the trace buffer content into a newly allocated buffer. The caller is
+ * responsible for freeing the returned buffer. Returns NULL on error.
+ */
+static char __maybe_unused *tracefs_read_trace(void)
+{
+ char *buf;
+ int fd;
+ ssize_t total = 0, ret;
+
+ buf = malloc(TRACE_BUFFER_SIZE);
+ if (!buf)
+ return NULL;
+
+ fd = open(TRACEFS_TRACE, O_RDONLY | O_CLOEXEC);
+ if (fd < 0) {
+ free(buf);
+ return NULL;
+ }
+
+ while (total < TRACE_BUFFER_SIZE - 1) {
+ ret = read(fd, buf + total, TRACE_BUFFER_SIZE - 1 - total);
+ if (ret <= 0)
+ break;
+ total += ret;
+ }
+ close(fd);
+ buf[total] = '\0';
+ return buf;
+}
+
+/* Counts the number of lines in @buf matching the basic regex @pattern. */
+static int __maybe_unused tracefs_count_matches(const char *buf,
+ const char *pattern)
+{
+ regex_t regex;
+ int count = 0;
+ const char *line, *end;
+
+ if (regcomp(&regex, pattern, 0) != 0)
+ return -EINVAL;
+
+ line = buf;
+ while (*line) {
+ end = strchr(line, '\n');
+ if (!end)
+ end = line + strlen(line);
+
+ /* Create a temporary NUL-terminated line. */
+ size_t len = end - line;
+ char *tmp = malloc(len + 1);
+
+ if (tmp) {
+ memcpy(tmp, line, len);
+ tmp[len] = '\0';
+ if (regexec(&regex, tmp, 0, NULL, 0) == 0)
+ count++;
+ free(tmp);
+ }
+
+ if (*end == '\n')
+ line = end + 1;
+ else
+ break;
+ }
+
+ regfree(&regex);
+ return count;
+}
+
+/*
+ * Extracts the value of a named field from a trace line in @buf. Searches for
+ * the first line matching @line_pattern, then extracts the value after
+ * "@field_name=" into @out. Stops at space or newline.
+ *
+ * Returns 0 on success, -ENOENT if no match.
+ */
+static int __maybe_unused tracefs_extract_field(const char *buf,
+ const char *line_pattern,
+ const char *field_name,
+ char *out, size_t out_size)
+{
+ regex_t regex;
+ const char *line, *end;
+
+ if (regcomp(&regex, line_pattern, 0) != 0)
+ return -EINVAL;
+
+ line = buf;
+ while (*line) {
+ end = strchr(line, '\n');
+ if (!end)
+ end = line + strlen(line);
+
+ size_t len = end - line;
+ char *tmp = malloc(len + 1);
+
+ if (tmp) {
+ const char *field, *val_start;
+ size_t field_len, val_len;
+
+ memcpy(tmp, line, len);
+ tmp[len] = '\0';
+
+ if (regexec(&regex, tmp, 0, NULL, 0) != 0) {
+ free(tmp);
+ goto next;
+ }
+
+ /*
+ * Find "field_name=" in the line, ensuring a word
+ * boundary before the field name to avoid substring
+ * matches (e.g., "port" in "sport").
+ */
+ field_len = strlen(field_name);
+ field = tmp;
+ while ((field = strstr(field, field_name))) {
+ if (field[field_len] == '=' &&
+ (field == tmp || field[-1] == ' '))
+ break;
+ field++;
+ }
+ if (!field) {
+ free(tmp);
+ regfree(&regex);
+ return -ENOENT;
+ }
+
+ val_start = field + field_len + 1;
+ val_len = 0;
+ while (val_start[val_len] &&
+ val_start[val_len] != ' ' &&
+ val_start[val_len] != '\n')
+ val_len++;
+
+ if (val_len >= out_size)
+ val_len = out_size - 1;
+ memcpy(out, val_start, val_len);
+ out[val_len] = '\0';
+
+ free(tmp);
+ regfree(&regex);
+ return 0;
+ }
+next:
+ if (*end == '\n')
+ line = end + 1;
+ else
+ break;
+ }
+
+ regfree(&regex);
+ return -ENOENT;
+}
+
+/*
+ * Common fixture setup for trace tests. Mounts tracefs if needed and sets a
+ * PID filter. The caller must create a mount namespace first
+ * (unshare(CLONE_NEWNS) + mount(MS_REC | MS_PRIVATE)) to isolate the tracefs
+ * mount; the trace buffer, per-event enable flags, and PID filter are global
+ * kernel state, scoped to the test by the PID filter.
+ *
+ * Returns 0 on success, -errno on failure (caller should SKIP).
+ */
+static int __maybe_unused tracefs_fixture_setup(void)
+{
+ int ret;
+
+ ret = tracefs_setup();
+ if (ret)
+ return ret;
+
+ return tracefs_set_pid_filter(getpid());
+}
+
+static void __maybe_unused tracefs_fixture_teardown(void)
+{
+ tracefs_clear_pid_filter();
+}
+
+/*
+ * Temporarily raises CAP_SYS_ADMIN effective capability, calls @func, then
+ * drops the capability. Returns the value from @func, or -EPERM if the
+ * capability manipulation fails.
+ */
+static int __maybe_unused tracefs_priv_call(int (*func)(void))
+{
+ const cap_value_t admin = CAP_SYS_ADMIN;
+ cap_t cap_p;
+ int ret;
+
+ cap_p = cap_get_proc();
+ if (!cap_p)
+ return -EPERM;
+
+ if (cap_set_flag(cap_p, CAP_EFFECTIVE, 1, &admin, CAP_SET) ||
+ cap_set_proc(cap_p)) {
+ cap_free(cap_p);
+ return -EPERM;
+ }
+
+ ret = func();
+
+ cap_set_flag(cap_p, CAP_EFFECTIVE, 1, &admin, CAP_CLEAR);
+ cap_set_proc(cap_p);
+ cap_free(cap_p);
+ return ret;
+}
+
+/* Read the trace buffer with elevated privileges. Returns NULL on failure. */
+static char __maybe_unused *tracefs_read_buf(void)
+{
+ /* Cannot use tracefs_priv_call() because the return type is char *. */
+ cap_t cap_p;
+ char *buf;
+ const cap_value_t admin = CAP_SYS_ADMIN;
+
+ cap_p = cap_get_proc();
+ if (!cap_p)
+ return NULL;
+
+ if (cap_set_flag(cap_p, CAP_EFFECTIVE, 1, &admin, CAP_SET) ||
+ cap_set_proc(cap_p)) {
+ cap_free(cap_p);
+ return NULL;
+ }
+
+ buf = tracefs_read_trace();
+
+ cap_set_flag(cap_p, CAP_EFFECTIVE, 1, &admin, CAP_CLEAR);
+ cap_set_proc(cap_p);
+ cap_free(cap_p);
+ return buf;
+}
+
+/* Clear the trace buffer with elevated privileges. Returns 0 on success. */
+static int __maybe_unused tracefs_clear_buf(void)
+{
+ return tracefs_priv_call(tracefs_clear);
+}
+
+/*
+ * Forks a child that creates a Landlock sandbox and performs an FS access. The
+ * parent waits for the child, then reads the trace buffer.
+ *
+ * Requires common.h and wrappers.h to be included before trace.h.
+ */
+static void __maybe_unused sandbox_child_fs_access(
+ struct __test_metadata *const _metadata, const char *rule_path,
+ __u64 handled_access, __u64 allowed_access, const char *access_path)
+{
+ pid_t pid;
+ int status;
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = handled_access,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = allowed_access,
+ };
+ int ruleset_fd, fd;
+
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open(rule_path, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0)) {
+ close(path_beneath.parent_fd);
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+
+ fd = open(access_path, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+}
+
+/*
+ * Forks a child that creates a Landlock sandbox allowing execute+read_dir for
+ * /usr and execute-only for ".", then execs ./true. The true binary opens "."
+ * on startup, triggering a read_dir denial with same_exec=0. The parent waits
+ * for the child to exit.
+ */
+static void __maybe_unused sandbox_child_exec_true(
+ struct __test_metadata *const _metadata, __u32 restrict_flags)
+{
+ pid_t pid;
+ int status;
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR |
+ LANDLOCK_ACCESS_FS_EXECUTE,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_EXECUTE |
+ LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd;
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd >= 0) {
+ landlock_add_rule(ruleset_fd,
+ LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0);
+ close(path_beneath.parent_fd);
+ }
+
+ path_beneath.allowed_access = LANDLOCK_ACCESS_FS_EXECUTE;
+ path_beneath.parent_fd =
+ open(".", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd >= 0) {
+ landlock_add_rule(ruleset_fd,
+ LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0);
+ close(path_beneath.parent_fd);
+ }
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, restrict_flags))
+ _exit(1);
+ close(ruleset_fd);
+
+ execl("./true", "./true", NULL);
+ _exit(1);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+}
diff --git a/tools/testing/selftests/landlock/trace_fs_test.c b/tools/testing/selftests/landlock/trace_fs_test.c
new file mode 100644
index 000000000000..64014ade3a0e
--- /dev/null
+++ b/tools/testing/selftests/landlock/trace_fs_test.c
@@ -0,0 +1,732 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Landlock tests - Filesystem tracepoints
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#define _GNU_SOURCE
+#include <assert.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <linux/landlock.h>
+#include <sched.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/types.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include "common.h"
+#include "trace.h"
+
+#define TRACE_TASK "trace_fs_test"
+
+/* Mirrors TRACE_SEQ_SIZE, conservatively larger than the usable buffer. */
+#define TRACE_SEQUENCE_SIZE 8192
+#define OCTAL_ESCAPE_LEN 4
+#define LONG_PATH_COMPONENT_COUNT 11
+#define LONG_PATH_COMPONENT_LEN 240
+#define LONG_PATH_LEN \
+ (LONG_PATH_COMPONENT_COUNT * (LONG_PATH_COMPONENT_LEN + 1) + \
+ sizeof("/tmp"))
+#define LONG_ESCAPED_PATH_LEN \
+ (LONG_PATH_COMPONENT_COUNT * LONG_PATH_COMPONENT_LEN * OCTAL_ESCAPE_LEN)
+
+static_assert(LONG_ESCAPED_PATH_LEN > TRACE_SEQUENCE_SIZE,
+ "escaped path must exceed the trace sequence");
+static_assert(LONG_PATH_LEN < PATH_MAX, "path must fit in PATH_MAX");
+
+static void create_long_path(struct __test_metadata *const _metadata,
+ char *path)
+{
+ size_t path_len;
+
+ strcpy(path, "/tmp");
+ path_len = strlen(path);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, mount("tmpfs", "/tmp", "tmpfs", 0, NULL));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ for (int i = 0; i < LONG_PATH_COMPONENT_COUNT; i++) {
+ path[path_len++] = '/';
+ memset(path + path_len, ' ', LONG_PATH_COMPONENT_LEN);
+ path_len += LONG_PATH_COMPONENT_LEN;
+ path[path_len] = '\0';
+ ASSERT_EQ(0, mkdir(path, 0700));
+ }
+}
+
+static void expect_truncated_path(struct __test_metadata *const _metadata,
+ const char *const trace,
+ const char *const event_regex)
+{
+ static const char marker[] = "\xe2\x80\xa6";
+ char *path;
+ size_t path_len;
+
+ path = malloc(TRACE_SEQUENCE_SIZE);
+ ASSERT_NE(NULL, path);
+ ASSERT_EQ(0, tracefs_extract_field(trace, event_regex, "path", path,
+ TRACE_SEQUENCE_SIZE));
+ EXPECT_EQ(path, strstr(path, "/tmp/"));
+ EXPECT_NE(NULL, strstr(path, "\\040"));
+
+ path_len = strlen(path);
+ ASSERT_LE(sizeof(marker) - 1, path_len);
+ EXPECT_STREQ(marker, path + path_len - (sizeof(marker) - 1));
+ free(path);
+}
+
+/*
+ * Like REGEX_DENY_ACCESS_FS(), but pins the logged field to a specific value
+ * ("0" or "1") so a test can tell a suppressed (quiet) denial from a logged
+ * one. The tracepoint fires for every denial; logged carries the audit
+ * verdict.
+ */
+#define REGEX_DENY_ACCESS_FS_LOGGED(task, log) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_access_fs: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=" log " " \
+ "blockers=[a-z_|]* " \
+ "dev=[0-9]\\+:[0-9]\\+ " \
+ "ino=[0-9]\\+ " \
+ "path=[^ ]*$"
+
+/* clang-format off */
+FIXTURE(trace_fs) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_fs)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE,
+ true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_fs)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, false);
+ tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false);
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * Baseline: verifies that without Landlock, the operation succeeds and no
+ * check_rule or deny_access trace events fire.
+ */
+TEST_F(trace_fs, unsandboxed)
+{
+ char *buf;
+ int count, status, fd;
+ pid_t pid;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ /*
+ * No sandbox: verify that a normal FS access does not produce
+ * Landlock trace events.
+ */
+ fd = open("/usr", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_EQ(0, count);
+ count = tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK));
+ EXPECT_EQ(0, count);
+
+ free(buf);
+}
+
+/*
+ * Verifies that adding a filesystem rule emits a landlock_add_rule_path_beneath
+ * event with the expected path and field values: the ruleset ID and
+ * access_rights are non-zero, and the path matches.
+ */
+TEST_F(trace_fs, add_rule_path_beneath)
+{
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE |
+ LANDLOCK_ACCESS_FS_WRITE_FILE |
+ LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+ char *buf, field_buf[64];
+ int ruleset_fd, count;
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ path_beneath.parent_fd = open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_LE(0, path_beneath.parent_fd);
+
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0));
+ ASSERT_EQ(0, close(path_beneath.parent_fd));
+ ASSERT_EQ(0, close(ruleset_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf,
+ REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK));
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 add_rule_path_beneath event, got %d\n%s",
+ count, buf);
+ }
+
+ /* Ruleset ID should be non-zero. */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK),
+ "ruleset", field_buf, sizeof(field_buf)));
+ EXPECT_STRNE("0", field_buf);
+
+ /* Access rights should be non-zero. */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK),
+ "access_rights", field_buf, sizeof(field_buf)));
+ EXPECT_STRNE("", field_buf);
+
+ /* Path should be /usr. */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK),
+ "path", field_buf, sizeof(field_buf)));
+ EXPECT_STREQ("/usr", field_buf);
+
+ free(buf);
+}
+
+/*
+ * Verifies that a path whose escaping exceeds the trace scratch sequence does
+ * not corrupt a sibling symbolic field.
+ */
+TEST_F(trace_fs, add_rule_path_beneath_escaped_path_overflow)
+{
+ static const char access_prefix[] = "execute|write_file|read_file|";
+ static const char access_suffix[] = "|ioctl_dev|resolve_unix";
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+ char path[PATH_MAX];
+ char *buf, field_buf[256];
+ size_t field_len;
+ int ruleset_fd, count;
+
+ create_long_path(_metadata, path);
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ path_beneath.parent_fd = open(path, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_LE(0, path_beneath.parent_fd);
+
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0));
+ ASSERT_EQ(0, close(path_beneath.parent_fd));
+ ASSERT_EQ(0, close(ruleset_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf,
+ REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK));
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 add_rule_path_beneath event, got %d\n%s",
+ count, buf);
+ }
+
+ /*
+ * The marker catches a full revert with any compiler. The symbolic
+ * field also catches scratch-sequence poisoning when the compiler
+ * evaluates the overflowing path first, as GCC currently does.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK),
+ "access_rights", field_buf, sizeof(field_buf)));
+ EXPECT_EQ(0,
+ strncmp(field_buf, access_prefix, sizeof(access_prefix) - 1));
+ EXPECT_EQ(NULL, strstr(field_buf, "|refer|"));
+ field_len = strlen(field_buf);
+ ASSERT_LE(sizeof(access_suffix) - 1, field_len);
+ EXPECT_STREQ(access_suffix,
+ field_buf + field_len - (sizeof(access_suffix) - 1));
+ expect_truncated_path(_metadata, buf,
+ REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK));
+
+ free(buf);
+}
+
+/*
+ * Verifies that an overflowing denied path does not corrupt its sibling
+ * symbolic blockers field.
+ */
+TEST_F(trace_fs, deny_access_fs_escaped_path_overflow)
+{
+ char path[PATH_MAX];
+ char *buf, field_buf[64];
+ int count, err;
+
+ create_long_path(_metadata, path);
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, path);
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK));
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 deny_access_fs event, got %d\n%s", count,
+ buf);
+ }
+
+ /*
+ * The marker catches a full revert with any compiler. The symbolic
+ * field also catches scratch-sequence poisoning when the compiler
+ * evaluates the overflowing path first, as GCC currently does.
+ */
+ err = tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK),
+ "blockers", field_buf, sizeof(field_buf));
+ ASSERT_EQ(0, err);
+ EXPECT_STREQ("read_dir", field_buf);
+ expect_truncated_path(_metadata, buf, REGEX_DENY_ACCESS_FS(TRACE_TASK));
+
+ free(buf);
+}
+
+/*
+ * Verifies that an allowed access emits check_rule events (rule matched during
+ * pathwalk) but does NOT emit deny_access events (no denial).
+ */
+TEST_F(trace_fs, allowed_access)
+{
+ char *buf, field_buf[64];
+ int count;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* Rule allows READ_DIR for /usr, access /usr which is allowed. */
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, "/usr");
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_LE(1, count);
+
+ /* Single-layer grants array, intersected with the request. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "grants", field_buf,
+ sizeof(field_buf)));
+ EXPECT_STREQ("{read_dir}", field_buf);
+
+ count = tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK));
+ EXPECT_EQ(0, count);
+
+ free(buf);
+}
+
+/*
+ * Verifies that accessing a path whose access type is not in the handled set
+ * does not emit landlock_check_rule events. The ruleset handles READ_FILE, but
+ * the directory open checks READ_DIR which is unhandled; Landlock has no
+ * opinion and no rule evaluation occurs.
+ */
+TEST_F(trace_fs, check_rule_unhandled)
+{
+ char *buf;
+ int count;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* Handles READ_FILE only; READ_DIR is unhandled. */
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_FILE,
+ LANDLOCK_ACCESS_FS_READ_FILE, "/tmp");
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* No check_rule events because READ_DIR is not in the handled set. */
+ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_EQ(0, count);
+
+ free(buf);
+}
+
+/*
+ * Verifies that nested domains (child sandboxed under a parent domain) emit
+ * check_rule events from both layers and produce a deny_access event when the
+ * inner domain's rule does not cover the access.
+ */
+TEST_F(trace_fs, check_rule_nested)
+{
+ char *buf, field_buf[64], *comma;
+ size_t first_len, second_len;
+ int count_rule, count_access, status;
+ pid_t pid;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd, fd;
+
+ /* First layer: allow /usr. */
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0)) {
+ close(path_beneath.parent_fd);
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+
+ /* Second layer: also allow /usr. */
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0)) {
+ close(path_beneath.parent_fd);
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(path_beneath.parent_fd);
+
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+
+ /* Access /usr which is allowed by both layers. */
+ fd = open("/usr", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+
+ /* Access /tmp which has no rule in either layer. */
+ fd = open("/tmp", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count_rule =
+ tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ EXPECT_LE(1, count_rule);
+
+ /*
+ * Both layers have the same rule, so the grants array must have two
+ * identical symbolic entries, e.g. {read_dir,read_dir}.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "grants", field_buf,
+ sizeof(field_buf)));
+ comma = strchr(field_buf, ',');
+ EXPECT_NE(0, !!comma);
+ if (comma) {
+ /*
+ * Verify both entries are identical: compare the substring
+ * before the comma with the substring after it (stripping the
+ * braces).
+ */
+ first_len = comma - field_buf - 1;
+ second_len = strlen(comma + 1) - 1;
+ EXPECT_EQ(first_len, second_len);
+ EXPECT_EQ(0, strncmp(field_buf + 1, comma + 1, first_len));
+ }
+
+ count_access =
+ tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK));
+ EXPECT_LE(1, count_access);
+
+ free(buf);
+}
+
+/*
+ * Verifies that a denied FS access emits a landlock_deny_access_fs trace event
+ * with the blocked access and path.
+ */
+TEST_F(trace_fs, deny_access_fs_denied)
+{
+ const char *const event_regex = REGEX_DENY_ACCESS_FS(TRACE_TASK);
+ char *buf, blockers[64];
+ int count;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /*
+ * Rule allows READ_DIR for /usr, but access /tmp which has no rule.
+ * READ_DIR access to /tmp is denied by absence and should emit a
+ * deny_access_fs event.
+ */
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, "/tmp");
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf, event_regex);
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 access denial, got %d\n%s", count, buf);
+ }
+ ASSERT_EQ(0, tracefs_extract_field(buf, event_regex, "blockers",
+ blockers, sizeof(blockers)));
+ EXPECT_STREQ("read_dir", blockers);
+
+ free(buf);
+}
+
+/*
+ * Verifies that a denied mount reports the singleton topology blocker rather
+ * than an empty access mask.
+ */
+TEST_F(trace_fs, deny_change_topology)
+{
+ const char *const event_regex = REGEX_DENY_ACCESS_FS(TRACE_TASK);
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_REFER,
+ };
+ char *buf, blockers[64];
+ int count, ruleset_fd, status;
+ pid_t pid;
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* Ensure that Landlock is the only expected mount denial. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (pid == 0) {
+ if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(2);
+ }
+ close(ruleset_fd);
+
+ if (mount(NULL, "/", NULL, MS_PRIVATE | MS_REC, NULL) != -1)
+ _exit(3);
+
+ if (errno != EPERM)
+ _exit(4);
+
+ _exit(0);
+ }
+ close(ruleset_fd);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+ count = tracefs_count_matches(buf, event_regex);
+ EXPECT_EQ(1, count)
+ {
+ TH_LOG("Expected 1 topology denial, got %d\n%s", count, buf);
+ }
+ ASSERT_EQ(0, tracefs_extract_field(buf, event_regex, "blockers",
+ blockers, sizeof(blockers)));
+ EXPECT_STREQ("change_topology", blockers);
+
+ free(buf);
+}
+
+/*
+ * A denied FS access covered by a quiet rule (LANDLOCK_ADD_RULE_QUIET with the
+ * access listed in quiet_access_fs) still emits a landlock_deny_access_fs
+ * event, but with logged=0, the same audit-logging verdict audit would apply to
+ * suppress the record.
+ */
+TEST_F(trace_fs, deny_access_fs_quiet)
+{
+ char *buf, field[64];
+ pid_t pid;
+ int status;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (pid == 0) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ .quiet_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = 0,
+ };
+ int ruleset_fd, fd;
+
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ /* Marks /tmp quiet without granting any access. */
+ path_beneath.parent_fd =
+ open("/tmp", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, LANDLOCK_ADD_RULE_QUIET)) {
+ close(path_beneath.parent_fd);
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0)) {
+ close(ruleset_fd);
+ _exit(1);
+ }
+ close(ruleset_fd);
+
+ /* Denied READ_DIR on the quiet /tmp: suppressed, logged=0. */
+ fd = open("/tmp", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+ _exit(0);
+ }
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* The event fires with the suppressed verdict. */
+ EXPECT_LE(1, tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS_LOGGED(
+ TRACE_TASK, "0")));
+ /* The quiet rule must not leave the denial logged. */
+ EXPECT_EQ(0, tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS_LOGGED(
+ TRACE_TASK, "1")));
+
+ /*
+ * Quiet suppresses only the logged verdict: the rest of the denial
+ * event stays populated (non-zero domain, non-empty blockers).
+ */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_ACCESS_FS_LOGGED(TRACE_TASK, "0"),
+ "domain", field, sizeof(field)));
+ EXPECT_STRNE("0", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_ACCESS_FS_LOGGED(TRACE_TASK, "0"),
+ "blockers", field, sizeof(field)));
+ EXPECT_STRNE("", field);
+
+ free(buf);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/trace_test.c b/tools/testing/selftests/landlock/trace_test.c
new file mode 100644
index 000000000000..f9b293a9dd56
--- /dev/null
+++ b/tools/testing/selftests/landlock/trace_test.c
@@ -0,0 +1,1629 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Landlock tests - Tracepoints
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <linux/landlock.h>
+#include <pthread.h>
+#include <sched.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/types.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include "common.h"
+#include "trace.h"
+
+#define TRACE_TASK "trace_test"
+
+/* clang-format off */
+FIXTURE(trace) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE,
+ true));
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_ADD_RULE_NET_PORT_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, true));
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_FREE_DOMAIN_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_FREE_RULESET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ /* Disables landlock events and clears PID filter. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false);
+ tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false);
+ tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false);
+ tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, false);
+ tracefs_enable_event(TRACEFS_ADD_RULE_NET_PORT_ENABLE, false);
+ tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false);
+ tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, false);
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, false);
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, false);
+ tracefs_enable_event(TRACEFS_FREE_DOMAIN_ENABLE, false);
+ tracefs_enable_event(TRACEFS_FREE_RULESET_ENABLE, false);
+ tracefs_clear_pid_filter();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * The mount namespace is cleaned up automatically when the test process
+ * (harness child) exits.
+ */
+}
+
+/*
+ * Verifies that no trace events are emitted when the tracepoints are disabled.
+ */
+TEST_F(trace, no_trace_when_disabled)
+{
+ char *buf;
+
+ /* Disable all landlock events. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false));
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE,
+ false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NET_PORT_ENABLE,
+ false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false));
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, false));
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, false));
+ ASSERT_EQ(0,
+ tracefs_enable_event(TRACEFS_DENY_ACCESS_NET_ENABLE, false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_PTRACE_ENABLE, false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_SCOPE_SIGNAL_ENABLE,
+ false));
+ ASSERT_EQ(0, tracefs_enable_event(
+ TRACEFS_DENY_SCOPE_ABSTRACT_UNIX_SOCKET_ENABLE,
+ false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_FREE_DOMAIN_ENABLE, false));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_FREE_RULESET_ENABLE, false));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * Trigger both allowed and denied accesses to verify neither check_rule
+ * nor check_access events fire when disabled.
+ */
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, "/tmp");
+
+ /* Read trace buffer and verify no landlock events at all. */
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(0, tracefs_count_matches(buf, "landlock_"))
+ {
+ TH_LOG("Expected 0 landlock events when disabled\n%s", buf);
+ }
+
+ free(buf);
+}
+
+/*
+ * Verifies that landlock_create_ruleset emits a trace event with the correct
+ * handled access masks.
+ */
+TEST_F(trace, create_ruleset)
+{
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE,
+ .handled_access_net = LANDLOCK_ACCESS_NET_BIND_TCP,
+ };
+ int ruleset_fd;
+ char *buf, *dot;
+ char field[64];
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, close(ruleset_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CREATE_RULESET(TRACE_TASK)))
+ {
+ TH_LOG("Expected 1 create_ruleset event\n%s", buf);
+ }
+
+ /* Verify handled_fs matches what we requested. */
+ EXPECT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "handled_fs", field, sizeof(field)));
+ EXPECT_STREQ("read_file", field);
+
+ /* Verify handled_net matches. */
+ EXPECT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "handled_net", field, sizeof(field)));
+ EXPECT_STREQ("bind_tcp", field);
+
+ /* Verify version is 0 at creation (no rules added yet). */
+ EXPECT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "ruleset", field, sizeof(field)));
+ /* Format is <hex>.<dec>; version is after the dot. */
+ dot = strchr(field, '.');
+ ASSERT_NE(0, !!dot);
+ EXPECT_STREQ("0", dot + 1);
+
+ free(buf);
+}
+
+/*
+ * Verifies that the ruleset version increments with each add_rule call and that
+ * create_domain records the correct version.
+ */
+TEST_F(trace, ruleset_version)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ const char *dot;
+ char field[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd;
+
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ /* First rule: version becomes 1. */
+ path_beneath.parent_fd =
+ open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0);
+ close(path_beneath.parent_fd);
+
+ /* Second rule: version becomes 2. */
+ path_beneath.parent_fd =
+ open("/tmp", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+ landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0);
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* Verify create_ruleset has version=0. */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "ruleset", field, sizeof(field)));
+ dot = strchr(field, '.');
+ ASSERT_NE(0, !!dot);
+ EXPECT_STREQ("0", dot + 1);
+
+ /* Verify two add_rule_path_beneath events were emitted. */
+ EXPECT_EQ(2, tracefs_count_matches(
+ buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK)))
+ {
+ TH_LOG("Expected 2 add_rule_path_beneath events\n%s", buf);
+ }
+
+ /*
+ * Verify create_domain records version=2 (after 2 add_rule calls). The
+ * ruleset field format is <hex_id>.<dec_version>.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "ruleset", field, sizeof(field)));
+ dot = strchr(field, '.');
+ ASSERT_NE(0, !!dot);
+ EXPECT_STREQ("2", dot + 1);
+
+ free(buf);
+}
+
+/*
+ * Verifies that landlock_create_domain emits a trace event linking the ruleset
+ * ID to the new domain ID.
+ */
+TEST_F(trace, create_domain)
+{
+ pid_t pid;
+ int status, check_count;
+ char *buf;
+ char parent_id[64], domain_id[64], check_domain[64];
+
+ /* Clear before the sandboxed child. */
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd, fd;
+
+ ruleset_fd = landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ path_beneath.parent_fd =
+ open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (path_beneath.parent_fd < 0)
+ _exit(1);
+
+ landlock_add_rule(ruleset_fd, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0);
+ close(path_beneath.parent_fd);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ /* Trigger a check_rule to verify domain_id correlation. */
+ fd = open("/usr", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* Verify create_domain event exists. */
+ EXPECT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK)))
+ {
+ TH_LOG("Expected 1 create_domain event\n%s", buf);
+ }
+
+ /* Extract the domain ID from create_domain. */
+ EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "domain", domain_id,
+ sizeof(domain_id)));
+
+ /* Verify domain ID is non-zero. */
+ EXPECT_NE(0, strcmp(domain_id, "0"));
+
+ /* Verify parent=0 (first restriction, no prior domain). */
+ EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "parent", parent_id,
+ sizeof(parent_id)));
+ EXPECT_STREQ("0", parent_id);
+
+ /*
+ * Verify the same domain ID appears in the check_rule event, confirming
+ * end-to-end correlation.
+ */
+ check_count =
+ tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK));
+ ASSERT_LE(1, check_count)
+ {
+ TH_LOG("Expected check_rule_inode events\n%s", buf);
+ }
+
+ EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK),
+ "domain", check_domain,
+ sizeof(check_domain)));
+ EXPECT_STREQ(domain_id, check_domain);
+
+ free(buf);
+}
+
+/* Builds a rule-less scope-based ruleset; returns the fd or -1. */
+static int build_enforce_ruleset(void)
+{
+ const struct landlock_ruleset_attr attr = {
+ .scoped = LANDLOCK_SCOPE_SIGNAL,
+ };
+
+ return landlock_create_ruleset(&attr, sizeof(attr), 0);
+}
+
+/*
+ * Verifies that nested landlock_restrict_self calls produce trace events with
+ * correct parent domain IDs: the second create_domain's parent should be the
+ * first domain's ID.
+ */
+TEST_F(trace, create_domain_nested)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ const char *after_first;
+ char first_domain[64], first_parent[64], second_parent[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ int ruleset_fd;
+
+ /* First restriction. */
+ ruleset_fd = build_enforce_ruleset();
+ if (ruleset_fd < 0)
+ _exit(1);
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ /* Second restriction (nested). */
+ ruleset_fd = build_enforce_ruleset();
+ if (ruleset_fd < 0)
+ _exit(1);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* Should have 2 create_domain events. */
+ EXPECT_EQ(2,
+ tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK)))
+ {
+ TH_LOG("Expected 2 create_domain events\n%s", buf);
+ }
+
+ /*
+ * Extract domain and parent from each create_domain event. The first
+ * event (parent=0) is the outer domain; the second (parent!=0) is the
+ * nested domain whose parent should match the first domain's ID.
+ */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "domain", first_domain,
+ sizeof(first_domain)));
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "parent", first_parent,
+ sizeof(first_parent)));
+ EXPECT_STREQ("0", first_parent);
+
+ /*
+ * Find the second create_domain by scanning past the first.
+ * tracefs_extract_field returns the first match, so search in the
+ * buffer after the first event.
+ *
+ * Skip past the first create_domain line. tracefs_extract_field matches
+ * the first line that matches the regex, so passing the buffer after
+ * the first matching line gives us the second event.
+ */
+ after_first = strstr(buf, "landlock_create_domain:");
+ ASSERT_NE(NULL, after_first);
+ after_first = strchr(after_first, '\n');
+ ASSERT_NE(NULL, after_first);
+
+ ASSERT_EQ(0, tracefs_extract_field(
+ after_first + 1, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "parent", second_parent, sizeof(second_parent)));
+
+ /* The second domain's parent should be the first domain's ID. */
+ EXPECT_STREQ(first_domain, second_parent);
+
+ free(buf);
+}
+
+/*
+ * Verifies that landlock_add_rule does not emit a trace event when the syscall
+ * fails (e.g., invalid ruleset fd).
+ */
+TEST_F(trace, add_rule_invalid_fd)
+{
+ struct landlock_path_beneath_attr path_beneath = {
+ .allowed_access = LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+ char *buf;
+
+ path_beneath.parent_fd = open("/usr", O_PATH | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_LE(0, path_beneath.parent_fd);
+
+ /* Invalid ruleset fd (-1). */
+ ASSERT_EQ(-1, landlock_add_rule(-1, LANDLOCK_RULE_PATH_BENEATH,
+ &path_beneath, 0));
+ ASSERT_EQ(0, close(path_beneath.parent_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(0, tracefs_count_matches(
+ buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK)))
+ {
+ TH_LOG("No add_rule_path_beneath event expected on invalid fd\n%s",
+ buf);
+ }
+
+ free(buf);
+}
+
+/*
+ * Verifies that landlock_create_domain does not emit a trace event when the
+ * syscall fails (e.g., invalid ruleset fd or unknown flags).
+ */
+TEST_F(trace, create_domain_invalid)
+{
+ int ruleset_fd;
+ char *buf;
+
+ ruleset_fd = build_enforce_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Clear the trace buffer after create_ruleset event. */
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* Invalid fd. */
+ ASSERT_EQ(-1, landlock_restrict_self(-1, 0));
+
+ /* Unknown flags. */
+ ASSERT_EQ(-1, landlock_restrict_self(ruleset_fd, -1));
+
+ ASSERT_EQ(0, close(ruleset_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(0,
+ tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK)))
+ {
+ TH_LOG("No create_domain event expected on error\n%s", buf);
+ }
+
+ free(buf);
+}
+
+/*
+ * Verifies that trace_landlock_free_domain fires when a domain is deallocated,
+ * with the correct denials count.
+ */
+TEST_F(trace, free_domain)
+{
+ char *buf;
+ int count;
+ char denials_field[32];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /*
+ * The domain is freed via a work queue (kworker), so the free_domain
+ * trace event is emitted from a different PID. Clear the PID filter
+ * BEFORE the child exits, so the kworker event passes the filter when
+ * it fires.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_clear_pid_filter();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, "/tmp");
+
+ /*
+ * Wait for the deferred deallocation work to run. The domain is freed
+ * asynchronously from a kworker; poll until the event appears or a
+ * timeout is reached.
+ */
+ for (int retry = 0; retry < 10; retry++) {
+ usleep(100000);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_NE(NULL, buf);
+
+ count = tracefs_count_matches(buf,
+ REGEX_FREE_DOMAIN(KWORKER_TASK));
+ if (count >= 1)
+ break;
+ free(buf);
+ buf = NULL;
+ }
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, tracefs_set_pid_filter(getpid()));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ ASSERT_NE(NULL, buf);
+ EXPECT_LE(1, count)
+ {
+ TH_LOG("Expected free_domain event, got %d\n%s", count, buf);
+ }
+
+ /* Verify denials count matches the single denial we triggered. */
+ EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_FREE_DOMAIN(KWORKER_TASK),
+ "denials", denials_field,
+ sizeof(denials_field)));
+ EXPECT_STREQ("1", denials_field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that deny_access_fs includes the enriched fields: same_exec and
+ * logged.
+ */
+TEST_F(trace, deny_access_fs_fields)
+{
+ char *buf;
+ char field_buf[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* Trigger a denial: rule for /usr, access /tmp. */
+ sandbox_child_fs_access(_metadata, "/usr", LANDLOCK_ACCESS_FS_READ_DIR,
+ LANDLOCK_ACCESS_FS_READ_DIR, "/tmp");
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* Verify the enriched fields are present and have valid values. */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_ACCESS_FS(TRACE_TASK), "same_exec",
+ field_buf, sizeof(field_buf)));
+ /* Child is the same exec that restricted itself. */
+ EXPECT_STREQ("1", field_buf);
+
+ /* Same exec with default flags: audit would log this denial. */
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_ACCESS_FS(TRACE_TASK), "logged",
+ field_buf, sizeof(field_buf)));
+ EXPECT_STREQ("1", field_buf);
+
+ free(buf);
+}
+
+/*
+ * Verifies that same_exec is 1 (true) for denials from the same executable that
+ * called landlock_restrict_self().
+ */
+TEST_F(trace, same_exec_before_exec)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ char field[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd, dir_fd;
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ /* No rules: all read_dir access is denied. */
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ close(ruleset_fd);
+
+ /* Trigger denial without exec (same executable). */
+ dir_fd = open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (dir_fd >= 0)
+ close(dir_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /* Should have at least one deny_access_fs denial. */
+ EXPECT_LE(1,
+ tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK)));
+
+ /* Verify same_exec=1 (same executable, no exec). */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK),
+ "same_exec", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+
+ /* Same exec with default flags: audit would log this denial. */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that same_exec is 0 (false) for denials from a process that has
+ * exec'd a new binary after landlock_restrict_self(). The sandboxed child
+ * exec's true which opens "." and triggers a read_dir denial. Covers the
+ * "trace-only" visibility condition: with same_exec=0 and the default
+ * log_new_exec=0, audit suppresses the denial (logged=0) but the trace event
+ * still fires.
+ */
+TEST_F(trace, same_exec_after_exec)
+{
+ char *buf;
+ char field[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ sandbox_child_exec_true(_metadata, 0);
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_LE(1, tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS("true")));
+
+ /* Verify same_exec=0 (different executable after exec). */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS("true"),
+ "same_exec", field, sizeof(field)));
+ EXPECT_STREQ("0", field);
+
+ /*
+ * same_exec=0 with default log_new_exec=0: audit suppresses (logged=0).
+ */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS("true"),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ("0", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that LANDLOCK_RESTRICT_SELF_LOG_SAME_EXEC_OFF suppresses logging
+ * (logged=0) for a denial from the same executable.
+ */
+TEST_F(trace, log_flags_same_exec_off)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ char field[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd, dir_fd;
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (ruleset_fd < 0)
+ _exit(1);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(
+ ruleset_fd,
+ LANDLOCK_RESTRICT_SELF_LOG_SAME_EXEC_OFF))
+ _exit(1);
+ close(ruleset_fd);
+
+ dir_fd = open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (dir_fd >= 0)
+ close(dir_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_LE(1,
+ tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK)));
+
+ /* Same-exec denial with LOG_SAME_EXEC_OFF: audit suppresses it. */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ("0", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that LANDLOCK_RESTRICT_SELF_LOG_NEW_EXEC_ON causes a post-exec
+ * denial to be logged (logged=1). The child exec's true so that the denial
+ * comes from a new executable (same_exec=0).
+ */
+TEST_F(trace, log_flags_new_exec_on)
+{
+ char *buf;
+ char field[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ sandbox_child_exec_true(_metadata,
+ LANDLOCK_RESTRICT_SELF_LOG_NEW_EXEC_ON);
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_LE(1, tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS("true")));
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS("true"),
+ "same_exec", field, sizeof(field)));
+ EXPECT_STREQ("0", field);
+
+ /* LOG_NEW_EXEC_ON: the post-exec denial (same_exec=0) is logged. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS("true"),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that denials suppressed by audit log flags are still counted in
+ * num_denials. The child restricts itself with default flags (log_same_exec=1,
+ * log_new_exec=0), then execs true which attempts to read a denied directory.
+ * After exec, same_exec=0 and log_new_exec=0, so audit suppresses the denial.
+ * But the trace event fires unconditionally and free_domain must report the
+ * correct denials count.
+ */
+TEST_F(trace, non_audit_visible_denial_counting)
+{
+ char *buf = NULL;
+ char denials_field[32];
+ int count;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, tracefs_clear());
+ tracefs_clear_pid_filter();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ sandbox_child_exec_true(_metadata, 0);
+
+ /* Wait for free_domain event with retry. */
+ for (int retry = 0; retry < 10; retry++) {
+ usleep(100000);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ if (!buf)
+ break;
+
+ count = tracefs_count_matches(buf,
+ REGEX_FREE_DOMAIN(KWORKER_TASK));
+ if (count >= 1)
+ break;
+ free(buf);
+ buf = NULL;
+ }
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, tracefs_set_pid_filter(getpid()));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * The denial happened after exec (same_exec=0), so audit would suppress
+ * it. But num_denials counts all denials regardless.
+ */
+ ASSERT_NE(NULL, buf)
+ {
+ TH_LOG("free_domain event not found after 10 retries");
+ }
+ EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_FREE_DOMAIN(KWORKER_TASK),
+ "denials", denials_field,
+ sizeof(denials_field)));
+ EXPECT_STREQ("1", denials_field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that landlock_add_rule_net_port emits a trace event with the correct
+ * port and allowed access mask fields.
+ */
+TEST_F(trace, add_rule_net_port_fields)
+{
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_net = LANDLOCK_ACCESS_NET_BIND_TCP,
+ };
+ struct landlock_net_port_attr net_port = {
+ .allowed_access = LANDLOCK_ACCESS_NET_BIND_TCP,
+ .port = 8080,
+ };
+ int ruleset_fd;
+ char *buf;
+ char field[64];
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NET_PORT,
+ &net_port, 0));
+ close(ruleset_fd);
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(1, tracefs_count_matches(buf,
+ REGEX_ADD_RULE_NET_PORT(TRACE_TASK)))
+ {
+ TH_LOG("Expected 1 add_rule_net_port event\n%s", buf);
+ }
+
+ /*
+ * Verify the port is in host endianness, matching the UAPI convention
+ * (landlock_net_port_attr.port). On little-endian, htons(8080) is
+ * 36895, so this comparison catches byte-order bugs.
+ */
+ EXPECT_EQ(0, tracefs_extract_field(buf,
+ REGEX_ADD_RULE_NET_PORT(TRACE_TASK),
+ "port", field, sizeof(field)));
+ EXPECT_STREQ("8080", field);
+ /*
+ * The allowed mask is the absolute value after transformation: the
+ * user-requested BIND_TCP plus all unhandled access rights (the other
+ * net access bits are unhandled because the ruleset only handles
+ * BIND_TCP).
+ */
+ EXPECT_EQ(0, tracefs_extract_field(
+ buf, REGEX_ADD_RULE_NET_PORT(TRACE_TASK),
+ "access_rights", field, sizeof(field)));
+ EXPECT_STREQ("bind_tcp|connect_tcp|bind_udp|connect_send_udp", field);
+
+ free(buf);
+}
+
+/*
+ * Verifies that LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF suppresses audit
+ * logging for child domains (logged=0) even though the child's own
+ * per-execution flags are the defaults, while the trace event still fires
+ * (tracing is unconditional). The parent creates a domain with
+ * LOG_SUBDOMAINS_OFF, then the child creates a sub-domain and triggers a
+ * denial.
+ */
+TEST_F(trace, log_flags_subdomains_off)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ char field[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+
+ if (pid == 0) {
+ struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int parent_fd, child_fd, dir_fd;
+
+ /* Parent domain with LOG_SUBDOMAINS_OFF. */
+ parent_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (parent_fd < 0)
+ _exit(1);
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(
+ parent_fd,
+ LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF))
+ _exit(1);
+ close(parent_fd);
+
+ /* Child sub-domain with default flags. */
+ child_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ if (child_fd < 0)
+ _exit(1);
+
+ if (landlock_restrict_self(child_fd, 0))
+ _exit(1);
+ close(child_fd);
+
+ /* Trigger a denial from the child domain. */
+ dir_fd = open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (dir_fd >= 0)
+ close(dir_fd);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ /*
+ * Trace fires unconditionally even though audit is disabled for the
+ * child domain (parent had LOG_SUBDOMAINS_OFF).
+ */
+ EXPECT_LE(1,
+ tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK)))
+ {
+ TH_LOG("Expected deny_access_fs event despite "
+ "LOG_SUBDOMAINS_OFF\n%s",
+ buf);
+ }
+
+ /*
+ * The child's per-execution flags default to logging, but the
+ * ancestor's LOG_SUBDOMAINS_OFF disables it, so audit suppresses this
+ * denial (logged=0). This is exactly the case the single logged field
+ * captures and the raw per-execution flags could not.
+ */
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ("0", field);
+
+ free(buf);
+}
+
+/* Verifies that landlock_free_ruleset fires when a ruleset FD is closed. */
+TEST_F(trace, free_ruleset_on_close)
+{
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_DIR,
+ };
+ int ruleset_fd;
+ char *buf;
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* Closing the FD should trigger free_ruleset. */
+ close(ruleset_fd);
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(1, tracefs_count_matches(buf, REGEX_FREE_RULESET(TRACE_TASK)))
+ {
+ TH_LOG("Expected 1 free_ruleset event\n%s", buf);
+ }
+
+ free(buf);
+}
+
+/*
+ * Counts landlock_enforce_domain lines, filtered by @domain (NULL matches any),
+ * @complete and @process_wide (a negative value matches any). Builds the
+ * anchored regex dynamically so a single helper covers every field assertion.
+ */
+static int count_enforce_matches(const char *buf, const char *domain,
+ int complete, int process_wide,
+ int no_new_privs)
+{
+ char pattern[512], dom[80], comp[8], pw[8], nnp[8];
+
+ if (domain)
+ snprintf(dom, sizeof(dom), "%s", domain);
+ else
+ snprintf(dom, sizeof(dom), "[0-9a-f]\\+");
+ if (complete < 0)
+ snprintf(comp, sizeof(comp), "[01]");
+ else
+ snprintf(comp, sizeof(comp), "%d", complete);
+ if (process_wide < 0)
+ snprintf(pw, sizeof(pw), "[01]");
+ else
+ snprintf(pw, sizeof(pw), "%d", process_wide);
+ if (no_new_privs < 0)
+ snprintf(nnp, sizeof(nnp), "[01]");
+ else
+ snprintf(nnp, sizeof(nnp), "%d", no_new_privs);
+
+ snprintf(pattern, sizeof(pattern),
+ TRACE_PREFIX(TRACE_TASK) "landlock_enforce_domain: "
+ "domain=%s "
+ "complete=%s process_wide=%s "
+ "no_new_privs=%s$",
+ dom, comp, pw, nnp);
+ return tracefs_count_matches(buf, pattern);
+}
+
+/* Idle sibling: waits on the barrier so it is a live thread, then sleeps. */
+static void *enforce_idle(void *arg)
+{
+ pthread_barrier_t *barrier = arg;
+
+ pthread_barrier_wait(barrier);
+ while (true)
+ sleep(1);
+ return NULL;
+}
+
+/*
+ * Child body: spawns @nthreads idle siblings (barrier-synchronized so they are
+ * live when the syscall runs), then enforces a domain with @flags. Returns 0
+ * on success; the process exits afterwards, reaping the siblings.
+ */
+static int child_enforce(int nthreads, __u32 flags)
+{
+ pthread_t threads[8];
+ pthread_barrier_t barrier;
+ int ruleset_fd, i;
+
+ if (nthreads > 0) {
+ if (pthread_barrier_init(&barrier, NULL, nthreads + 1))
+ return 1;
+ for (i = 0; i < nthreads; i++)
+ if (pthread_create(&threads[i], NULL, enforce_idle,
+ &barrier))
+ return 1;
+ pthread_barrier_wait(&barrier);
+ }
+
+ ruleset_fd = build_enforce_ruleset();
+ if (ruleset_fd < 0)
+ return 1;
+
+ /*
+ * LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS sets no_new_privs itself, so skip
+ * the prctl() to exercise that path; otherwise Landlock requires
+ * no_new_privs up front.
+ */
+ if (!(flags & LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS))
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, flags))
+ return 1;
+ close(ruleset_fd);
+ return 0;
+}
+
+/*
+ * Runs in a spawned thread after the group leader called pthread_exit(). The
+ * leader lingers as an un-reaped zombie, so get_nr_threads() still counts it
+ * and this non-leader is not the only thread; enforcing here therefore reports
+ * process_wide=0.
+ */
+static void *enforce_nonleader(void *arg)
+{
+ int ruleset_fd;
+
+ ruleset_fd = build_enforce_ruleset();
+ if (ruleset_fd < 0)
+ _exit(1);
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(ruleset_fd, 0))
+ _exit(1);
+ _exit(0);
+}
+
+/*
+ * Collapses the enforce_domain field cases into one parametrized test. Each
+ * variant runs child_enforce(nthreads, flags) and checks the resulting
+ * enforce_domain events. The flags column also selects how no_new_privs is
+ * set: with LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS child_enforce() skips the
+ * prctl() so the flag sets it (and, with TSYNC, propagates to the siblings);
+ * otherwise a prior prctl() sets it on the caller (and TSYNC propagates that).
+ */
+
+/* clang-format off */
+FIXTURE(trace_enforce) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(trace_enforce)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(trace_enforce)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false);
+ tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false);
+ tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE_VARIANT(trace_enforce) {
+ /* clang-format on */
+ /* Inputs to child_enforce(). */
+ int nthreads;
+ __u32 flags;
+ /* Expected enforce_domain event counts. */
+ int total;
+ int complete;
+ int process_wide;
+ int no_new_privs;
+};
+
+/* clang-format off */
+
+/* Single thread, no flags: prctl-backed no_new_privs. */
+FIXTURE_VARIANT_ADD(trace_enforce, single) {
+ .nthreads = 0, .flags = 0,
+ .total = 1, .complete = 1, .process_wide = 1, .no_new_privs = 1,
+};
+
+/* Single thread: the NO_NEW_PRIVS flag sets no_new_privs (no prctl). */
+FIXTURE_VARIANT_ADD(trace_enforce, no_new_privs) {
+ .nthreads = 0, .flags = LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS,
+ .total = 1, .complete = 1, .process_wide = 1, .no_new_privs = 1,
+};
+
+/* TSYNC on a lone thread still concludes, process-wide. */
+FIXTURE_VARIANT_ADD(trace_enforce, tsync_single) {
+ .nthreads = 0, .flags = LANDLOCK_RESTRICT_SELF_TSYNC,
+ .total = 1, .complete = 1, .process_wide = 1, .no_new_privs = 1,
+};
+
+/* TSYNC sweeps N siblings; the caller's prctl-backed nnp propagates to all. */
+FIXTURE_VARIANT_ADD(trace_enforce, tsync_multithread) {
+ .nthreads = 3, .flags = LANDLOCK_RESTRICT_SELF_TSYNC,
+ .total = 4, .complete = 1, .process_wide = 4, .no_new_privs = 4,
+};
+
+/* TSYNC + NO_NEW_PRIVS flag sets nnp on the caller and every swept sibling. */
+FIXTURE_VARIANT_ADD(trace_enforce, tsync_no_new_privs) {
+ .nthreads = 3,
+ .flags = LANDLOCK_RESTRICT_SELF_TSYNC | LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS,
+ .total = 4, .complete = 1, .process_wide = 4, .no_new_privs = 4,
+};
+
+/* Non-TSYNC on a multi-threaded process enforces only the caller. */
+FIXTURE_VARIANT_ADD(trace_enforce, multithread_non_tsync) {
+ .nthreads = 3, .flags = 0,
+ .total = 1, .complete = 1, .process_wide = 0, .no_new_privs = 1,
+};
+
+/* clang-format on */
+
+/*
+ * One create_domain and variant->total enforce_domain events sharing that
+ * domain ID; complete=1 marks the single concluding event, and the process_wide
+ * / no_new_privs counts match the variant. Counts are order-independent,
+ * evaluated after the syscall returns.
+ */
+TEST_F(trace_enforce, enforce)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+ char domain[64];
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (pid == 0)
+ _exit(child_enforce(variant->nthreads, variant->flags));
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK)));
+ EXPECT_EQ(variant->total, count_enforce_matches(buf, NULL, -1, -1, -1))
+ {
+ TH_LOG("Expected %d enforce_domain events\n%s", variant->total,
+ buf);
+ }
+ EXPECT_EQ(variant->complete,
+ count_enforce_matches(buf, NULL, 1, -1, -1));
+ EXPECT_EQ(variant->total - variant->complete,
+ count_enforce_matches(buf, NULL, 0, -1, -1));
+ EXPECT_EQ(variant->process_wide,
+ count_enforce_matches(buf, NULL, -1, 1, -1));
+ EXPECT_EQ(variant->total - variant->process_wide,
+ count_enforce_matches(buf, NULL, -1, 0, -1));
+ EXPECT_EQ(variant->no_new_privs,
+ count_enforce_matches(buf, NULL, -1, -1, 1));
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "domain", domain, sizeof(domain)));
+ EXPECT_EQ(variant->total,
+ count_enforce_matches(buf, domain, -1, -1, -1));
+
+ free(buf);
+}
+
+/*
+ * A non-leader thread enforcing a domain while the group leader lingers as an
+ * un-reaped zombie reports process_wide=0: get_nr_threads() counts the zombie
+ * leader, so the group is not single-threaded. This is the reachable half of
+ * the caveat that process_wide==0 never proves the process is multi-threaded
+ * (get_nr_threads(), unlike the leader-relative thread_group_empty(), counts
+ * the zombie leader).
+ */
+TEST_F(trace, enforce_single_non_leader)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (pid == 0) {
+ pthread_t worker;
+
+ if (pthread_create(&worker, NULL, enforce_nonleader, NULL))
+ _exit(1);
+ /* Leader leaves; the worker enforces as a non-leader. */
+ pthread_exit(NULL);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK)));
+ EXPECT_EQ(1, count_enforce_matches(buf, NULL, 1, 0, -1))
+ {
+ TH_LOG("Expected complete=1 process_wide=0 for non-leader\n%s",
+ buf);
+ }
+
+ free(buf);
+}
+
+/*
+ * Verifies the flags-only path (ruleset_fd == -1) creates no domain and emits
+ * neither create_domain nor enforce_domain, with and without TSYNC.
+ */
+TEST_F(trace, enforce_flags_only)
+{
+ pid_t pid;
+ int status;
+ char *buf;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (pid == 0) {
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ if (landlock_restrict_self(
+ -1, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF))
+ _exit(1);
+ if (landlock_restrict_self(
+ -1, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF |
+ LANDLOCK_RESTRICT_SELF_TSYNC))
+ _exit(1);
+ _exit(0);
+ }
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ EXPECT_EQ(0,
+ tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK)));
+ EXPECT_EQ(0, count_enforce_matches(buf, NULL, -1, -1, -1))
+ {
+ TH_LOG("No enforce_domain expected on flags-only path\n%s",
+ buf);
+ }
+
+ free(buf);
+}
+
+static void enforce_nop_handler(int sig)
+{
+}
+
+struct abort_signaler_data {
+ pthread_t target;
+ volatile bool stop;
+};
+
+/*
+ * Hammers the target thread with SIGUSR1 to interrupt the TSYNC prepare wait.
+ */
+static void *abort_signaler(void *arg)
+{
+ struct abort_signaler_data *data = arg;
+
+ while (!data->stop)
+ pthread_kill(data->target, SIGUSR1);
+ return NULL;
+}
+
+/*
+ * Child body for the abort test: with idle siblings and a signaler interrupting
+ * it, repeatedly enforces under TSYNC. An interrupted attempt aborts its
+ * just-created domain (create_domain + free_domain, zero enforce_domain) while
+ * -ERESTARTNOINTR transparently restarts the syscall, so a successful retry may
+ * add its own full lifecycle.
+ */
+static int child_abort(int nsiblings, int attempts)
+{
+ pthread_t threads[200];
+ pthread_t signaler;
+ pthread_barrier_t barrier;
+ struct abort_signaler_data data = {};
+ struct sigaction sa = {};
+ int i;
+
+ sa.sa_handler = enforce_nop_handler;
+ if (sigaction(SIGUSR1, &sa, NULL))
+ return 1;
+
+ if (pthread_barrier_init(&barrier, NULL, nsiblings + 1))
+ return 1;
+ for (i = 0; i < nsiblings; i++)
+ if (pthread_create(&threads[i], NULL, enforce_idle, &barrier))
+ return 1;
+ pthread_barrier_wait(&barrier);
+
+ data.target = pthread_self();
+ if (pthread_create(&signaler, NULL, abort_signaler, &data))
+ return 1;
+
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0);
+ for (i = 0; i < attempts; i++) {
+ int ruleset_fd = build_enforce_ruleset();
+
+ if (ruleset_fd < 0)
+ break;
+ /*
+ * Ignore the result: an abort returns an error, that is fine.
+ */
+ landlock_restrict_self(ruleset_fd,
+ LANDLOCK_RESTRICT_SELF_TSYNC);
+ close(ruleset_fd);
+ }
+
+ data.stop = true;
+ pthread_join(signaler, NULL);
+ return 0;
+}
+
+/*
+ * Verifies the abort contract: a domain aborted by a thread-sync failure emits
+ * create_domain and free_domain but zero enforce_domain. The signal race is
+ * probabilistic and -ERESTARTNOINTR may add a successful retry's lifecycle, so
+ * events are grouped by domain ID and the test SKIPs if no abort occurred.
+ */
+TEST_F(trace, enforce_abort)
+{
+ pid_t pid;
+ int status, retry;
+ char *buf = NULL;
+ const char *cursor;
+ char domain[64];
+ bool abort_found = false;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ /* free_domain fires from a kworker, so widen the filter first. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_clear_pid_filter();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ pid = fork();
+ ASSERT_LE(0, pid);
+ if (pid == 0)
+ /*
+ * Match tsync_test's NUM_IDLE_THREADS: enough siblings that
+ * credential preparation runs in several serialized waves,
+ * giving the signaler a window to interrupt the thread-sync
+ * wait and abort the operation. A handful of threads finishes
+ * in a single wave, leaving no window (the abort never fires).
+ */
+ _exit(child_abort(200, 8));
+
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ /* Poll for the asynchronous free_domain events. */
+ for (retry = 0; retry < 10; retry++) {
+ usleep(100000);
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ free(buf);
+ buf = tracefs_read_trace();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_NE(NULL, buf);
+ }
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, tracefs_set_pid_filter(getpid()));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * Walk every create_domain and look for one whose domain ID has zero
+ * enforce_domain events but a matching free_domain: that is an aborted
+ * domain (created, never enforced, freed).
+ */
+ cursor = buf;
+ while (tracefs_extract_field(cursor, REGEX_CREATE_DOMAIN(TRACE_TASK),
+ "domain", domain, sizeof(domain)) == 0) {
+ const char *cd, *nl;
+ char free_pattern[256];
+
+ if (count_enforce_matches(buf, domain, -1, -1, -1) == 0) {
+ snprintf(
+ free_pattern, sizeof(free_pattern),
+ TRACE_PREFIX(
+ KWORKER_TASK) "landlock_free_domain: "
+ "domain=%s denials=[0-9]\\+$",
+ domain);
+ if (tracefs_count_matches(buf, free_pattern) >= 1)
+ abort_found = true;
+ }
+
+ cd = strstr(cursor, "landlock_create_domain:");
+ if (!cd)
+ break;
+ nl = strchr(cd, '\n');
+ if (!nl)
+ break;
+ cursor = nl + 1;
+ }
+
+ if (!abort_found) {
+ free(buf);
+ SKIP(return, "signal race did not produce a thread-sync abort");
+ }
+
+ free(buf);
+}
+
+/*
+ * The following tests are intentionally elided because the underlying kernel
+ * mechanisms are already validated by audit tests:
+ *
+ * - Domain ID monotonicity: validated by audit_test.c:layers. The same
+ * landlock_get_id_range() function serves both audit and trace.
+ *
+ * - Domain deallocation order (LIFO): validated by audit_test.c:layers. Trace
+ * events fire from the same free_domain_work() code path.
+ *
+ * - Max-layer stacking (16 domains): validated by audit_test.c:layers.
+ *
+ * - IPv6 network tests: IPv6 hook dispatch uses the same
+ * current_check_access_socket() as IPv4, validated by net_test.c:audit tests.
+ *
+ * - Per-access-right full matrix (all 16 FS rights): hook dispatch is validated
+ * by fs_test.c:audit tests. Trace tests verify representative samples to
+ * ensure bitmask encoding is correct.
+ *
+ * - Combined log flag variants (e.g., LOG_SUBDOMAINS_OFF + LOG_NEW_EXEC_ON):
+ * individual flag tests above cover each flag's effect on trace fields. Flag
+ * combination logic is validated by audit_test.c:audit_flags tests.
+ *
+ * - fs.refer multi-record denials and fs.change_topology (mount):
+ * trace_denial() uses the same code path for all FS request types. The
+ * DENTRY union member is validated by the deny_access_fs_fields
+ * test. Audit tests in fs_test.c cover refer and mount denial specifics.
+ */
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/true.c b/tools/testing/selftests/landlock/true.c
index 3f9ccbf52783..1e39b664512d 100644
--- a/tools/testing/selftests/landlock/true.c
+++ b/tools/testing/selftests/landlock/true.c
@@ -1,5 +1,15 @@
// SPDX-License-Identifier: GPL-2.0
+/*
+ * Minimal helper for Landlock selftests. Opens its own working directory
+ * before exiting, which may trigger access denials depending on the sandbox
+ * configuration.
+ */
+
+#include <fcntl.h>
+#include <unistd.h>
+
int main(void)
{
+ close(open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC));
return 0;
}
diff --git a/tools/testing/selftests/landlock/tsync_test.c b/tools/testing/selftests/landlock/tsync_test.c
index 9cf1491bbaaf..2b53596c986e 100644
--- a/tools/testing/selftests/landlock/tsync_test.c
+++ b/tools/testing/selftests/landlock/tsync_test.c
@@ -62,32 +62,104 @@ static void *idle(void *data)
pthread_cleanup_pop(1);
}
-TEST(multi_threaded_success)
+FIXTURE(multi_threaded)
{
- pthread_t t1, t2;
- bool no_new_privs1, no_new_privs2;
- const int ruleset_fd = create_ruleset(_metadata);
+ int ruleset_fd;
+};
+
+FIXTURE_VARIANT(multi_threaded)
+{
+ const __u32 restrict_flags;
+ /* Sets no_new_privs with prctl(2) before the enforcement. */
+ const bool prior_no_new_privs;
+ /* Enforces the maximum number of allowed layers beforehand. */
+ const bool max_layers;
+ const int expected_errno;
+ /* Expected no_new_privs state of all threads after the call. */
+ const bool expected_no_new_privs;
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(multi_threaded, success) {
+ /* clang-format on */
+ .restrict_flags = LANDLOCK_RESTRICT_SELF_TSYNC,
+ .prior_no_new_privs = true,
+ .expected_no_new_privs = true,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(multi_threaded, no_new_privs) {
+ /* clang-format on */
+ .restrict_flags = LANDLOCK_RESTRICT_SELF_TSYNC |
+ LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS,
+ .expected_no_new_privs = true,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(multi_threaded, no_new_privs_max_layers) {
+ /* clang-format on */
+ .restrict_flags = LANDLOCK_RESTRICT_SELF_TSYNC |
+ LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS,
+ .max_layers = true,
+ .expected_errno = E2BIG,
+ .expected_no_new_privs = false,
+};
+
+FIXTURE_SETUP(multi_threaded)
+{
+ self->ruleset_fd = create_ruleset(_metadata);
+
+ if (variant->max_layers) {
+ /* Enforces the maximum number of allowed layers. */
+ for (int i = 0; i < LANDLOCK_MAX_NUM_LAYERS; i++)
+ ASSERT_EQ(0,
+ landlock_restrict_self(self->ruleset_fd, 0));
+ }
disable_caps(_metadata);
+}
+
+FIXTURE_TEARDOWN(multi_threaded)
+{
+ EXPECT_EQ(0, close(self->ruleset_fd));
+}
+
+TEST_F(multi_threaded, restrict)
+{
+ pthread_t t1, t2;
+ bool no_new_privs1, no_new_privs2;
ASSERT_EQ(0, pthread_create(&t1, NULL, idle, &no_new_privs1));
ASSERT_EQ(0, pthread_create(&t2, NULL, idle, &no_new_privs2));
- ASSERT_EQ(0, prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0));
+ if (variant->prior_no_new_privs) {
+ ASSERT_EQ(0, prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0));
+ } else {
+ /* No prior prctl(2) PR_SET_NO_NEW_PRIVS call. */
+ ASSERT_EQ(0, prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0));
+ }
- EXPECT_EQ(0, landlock_restrict_self(ruleset_fd,
- LANDLOCK_RESTRICT_SELF_TSYNC));
+ if (variant->expected_errno) {
+ EXPECT_EQ(-1, landlock_restrict_self(self->ruleset_fd,
+ variant->restrict_flags));
+ EXPECT_EQ(variant->expected_errno, errno);
+ } else {
+ EXPECT_EQ(0, landlock_restrict_self(self->ruleset_fd,
+ variant->restrict_flags));
+ }
+
+ /* Checks the no_new_privs state of the calling thread. */
+ EXPECT_EQ(variant->expected_no_new_privs,
+ prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0));
ASSERT_EQ(0, pthread_cancel(t1));
ASSERT_EQ(0, pthread_cancel(t2));
ASSERT_EQ(0, pthread_join(t1, NULL));
ASSERT_EQ(0, pthread_join(t2, NULL));
- /* The no_new_privs flag was implicitly enabled on all threads. */
- EXPECT_TRUE(no_new_privs1);
- EXPECT_TRUE(no_new_privs2);
-
- EXPECT_EQ(0, close(ruleset_fd));
+ /* Checks the no_new_privs state of the sibling threads. */
+ EXPECT_EQ(variant->expected_no_new_privs, no_new_privs1);
+ EXPECT_EQ(variant->expected_no_new_privs, no_new_privs2);
}
TEST(multi_threaded_success_despite_diverging_domains)
diff --git a/tools/testing/selftests/livepatch/functions.sh b/tools/testing/selftests/livepatch/functions.sh
index 30dc677b2f45..a65b7b1ac8ad 100644
--- a/tools/testing/selftests/livepatch/functions.sh
+++ b/tools/testing/selftests/livepatch/functions.sh
@@ -126,6 +126,20 @@ function set_ftrace_enabled() {
echo "livepatch: kernel.ftrace_enabled = $result" > /dev/kmsg
}
+# ftrace_disable_supported() - probe whether kernel.ftrace_enabled=0
+# can still disable ftrace on this kernel. Newer kernels deprecate
+# the knob and always refuse the write with -EOPNOTSUPP.
+function ftrace_disable_supported() {
+ local orig result
+
+ orig=$(sysctl --values kernel.ftrace_enabled)
+ sysctl -q kernel.ftrace_enabled=0 &> /dev/null
+ result=$(sysctl --values kernel.ftrace_enabled)
+ sysctl -q "kernel.ftrace_enabled=$orig" &> /dev/null
+
+ [[ "$result" == "0" ]]
+}
+
function cleanup() {
pop_config
}
diff --git a/tools/testing/selftests/livepatch/test-ftrace.sh b/tools/testing/selftests/livepatch/test-ftrace.sh
index d2c3dea63104..cd27148510f8 100755
--- a/tools/testing/selftests/livepatch/test-ftrace.sh
+++ b/tools/testing/selftests/livepatch/test-ftrace.sh
@@ -12,29 +12,32 @@ setup_config
# - turn ftrace_enabled OFF and verify livepatches can't load
# - turn ftrace_enabled ON and verify livepatch can load
# - verify that ftrace_enabled can't be turned OFF while a livepatch is loaded
+# (skipped on kernels where the sysctl is deprecated and always refuses 0)
start_test "livepatch interaction with ftrace_enabled sysctl"
-set_ftrace_enabled 0
-load_failing_mod $MOD_LIVEPATCH
+if ftrace_disable_supported; then
-set_ftrace_enabled 1
-load_lp $MOD_LIVEPATCH
-if [[ "$(cat /proc/cmdline)" != "$MOD_LIVEPATCH: this has been live patched" ]] ; then
- echo -e "FAIL\n\n"
- die "livepatch kselftest(s) failed"
-fi
+ set_ftrace_enabled 0
+ load_failing_mod $MOD_LIVEPATCH
-# Check that ftrace could not get disabled when a livepatch is enabled
-set_ftrace_enabled --fail 0
-if [[ "$(cat /proc/cmdline)" != "$MOD_LIVEPATCH: this has been live patched" ]] ; then
- echo -e "FAIL\n\n"
- die "livepatch kselftest(s) failed"
-fi
-disable_lp $MOD_LIVEPATCH
-unload_lp $MOD_LIVEPATCH
+ set_ftrace_enabled 1
+ load_lp $MOD_LIVEPATCH
+ if [[ "$(cat /proc/cmdline)" != "$MOD_LIVEPATCH: this has been live patched" ]] ; then
+ echo -e "FAIL\n\n"
+ die "livepatch kselftest(s) failed"
+ fi
-check_result "livepatch: kernel.ftrace_enabled = 0
+ # Check that ftrace could not get disabled when a livepatch is enabled
+ set_ftrace_enabled --fail 0
+ if [[ "$(cat /proc/cmdline)" != "$MOD_LIVEPATCH: this has been live patched" ]] ; then
+ echo -e "FAIL\n\n"
+ die "livepatch kselftest(s) failed"
+ fi
+ disable_lp $MOD_LIVEPATCH
+ unload_lp $MOD_LIVEPATCH
+
+ check_result "livepatch: kernel.ftrace_enabled = 0
% insmod test_modules/$MOD_LIVEPATCH.ko
livepatch: enabling patch '$MOD_LIVEPATCH'
livepatch: '$MOD_LIVEPATCH': initializing patching transition
@@ -60,6 +63,14 @@ livepatch: '$MOD_LIVEPATCH': completing unpatching transition
livepatch: '$MOD_LIVEPATCH': unpatching complete
% rmmod $MOD_LIVEPATCH"
+else
+
+ set_ftrace_enabled --fail 0
+ check_result "livepatch: sysctl: setting key \"kernel.ftrace_enabled\": \
+Operation not supported"
+
+fi
+
# - verify livepatch can load
# - check if traces have a patched function
diff --git a/tools/testing/selftests/liveupdate/.gitignore b/tools/testing/selftests/liveupdate/.gitignore
index 661827083ab6..47c670bf532b 100644
--- a/tools/testing/selftests/liveupdate/.gitignore
+++ b/tools/testing/selftests/liveupdate/.gitignore
@@ -3,7 +3,9 @@
!/**/
!*.c
!*.h
+!*.mk
!*.sh
!.gitignore
!config
+!config.*
!Makefile
diff --git a/tools/testing/selftests/liveupdate/Makefile b/tools/testing/selftests/liveupdate/Makefile
index 30689d22cb02..634211c66652 100644
--- a/tools/testing/selftests/liveupdate/Makefile
+++ b/tools/testing/selftests/liveupdate/Makefile
@@ -1,7 +1,5 @@
# SPDX-License-Identifier: GPL-2.0-only
-LIB_C += luo_test_utils.c
-
TEST_GEN_PROGS += liveupdate
TEST_GEN_PROGS_EXTENDED += luo_kexec_simple
@@ -12,25 +10,21 @@ TEST_GEN_PROGS_EXTENDED += luo_stress_files
TEST_FILES += do_kexec.sh
include ../lib.mk
+include lib/libliveupdate.mk
CFLAGS += $(KHDR_INCLUDES)
CFLAGS += -Wall -O2 -Wno-unused-function
CFLAGS += -MD
-LIB_O := $(patsubst %.c, $(OUTPUT)/%.o, $(LIB_C))
TEST_O := $(patsubst %, %.o, $(TEST_GEN_PROGS))
TEST_O += $(patsubst %, %.o, $(TEST_GEN_PROGS_EXTENDED))
-TEST_DEP_FILES := $(patsubst %.o, %.d, $(LIB_O))
+TEST_DEP_FILES := $(patsubst %.o, %.d, $(LIBLIVEUPDATE_O))
TEST_DEP_FILES += $(patsubst %.o, %.d, $(TEST_O))
-include $(TEST_DEP_FILES)
-$(LIB_O): $(OUTPUT)/%.o: %.c
- $(CC) $(CFLAGS) $(CPPFLAGS) $(TARGET_ARCH) -c $< -o $@
-
-$(TEST_GEN_PROGS) $(TEST_GEN_PROGS_EXTENDED): $(OUTPUT)/%: %.o $(LIB_O)
- $(CC) $(CFLAGS) $(CPPFLAGS) $(LDFLAGS) $(TARGET_ARCH) $< $(LIB_O) $(LDLIBS) -o $@
+$(TEST_GEN_PROGS) $(TEST_GEN_PROGS_EXTENDED): $(OUTPUT)/%: %.o $(LIBLIVEUPDATE_O)
+ $(CC) $(CFLAGS) $(CPPFLAGS) $(LDFLAGS) $(TARGET_ARCH) $< $(LIBLIVEUPDATE_O) $(LDLIBS) -o $@
-EXTRA_CLEAN += $(LIB_O)
EXTRA_CLEAN += $(TEST_O)
EXTRA_CLEAN += $(TEST_DEP_FILES)
diff --git a/tools/testing/selftests/liveupdate/config b/tools/testing/selftests/liveupdate/config
index 91d03f9a6a39..016d009dba13 100644
--- a/tools/testing/selftests/liveupdate/config
+++ b/tools/testing/selftests/liveupdate/config
@@ -1,4 +1,5 @@
CONFIG_BLK_DEV_INITRD=y
+CONFIG_DEVTMPFS=y
CONFIG_KEXEC_FILE=y
CONFIG_KEXEC_HANDOVER=y
CONFIG_KEXEC_HANDOVER_ENABLE_DEFAULT=y
diff --git a/tools/testing/selftests/liveupdate/config.aarch64 b/tools/testing/selftests/liveupdate/config.aarch64
new file mode 100644
index 000000000000..445716403925
--- /dev/null
+++ b/tools/testing/selftests/liveupdate/config.aarch64
@@ -0,0 +1,2 @@
+CONFIG_SERIAL_AMBA_PL011=y
+CONFIG_SERIAL_AMBA_PL011_CONSOLE=y
diff --git a/tools/testing/selftests/liveupdate/config.x86_64 b/tools/testing/selftests/liveupdate/config.x86_64
new file mode 100644
index 000000000000..810d9c9d213e
--- /dev/null
+++ b/tools/testing/selftests/liveupdate/config.x86_64
@@ -0,0 +1,2 @@
+CONFIG_SERIAL_8250=y
+CONFIG_SERIAL_8250_CONSOLE=y
diff --git a/tools/testing/selftests/liveupdate/init.c b/tools/testing/selftests/liveupdate/init.c
new file mode 100644
index 000000000000..fb08bd58b9b9
--- /dev/null
+++ b/tools/testing/selftests/liveupdate/init.c
@@ -0,0 +1,179 @@
+// SPDX-License-Identifier: GPL-2.0
+
+/*
+ * Copyright (c) 2025, Google LLC.
+ * Pasha Tatashin <pasha.tatashin@soleen.com>
+ */
+#include <fcntl.h>
+#include <linux/kexec.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/reboot.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#define COMMAND_LINE_SIZE 2048
+#define KERNEL_IMAGE "/kernel"
+#define INITRD_IMAGE "/initrd.img"
+#define TEST_BINARY "/test_binary"
+
+static int mount_filesystems(void)
+{
+ if (mount("devtmpfs", "/dev", "devtmpfs", 0, NULL) < 0) {
+ fprintf(stderr, "INIT: Warning: Failed to mount devtmpfs\n");
+ return -1;
+ }
+
+ if (mount("debugfs", "/debugfs", "debugfs", 0, NULL) < 0) {
+ fprintf(stderr, "INIT: Failed to mount debugfs\n");
+ return -1;
+ }
+
+ if (mount("proc", "/proc", "proc", 0, NULL) < 0) {
+ fprintf(stderr, "INIT: Failed to mount proc\n");
+ return -1;
+ }
+
+ return 0;
+}
+
+static long kexec_file_load(int kernel_fd, int initrd_fd,
+ unsigned long cmdline_len, const char *cmdline,
+ unsigned long flags)
+{
+ return syscall(__NR_kexec_file_load, kernel_fd, initrd_fd, cmdline_len,
+ cmdline, flags);
+}
+
+static int kexec_load(void)
+{
+ char cmdline[COMMAND_LINE_SIZE];
+ int kernel_fd, initrd_fd, err;
+ ssize_t len;
+ int fd;
+
+ fd = open("/proc/cmdline", O_RDONLY);
+ if (fd < 0) {
+ fprintf(stderr, "INIT: Failed to read /proc/cmdline\n");
+
+ return -1;
+ }
+
+ len = read(fd, cmdline, sizeof(cmdline) - 1);
+ close(fd);
+ if (len < 0)
+ return -1;
+
+ cmdline[len] = 0;
+ if (len > 0 && cmdline[len - 1] == '\n')
+ cmdline[len - 1] = 0;
+
+ strncat(cmdline, " luo_stage=2", sizeof(cmdline) - strlen(cmdline) - 1);
+
+ kernel_fd = open(KERNEL_IMAGE, O_RDONLY);
+ if (kernel_fd < 0) {
+ fprintf(stderr, "INIT: Failed to open kernel image\n");
+ return -1;
+ }
+
+ initrd_fd = open(INITRD_IMAGE, O_RDONLY);
+ if (initrd_fd < 0) {
+ fprintf(stderr, "INIT: Failed to open initrd image\n");
+ close(kernel_fd);
+ return -1;
+ }
+
+ err = kexec_file_load(kernel_fd, initrd_fd, strlen(cmdline) + 1,
+ cmdline, 0);
+
+ close(initrd_fd);
+ close(kernel_fd);
+
+ return err;
+}
+
+static int run_test(int stage)
+{
+ char stage_arg[32];
+ int status;
+ pid_t pid;
+
+ snprintf(stage_arg, sizeof(stage_arg), "%d", stage);
+
+ pid = fork();
+ if (pid < 0)
+ return -1;
+
+ if (!pid) {
+ char *const argv[] = {TEST_BINARY, "-s", stage_arg, NULL};
+
+ execve(TEST_BINARY, argv, NULL);
+ fprintf(stderr, "INIT: execve failed\n");
+ _exit(1);
+ }
+
+ waitpid(pid, &status, 0);
+
+ return (WIFEXITED(status) && WEXITSTATUS(status) == 0) ? 0 : -1;
+}
+
+static int get_current_stage(void)
+{
+ char cmdline[COMMAND_LINE_SIZE];
+ ssize_t len;
+ int fd;
+
+ fd = open("/proc/cmdline", O_RDONLY);
+ if (fd < 0)
+ return -1;
+
+ len = read(fd, cmdline, sizeof(cmdline) - 1);
+ close(fd);
+
+ if (len < 0)
+ return -1;
+
+ cmdline[len] = 0;
+
+ return strstr(cmdline, "luo_stage=2") ? 2 : 1;
+}
+
+int main(int argc, char *argv[])
+{
+ int current_stage;
+ int err;
+
+ if (mount_filesystems())
+ goto err_reboot;
+
+ current_stage = get_current_stage();
+ if (current_stage < 0) {
+ fprintf(stderr, "INIT: Failed to read cmdline");
+ goto err_reboot;
+ }
+
+ printf("INIT: Starting Stage %d\n", current_stage);
+
+ if (current_stage == 1 && kexec_load()) {
+ fprintf(stderr, "INIT: Failed to load kexec kernel\n");
+ goto err_reboot;
+ }
+
+ if (run_test(current_stage)) {
+ fprintf(stderr, "INIT: Test binary returned failure\n");
+ goto err_reboot;
+ }
+
+ printf("INIT: Stage %d completed successfully.\n", current_stage);
+ reboot(current_stage == 1 ? RB_KEXEC : RB_AUTOBOOT);
+
+ return 0;
+
+err_reboot:
+ reboot(RB_AUTOBOOT);
+
+ return -1;
+}
diff --git a/tools/testing/selftests/liveupdate/luo_test_utils.h b/tools/testing/selftests/liveupdate/lib/include/libliveupdate.h
index 6a0d85386613..fa07fed08364 100644
--- a/tools/testing/selftests/liveupdate/luo_test_utils.h
+++ b/tools/testing/selftests/liveupdate/lib/include/libliveupdate.h
@@ -7,13 +7,13 @@
* Utility functions for LUO kselftests.
*/
-#ifndef LUO_TEST_UTILS_H
-#define LUO_TEST_UTILS_H
+#ifndef SELFTESTS_LIVEUPDATE_LIB_LIVEUPDATE_H
+#define SELFTESTS_LIVEUPDATE_LIB_LIVEUPDATE_H
#include <errno.h>
#include <string.h>
#include <linux/liveupdate.h>
-#include "../kselftest.h"
+#include "../../../kselftest.h"
#define LUO_DEVICE "/dev/liveupdate"
@@ -25,8 +25,11 @@ int luo_open_device(void);
int luo_create_session(int luo_fd, const char *name);
int luo_retrieve_session(int luo_fd, const char *name);
int luo_session_finish(int session_fd);
+int luo_get_session_name(int session_fd, char *name, size_t name_len);
int luo_ensure_nofile_limit(long min_limit);
+int luo_session_preserve_fd(int session_fd, int fd, __u64 token);
+int luo_session_retrieve_fd(int session_fd, __u64 token);
int create_and_preserve_memfd(int session_fd, int token, const char *data);
int restore_and_verify_memfd(int session_fd, int token, const char *expected_data);
@@ -43,4 +46,4 @@ typedef void (*luo_test_stage2_fn)(int luo_fd, int state_session_fd);
int luo_test(int argc, char *argv[], const char *state_session_name,
luo_test_stage1_fn stage1, luo_test_stage2_fn stage2);
-#endif /* LUO_TEST_UTILS_H */
+#endif /* SELFTESTS_LIVEUPDATE_LIB_LIVEUPDATE_H */
diff --git a/tools/testing/selftests/liveupdate/lib/libliveupdate.mk b/tools/testing/selftests/liveupdate/lib/libliveupdate.mk
new file mode 100644
index 000000000000..634cd4c16c47
--- /dev/null
+++ b/tools/testing/selftests/liveupdate/lib/libliveupdate.mk
@@ -0,0 +1,20 @@
+include $(top_srcdir)/scripts/subarch.include
+ARCH ?= $(SUBARCH)
+
+LIBLIVEUPDATE_SRCDIR := $(selfdir)/liveupdate/lib
+
+LIBLIVEUPDATE_C := lu_utils.c
+
+LIBLIVEUPDATE_OUTPUT := $(OUTPUT)/libliveupdate
+
+LIBLIVEUPDATE_O := $(patsubst %.c, $(LIBLIVEUPDATE_OUTPUT)/%.o, $(LIBLIVEUPDATE_C))
+
+CFLAGS += -I$(LIBLIVEUPDATE_SRCDIR)/include
+
+$(LIBLIVEUPDATE_OUTPUT):
+ $(Q)mkdir -p $@
+
+$(LIBLIVEUPDATE_O): $(LIBLIVEUPDATE_OUTPUT)/%.o : $(LIBLIVEUPDATE_SRCDIR)/%.c | $(LIBLIVEUPDATE_OUTPUT)
+ $(CC) $(CFLAGS) $(CPPFLAGS) $(TARGET_ARCH) -c $< -o $@
+
+EXTRA_CLEAN += $(LIBLIVEUPDATE_OUTPUT)
diff --git a/tools/testing/selftests/liveupdate/luo_test_utils.c b/tools/testing/selftests/liveupdate/lib/lu_utils.c
index 333a3530051b..74d41115c281 100644
--- a/tools/testing/selftests/liveupdate/luo_test_utils.c
+++ b/tools/testing/selftests/liveupdate/lib/lu_utils.c
@@ -21,8 +21,9 @@
#include <sys/stat.h>
#include <errno.h>
#include <stdarg.h>
+#include <linux/unistd.h>
-#include "luo_test_utils.h"
+#include <libliveupdate.h>
int luo_open_device(void)
{
@@ -59,7 +60,7 @@ int luo_create_session(int luo_fd, const char *name)
snprintf((char *)arg.name, LIVEUPDATE_SESSION_NAME_LENGTH, "%.*s",
LIVEUPDATE_SESSION_NAME_LENGTH - 1, name);
- if (ioctl(luo_fd, LIVEUPDATE_IOCTL_CREATE_SESSION, &arg) < 0)
+ if (ioctl(luo_fd, LIVEUPDATE_IOCTL_CREATE_SESSION, &arg))
return -errno;
return arg.fd;
@@ -72,16 +73,58 @@ int luo_retrieve_session(int luo_fd, const char *name)
snprintf((char *)arg.name, LIVEUPDATE_SESSION_NAME_LENGTH, "%.*s",
LIVEUPDATE_SESSION_NAME_LENGTH - 1, name);
- if (ioctl(luo_fd, LIVEUPDATE_IOCTL_RETRIEVE_SESSION, &arg) < 0)
+ if (ioctl(luo_fd, LIVEUPDATE_IOCTL_RETRIEVE_SESSION, &arg))
return -errno;
return arg.fd;
}
+int luo_session_preserve_fd(int session_fd, int fd, __u64 token)
+{
+ struct liveupdate_session_preserve_fd arg = {
+ .size = sizeof(arg),
+ .fd = fd,
+ .token = token,
+ };
+
+ if (ioctl(session_fd, LIVEUPDATE_SESSION_PRESERVE_FD, &arg))
+ return -errno;
+
+ return 0;
+}
+
+int luo_session_retrieve_fd(int session_fd, __u64 token)
+{
+ struct liveupdate_session_retrieve_fd arg = {
+ .size = sizeof(arg),
+ .token = token,
+ };
+
+ if (ioctl(session_fd, LIVEUPDATE_SESSION_RETRIEVE_FD, &arg))
+ return -errno;
+
+ return arg.fd;
+}
+
+/* Helper function to get a session name via ioctl. */
+int luo_get_session_name(int session_fd, char *name, size_t name_len)
+{
+ struct liveupdate_session_get_name args = {};
+
+ args.size = sizeof(args);
+
+ if (ioctl(session_fd, LIVEUPDATE_SESSION_GET_NAME, &args))
+ return -errno;
+
+ strncpy(name, (char *)args.name, name_len - 1);
+ name[name_len - 1] = '\0';
+
+ return 0;
+}
+
int create_and_preserve_memfd(int session_fd, int token, const char *data)
{
- struct liveupdate_session_preserve_fd arg = { .size = sizeof(arg) };
- long page_size = sysconf(_SC_PAGE_SIZE);
+ long page_size = getpagesize();
void *map = MAP_FAILED;
int mfd = -1, ret = -1;
@@ -99,9 +142,8 @@ int create_and_preserve_memfd(int session_fd, int token, const char *data)
snprintf(map, page_size, "%s", data);
munmap(map, page_size);
- arg.fd = mfd;
- arg.token = token;
- if (ioctl(session_fd, LIVEUPDATE_SESSION_PRESERVE_FD, &arg) < 0)
+ ret = luo_session_preserve_fd(session_fd, mfd, token);
+ if (ret)
goto out;
ret = 0;
@@ -116,15 +158,13 @@ out:
int restore_and_verify_memfd(int session_fd, int token,
const char *expected_data)
{
- struct liveupdate_session_retrieve_fd arg = { .size = sizeof(arg) };
- long page_size = sysconf(_SC_PAGE_SIZE);
+ long page_size = getpagesize();
void *map = MAP_FAILED;
int mfd = -1, ret = -1;
- arg.token = token;
- if (ioctl(session_fd, LIVEUPDATE_SESSION_RETRIEVE_FD, &arg) < 0)
- return -errno;
- mfd = arg.fd;
+ mfd = luo_session_retrieve_fd(session_fd, token);
+ if (mfd < 0)
+ return mfd;
map = mmap(NULL, page_size, PROT_READ, MAP_SHARED, mfd, 0);
if (map == MAP_FAILED)
@@ -228,16 +268,11 @@ void daemonize_and_wait(void)
static int parse_stage_args(int argc, char *argv[])
{
- static struct option long_options[] = {
- {"stage", required_argument, 0, 's'},
- {0, 0, 0, 0}
- };
- int option_index = 0;
int stage = 1;
int opt;
optind = 1;
- while ((opt = getopt_long(argc, argv, "s:", long_options, &option_index)) != -1) {
+ while ((opt = getopt(argc, argv, "s:")) != -1) {
switch (opt) {
case 's':
stage = atoi(optarg);
@@ -248,6 +283,7 @@ static int parse_stage_args(int argc, char *argv[])
fail_exit("Unknown argument");
}
}
+
return stage;
}
@@ -275,7 +311,7 @@ int luo_test(int argc, char *argv[],
fail_exit("Failed to check for state session");
if (target_stage != detected_stage) {
- ksft_exit_fail_msg("Stage mismatch Requested --stage %d, but system is in stage %d.\n"
+ ksft_exit_fail_msg("Stage mismatch Requested stage %d, but system is in stage %d.\n"
"(State session %s: %s)\n",
target_stage, detected_stage, state_session_name,
(detected_stage == 2) ? "EXISTS" : "MISSING");
diff --git a/tools/testing/selftests/liveupdate/liveupdate.c b/tools/testing/selftests/liveupdate/liveupdate.c
index 502fb3567e38..2dedd5fc2534 100644
--- a/tools/testing/selftests/liveupdate/liveupdate.c
+++ b/tools/testing/selftests/liveupdate/liveupdate.c
@@ -24,9 +24,9 @@
#include <sys/ioctl.h>
#include <unistd.h>
+#include <libliveupdate.h>
#include <linux/liveupdate.h>
-#include "luo_test_utils.h"
#include "../kselftest.h"
#include "../kselftest_harness.h"
@@ -89,36 +89,6 @@ TEST_F(liveupdate_device, exclusive_open)
EXPECT_EQ(errno, EBUSY);
}
-/* Helper function to create a LUO session via ioctl. */
-static int create_session(int lu_fd, const char *name)
-{
- struct liveupdate_ioctl_create_session args = {};
-
- args.size = sizeof(args);
- strncpy((char *)args.name, name, sizeof(args.name) - 1);
-
- if (ioctl(lu_fd, LIVEUPDATE_IOCTL_CREATE_SESSION, &args))
- return -errno;
-
- return args.fd;
-}
-
-/* Helper function to get a session name via ioctl. */
-static int get_session_name(int session_fd, char *name, size_t name_len)
-{
- struct liveupdate_session_get_name args = {};
-
- args.size = sizeof(args);
-
- if (ioctl(session_fd, LIVEUPDATE_SESSION_GET_NAME, &args))
- return -errno;
-
- strncpy(name, (char *)args.name, name_len - 1);
- name[name_len - 1] = '\0';
-
- return 0;
-}
-
/*
* Test Case: Create Duplicate Session
*
@@ -135,10 +105,10 @@ TEST_F(liveupdate_device, create_duplicate_session)
ASSERT_GE(self->fd1, 0);
- session_fd1 = create_session(self->fd1, "duplicate-session-test");
+ session_fd1 = luo_create_session(self->fd1, "duplicate-session-test");
ASSERT_GE(session_fd1, 0);
- session_fd2 = create_session(self->fd1, "duplicate-session-test");
+ session_fd2 = luo_create_session(self->fd1, "duplicate-session-test");
EXPECT_LT(session_fd2, 0);
EXPECT_EQ(-session_fd2, EEXIST);
@@ -160,30 +130,16 @@ TEST_F(liveupdate_device, create_distinct_sessions)
ASSERT_GE(self->fd1, 0);
- session_fd1 = create_session(self->fd1, "distinct-session-1");
+ session_fd1 = luo_create_session(self->fd1, "distinct-session-1");
ASSERT_GE(session_fd1, 0);
- session_fd2 = create_session(self->fd1, "distinct-session-2");
+ session_fd2 = luo_create_session(self->fd1, "distinct-session-2");
ASSERT_GE(session_fd2, 0);
ASSERT_EQ(close(session_fd1), 0);
ASSERT_EQ(close(session_fd2), 0);
}
-static int preserve_fd(int session_fd, int fd_to_preserve, __u64 token)
-{
- struct liveupdate_session_preserve_fd args = {};
-
- args.size = sizeof(args);
- args.fd = fd_to_preserve;
- args.token = token;
-
- if (ioctl(session_fd, LIVEUPDATE_SESSION_PRESERVE_FD, &args))
- return -errno;
-
- return 0;
-}
-
/*
* Test Case: Preserve MemFD
*
@@ -201,14 +157,14 @@ TEST_F(liveupdate_device, preserve_memfd)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, "preserve-memfd-test");
+ session_fd = luo_create_session(self->fd1, "preserve-memfd-test");
ASSERT_GE(session_fd, 0);
mem_fd = memfd_create("test-memfd", 0);
ASSERT_GE(mem_fd, 0);
ASSERT_EQ(write(mem_fd, test_str, strlen(test_str)), strlen(test_str));
- ASSERT_EQ(preserve_fd(session_fd, mem_fd, 0x1234), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd, mem_fd, 0x1234), 0);
ASSERT_EQ(close(session_fd), 0);
ASSERT_EQ(lseek(mem_fd, 0, SEEK_SET), 0);
@@ -236,7 +192,7 @@ TEST_F(liveupdate_device, preserve_multiple_memfds)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, "preserve-multi-memfd-test");
+ session_fd = luo_create_session(self->fd1, "preserve-multi-memfd-test");
ASSERT_GE(session_fd, 0);
mem_fd1 = memfd_create("test-memfd-1", 0);
@@ -247,8 +203,8 @@ TEST_F(liveupdate_device, preserve_multiple_memfds)
ASSERT_EQ(write(mem_fd1, test_str1, strlen(test_str1)), strlen(test_str1));
ASSERT_EQ(write(mem_fd2, test_str2, strlen(test_str2)), strlen(test_str2));
- ASSERT_EQ(preserve_fd(session_fd, mem_fd1, 0xAAAA), 0);
- ASSERT_EQ(preserve_fd(session_fd, mem_fd2, 0xBBBB), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd, mem_fd1, 0xAAAA), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd, mem_fd2, 0xBBBB), 0);
memset(read_buf, 0, sizeof(read_buf));
ASSERT_EQ(lseek(mem_fd1, 0, SEEK_SET), 0);
@@ -284,9 +240,9 @@ TEST_F(liveupdate_device, preserve_complex_scenario)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd1 = create_session(self->fd1, "complex-session-1");
+ session_fd1 = luo_create_session(self->fd1, "complex-session-1");
ASSERT_GE(session_fd1, 0);
- session_fd2 = create_session(self->fd1, "complex-session-2");
+ session_fd2 = luo_create_session(self->fd1, "complex-session-2");
ASSERT_GE(session_fd2, 0);
mem_fd_data1 = memfd_create("data1", 0);
@@ -303,10 +259,10 @@ TEST_F(liveupdate_device, preserve_complex_scenario)
mem_fd_empty2 = memfd_create("empty2", 0);
ASSERT_GE(mem_fd_empty2, 0);
- ASSERT_EQ(preserve_fd(session_fd1, mem_fd_data1, 0x1111), 0);
- ASSERT_EQ(preserve_fd(session_fd1, mem_fd_empty1, 0x2222), 0);
- ASSERT_EQ(preserve_fd(session_fd2, mem_fd_data2, 0x3333), 0);
- ASSERT_EQ(preserve_fd(session_fd2, mem_fd_empty2, 0x4444), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd1, mem_fd_data1, 0x1111), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd1, mem_fd_empty1, 0x2222), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd2, mem_fd_data2, 0x3333), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd2, mem_fd_empty2, 0x4444), 0);
ASSERT_EQ(lseek(mem_fd_data1, 0, SEEK_SET), 0);
ASSERT_EQ(read(mem_fd_data1, read_buf, sizeof(read_buf)), strlen(data1));
@@ -349,13 +305,13 @@ TEST_F(liveupdate_device, preserve_unsupported_fd)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, "unsupported-fd-test");
+ session_fd = luo_create_session(self->fd1, "unsupported-fd-test");
ASSERT_GE(session_fd, 0);
unsupported_fd = open("/dev/null", O_RDWR);
ASSERT_GE(unsupported_fd, 0);
- ret = preserve_fd(session_fd, unsupported_fd, 0xDEAD);
+ ret = luo_session_preserve_fd(session_fd, unsupported_fd, 0xDEAD);
EXPECT_EQ(ret, -ENOENT);
ASSERT_EQ(close(unsupported_fd), 0);
@@ -379,23 +335,23 @@ TEST_F(liveupdate_device, prevent_double_preservation)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd1 = create_session(self->fd1, "double-preserve-session-1");
+ session_fd1 = luo_create_session(self->fd1, "double-preserve-session-1");
ASSERT_GE(session_fd1, 0);
- session_fd2 = create_session(self->fd1, "double-preserve-session-2");
+ session_fd2 = luo_create_session(self->fd1, "double-preserve-session-2");
ASSERT_GE(session_fd2, 0);
mem_fd = memfd_create("test-memfd", 0);
ASSERT_GE(mem_fd, 0);
/* First preservation should succeed */
- ASSERT_EQ(preserve_fd(session_fd1, mem_fd, 0x1111), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd1, mem_fd, 0x1111), 0);
/* Second preservation in a different session should fail with EBUSY */
- ret = preserve_fd(session_fd2, mem_fd, 0x2222);
+ ret = luo_session_preserve_fd(session_fd2, mem_fd, 0x2222);
EXPECT_EQ(ret, -EBUSY);
/* Second preservation in the same session (different token) should fail with EBUSY */
- ret = preserve_fd(session_fd1, mem_fd, 0x3333);
+ ret = luo_session_preserve_fd(session_fd1, mem_fd, 0x3333);
EXPECT_EQ(ret, -EBUSY);
ASSERT_EQ(close(mem_fd), 0);
@@ -441,7 +397,7 @@ TEST_F(liveupdate_device, create_session_empty_name)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, "");
+ session_fd = luo_create_session(self->fd1, "");
EXPECT_EQ(session_fd, -EINVAL);
}
@@ -462,10 +418,10 @@ TEST_F(liveupdate_device, get_session_name)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, session_name);
+ session_fd = luo_create_session(self->fd1, session_name);
ASSERT_GE(session_fd, 0);
- ASSERT_EQ(get_session_name(session_fd, name_buf, sizeof(name_buf)), 0);
+ ASSERT_EQ(luo_get_session_name(session_fd, name_buf, sizeof(name_buf)), 0);
ASSERT_STREQ(name_buf, session_name);
ASSERT_EQ(close(session_fd), 0);
@@ -491,10 +447,10 @@ TEST_F(liveupdate_device, get_session_name_max_length)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, long_name);
+ session_fd = luo_create_session(self->fd1, long_name);
ASSERT_GE(session_fd, 0);
- ASSERT_EQ(get_session_name(session_fd, name_buf, sizeof(name_buf)), 0);
+ ASSERT_EQ(luo_get_session_name(session_fd, name_buf, sizeof(name_buf)), 0);
ASSERT_STREQ(name_buf, long_name);
ASSERT_EQ(close(session_fd), 0);
@@ -528,7 +484,7 @@ TEST_F(liveupdate_device, preserve_many_sessions)
char name[64];
snprintf(name, sizeof(name), "many-session-%d", i);
- session_fds[i] = create_session(self->fd1, name);
+ session_fds[i] = luo_create_session(self->fd1, name);
ASSERT_GE(session_fds[i], 0);
}
@@ -554,7 +510,7 @@ TEST_F(liveupdate_device, preserve_many_files)
SKIP(return, "%s does not exist", LIVEUPDATE_DEV);
ASSERT_GE(self->fd1, 0);
- session_fd = create_session(self->fd1, "many-files-test");
+ session_fd = luo_create_session(self->fd1, "many-files-test");
ASSERT_GE(session_fd, 0);
ret = luo_ensure_nofile_limit(MANY_FILES + 10);
@@ -565,7 +521,7 @@ TEST_F(liveupdate_device, preserve_many_files)
for (i = 0; i < MANY_FILES; i++) {
mem_fds[i] = memfd_create("test-memfd", 0);
ASSERT_GE(mem_fds[i], 0);
- ASSERT_EQ(preserve_fd(session_fd, mem_fds[i], i), 0);
+ ASSERT_EQ(luo_session_preserve_fd(session_fd, mem_fds[i], i), 0);
}
for (i = 0; i < MANY_FILES; i++)
diff --git a/tools/testing/selftests/liveupdate/luo_kexec_simple.c b/tools/testing/selftests/liveupdate/luo_kexec_simple.c
index d7ac1f3dc4cb..786ac93b9ae3 100644
--- a/tools/testing/selftests/liveupdate/luo_kexec_simple.c
+++ b/tools/testing/selftests/liveupdate/luo_kexec_simple.c
@@ -8,7 +8,7 @@
* across a single kexec reboot.
*/
-#include "luo_test_utils.h"
+#include <libliveupdate.h>
#define TEST_SESSION_NAME "test-session"
#define TEST_MEMFD_TOKEN 0x1A
diff --git a/tools/testing/selftests/liveupdate/luo_multi_session.c b/tools/testing/selftests/liveupdate/luo_multi_session.c
index 0ee2d795beef..aac24a5f5ce3 100644
--- a/tools/testing/selftests/liveupdate/luo_multi_session.c
+++ b/tools/testing/selftests/liveupdate/luo_multi_session.c
@@ -9,7 +9,7 @@
* files.
*/
-#include "luo_test_utils.h"
+#include <libliveupdate.h>
#define SESSION_EMPTY_1 "multi-test-empty-1"
#define SESSION_EMPTY_2 "multi-test-empty-2"
diff --git a/tools/testing/selftests/liveupdate/luo_stress_files.c b/tools/testing/selftests/liveupdate/luo_stress_files.c
index 0cdf9cd4bac7..a0d48490f4ed 100644
--- a/tools/testing/selftests/liveupdate/luo_stress_files.c
+++ b/tools/testing/selftests/liveupdate/luo_stress_files.c
@@ -10,7 +10,8 @@
#include <stdio.h>
#include <unistd.h>
-#include "luo_test_utils.h"
+
+#include <libliveupdate.h>
#define NUM_FILES 500
#define STATE_SESSION_NAME "kexec_many_files_state"
diff --git a/tools/testing/selftests/liveupdate/luo_stress_sessions.c b/tools/testing/selftests/liveupdate/luo_stress_sessions.c
index f201b1839d1d..278aebabe0e9 100644
--- a/tools/testing/selftests/liveupdate/luo_stress_sessions.c
+++ b/tools/testing/selftests/liveupdate/luo_stress_sessions.c
@@ -10,7 +10,8 @@
#include <stdio.h>
#include <unistd.h>
-#include "luo_test_utils.h"
+
+#include <libliveupdate.h>
#define NUM_SESSIONS 2000
#define STATE_SESSION_NAME "kexec_many_state"
diff --git a/tools/testing/selftests/liveupdate/run-vmtests.sh b/tools/testing/selftests/liveupdate/run-vmtests.sh
new file mode 100755
index 000000000000..d656ce58c5a7
--- /dev/null
+++ b/tools/testing/selftests/liveupdate/run-vmtests.sh
@@ -0,0 +1,97 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+
+SCRIPT_DIR=$(dirname "$(realpath "$0")")
+TEST_RUNNER="$SCRIPT_DIR/vmtest.sh"
+
+TARGETS=("x86_64" "aarch64")
+
+GREEN='\033[0;32m'
+RED='\033[0;31m'
+YELLOW='\033[1;33m'
+NC='\033[0m'
+
+passed=0
+failed=0
+skipped=0
+
+TEST_NAMES=(
+ "luo_kexec_simple"
+ "luo_multi_session"
+ "luo_stress_files"
+ "luo_stress_sessions"
+)
+
+function usage() {
+ cat <<EOF
+$0 [-k] [-o output_dir] [-h]
+Options:
+ -k) keep logs
+ -o) specify output directory
+ -h) display this help
+EOF
+}
+
+function pass() {
+ echo -e "${GREEN}PASS${NC}"
+ ((passed++))
+}
+
+function skip() {
+ echo -e "${YELLOW}SKIP${NC}"
+ ((skipped++))
+}
+
+function fail() {
+ echo -e "${RED}FAIL${NC}"
+ ((failed++))
+}
+
+function main() {
+ while getopts 'hko:' opt; do
+ case $opt in
+ k) keep_logs=1 ;;
+ o) output_dir=$OPTARG ;;
+ h) usage; exit 0 ;;
+ *) echo Unknown argument "$opt"
+ usage; exit 1 ;;
+ esac
+ done
+
+ if [ -n "$keep_logs" ]; then
+ if [ -z "$output_dir" ]; then
+ output_dir="$SCRIPT_DIR/results_$(date +%Y%m%d_%H%M%S)"
+ fi;
+ mkdir -p "$output_dir"
+ else
+ output_dir=$(mktemp -d /tmp/luo.XXXXXXXX)
+ trap 'rm -fr "$output_dir"' EXIT
+ fi
+
+ for arch in "${TARGETS[@]}"; do
+ for test_name in "${TEST_NAMES[@]}"; do
+ log="$output_dir/${arch}_${test_name}.log"
+
+ printf " -> %-8s %-24s ... " "$arch" "$test_name"
+
+ "$TEST_RUNNER" -t "$arch" -T "$test_name" &> "$log"
+ exit_code=$?
+
+ case $exit_code in
+ 0) pass;;
+ 4) skip;;
+ *) fail;;
+ esac
+ done
+ echo ""
+ done
+
+ echo "SUMMARY: PASS=$passed SKIP=$skipped FAIL=$failed"
+ if [ -n "$keep_logs" ]; then
+ echo "Logs: $output_dir"
+ fi
+
+ exit $((failed != 0))
+}
+
+main "$@"
diff --git a/tools/testing/selftests/liveupdate/vmtest.sh b/tools/testing/selftests/liveupdate/vmtest.sh
new file mode 100755
index 000000000000..64fd2ab55b76
--- /dev/null
+++ b/tools/testing/selftests/liveupdate/vmtest.sh
@@ -0,0 +1,263 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+
+set -ue
+
+CROSS_COMPILE="${CROSS_COMPILE:-""}"
+
+test_dir=$(realpath "$(dirname "$0")")
+kernel_dir=$(realpath "$test_dir/../../../..")
+
+workspace_dir=""
+headers_dir=""
+initrd=""
+KEEP_WORKSPACE=0
+
+source "$test_dir/../kselftest/ktap_helpers.sh"
+
+function get_arch_conf() {
+ local arch=$1
+ if [[ "$arch" == "arm64" ]]; then
+ QEMU_CMD="qemu-system-aarch64 -M virt -cpu max"
+ KERNEL_IMAGE="Image"
+ KERNEL_CMDLINE="console=ttyAMA0"
+ elif [[ "$arch" == "x86" ]]; then
+ QEMU_CMD="qemu-system-x86_64"
+ KERNEL_IMAGE="bzImage"
+ KERNEL_CMDLINE="console=ttyS0"
+ else
+ echo "Unsupported architecture: $arch"
+ exit 1
+ fi
+}
+
+function usage() {
+ cat <<EOF
+$0 [-d build_dir] [-j jobs] [-t target_arch] [-T test_name] [-w workspace_dir] [-k] [-h]
+Options:
+ -d) path to the kernel build directory (default: .luo_test_build.<arch>)
+ -j) number of jobs for compilation
+ -t) run test for target_arch (aarch64, x86_64)
+ -T) test name to run (default: luo_kexec_simple)
+ -w) custom workspace directory (default: creates temp dir)
+ -k) keep workspace directory after successful test
+ -h) display this help
+EOF
+}
+
+function cleanup() {
+ if [ "$KEEP_WORKSPACE" -eq 1 ]; then
+ echo "# Workspace preserved at: $workspace_dir"
+ else
+ rm -fr "$workspace_dir"
+ fi
+
+ ktap_finished
+}
+
+function skip() {
+ local msg=${1:-""}
+ ktap_test_skip "$msg"
+ exit "$KSFT_SKIP"
+}
+
+function fail() {
+ local msg=${1:-""}
+ ktap_test_fail "$msg"
+ exit "$KSFT_FAIL"
+}
+
+function detect_cross_compile() {
+ local target=$1
+ local host=$(uname -m)
+
+ [[ "$host" == "arm64" ]] && host="aarch64"
+ [[ "$target" == "arm64" ]] && target="aarch64"
+
+ if [[ "$host" == "$target" ]]; then
+ CROSS_COMPILE=""
+ return
+ fi
+
+ if [[ -n "$CROSS_COMPILE" ]]; then
+ return
+ fi
+
+ local candidate=""
+ case "$target" in
+ aarch64) candidate="aarch64-linux-gnu-" ;;
+ x86_64) candidate="x86_64-linux-gnu-" ;;
+ *) skip "Auto-detection for target '$target' not supported. Please set CROSS_COMPILE manually." ;;
+ esac
+
+ if command -v "${candidate}gcc" &> /dev/null; then
+ CROSS_COMPILE="$candidate"
+ else
+ skip "Compiler '${candidate}gcc' not found. Please install it (e.g., 'apt install gcc-aarch64-linux-gnu') or set CROSS_COMPILE."
+ fi
+}
+
+function build_kernel() {
+ local build_dir=$1
+ local make_cmd=$2
+ local kimage=$3
+ local target_arch=$4
+
+ local luo_config="$build_dir/luo.config"
+ local kconfig="$build_dir/.config"
+ local common_conf="$test_dir/config"
+ local arch_conf="$test_dir/config.$target_arch"
+
+ echo "# Building kernel in: $build_dir"
+
+ cat "$arch_conf" "$common_conf" | tee "$kconfig" > "$luo_config"
+ $make_cmd olddefconfig
+
+ # verify that kernel confiration has all necessary options
+ while read -r opt ; do
+ grep "$opt" "$kconfig" &>/dev/null || skip "$opt is missing"
+ done < "$luo_config"
+
+ $make_cmd "$kimage"
+ $make_cmd headers_install INSTALL_HDR_PATH="$headers_dir"
+}
+
+function mkinitrd() {
+ local build_dir=$1
+ local kernel_path=$2
+ local test_name=$3
+
+ # Compile the test binary and the init process
+ "$CROSS_COMPILE"gcc -static -O2 -nostdinc -nostdlib \
+ -I "$headers_dir/include" \
+ -I "$kernel_dir/tools/include/nolibc" \
+ -I "$test_dir/lib/include" \
+ -o "$workspace_dir/test_binary" \
+ "$test_dir/$test_name.c" "$test_dir/lib/lu_utils.c"
+
+ "$CROSS_COMPILE"gcc -s -static -Os -nostdinc -nostdlib \
+ -fno-asynchronous-unwind-tables -fno-ident \
+ -fno-stack-protector \
+ -I "$headers_dir/include" \
+ -I "$kernel_dir/tools/include/nolibc" \
+ -o "$workspace_dir/init" "$test_dir/init.c"
+
+ cat > "$workspace_dir/cpio_list_inner" <<EOF
+dir /dev 0755 0 0
+dir /proc 0755 0 0
+dir /debugfs 0755 0 0
+nod /dev/console 0600 0 0 c 5 1
+file /init $workspace_dir/init 0755 0 0
+file /test_binary $workspace_dir/test_binary 0755 0 0
+EOF
+
+ # Generate inner_initrd.cpio
+ "$build_dir/usr/gen_init_cpio" "$workspace_dir/cpio_list_inner" > "$workspace_dir/inner_initrd.cpio"
+
+ cat > "$workspace_dir/cpio_list" <<EOF
+dir /dev 0755 0 0
+dir /proc 0755 0 0
+dir /debugfs 0755 0 0
+nod /dev/console 0600 0 0 c 5 1
+file /init $workspace_dir/init 0755 0 0
+file /kernel $kernel_path 0644 0 0
+file /test_binary $workspace_dir/test_binary 0755 0 0
+file /initrd.img $workspace_dir/inner_initrd.cpio 0644 0 0
+EOF
+
+ # Generate the final initrd
+ "$build_dir/usr/gen_init_cpio" "$workspace_dir/cpio_list" > "$initrd"
+}
+
+function run_qemu() {
+ local qemu_cmd=$1
+ local cmdline=$2
+ local kernel_path=$3
+ local serial="$workspace_dir/qemu.serial"
+
+ cmdline="$cmdline liveupdate=on panic=-1"
+
+ echo "# Serial Log: $serial"
+ timeout 30s \
+ $qemu_cmd -m 1G -smp 2 -no-reboot -nographic -nodefaults \
+ -accel tcg -accel hvf -accel kvm \
+ -serial file:"$serial" \
+ -append "$cmdline" \
+ -kernel "$kernel_path" \
+ -initrd "$initrd"
+
+ grep "TEST PASSED" "$serial" &> /dev/null || fail "Liveupdate failed"
+}
+
+function target_to_arch() {
+ local target=$1
+ case $target in
+ aarch64) echo "arm64" ;;
+ x86_64) echo "x86" ;;
+ *) skip "architecture $target is not supported"
+ esac
+}
+
+function main() {
+ local build_dir=""
+ local jobs=$(nproc)
+ local target="$(uname -m)"
+ local test_name="luo_kexec_simple"
+ local workspace_arg=""
+
+ set -o errtrace
+ trap fail ERR
+
+ while getopts 'hd:j:t:T:w:k' opt; do
+ case $opt in
+ d) build_dir="$OPTARG" ;;
+ j) jobs="$OPTARG" ;;
+ t) target="$OPTARG" ;;
+ T) test_name="$OPTARG" ;;
+ w) workspace_arg="$OPTARG" ;;
+ k) KEEP_WORKSPACE=1 ;;
+ h) usage; exit 0 ;;
+ *) echo "Unknown argument $opt"; usage; exit 1 ;;
+ esac
+ done
+
+ ktap_print_header
+ ktap_set_plan 1
+ trap cleanup EXIT
+
+ if [ -n "$workspace_arg" ]; then
+ workspace_dir="$(realpath -m "$workspace_arg")"
+ mkdir -p "$workspace_dir"
+ else
+ workspace_dir=$(mktemp -d /tmp/luo-test.XXXXXXXX)
+ fi
+
+ echo "# Workspace created at: $workspace_dir"
+ headers_dir="$workspace_dir/usr"
+ initrd="$workspace_dir/initrd.cpio"
+
+ detect_cross_compile "$target"
+
+ local arch=$(target_to_arch "$target")
+
+ if [ -z "$build_dir" ]; then
+ build_dir="$kernel_dir/.luo_test_build.$arch"
+ fi
+
+ mkdir -p "$build_dir"
+ build_dir=$(realpath "$build_dir")
+ get_arch_conf "$arch"
+
+ local make_cmd="make -s ARCH=$arch CROSS_COMPILE=$CROSS_COMPILE -j$jobs"
+ local make_cmd_build="$make_cmd -C $kernel_dir O=$build_dir"
+
+ build_kernel "$build_dir" "$make_cmd_build" "$KERNEL_IMAGE" "$target"
+
+ local final_kernel="$build_dir/arch/$arch/boot/$KERNEL_IMAGE"
+ mkinitrd "$build_dir" "$final_kernel" "$test_name"
+
+ run_qemu "$QEMU_CMD" "$KERNEL_CMDLINE" "$final_kernel"
+ ktap_test_pass "$test_name succeeded"
+}
+
+main "$@"
diff --git a/tools/testing/selftests/lsm/common.c b/tools/testing/selftests/lsm/common.c
index 9ad258912646..927dce4f04cb 100644
--- a/tools/testing/selftests/lsm/common.c
+++ b/tools/testing/selftests/lsm/common.c
@@ -76,7 +76,7 @@ int attr_lsm_count(void)
return 0;
if (read_sysfs_lsms(names, sysconf(_SC_PAGESIZE)))
- return 0;
+ goto out;
if (strstr(names, "selinux"))
count++;
@@ -85,5 +85,7 @@ int attr_lsm_count(void)
if (strstr(names, "apparmor"))
count++;
+out:
+ free(names);
return count;
}
diff --git a/tools/testing/selftests/mincore/mincore_selftest.c b/tools/testing/selftests/mincore/mincore_selftest.c
index cdd022c1c497..18dfbf318b1f 100644
--- a/tools/testing/selftests/mincore/mincore_selftest.c
+++ b/tools/testing/selftests/mincore/mincore_selftest.c
@@ -242,8 +242,10 @@ TEST(check_file_mmap)
}
/*
- * Touch a page in the middle of the mapping. We expect the next
- * few pages (the readahead window) to be populated too.
+ * Touch a page in the middle of the mapping. We expect some
+ * surrounding pages (the readahead window) to be populated too.
+ * Depending on the page size and readahead setting, the pages may
+ * land before the faulted page rather than after it.
*/
addr[FILE_SIZE / 2] = 1;
retval = mincore(addr, FILE_SIZE, vec);
@@ -252,6 +254,12 @@ TEST(check_file_mmap)
TH_LOG("Page not found in memory after use");
}
+ i = FILE_SIZE / 2 / page_size - 1;
+ while (i >= 0 && vec[i]) {
+ ra_pages++;
+ i--;
+ }
+
i = FILE_SIZE / 2 / page_size + 1;
while (i < vec_size && vec[i]) {
ra_pages++;
diff --git a/tools/testing/selftests/mm/.gitignore b/tools/testing/selftests/mm/.gitignore
index 9ccd9e1447e6..fcd892ed21e3 100644
--- a/tools/testing/selftests/mm/.gitignore
+++ b/tools/testing/selftests/mm/.gitignore
@@ -1,68 +1,10 @@
# SPDX-License-Identifier: GPL-2.0-only
-cow
-hugepage-mmap
-hugepage-mremap
-hugepage-shm
-hugepage-vmemmap
-hugetlb-mmap
-hugetlb-mremap
-hugetlb-shm
-hugetlb-vmemmap
-hugetlb-madvise
-hugetlb-read-hwpoison
-hugetlb-soft-offline
-khugepaged
-map_hugetlb
-map_populate
-thuge-gen
-compaction_test
-memory-failure
-migration
-mlock2-tests
-mrelease_test
-mremap_dontunmap
-mremap_test
-on-fault-limit
-transhuge-stress
-pagemap_ioctl
-pfnmap
-process_madv
-*.tmp*
-protection_keys
-protection_keys_32
-protection_keys_64
-madv_populate
-uffd-stress
-uffd-unit-tests
-uffd-wp-mremap
-mlock-intersect-test
-mlock-random-test
-virtual_address_range
-gup_test
-va_128TBswitch
-map_fixed_noreplace
-write_to_hugetlbfs
-hmm-tests
-memfd_secret
-soft-dirty
-split_huge_page_test
-ksm_tests
-local_config.h
-local_config.mk
-ksm_functional_tests
-mdwe_test
-gup_longterm
-mkdirty
-va_high_addr_switch
-hugetlb_fault_after_madv
-hugetlb_madv_vs_map
-mseal_test
-droppable
-hugetlb_dio
-pkey_sighandler_tests_32
-pkey_sighandler_tests_64
-guard-regions
-merge
-prctl_thp_disable
-rmap
-folio_split_race_test
+*
+!/**/
+!*.c
+!*.h
+!*.sh
+!.gitignore
+!Makefile
+!config
+!settings
diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile
index e6df968f0971..2d5366196e30 100644
--- a/tools/testing/selftests/mm/Makefile
+++ b/tools/testing/selftests/mm/Makefile
@@ -5,7 +5,7 @@
# script so kunit knows to run it, and add it to the list below.
# If you do not YOUR TESTS WILL NOT RUN IN THE CI.
-LOCAL_HDRS += $(selfdir)/mm/local_config.h $(top_srcdir)/mm/gup_test.h
+LOCAL_HDRS += $(selfdir)/mm/local_config.h_gen $(top_srcdir)/mm/gup_test.h
LOCAL_HDRS += $(selfdir)/mm/mseal_helpers.h
include local_config.mk
@@ -66,7 +66,6 @@ TEST_GEN_FILES += hugetlb-mremap
TEST_GEN_FILES += hugetlb-read-hwpoison
TEST_GEN_FILES += hugetlb-shm
TEST_GEN_FILES += hugetlb-soft-offline
-TEST_GEN_FILES += hugetlb-vmemmap
TEST_GEN_FILES += khugepaged
TEST_GEN_FILES += madv_populate
TEST_GEN_FILES += map_fixed_noreplace
@@ -150,6 +149,7 @@ TEST_PROGS += ksft_gup_test.sh
TEST_PROGS += ksft_hmm.sh
TEST_PROGS += ksft_hugetlb.sh
TEST_PROGS += ksft_hugevm.sh
+TEST_PROGS += ksft_kmemleak_confirm.sh
TEST_PROGS += ksft_kmemleak_dedup.sh
TEST_PROGS += ksft_ksm.sh
TEST_PROGS += ksft_ksm_numa.sh
@@ -262,11 +262,11 @@ $(OUTPUT)/migration: LDLIBS += -lnuma
$(OUTPUT)/rmap: LDLIBS += -lnuma
-local_config.mk local_config.h: check_config.sh
+local_config.mk local_config.h_gen: check_config.sh
$(call msg,CHK,config,$@)
$(Q)CC="$(CC)" CFLAGS="$(CFLAGS)" ./check_config.sh
-EXTRA_CLEAN += local_config.mk local_config.h
+EXTRA_CLEAN += local_config.mk local_config.h_gen
ifeq ($(IOURING_EXTRA_LIBS),)
all: warn_missing_liburing
diff --git a/tools/testing/selftests/mm/check_config.sh b/tools/testing/selftests/mm/check_config.sh
index 32beaefe279e..1c603261e93d 100755
--- a/tools/testing/selftests/mm/check_config.sh
+++ b/tools/testing/selftests/mm/check_config.sh
@@ -4,7 +4,7 @@
# Probe for libraries and create header files to record the results. Both C
# header files and Makefile include fragments are created.
-OUTPUT_H_FILE=local_config.h
+OUTPUT_H_FILE=local_config.h_gen
OUTPUT_MKFILE=local_config.mk
tmpname=$(mktemp)
diff --git a/tools/testing/selftests/mm/compaction_test.c b/tools/testing/selftests/mm/compaction_test.c
index 5b582588e015..30d4ace7155a 100644
--- a/tools/testing/selftests/mm/compaction_test.c
+++ b/tools/testing/selftests/mm/compaction_test.c
@@ -29,30 +29,34 @@ struct map_list {
int read_memory_info(unsigned long *memfree, unsigned long *hugepagesize)
{
- char buffer[256] = {0};
- char *cmd = "cat /proc/meminfo | grep -i memfree | grep -o '[0-9]*'";
- FILE *cmdfile = popen(cmd, "r");
+ char buffer[256];
+ int found = 0;
+ FILE *file;
+ int ret = -1;
- if (!(fgets(buffer, sizeof(buffer), cmdfile))) {
- ksft_print_msg("Failed to read meminfo: %s\n", strerror(errno));
+ file = fopen("/proc/meminfo", "r");
+ if (!file) {
+ ksft_print_msg("Failed to open /proc/meminfo: %s\n",
+ strerror(errno));
return -1;
}
- pclose(cmdfile);
-
- *memfree = atoll(buffer);
- cmd = "cat /proc/meminfo | grep -i hugepagesize | grep -o '[0-9]*'";
- cmdfile = popen(cmd, "r");
-
- if (!(fgets(buffer, sizeof(buffer), cmdfile))) {
- ksft_print_msg("Failed to read meminfo: %s\n", strerror(errno));
- return -1;
+ while (fgets(buffer, sizeof(buffer), file) && found != 2) {
+ if (sscanf(buffer, "MemFree: %lu kB", memfree) == 1 ||
+ sscanf(buffer, "Hugepagesize: %lu kB", hugepagesize) == 1)
+ found++;
}
- pclose(cmdfile);
- *hugepagesize = atoll(buffer);
+ if (ferror(file))
+ ksft_print_msg("Failed to read /proc/meminfo: %s\n",
+ strerror(errno));
+ else if (found != 2)
+ ksft_print_msg("Failed to parse /proc/meminfo\n");
+ else
+ ret = 0;
- return 0;
+ fclose(file);
+ return ret;
}
int prereq(void)
diff --git a/tools/testing/selftests/mm/cow.c b/tools/testing/selftests/mm/cow.c
index 0c627ea89ff7..8aa5249d9bef 100644
--- a/tools/testing/selftests/mm/cow.c
+++ b/tools/testing/selftests/mm/cow.c
@@ -21,7 +21,7 @@
#include <sys/wait.h>
#include <linux/memfd.h>
-#include "local_config.h"
+#include "local_config.h_gen"
#ifdef LOCAL_CONFIG_HAVE_LIBURING
#include <liburing.h>
#endif /* LOCAL_CONFIG_HAVE_LIBURING */
@@ -1718,8 +1718,13 @@ static void run_with_tmpfile(non_anon_test_fn fn, const char *desc)
/* File consists of a single page filled with zeroes. */
if (fallocate(fd, 0, 0, pagesize)) {
- ksft_perror("fallocate() failed");
- log_test_result(KSFT_FAIL);
+ if (errno == EOPNOTSUPP) {
+ ksft_print_msg("fallocate() not supported by filesystem\n");
+ log_test_result(KSFT_SKIP);
+ } else {
+ ksft_perror("fallocate() failed");
+ log_test_result(KSFT_FAIL);
+ }
goto close;
}
diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c
index 6329e37fff4c..45b84f7b364e 100644
--- a/tools/testing/selftests/mm/folio_split_race_test.c
+++ b/tools/testing/selftests/mm/folio_split_race_test.c
@@ -182,7 +182,7 @@ static uint64_t run_iteration(void)
for (i = 0; i < TOTAL_PAGES; i++)
fill_page(mmap_base, i);
- if (!check_huge_shmem(mmap_base, NR_PMD_PAGE, pmd_pagesize))
+ if (!check_huge_shmem(mmap_base, FILE_SIZE, NR_PMD_PAGE, pmd_pagesize))
ksft_exit_fail_msg("No shmem THP is allocated\n");
if (pthread_barrier_init(&ctl.barrier, NULL, NUM_READER_THREADS + 1) != 0)
diff --git a/tools/testing/selftests/mm/guard-regions.c b/tools/testing/selftests/mm/guard-regions.c
index b21df3040b1c..5c8ec3ca75d7 100644
--- a/tools/testing/selftests/mm/guard-regions.c
+++ b/tools/testing/selftests/mm/guard-regions.c
@@ -1912,7 +1912,7 @@ TEST_F(guard_regions, hole_punch)
{
const unsigned long page_size = self->page_size;
char *ptr;
- int i;
+ int i, ret;
if (variant->backing == ANON_BACKED)
SKIP(return, "Truncation test specific to file-backed");
@@ -1944,8 +1944,12 @@ TEST_F(guard_regions, hole_punch)
}
/* Now hole punch the guarded region. */
- ASSERT_EQ(madvise(&ptr[3 * page_size], 4 * page_size,
- MADV_REMOVE), 0);
+ ret = madvise(&ptr[3 * page_size], 4 * page_size, MADV_REMOVE);
+ if (ret == -1 && errno == EOPNOTSUPP) {
+ ASSERT_EQ(munmap(ptr, 10 * page_size), 0);
+ SKIP(return, "MADV_REMOVE not supported by filesystem");
+ }
+ ASSERT_EQ(ret, 0);
/* Ensure guard regions remain. */
for (i = 0; i < 10; i++) {
diff --git a/tools/testing/selftests/mm/gup_longterm.c b/tools/testing/selftests/mm/gup_longterm.c
index eb8963e9d98f..510de93be681 100644
--- a/tools/testing/selftests/mm/gup_longterm.c
+++ b/tools/testing/selftests/mm/gup_longterm.c
@@ -21,7 +21,7 @@
#include <linux/magic.h>
#include <linux/memfd.h>
-#include "local_config.h"
+#include "local_config.h_gen"
#ifdef LOCAL_CONFIG_HAVE_LIBURING
#include <liburing.h>
#endif /* LOCAL_CONFIG_HAVE_LIBURING */
@@ -196,7 +196,7 @@ static void do_test(int fd, size_t size, enum test_type type, bool shared)
args.flags |= rw ? PIN_LONGTERM_TEST_FLAG_USE_WRITE : 0;
ret = ioctl(gup_fd, PIN_LONGTERM_TEST_START, &args);
if (ret && errno == EINVAL) {
- ksft_print_msg("PIN_LONGTERM_TEST_START failed (EINVAL)n");
+ ksft_print_msg("PIN_LONGTERM_TEST_START failed (EINVAL)\n");
result = KSFT_SKIP;
break;
} else if (ret && errno == EFAULT) {
diff --git a/tools/testing/selftests/mm/hmm-tests.c b/tools/testing/selftests/mm/hmm-tests.c
index 2f2b9879d100..e2642eca0d02 100644
--- a/tools/testing/selftests/mm/hmm-tests.c
+++ b/tools/testing/selftests/mm/hmm-tests.c
@@ -29,6 +29,10 @@
#include <sys/mman.h>
#include <sys/ioctl.h>
#include <sys/time.h>
+#include <sys/syscall.h>
+#include <sys/eventfd.h>
+#include <linux/userfaultfd.h>
+#include <poll.h>
/*
* This is a private UAPI to the kernel test module so it isn't exported
@@ -225,6 +229,53 @@ static void hmm_buffer_free(struct hmm_buffer *buffer)
}
/*
+ * Allocate a buffer structure with memory mapping and mirror.
+ *
+ * @mmap_size: total size of the mmap region (may differ from @mirror_size
+ * for alignment padding in THP tests).
+ * @mirror_size: size of the mirror data buffer (the actual working set).
+ * @prot: protection flags for the mmap (e.g. PROT_READ | PROT_WRITE).
+ * @flags: flags for the mmap (e.g. MAP_PRIVATE, MAP_SHARED,
+ * MAP_ANONYMOUS, MAP_HUGETLB).
+ * @fd: file descriptor for the mmap; pass -1 for MAP_ANONYMOUS.
+ *
+ * All internal allocations are checked; returns NULL and cleans up on any
+ * failure. Caller must ASSERT_NE or otherwise check the return value.
+ */
+static struct hmm_buffer *hmm_buffer_alloc(unsigned long mmap_size,
+ unsigned long mirror_size,
+ int prot, int flags,
+ int fd)
+{
+ struct hmm_buffer *buffer;
+
+ buffer = malloc(sizeof(*buffer));
+ if (!buffer) {
+ perror("malloc buffer");
+ return NULL;
+ }
+
+ buffer->fd = fd;
+ buffer->size = mmap_size;
+ buffer->mirror = malloc(mirror_size);
+ if (!buffer->mirror) {
+ perror("malloc mirror");
+ free(buffer);
+ return NULL;
+ }
+
+ buffer->ptr = mmap(NULL, mmap_size, prot, flags, fd, 0);
+ if (buffer->ptr == MAP_FAILED) {
+ perror("mmap");
+ free(buffer->mirror);
+ free(buffer);
+ return NULL;
+ }
+
+ return buffer;
+}
+
+/*
* Create a temporary file that will be deleted on close.
*/
static int hmm_create_file(unsigned long size)
@@ -314,20 +365,11 @@ TEST_F(hmm, anon_read)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/*
* Initialize buffer in system memory but leave the first two pages
* zero (pte_none and pfn_zero).
@@ -377,20 +419,11 @@ TEST_F(hmm, anon_read_prot)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -436,20 +469,11 @@ TEST_F(hmm, anon_write)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize data that the device will write to buffer->ptr. */
for (i = 0, ptr = buffer->mirror; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -484,20 +508,11 @@ TEST_F(hmm, anon_write_prot)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Simulate a device reading a zero page of memory. */
ret = hmm_dmirror_cmd(self->fd, HMM_DMIRROR_READ, buffer, 1);
ASSERT_EQ(ret, 0);
@@ -559,20 +574,10 @@ TEST_F(hmm, anon_write_child)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size * 2, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = size * 2;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size * 2,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
old_ptr = buffer->ptr;
if (use_thp) {
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
@@ -661,20 +666,11 @@ TEST_F(hmm, anon_write_child_shared)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_SHARED | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_SHARED | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer->ptr so we can tell if it is written. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -736,20 +732,10 @@ TEST_F(hmm, anon_write_huge)
size = 2 * read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
size /= 2;
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
@@ -795,22 +781,11 @@ TEST_F(hmm, anon_write_hugetlbfs)
size = ALIGN(TWOMEG, default_hsize);
npages = size >> self->page_shift;
- buffer = malloc(sizeof(*buffer));
- ASSERT_NE(buffer, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB,
- -1, 0);
- if (buffer->ptr == MAP_FAILED) {
- free(buffer);
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB, -1);
+ if (!buffer)
SKIP(return, "Huge page could not be allocated");
- }
-
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
/* Initialize data that the device will write to buffer->ptr. */
for (i = 0, ptr = buffer->mirror; i < size / sizeof(*ptr); ++i)
@@ -852,14 +827,11 @@ TEST_F(hmm, file_read)
fd = hmm_create_file(size);
ASSERT_GE(fd, 0);
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ,
+ MAP_SHARED, fd);
ASSERT_NE(buffer, NULL);
- buffer->fd = fd;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
/* Write initial contents of the file. */
for (i = 0, ptr = buffer->mirror; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -867,12 +839,6 @@ TEST_F(hmm, file_read)
ASSERT_EQ(len, size);
memset(buffer->mirror, 0, size);
- buffer->ptr = mmap(NULL, size,
- PROT_READ,
- MAP_SHARED,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Simulate a device reading system memory. */
ret = hmm_dmirror_cmd(self->fd, HMM_DMIRROR_READ, buffer, npages);
ASSERT_EQ(ret, 0);
@@ -907,20 +873,11 @@ TEST_F(hmm, file_write)
fd = hmm_create_file(size);
ASSERT_GE(fd, 0);
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_SHARED, fd);
ASSERT_NE(buffer, NULL);
- buffer->fd = fd;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_SHARED,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize data that the device will write to buffer->ptr. */
for (i = 0, ptr = buffer->mirror; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -960,20 +917,11 @@ TEST_F(hmm, migrate)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1010,20 +958,11 @@ TEST_F(hmm, migrate_file_private)
fd = hmm_create_file(size);
ASSERT_GE(fd, 0);
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE, fd);
ASSERT_NE(buffer, NULL);
- buffer->fd = fd;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1058,20 +997,11 @@ TEST_F(hmm, migrate_fault)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1114,18 +1044,11 @@ TEST_F(hmm, migrate_release)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size, PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS, buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1164,20 +1087,11 @@ TEST_F(hmm, migrate_shared)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_SHARED | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_SHARED | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Migrate memory to device. */
ret = hmm_migrate_sys_to_dev(self->fd, buffer, npages);
ASSERT_EQ(ret, -ENOENT);
@@ -1201,20 +1115,10 @@ TEST_F(hmm2, migrate_mixed)
npages = 6;
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_NONE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- /* Reserve a range of addresses. */
- buffer->ptr = mmap(NULL, size,
- PROT_NONE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
p = buffer->ptr;
/* Migrating a protected area should be an error. */
@@ -1295,20 +1199,11 @@ TEST_F(hmm, migrate_multiple)
size = npages << self->page_shift;
for (c = 0; c < NTIMES; c++) {
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1354,20 +1249,11 @@ TEST_F(hmm, anon_read_multiple)
size = npages << self->page_shift;
for (c = 0; c < NTIMES; c++) {
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i + c;
@@ -1420,20 +1306,11 @@ TEST_F(hmm, anon_teardown)
int *ptr;
int rc;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i + c;
@@ -1474,22 +1351,11 @@ TEST_F(hmm, mixedmap)
npages = 1;
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, npages,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE, self->fd);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(npages);
- ASSERT_NE(buffer->mirror, NULL);
-
-
- /* Reserve a range of addresses. */
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE,
- self->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Simulate a device snapshotting CPU pagetables. */
ret = hmm_dmirror_cmd(self->fd, HMM_DMIRROR_SNAPSHOT, buffer, npages);
ASSERT_EQ(ret, 0);
@@ -1519,20 +1385,10 @@ TEST_F(hmm2, snapshot)
npages = 7;
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, npages,
+ PROT_NONE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(npages);
- ASSERT_NE(buffer->mirror, NULL);
-
- /* Reserve a range of addresses. */
- buffer->ptr = mmap(NULL, size,
- PROT_NONE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
p = buffer->ptr;
/* Punch a hole after the first page address. */
@@ -1626,21 +1482,11 @@ TEST_F(hmm, compound)
size = ALIGN(TWOMEG, default_hsize);
npages = size >> self->page_shift;
- buffer = malloc(sizeof(*buffer));
- ASSERT_NE(buffer, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB,
- -1, 0);
- if (buffer->ptr == MAP_FAILED) {
- free(buffer);
- return;
- }
-
- buffer->size = size;
- buffer->mirror = malloc(npages);
- ASSERT_NE(buffer->mirror, NULL);
+ buffer = hmm_buffer_alloc(size, npages,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB, -1);
+ if (!buffer)
+ SKIP(return, "Huge page could not be allocated");
/* Initialize the pages the device will snapshot in buffer->ptr. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
@@ -1707,21 +1553,11 @@ TEST_F(hmm2, double_map)
npages = 6;
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- /* Reserve a range of addresses. */
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1783,20 +1619,11 @@ TEST_F(hmm, exclusive)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1837,20 +1664,11 @@ TEST_F(hmm, exclusive_mprotect)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1892,20 +1710,11 @@ TEST_F(hmm, exclusive_cow)
ASSERT_NE(npages, 0);
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -1993,20 +1802,11 @@ TEST_F(hmm, hmm_gup_test)
npages = 4;
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -2079,20 +1879,11 @@ TEST_F(hmm, hmm_cow_in_device)
npages = 4;
size = npages << self->page_shift;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
-
- buffer->ptr = mmap(NULL, size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
/* Initialize buffer in system memory. */
for (i = 0, ptr = buffer->ptr; i < size / sizeof(*ptr); ++i)
ptr[i] = i;
@@ -2152,21 +1943,12 @@ TEST_F(hmm, migrate_anon_huge_empty)
size = read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, size,
+ PROT_READ,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, size);
- buffer->ptr = mmap(NULL, 2 * size,
- PROT_READ,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
ret = madvise(map, size, MADV_HUGEPAGE);
@@ -2204,21 +1986,12 @@ TEST_F(hmm, migrate_anon_huge_zero)
size = read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, size,
+ PROT_READ,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, size);
- buffer->ptr = mmap(NULL, 2 * size,
- PROT_READ,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
ret = madvise(map, size, MADV_HUGEPAGE);
@@ -2267,21 +2040,12 @@ TEST_F(hmm, migrate_anon_huge_free)
size = read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, size);
- buffer->ptr = mmap(NULL, 2 * size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
ret = madvise(map, size, MADV_HUGEPAGE);
@@ -2329,21 +2093,12 @@ TEST_F(hmm, migrate_anon_huge_fault)
size = read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, size);
- buffer->ptr = mmap(NULL, 2 * size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
old_ptr = buffer->ptr;
@@ -2425,21 +2180,11 @@ TEST_F(hmm, migrate_partial_unmap_fault)
for (use_thp = 0; use_thp < 2; ++use_thp) {
for (j = 0; j < ARRAY_SIZE(offsets); ++j) {
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, size);
-
- buffer->ptr = mmap(NULL, 2 * size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
if (use_thp)
@@ -2505,21 +2250,11 @@ TEST_F(hmm, migrate_remap_fault)
if (dont_unmap)
flags |= MREMAP_DONTUNMAP;
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(8 * size, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 8 * size;
- buffer->mirror = malloc(size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, size);
-
- buffer->ptr = mmap(NULL, buffer->size,
- PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS,
- buffer->fd, 0);
- ASSERT_NE(buffer->ptr, MAP_FAILED);
-
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)buffer->ptr, size);
if (use_thp)
@@ -2590,18 +2325,13 @@ TEST_F(hmm, migrate_anon_huge_err)
size = read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, 2 * size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(2 * size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, 2 * size);
- old_ptr = mmap(NULL, 2 * size, PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS, buffer->fd, 0);
- ASSERT_NE(old_ptr, MAP_FAILED);
+ old_ptr = buffer->ptr;
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)old_ptr, size);
@@ -2685,18 +2415,13 @@ TEST_F(hmm, migrate_anon_huge_zero_err)
size = read_pmd_pagesize();
- buffer = malloc(sizeof(*buffer));
+ buffer = hmm_buffer_alloc(2 * size, 2 * size,
+ PROT_READ,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
ASSERT_NE(buffer, NULL);
-
- buffer->fd = -1;
- buffer->size = 2 * size;
- buffer->mirror = malloc(2 * size);
- ASSERT_NE(buffer->mirror, NULL);
memset(buffer->mirror, 0xFF, 2 * size);
- old_ptr = mmap(NULL, 2 * size, PROT_READ,
- MAP_PRIVATE | MAP_ANONYMOUS, buffer->fd, 0);
- ASSERT_NE(old_ptr, MAP_FAILED);
+ old_ptr = buffer->ptr;
npages = size >> self->page_shift;
map = (void *)ALIGN((uintptr_t)old_ptr, size);
@@ -2766,19 +2491,6 @@ static double get_time_ms(void)
return (tv.tv_sec * 1000.0) + (tv.tv_usec / 1000.0);
}
-static inline struct hmm_buffer *hmm_buffer_alloc(unsigned long size)
-{
- struct hmm_buffer *buffer;
-
- buffer = malloc(sizeof(*buffer));
-
- buffer->fd = -1;
- buffer->size = size;
- buffer->mirror = malloc(size);
- memset(buffer->mirror, 0xFF, size);
- return buffer;
-}
-
static void print_benchmark_results(const char *test_name, size_t buffer_size,
struct benchmark_results *thp,
struct benchmark_results *regular)
@@ -2823,14 +2535,14 @@ static inline int run_migration_benchmark(int fd, int use_thp, size_t buffer_siz
int ret, i;
int *ptr;
- buffer = hmm_buffer_alloc(buffer_size);
-
- /* Map memory */
- buffer->ptr = mmap(NULL, buffer_size, PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
-
- if (buffer->ptr == MAP_FAILED)
- return -1;
+ buffer = hmm_buffer_alloc(buffer_size, buffer_size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1);
+ if (!buffer) {
+ ret = -1;
+ goto cleanup;
+ }
+ memset(buffer->mirror, 0xFF, buffer_size);
/* Apply THP hint if requested */
if (use_thp)
@@ -2839,7 +2551,7 @@ static inline int run_migration_benchmark(int fd, int use_thp, size_t buffer_siz
ret = madvise(buffer->ptr, buffer_size, MADV_NOHUGEPAGE);
if (ret)
- return ret;
+ goto cleanup;
/* Initialize memory to make sure pages are allocated */
ptr = (int *)buffer->ptr;
@@ -2849,11 +2561,11 @@ static inline int run_migration_benchmark(int fd, int use_thp, size_t buffer_siz
/* Warmup iteration */
ret = hmm_migrate_sys_to_dev(fd, buffer, npages);
if (ret)
- return ret;
+ goto cleanup;
ret = hmm_migrate_dev_to_sys(fd, buffer, npages);
if (ret)
- return ret;
+ goto cleanup;
/* Benchmark iterations */
for (i = 0; i < iterations; i++) {
@@ -2862,7 +2574,7 @@ static inline int run_migration_benchmark(int fd, int use_thp, size_t buffer_siz
ret = hmm_migrate_sys_to_dev(fd, buffer, npages);
if (ret)
- return ret;
+ goto cleanup;
end = get_time_ms();
s2d_total += (end - start);
@@ -2872,7 +2584,7 @@ static inline int run_migration_benchmark(int fd, int use_thp, size_t buffer_siz
ret = hmm_migrate_dev_to_sys(fd, buffer, npages);
if (ret)
- return ret;
+ goto cleanup;
end = get_time_ms();
d2s_total += (end - start);
@@ -2886,9 +2598,9 @@ static inline int run_migration_benchmark(int fd, int use_thp, size_t buffer_siz
results->throughput_d2s = (buffer_size / (1024.0 * 1024.0 * 1024.0)) /
(results->dev_to_sys_time / 1000.0);
- /* Cleanup */
+cleanup:
hmm_buffer_free(buffer);
- return 0;
+ return ret;
}
/*
@@ -2949,4 +2661,150 @@ TEST_F_TIMEOUT(hmm, benchmark_thp_migration, 120)
&thp_results, &regular_results);
}
}
+/*
+ * Test that HMM can fault in pages backed by userfaultfd using the
+ * hmm_range_fault_unlocked_timeout() path with no timeout. This exercises
+ * the lock-drop retry logic in the HMM framework.
+ */
+struct uffd_thread_args {
+ int uffd;
+ int stop_fd;
+ void *page_buffer;
+ unsigned long page_size;
+};
+
+static void *uffd_handler_thread(void *arg)
+{
+ struct uffd_thread_args *args = arg;
+ struct uffd_msg msg;
+ struct uffdio_copy copy;
+ struct pollfd pollfd[2];
+ int ret;
+
+ pollfd[0].fd = args->uffd;
+ pollfd[0].events = POLLIN;
+ pollfd[1].fd = args->stop_fd;
+ pollfd[1].events = POLLIN;
+
+ while (1) {
+ ret = poll(pollfd, 2, -1);
+ if (ret <= 0)
+ break;
+ if (pollfd[1].revents)
+ break;
+ if (!(pollfd[0].revents & POLLIN))
+ break;
+
+ ret = read(args->uffd, &msg, sizeof(msg));
+ if (ret != sizeof(msg))
+ break;
+
+ if (msg.event != UFFD_EVENT_PAGEFAULT)
+ break;
+
+ /* Fill the page with a known pattern */
+ memset(args->page_buffer, 0xAB, args->page_size);
+
+ copy.dst = msg.arg.pagefault.address & ~(args->page_size - 1);
+ copy.src = (unsigned long)args->page_buffer;
+ copy.len = args->page_size;
+ copy.mode = 0;
+ copy.copy = 0;
+
+ ret = ioctl(args->uffd, UFFDIO_COPY, &copy);
+ if (ret < 0)
+ break;
+ }
+
+ return NULL;
+}
+
+TEST_F(hmm, userfaultfd_read)
+{
+ struct hmm_buffer *buffer;
+ struct uffd_thread_args uffd_args;
+ unsigned long npages;
+ unsigned long size;
+ unsigned long i;
+ unsigned char *ptr;
+ pthread_t thread;
+ int uffd;
+ int stop_fd;
+ int ret;
+ struct uffdio_api api;
+ struct uffdio_register reg;
+ uint64_t stop = 1;
+ ssize_t nwrite;
+
+ npages = 4;
+ size = npages << self->page_shift;
+
+ /* Create userfaultfd */
+ uffd = syscall(__NR_userfaultfd, O_CLOEXEC | O_NONBLOCK);
+ if (uffd < 0)
+ SKIP(return, "userfaultfd not available");
+
+ api.api = UFFD_API;
+ api.features = 0;
+ ret = ioctl(uffd, UFFDIO_API, &api);
+ ASSERT_EQ(ret, 0);
+
+ buffer = malloc(sizeof(*buffer));
+ ASSERT_NE(buffer, NULL);
+
+ buffer->fd = -1;
+ buffer->size = size;
+ buffer->mirror = malloc(size);
+ ASSERT_NE(buffer->mirror, NULL);
+
+ /* Create anonymous mapping */
+ buffer->ptr = mmap(NULL, size,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS,
+ -1, 0);
+ ASSERT_NE(buffer->ptr, MAP_FAILED);
+
+ /* Register the region with userfaultfd */
+ reg.range.start = (unsigned long)buffer->ptr;
+ reg.range.len = size;
+ reg.mode = UFFDIO_REGISTER_MODE_MISSING;
+ ret = ioctl(uffd, UFFDIO_REGISTER, &reg);
+ ASSERT_EQ(ret, 0);
+
+ /* Set up the handler thread */
+ uffd_args.uffd = uffd;
+ stop_fd = eventfd(0, EFD_CLOEXEC);
+ ASSERT_GE(stop_fd, 0);
+ uffd_args.stop_fd = stop_fd;
+ uffd_args.page_buffer = malloc(self->page_size);
+ ASSERT_NE(uffd_args.page_buffer, NULL);
+ uffd_args.page_size = self->page_size;
+
+ ret = pthread_create(&thread, NULL, uffd_handler_thread, &uffd_args);
+ ASSERT_EQ(ret, 0);
+
+ /*
+ * Use the unlocked read path which allows the mmap lock to be
+ * dropped during the fault, enabling userfaultfd resolution.
+ */
+ ret = hmm_dmirror_cmd(self->fd, HMM_DMIRROR_READ_UNLOCKED,
+ buffer, npages);
+ ASSERT_EQ(ret, 0);
+ ASSERT_EQ(buffer->cpages, npages);
+
+ /* Verify the device read the data filled by the uffd handler */
+ ptr = buffer->mirror;
+ for (i = 0; i < size; ++i)
+ ASSERT_EQ(ptr[i], (unsigned char)0xAB);
+
+ nwrite = write(stop_fd, &stop, sizeof(stop));
+ ASSERT_EQ(nwrite, sizeof(stop));
+ pthread_join(thread, NULL);
+ close(stop_fd);
+ free(uffd_args.page_buffer);
+ close(uffd);
+ hmm_buffer_free(buffer);
+}
+
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c
index 2eab2110ac6a..d7917dce3aba 100644
--- a/tools/testing/selftests/mm/hugepage_settings.c
+++ b/tools/testing/selftests/mm/hugepage_settings.c
@@ -437,7 +437,7 @@ void hugetlb_set_nr_pages(unsigned long size, unsigned long nr)
hugetlb_sysfs_path(path, sizeof(path), size, "nr_hugepages");
- write_num(path, nr);
+ write_num_ignore_einval(path, nr);
}
unsigned long hugetlb_free_pages(unsigned long size)
diff --git a/tools/testing/selftests/mm/hugetlb-vmemmap.c b/tools/testing/selftests/mm/hugetlb-vmemmap.c
deleted file mode 100644
index 507df78a158d..000000000000
--- a/tools/testing/selftests/mm/hugetlb-vmemmap.c
+++ /dev/null
@@ -1,132 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * A test case of using hugepage memory in a user application using the
- * mmap system call with MAP_HUGETLB flag. Before running this program
- * make sure the administrator has allocated enough default sized huge
- * pages to cover the 2 MB allocation.
- */
-#include <stdlib.h>
-#include <stdio.h>
-#include <unistd.h>
-#include <sys/mman.h>
-#include <fcntl.h>
-#include "vm_util.h"
-#include "hugepage_settings.h"
-
-#define PAGE_COMPOUND_HEAD (1UL << 15)
-#define PAGE_COMPOUND_TAIL (1UL << 16)
-#define PAGE_HUGE (1UL << 17)
-
-#define HEAD_PAGE_FLAGS (PAGE_COMPOUND_HEAD | PAGE_HUGE)
-#define TAIL_PAGE_FLAGS (PAGE_COMPOUND_TAIL | PAGE_HUGE)
-
-#define PM_PFRAME_BITS 55
-#define PM_PFRAME_MASK ~((1UL << PM_PFRAME_BITS) - 1)
-
-static size_t pagesize;
-static size_t maplength;
-
-static void write_bytes(char *addr, size_t length)
-{
- unsigned long i;
-
- for (i = 0; i < length; i++)
- *(addr + i) = (char)i;
-}
-
-static unsigned long virt_to_pfn(void *addr)
-{
- int fd;
- unsigned long pagemap;
-
- fd = open("/proc/self/pagemap", O_RDONLY);
- if (fd < 0)
- return -1UL;
-
- lseek(fd, (unsigned long)addr / pagesize * sizeof(pagemap), SEEK_SET);
- read(fd, &pagemap, sizeof(pagemap));
- close(fd);
-
- return pagemap & ~PM_PFRAME_MASK;
-}
-
-static int check_page_flags(unsigned long pfn)
-{
- int fd, i;
- unsigned long pageflags;
-
- fd = open("/proc/kpageflags", O_RDONLY);
- if (fd < 0)
- return -1;
-
- lseek(fd, pfn * sizeof(pageflags), SEEK_SET);
-
- read(fd, &pageflags, sizeof(pageflags));
- if ((pageflags & HEAD_PAGE_FLAGS) != HEAD_PAGE_FLAGS) {
- close(fd);
- ksft_print_msg("Head page flags (%lx) is invalid\n", pageflags);
- return -1;
- }
-
- /*
- * pages other than the first page must be tail and shouldn't be head;
- * this also verifies kernel has correctly set the fake page_head to tail
- * while hugetlb_free_vmemmap is enabled.
- */
- for (i = 1; i < maplength / pagesize; i++) {
- read(fd, &pageflags, sizeof(pageflags));
- if ((pageflags & TAIL_PAGE_FLAGS) != TAIL_PAGE_FLAGS ||
- (pageflags & HEAD_PAGE_FLAGS) == HEAD_PAGE_FLAGS) {
- close(fd);
- ksft_print_msg("Tail page flags (%lx) is invalid\n", pageflags);
- return -1;
- }
- }
-
- close(fd);
-
- return 0;
-}
-
-int main(int argc, char **argv)
-{
- void *addr;
- unsigned long pfn;
- int ret;
-
- ksft_print_header();
- ksft_set_plan(1);
-
- if (!hugetlb_setup_default(1))
- ksft_exit_skip("Not enough free huge pages\n");
-
- pagesize = psize();
- maplength = default_huge_page_size();
- if (!maplength)
- ksft_exit_skip("Unable to determine huge page size\n");
-
- addr = mmap(NULL, maplength, PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB, -1, 0);
- if (addr == MAP_FAILED)
- ksft_exit_fail_perror("mmap");
-
- /* Trigger allocation of HugeTLB page. */
- write_bytes(addr, maplength);
-
- pfn = virt_to_pfn(addr);
- if (pfn == -1UL) {
- ksft_perror("virt_to_pfn");
- munmap(addr, maplength);
- ksft_exit_fail();
- }
-
- ksft_print_msg("Returned address is %p whose pfn is %lx\n", addr, pfn);
-
- ret = check_page_flags(pfn);
-
- if (munmap(addr, maplength))
- ksft_exit_fail_perror("munmap");
-
- ksft_test_result(!ret, "HugeTLB vmemmap page flags\n");
- ksft_finished();
-}
diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c
index 10e8dedcb087..1d2d6bd72fd2 100644
--- a/tools/testing/selftests/mm/khugepaged.c
+++ b/tools/testing/selftests/mm/khugepaged.c
@@ -26,9 +26,11 @@
#define BASE_ADDR ((void *)(1UL << 30))
static unsigned long hpage_pmd_size;
+static int hpage_pmd_order;
static unsigned long page_size;
static int hpage_pmd_nr;
static int anon_order;
+static int collapse_order;
#define PID_SMAPS "/proc/self/smaps"
#define TEST_FILE "collapse_test_file"
@@ -51,7 +53,7 @@ struct mem_ops {
void *(*setup_area)(int nr_hpages);
void (*cleanup_area)(void *p, unsigned long size);
void (*fault)(void *p, unsigned long start, unsigned long end);
- bool (*check_huge)(void *addr, int nr_hpages);
+ bool (*check_huge)(void *addr, size_t len, int nr_hpages, unsigned long hpage_size);
const char *name;
};
@@ -69,6 +71,7 @@ struct collapse_context {
};
static struct collapse_context *khugepaged_context;
+static struct collapse_context *mthp_khugepaged_context;
static struct collapse_context *madvise_context;
struct file_info {
@@ -121,7 +124,8 @@ static void get_finfo(const char *dir)
char *str, *end;
finfo.dir = dir;
- stat(finfo.dir, &path_stat);
+ if (stat(finfo.dir, &path_stat))
+ ksft_exit_fail_perror("stat()");
if (!S_ISDIR(path_stat.st_mode))
ksft_exit_fail_msg("%s: Not a directory (%s)\n", __func__, finfo.dir);
if (snprintf(finfo.path, sizeof(finfo.path), "%s/" TEST_FILE,
@@ -138,8 +142,8 @@ static void get_finfo(const char *dir)
major(path_stat.st_dev), minor(path_stat.st_dev))
>= sizeof(path))
ksft_exit_fail_msg("%s: Pathname is too long\n", __func__);
- if (read_file(path, buf, sizeof(buf)) < 0)
- ksft_exit_fail_perror("read_file(read_num)");
+ if (!read_file(path, buf, sizeof(buf)))
+ ksft_exit_fail_perror("read_file(uevent)");
if (strstr(buf, "DEVTYPE=disk")) {
/* Found it */
if (snprintf(finfo.dev_queue_read_ahead_path,
@@ -276,7 +280,7 @@ static void *alloc_hpage(struct mem_ops *ops)
ksft_print_msg("Allocate huge page...");
if (madvise_collapse_retry(p, hpage_pmd_size))
ksft_exit_fail_perror("madvise(MADV_COLLAPSE)");
- if (!ops->check_huge(p, 1))
+ if (!ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
ksft_exit_fail_perror("madvise(MADV_COLLAPSE)");
if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE))
ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)");
@@ -310,9 +314,10 @@ static void anon_fault(void *p, unsigned long start, unsigned long end)
fill_memory(p, start, end);
}
-static bool anon_check_huge(void *addr, int nr_hpages)
+static bool anon_check_huge(void *addr, size_t len, int nr_hpages,
+ unsigned long hpage_size)
{
- return check_huge_anon(addr, nr_hpages, hpage_pmd_size);
+ return check_huge_anon(addr, len, nr_hpages, hpage_size);
}
static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup)
@@ -412,13 +417,14 @@ static void file_fault_write(void *p, unsigned long start, unsigned long end)
ksft_exit_fail_perror("madvise(MADV_POPULATE_WRITE)");
}
-static bool file_check_huge(void *addr, int nr_hpages)
+static bool file_check_huge(void *addr, size_t len, int nr_hpages,
+ unsigned long hpage_size)
{
switch (finfo.type) {
case VMA_FILE:
- return check_huge_file(addr, nr_hpages, hpage_pmd_size);
+ return check_huge_file(addr, len, nr_hpages, hpage_size);
case VMA_SHMEM:
- return check_huge_shmem(addr, nr_hpages, hpage_pmd_size);
+ return check_huge_shmem(addr, len, nr_hpages, hpage_size);
default:
exit(EXIT_FAILURE);
return false;
@@ -448,9 +454,10 @@ static void shmem_cleanup_area(void *p, unsigned long size)
close(finfo.fd);
}
-static bool shmem_check_huge(void *addr, int nr_hpages)
+static bool shmem_check_huge(void *addr, size_t len, int nr_hpages,
+ unsigned long hpage_size)
{
- return check_huge_shmem(addr, nr_hpages, hpage_pmd_size);
+ return check_huge_shmem(addr, len, nr_hpages, hpage_size);
}
static struct mem_ops __anon_ops = {
@@ -533,7 +540,7 @@ static void __madvise_collapse(const char *msg, char *p, int nr_hpages,
ret = madvise_collapse_retry(p, nr_hpages * hpage_pmd_size);
if (((bool)ret) == expect)
fail("Fail: Bad return value");
- else if (!ops->check_huge(p, expect ? nr_hpages : 0))
+ else if (!ops->check_huge(p, nr_hpages * hpage_pmd_size, expect ? nr_hpages : 0, hpage_pmd_size))
fail("Fail: check_huge()");
else
success("OK");
@@ -545,30 +552,31 @@ static void madvise_collapse(const char *msg, char *p, int nr_hpages,
struct mem_ops *ops, bool expect)
{
/* Sanity check */
- if (!ops->check_huge(p, 0))
+ if (!ops->check_huge(p, nr_hpages * hpage_pmd_size, 0, hpage_pmd_size))
ksft_exit_fail_msg("Unexpected huge page\n");
__madvise_collapse(msg, p, nr_hpages, ops, expect);
}
#define TICK 500000
-static bool wait_for_scan(const char *msg, char *p, int nr_hpages,
- struct mem_ops *ops)
+static bool wait_for_scan(const char *msg, char *p, size_t len,
+ int nr_hpages, int collap_order, struct mem_ops *ops)
{
+ unsigned long hpage_size = page_size << collap_order;
int full_scans;
int timeout = 6; /* 3 seconds */
/* Sanity check */
- if (!ops->check_huge(p, 0))
+ if (!ops->check_huge(p, len, 0, hpage_size))
ksft_exit_fail_msg("Unexpected huge page\n");
- madvise(p, nr_hpages * hpage_pmd_size, MADV_HUGEPAGE);
+ madvise(p, len, MADV_HUGEPAGE);
/* Wait until the second full_scan completed */
full_scans = thp_read_num("khugepaged/full_scans") + 2;
ksft_print_msg("%s...", msg);
while (timeout--) {
- if (ops->check_huge(p, nr_hpages))
+ if (ops->check_huge(p, len, nr_hpages, hpage_size))
break;
if (thp_read_num("khugepaged/full_scans") >= full_scans)
break;
@@ -582,6 +590,8 @@ static bool wait_for_scan(const char *msg, char *p, int nr_hpages,
static void khugepaged_collapse(const char *msg, char *p, int nr_hpages,
struct mem_ops *ops, bool expect)
{
+ size_t len = nr_hpages * hpage_pmd_size;
+
/*
* read&write file collapse fails since khugepaged does not flush
* the target dirty folios
@@ -589,7 +599,7 @@ static void khugepaged_collapse(const char *msg, char *p, int nr_hpages,
if (!is_tmpfs(ops) && ops == &__read_write_file_write_ops)
expect = false;
- if (wait_for_scan(msg, p, nr_hpages, ops)) {
+ if (wait_for_scan(msg, p, len, nr_hpages, hpage_pmd_order, ops)) {
if (expect)
fail("Timeout");
else
@@ -605,10 +615,54 @@ static void khugepaged_collapse(const char *msg, char *p, int nr_hpages,
if (ops != &__anon_ops)
ops->fault(p, 0, nr_hpages * hpage_pmd_size);
- if (ops->check_huge(p, expect ? nr_hpages : 0))
+ if (ops->check_huge(p, len, expect ? nr_hpages : 0, hpage_pmd_size))
+ success("OK");
+ else
+ fail("Fail");
+}
+
+static void mthp_khugepaged_collapse(const char *msg, char *p, int nr_hpages,
+ struct mem_ops *ops, bool expect)
+{
+ unsigned long hpage_size = page_size << collapse_order;
+ struct thp_settings settings = *thp_current_settings();
+ /* mTHP collpase only allocates PMD sized memory */
+ size_t len = hpage_pmd_size;
+
+ /* Set mTHP setting for mTHP collapse */
+ if (ops == &__anon_ops) {
+ settings.thp_enabled = THP_NEVER;
+ settings.hugepages[collapse_order].enabled = THP_MADVISE;
+ }
+
+ thp_push_settings(&settings);
+
+ if (wait_for_scan(msg, p, len, nr_hpages, collapse_order, ops)) {
+ if (expect)
+ fail("Timeout");
+ else
+ success("OK");
+
+ /* Restore THP settings for mTHP collapse. */
+ thp_pop_settings();
+ return;
+ }
+
+ /*
+ * For file and shmem memory, khugepaged only retracts pte entries after
+ * putting the new hugepage in the page cache. The hugepage must be
+ * subsequently refaulted to install the pmd mapping for the mm.
+ */
+ if (ops != &__anon_ops)
+ ops->fault(p, 0, nr_hpages * hpage_size);
+
+ if (ops->check_huge(p, len, expect ? nr_hpages : 0, hpage_size))
success("OK");
else
fail("Fail");
+
+ /* Restore THP settings for mTHP collapse. */
+ thp_pop_settings();
}
static struct collapse_context __khugepaged_context = {
@@ -617,6 +671,12 @@ static struct collapse_context __khugepaged_context = {
.name = "khugepaged",
};
+static struct collapse_context __mthp_khugepaged_context = {
+ .collapse = &mthp_khugepaged_collapse,
+ .enforce_pte_scan_limits = true,
+ .name = "mthp_khugepaged",
+};
+
static struct collapse_context __madvise_context = {
.collapse = &madvise_collapse,
.enforce_pte_scan_limits = false,
@@ -634,7 +694,7 @@ static void alloc_at_fault(void)
p = alloc_mapping(1);
*p = 1;
ksft_print_msg("Allocate huge page on fault...");
- if (check_huge_anon(p, 1, hpage_pmd_size))
+ if (check_huge_anon(p, hpage_pmd_size, 1, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -643,7 +703,7 @@ static void alloc_at_fault(void)
madvise(p, page_size, MADV_DONTNEED);
ksft_print_msg("Split huge PMD on MADV_DONTNEED...");
- if (check_huge_anon(p, 0, hpage_pmd_size))
+ if (check_huge_anon(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -655,10 +715,17 @@ static void alloc_at_fault(void)
static void collapse_full(struct collapse_context *c, struct mem_ops *ops)
{
void *p;
- int nr_hpages = 4;
+ int nr_pmds = 4, nr_hpages = 4;
unsigned long size = nr_hpages * hpage_pmd_size;
- p = ops->setup_area(nr_hpages);
+ /* Only try 1 PMD sized range for mTHP collapse. */
+ if (c == &__mthp_khugepaged_context) {
+ nr_pmds = 1;
+ nr_hpages = 1 << (hpage_pmd_order - collapse_order);
+ size = hpage_pmd_size;
+ }
+
+ p = ops->setup_area(nr_pmds);
ops->fault(p, 0, size);
c->collapse("Collapse multiple fully populated PTE table", p, nr_hpages,
ops, true);
@@ -670,10 +737,31 @@ static void collapse_full(struct collapse_context *c, struct mem_ops *ops)
static void collapse_empty(struct collapse_context *c, struct mem_ops *ops)
{
+ int nr_hpages = 1;
+ void *p;
+
+ if (c == &__mthp_khugepaged_context)
+ nr_hpages = 1 << (hpage_pmd_order - collapse_order);
+
+ p = ops->setup_area(1);
+ c->collapse("Do not collapse empty PTE table", p, nr_hpages, ops, false);
+ ops->cleanup_area(p, hpage_pmd_size);
+ ksft_test_result_report(exit_status, "%s\n", __func__);
+}
+
+static void collapse_single_mthp(struct collapse_context *c, struct mem_ops *ops)
+{
+ unsigned long hpage_size = page_size << collapse_order;
void *p;
p = ops->setup_area(1);
- c->collapse("Do not collapse empty PTE table", p, 1, ops, false);
+ /*
+ * Only fault collapse_order sized ranges, and only check 1
+ * collapse_order sized huge page.
+ */
+ ops->fault(p, 0, hpage_size);
+ c->collapse("Collapse PTE table with half PTE entries present",
+ p, 1, ops, true);
ops->cleanup_area(p, hpage_pmd_size);
ksft_test_result_report(exit_status, "%s\n", __func__);
}
@@ -815,7 +903,7 @@ static void collapse_single_pte_entry_compound(struct collapse_context *c, struc
madvise(p, hpage_pmd_size, MADV_NOHUGEPAGE);
ksft_print_msg("Split huge page leaving single PTE mapping compound page...");
madvise(p + page_size, hpage_pmd_size - page_size, MADV_DONTNEED);
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -836,7 +924,7 @@ static void collapse_full_of_compound(struct collapse_context *c, struct mem_ops
ksft_print_msg("Split huge page leaving single PTE page table full of compound pages...");
madvise(p, page_size, MADV_NOHUGEPAGE);
madvise(p, hpage_pmd_size, MADV_NOHUGEPAGE);
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -858,7 +946,7 @@ static void collapse_compound_extreme(struct collapse_context *c, struct mem_ops
for (i = 0; i < hpage_pmd_nr; i++) {
madvise(BASE_ADDR, hpage_pmd_size, MADV_HUGEPAGE);
ops->fault(BASE_ADDR, 0, hpage_pmd_size);
- if (!ops->check_huge(BASE_ADDR, 1))
+ if (!ops->check_huge(BASE_ADDR, hpage_pmd_size, 1, hpage_pmd_size))
ksft_exit_fail_msg("Failed to allocate huge page\n");
madvise(BASE_ADDR, hpage_pmd_size, MADV_NOHUGEPAGE);
@@ -881,7 +969,7 @@ static void collapse_compound_extreme(struct collapse_context *c, struct mem_ops
ops->cleanup_area(BASE_ADDR, hpage_pmd_size);
ops->fault(p, 0, hpage_pmd_size);
- if (!ops->check_huge(p, 1))
+ if (!ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -903,7 +991,7 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops)
ksft_print_msg("Allocate small page...");
ops->fault(p, 0, page_size);
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -911,7 +999,7 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops)
ksft_print_msg("Share small page over fork()...");
if (!fork()) {
/* Do not touch settings on child exit */
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -929,7 +1017,7 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops)
exit_status = WEXITSTATUS(wstatus);
ksft_print_msg("Check if parent still has small page...");
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -947,7 +1035,7 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o
ksft_print_msg("Share huge page over fork()...");
if (!fork()) {
/* Do not touch settings on child exit */
- if (ops->check_huge(p, 1))
+ if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -955,7 +1043,7 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o
ksft_print_msg("Split huge page PMD in child process...");
madvise(p, page_size, MADV_NOHUGEPAGE);
madvise(p, hpage_pmd_size, MADV_NOHUGEPAGE);
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -976,7 +1064,7 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o
exit_status = WEXITSTATUS(wstatus);
ksft_print_msg("Check if parent still has huge page...");
- if (ops->check_huge(p, 1))
+ if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -995,7 +1083,7 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops
ksft_print_msg("Share huge page over fork()...");
if (!fork()) {
/* Do not touch settings on child exit */
- if (ops->check_huge(p, 1))
+ if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -1003,7 +1091,7 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops
ksft_print_msg("Trigger CoW on page %d of %d...",
hpage_pmd_nr - max_ptes_shared - 1, hpage_pmd_nr);
ops->fault(p, 0, (hpage_pmd_nr - max_ptes_shared - 1) * page_size);
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -1016,7 +1104,7 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops
hpage_pmd_nr - max_ptes_shared, hpage_pmd_nr);
ops->fault(p, 0, (hpage_pmd_nr - max_ptes_shared) *
page_size);
- if (ops->check_huge(p, 0))
+ if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -1034,7 +1122,7 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops
exit_status = WEXITSTATUS(wstatus);
ksft_print_msg("Check if parent still has huge page...");
- if (ops->check_huge(p, 1))
+ if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
success("OK");
else
fail("Fail");
@@ -1075,8 +1163,8 @@ static void madvise_retracted_page_tables(struct collapse_context *c,
ops->fault(p, 0, size);
/* Let khugepaged collapse and leave pmd cleared */
- if (wait_for_scan("Collapse and leave PMD cleared", p, nr_hpages,
- ops)) {
+ if (wait_for_scan("Collapse and leave PMD cleared", p, size, nr_hpages,
+ hpage_pmd_order, ops)) {
fail("Timeout");
return;
}
@@ -1092,17 +1180,19 @@ static void usage(void)
{
fprintf(stderr, "\nUsage: ./khugepaged [OPTIONS] <test type> [dir]\n\n");
fprintf(stderr, "\t<test type>\t: <context>:<mem_type>\n");
- fprintf(stderr, "\t<context>\t: [all|khugepaged|madvise]\n");
+ fprintf(stderr, "\t<context>\t: [all|khugepaged|mthp_khugepaged|madvise]\n");
fprintf(stderr, "\t<mem_type>\t: [all|anon|file|shmem]\n");
fprintf(stderr, "\n\t\"file,all\" mem_type requires [dir] argument\n");
fprintf(stderr, "\n\t\"file,all\" mem_type requires a file system\n");
fprintf(stderr, "\twith PMD-sized large folio support\n");
fprintf(stderr, "\n\tif [dir] is a (sub)directory of a tmpfs mount, tmpfs must be\n");
fprintf(stderr, "\tmounted with huge=advise option for khugepaged tests to work\n");
+ fprintf(stderr, "\n\tmthp_khugepaged only supports anon mem_type now.\n");
fprintf(stderr, "\n\tSupported Options:\n");
fprintf(stderr, "\t\t-h: This help message.\n");
fprintf(stderr, "\t\t-s: mTHP size, expressed as page order.\n");
fprintf(stderr, "\t\t Defaults to 0. Use this size for anon or shmem allocations.\n");
+ fprintf(stderr, "\t\t-c: collapse order for mTHP collapse, expressed as page order.\n");
exit(1);
}
@@ -1112,11 +1202,14 @@ static void parse_test_type(int argc, char **argv)
char *buf;
const char *token;
- while ((opt = getopt(argc, argv, "s:h")) != -1) {
+ while ((opt = getopt(argc, argv, "s:c:h")) != -1) {
switch (opt) {
case 's':
anon_order = atoi(optarg);
break;
+ case 'c':
+ collapse_order = atoi(optarg);
+ break;
case 'h':
default:
usage();
@@ -1142,6 +1235,10 @@ static void parse_test_type(int argc, char **argv)
madvise_context = &__madvise_context;
} else if (!strcmp(token, "khugepaged")) {
khugepaged_context = &__khugepaged_context;
+ } else if (!strcmp(token, "mthp_khugepaged")) {
+ mthp_khugepaged_context = &__mthp_khugepaged_context;
+ if (collapse_order <= 0 || collapse_order >= hpage_pmd_order)
+ usage();
} else if (!strcmp(token, "madvise")) {
madvise_context = &__madvise_context;
} else {
@@ -1157,14 +1254,20 @@ static void parse_test_type(int argc, char **argv)
read_write_file_write_ops = &__read_write_file_write_ops;
anon_ops = &__anon_ops;
shmem_ops = &__shmem_ops;
+ if (mthp_khugepaged_context)
+ usage();
} else if (!strcmp(buf, "anon")) {
anon_ops = &__anon_ops;
} else if (!strcmp(buf, "file")) {
read_only_file_ops = &__read_only_file_ops;
read_write_file_read_ops = &__read_write_file_read_ops;
read_write_file_write_ops = &__read_write_file_write_ops;
+ if (mthp_khugepaged_context)
+ usage();
} else if (!strcmp(buf, "shmem")) {
shmem_ops = &__shmem_ops;
+ if (mthp_khugepaged_context)
+ usage();
} else {
usage();
}
@@ -1207,7 +1310,6 @@ static int nr_test_cases;
int main(int argc, char **argv)
{
- int hpage_pmd_order;
struct thp_settings default_settings = {
.thp_enabled = THP_MADVISE,
.thp_defrag = THP_DEFRAG_ALWAYS,
@@ -1233,10 +1335,6 @@ int main(int argc, char **argv)
if (!thp_is_enabled())
ksft_exit_skip("Transparent Hugepages not available\n");
- parse_test_type(argc, argv);
-
- setbuf(stdout, NULL);
-
page_size = getpagesize();
hpage_pmd_size = read_pmd_pagesize();
if (!hpage_pmd_size)
@@ -1244,6 +1342,10 @@ int main(int argc, char **argv)
hpage_pmd_nr = hpage_pmd_size / page_size;
hpage_pmd_order = __builtin_ctz(hpage_pmd_nr);
+ parse_test_type(argc, argv);
+
+ setbuf(stdout, NULL);
+
default_settings.khugepaged.max_ptes_none = hpage_pmd_nr - 1;
default_settings.khugepaged.max_ptes_swap = hpage_pmd_nr / 8;
default_settings.khugepaged.max_ptes_shared = hpage_pmd_nr / 2;
@@ -1261,6 +1363,7 @@ int main(int argc, char **argv)
TEST(collapse_full, khugepaged_context, read_write_file_read_ops);
TEST(collapse_full, khugepaged_context, read_write_file_write_ops);
TEST(collapse_full, khugepaged_context, shmem_ops);
+ TEST(collapse_full, mthp_khugepaged_context, anon_ops);
TEST(collapse_full, madvise_context, anon_ops);
TEST(collapse_full, madvise_context, read_only_file_ops);
TEST(collapse_full, madvise_context, read_write_file_read_ops);
@@ -1268,8 +1371,11 @@ int main(int argc, char **argv)
TEST(collapse_full, madvise_context, shmem_ops);
TEST(collapse_empty, khugepaged_context, anon_ops);
+ TEST(collapse_empty, mthp_khugepaged_context, anon_ops);
TEST(collapse_empty, madvise_context, anon_ops);
+ TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops);
+
TEST(collapse_single_pte_entry, khugepaged_context, anon_ops);
TEST(collapse_single_pte_entry, khugepaged_context, read_only_file_ops);
TEST(collapse_single_pte_entry, khugepaged_context, read_write_file_read_ops);
diff --git a/tools/testing/selftests/mm/ksft_kmemleak_confirm.sh b/tools/testing/selftests/mm/ksft_kmemleak_confirm.sh
new file mode 100755
index 000000000000..72ded5e6794c
--- /dev/null
+++ b/tools/testing/selftests/mm/ksft_kmemleak_confirm.sh
@@ -0,0 +1,130 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Functional test for kmemleak's N-consecutive-scan leak confirmation
+# (the min_unref_scans module parameter).
+#
+# kmemleak only reports an object once it has stayed unreferenced for
+# min_unref_scans consecutive scans. A threshold of 1 reports on the first
+# scan (historical behaviour); higher values filter transient false
+# positives where a live object's only reference is briefly invisible to a
+# single scan (e.g. an RCU tree update in flight while the scan runs). The
+# test loads samples/kmemleak's helper module to create orphan allocations
+# and, counting only those orphans (matched by their [kmemleak_test]
+# backtrace so unrelated leaks already present on the system are ignored),
+# checks that:
+# - a freshly allocated object is greyed on its first scan (its checksum
+# settles then), so nothing can be reported before that priming scan;
+# each case below primes once first,
+# - at min_unref_scans=1 one scan after priming reports the orphans,
+# - raising the threshold to 2 needs two scans after priming: one is not
+# enough, the second reports,
+# - the parameter reads back what was written.
+#
+# The "one post-prime scan is not enough at min_unref_scans=2" check is the
+# core regression test: raising min_unref_scans must push the report
+# strictly later. Like ksft_kmemleak_dedup.sh, if the module yields no
+# detectable orphan at all in the running environment the test skips rather
+# than failing.
+#
+# Author: Breno Leitao <leitao@debian.org>
+
+# KTAP output helpers (ktap_skip_all, ktap_exit_fail_msg, ktap_test_pass, ...).
+DIR="$(dirname "$(readlink -f "$0")")"
+# shellcheck source=../kselftest/ktap_helpers.sh
+source "${DIR}"/../kselftest/ktap_helpers.sh
+
+KMEMLEAK=/sys/kernel/debug/kmemleak
+PARAM=/sys/module/kmemleak/parameters/min_unref_scans
+MODULE=kmemleak-test
+AGE=6 # seconds; must exceed kmemleak's 5s minimum object age
+
+ktap_print_header
+
+[ "$(id -u)" -eq 0 ] || { ktap_skip_all "must run as root"; exit "$KSFT_SKIP"; }
+[ -r "$KMEMLEAK" ] ||
+ { ktap_skip_all "no kmemleak debugfs (CONFIG_DEBUG_KMEMLEAK)"; exit "$KSFT_SKIP"; }
+[ -w "$PARAM" ] ||
+ { ktap_skip_all "min_unref_scans module parameter not present"; exit "$KSFT_SKIP"; }
+modinfo "$MODULE" >/dev/null 2>&1 ||
+ { ktap_skip_all "$MODULE not built (CONFIG_SAMPLE_KMEMLEAK)"; exit "$KSFT_SKIP"; }
+
+# kmemleak can be present but disabled at runtime (kmemleak=off boot arg,
+# or it self-disabled after an internal error); a "scan" then returns
+# EPERM. Probe once and skip if so.
+echo scan > "$KMEMLEAK" 2>/dev/null ||
+ { ktap_skip_all "kmemleak is disabled (check dmesg or kmemleak= boot arg)"; exit "$KSFT_SKIP"; }
+
+prev=$(cat "$PARAM")
+# shellcheck disable=SC2317 # invoked indirectly via trap
+cleanup() {
+ echo "$prev" > "$PARAM" 2>/dev/null # restore the parameter
+ echo scan=on > "$KMEMLEAK" 2>/dev/null # re-enable auto scan
+ rmmod "$MODULE" 2>/dev/null
+ echo clear > "$KMEMLEAK" 2>/dev/null
+}
+trap cleanup EXIT
+
+# Stop the automatic scan thread: only our manual scans should advance an
+# object's consecutive-unreferenced run. An auto scan landing between two
+# manual scans would change the result and make the test flaky.
+echo scan=off > "$KMEMLEAK" 2>/dev/null
+
+# Create a fresh, aged set of orphan objects from the helper module's init
+# path (its kmalloc/vmalloc/percpu allocations are dropped right away).
+# Pre-existing reported leaks are greyed first ("clear") so only our
+# orphans are counted. The module is left loaded on purpose: once it is
+# unloaded its symbols are gone, so the orphan backtraces no longer resolve
+# to [kmemleak_test] and could not be matched below.
+gen_orphans() {
+ rmmod "$MODULE" 2>/dev/null
+ echo clear > "$KMEMLEAK"
+ modprobe "$MODULE" ||
+ { ktap_skip_all "failed to load $MODULE"; exit "$KSFT_SKIP"; }
+ sleep "$AGE"
+}
+
+scan() { echo scan > "$KMEMLEAK"; }
+
+# Number of helper-module orphans currently reported by kmemleak. Matching
+# the module's own backtrace ([kmemleak_test]) keeps the count immune to
+# unrelated leaks on the running system. kmemleak only lists an object here
+# once it has been reported, so this reflects the confirmation gating.
+count_orphans() {
+ c=$(grep -c '\[kmemleak_test\]' "$KMEMLEAK" 2>/dev/null)
+ echo "${c:-0}"
+}
+
+# 0) the parameter reads back what was written.
+echo 3 > "$PARAM"
+[ "$(cat "$PARAM")" = "3" ] || ktap_exit_fail_msg "min_unref_scans did not read back as 3"
+
+# Priming scan: kmemleak greys a freshly allocated object on its first scan
+# (its checksum settles then), so nothing can be reported until a second
+# scan. Every case below runs this priming scan before counting.
+prime() { scan; }
+
+# 1) min_unref_scans=1: one scan after priming reports the orphans. This
+# also establishes that the helper produces detectable orphans here.
+echo 1 > "$PARAM"
+gen_orphans
+prime
+scan
+first=$(count_orphans)
+[ "$first" -gt 0 ] ||
+ { ktap_skip_all "$MODULE produced no detectable orphans (cannot test min_unref_scans)"; exit "$KSFT_SKIP"; }
+
+# 2) min_unref_scans=2: after priming, one scan is not enough (still
+# gated), the second reports. The gated-scan-zero check is the core
+# regression.
+echo 2 > "$PARAM"
+gen_orphans
+prime
+scan; s1=$(count_orphans)
+scan; s2=$(count_orphans)
+[ "$s1" -eq 0 ] || ktap_exit_fail_msg "min_unref_scans=2: $s1 orphan(s) after 1 post-prime scan (must be 0)"
+[ "$s2" -gt 0 ] || ktap_exit_fail_msg "min_unref_scans=2: no report after 2 post-prime scans (false negative)"
+
+ktap_set_plan 1
+ktap_test_pass "min_unref_scans=1 reported $first orphan(s) one scan after priming; =2 held them one scan longer ($s1 after one scan, $s2 after two); param read-back ok"
+ktap_finished
diff --git a/tools/testing/selftests/mm/ksm_tests.c b/tools/testing/selftests/mm/ksm_tests.c
index a050f4840cfa..5fd7792a0d47 100644
--- a/tools/testing/selftests/mm/ksm_tests.c
+++ b/tools/testing/selftests/mm/ksm_tests.c
@@ -288,8 +288,8 @@ static bool assert_ksm_pages_count(long dupl_page_count)
static int ksm_save_def(struct ksm_sysfs *ksm_sysfs)
{
if (ksm_read_sysfs(KSM_FP("max_page_sharing"), &ksm_sysfs->max_page_sharing) ||
- numa_available() ? 0 :
- ksm_read_sysfs(KSM_FP("merge_across_nodes"), &ksm_sysfs->merge_across_nodes) ||
+ (numa_available() ? 0 :
+ ksm_read_sysfs(KSM_FP("merge_across_nodes"), &ksm_sysfs->merge_across_nodes)) ||
ksm_read_sysfs(KSM_FP("sleep_millisecs"), &ksm_sysfs->sleep_millisecs) ||
ksm_read_sysfs(KSM_FP("pages_to_scan"), &ksm_sysfs->pages_to_scan) ||
ksm_read_sysfs(KSM_FP("run"), &ksm_sysfs->run) ||
@@ -304,8 +304,8 @@ static int ksm_save_def(struct ksm_sysfs *ksm_sysfs)
static int ksm_restore(struct ksm_sysfs *ksm_sysfs)
{
if (ksm_write_sysfs(KSM_FP("max_page_sharing"), ksm_sysfs->max_page_sharing) ||
- numa_available() ? 0 :
- ksm_write_sysfs(KSM_FP("merge_across_nodes"), ksm_sysfs->merge_across_nodes) ||
+ (numa_available() ? 0 :
+ ksm_write_sysfs(KSM_FP("merge_across_nodes"), ksm_sysfs->merge_across_nodes)) ||
ksm_write_sysfs(KSM_FP("pages_to_scan"), ksm_sysfs->pages_to_scan) ||
ksm_write_sysfs(KSM_FP("run"), ksm_sysfs->run) ||
ksm_write_sysfs(KSM_FP("sleep_millisecs"), ksm_sysfs->sleep_millisecs) ||
@@ -440,9 +440,9 @@ static int get_next_mem_node(int node)
mem_node = i % (max_node + 1);
node_size = numa_node_size(mem_node, NULL);
if (node_size > 0)
- break;
+ return mem_node;
}
- return mem_node;
+ return -ENODEV;
}
static int get_first_mem_node(void)
@@ -455,8 +455,8 @@ static int check_ksm_numa_merge(int merge_type, int mapping, int prot, int timeo
{
void *numa1_map_ptr, *numa2_map_ptr;
struct timespec start_time;
+ int first_node, second_node;
int page_count = 2;
- int first_node;
if (clock_gettime(CLOCK_MONOTONIC_RAW, &start_time)) {
ksft_perror("clock_gettime");
@@ -467,17 +467,19 @@ static int check_ksm_numa_merge(int merge_type, int mapping, int prot, int timeo
ksft_print_msg("NUMA support not enabled\n");
return KSFT_SKIP;
}
- if (numa_num_configured_nodes() <= 1) {
- ksft_print_msg("At least 2 NUMA nodes must be available\n");
+ first_node = get_first_mem_node();
+ second_node = get_next_mem_node(first_node);
+
+ if (second_node < 0) {
+ ksft_print_msg("At least 2 NUMA nodes with memory must be available\n");
return KSFT_SKIP;
}
if (ksm_write_sysfs(KSM_FP("merge_across_nodes"), merge_across_nodes))
return KSFT_FAIL;
/* allocate 2 pages in 2 different NUMA nodes and fill them with the same data */
- first_node = get_first_mem_node();
numa1_map_ptr = numa_alloc_onnode(page_size, first_node);
- numa2_map_ptr = numa_alloc_onnode(page_size, get_next_mem_node(first_node));
+ numa2_map_ptr = numa_alloc_onnode(page_size, second_node);
if (!numa1_map_ptr || !numa2_map_ptr) {
ksft_perror("numa_alloc_onnode");
return KSFT_FAIL;
@@ -844,8 +846,8 @@ int main(int argc, char *argv[])
if (ksm_write_sysfs(KSM_FP("run"), 2) ||
ksm_write_sysfs(KSM_FP("sleep_millisecs"), 0) ||
- numa_available() ? 0 :
- ksm_write_sysfs(KSM_FP("merge_across_nodes"), 1) ||
+ (numa_available() ? 0 :
+ ksm_write_sysfs(KSM_FP("merge_across_nodes"), 1)) ||
ksm_write_sysfs(KSM_FP("pages_to_scan"), page_count))
ksft_exit_fail_msg("Cannot set up KSM tunables\n");
diff --git a/tools/testing/selftests/mm/memfd_secret.c b/tools/testing/selftests/mm/memfd_secret.c
index aac4f795c327..c55d84c5e613 100644
--- a/tools/testing/selftests/mm/memfd_secret.c
+++ b/tools/testing/selftests/mm/memfd_secret.c
@@ -57,33 +57,6 @@ static void test_file_apis(int fd)
pass("file IO is blocked as expected\n");
}
-static void test_mlock_limit(int fd)
-{
- size_t len;
- char *mem;
-
- len = mlock_limit_cur;
- if (len % page_size != 0)
- len = (len/page_size) * page_size;
-
- mem = mmap(NULL, len, prot, mode, fd, 0);
- if (mem == MAP_FAILED) {
- fail("unable to mmap secret memory\n");
- return;
- }
- munmap(mem, len);
-
- len = mlock_limit_max * 2;
- mem = mmap(NULL, len, prot, mode, fd, 0);
- if (mem != MAP_FAILED) {
- fail("unexpected mlock limit violation\n");
- munmap(mem, len);
- return;
- }
-
- pass("mlock limit is respected\n");
-}
-
static void test_vmsplice(int fd, const char *desc)
{
ssize_t transferred;
@@ -297,7 +270,7 @@ static void prepare(void)
strerror(errno));
}
-#define NUM_TESTS 6
+#define NUM_TESTS 5
int main(int argc, char *argv[])
{
@@ -319,7 +292,6 @@ int main(int argc, char *argv[])
if (ftruncate(fd, page_size))
ksft_exit_fail_msg("ftruncate failed: %s\n", strerror(errno));
- test_mlock_limit(fd);
test_file_apis(fd);
/*
* We have to run the first vmsplice test before any secretmem page was
diff --git a/tools/testing/selftests/mm/memory-failure.c b/tools/testing/selftests/mm/memory-failure.c
index 032ed952057c..f3cb578b1609 100644
--- a/tools/testing/selftests/mm/memory-failure.c
+++ b/tools/testing/selftests/mm/memory-failure.c
@@ -46,7 +46,7 @@ FIXTURE(memory_failure)
unsigned long pfn;
int pagemap_fd;
int kpageflags_fd;
- bool triggered;
+ bool injection_attempted;
};
FIXTURE_VARIANT(memory_failure)
@@ -122,13 +122,6 @@ static void teardown_sighandler(void)
sigaction(SIGBUS, &sa, NULL);
}
-FIXTURE_TEARDOWN(memory_failure)
-{
- close(self->kpageflags_fd);
- close(self->pagemap_fd);
- teardown_sighandler();
-}
-
static void prepare(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure) * self,
void *vaddr)
{
@@ -200,8 +193,7 @@ static void check(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure
ASSERT_EQ(pfn_flags & KPF_HWPOISON, KPF_HWPOISON);
}
-static void cleanup(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure) * self,
- void *vaddr)
+static void cleanup(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure) * self)
{
unsigned long size;
uint64_t pfn_flags;
@@ -217,6 +209,20 @@ static void cleanup(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failu
ASSERT_EQ(size, self->corrupted_size);
}
+FIXTURE_TEARDOWN(memory_failure)
+{
+ /*
+ * Injection may poison the page before failing or delivering SIGBUS, so
+ * clean up after every injection attempt.
+ */
+ if (self->injection_attempted)
+ cleanup(_metadata, self);
+
+ close(self->kpageflags_fd);
+ close(self->pagemap_fd);
+ teardown_sighandler();
+}
+
TEST_F(memory_failure, anon)
{
char *addr;
@@ -231,8 +237,8 @@ TEST_F(memory_failure, anon)
prepare(_metadata, self, addr);
ret = sigsetjmp(signal_jmp_buf, 1);
- if (!self->triggered) {
- self->triggered = true;
+ if (!self->injection_attempted) {
+ self->injection_attempted = true;
ASSERT_EQ(variant->inject(self, addr), 0);
FORCE_READ(*addr);
}
@@ -242,8 +248,6 @@ TEST_F(memory_failure, anon)
else
check(_metadata, self, addr, MADV_SOFT_ANON, ret);
- cleanup(_metadata, self, addr);
-
ASSERT_EQ(munmap(addr, self->page_size), 0);
}
@@ -283,8 +287,10 @@ TEST_F(memory_failure, clean_pagecache)
if (fd < 0)
SKIP(return, "failed to open test file.\n");
fs_type = get_fs_type(fd);
- if (!fs_type || fs_type == TMPFS_MAGIC)
+ if (!fs_type || fs_type == TMPFS_MAGIC) {
+ close(fd);
SKIP(return, "unsupported filesystem :%x\n", fs_type);
+ }
addr = mmap(0, self->page_size, PROT_READ | PROT_WRITE,
MAP_SHARED, fd, 0);
@@ -296,8 +302,8 @@ TEST_F(memory_failure, clean_pagecache)
prepare(_metadata, self, addr);
ret = sigsetjmp(signal_jmp_buf, 1);
- if (!self->triggered) {
- self->triggered = true;
+ if (!self->injection_attempted) {
+ self->injection_attempted = true;
ASSERT_EQ(variant->inject(self, addr), 0);
FORCE_READ(*addr);
}
@@ -307,8 +313,6 @@ TEST_F(memory_failure, clean_pagecache)
else
check(_metadata, self, addr, MADV_SOFT_CLEAN_PAGECACHE, ret);
- cleanup(_metadata, self, addr);
-
ASSERT_EQ(munmap(addr, self->page_size), 0);
ASSERT_EQ(close(fd), 0);
@@ -325,8 +329,16 @@ TEST_F(memory_failure, dirty_pagecache)
if (fd < 0)
SKIP(return, "failed to open test file.\n");
fs_type = get_fs_type(fd);
- if (!fs_type || fs_type == TMPFS_MAGIC)
+ /*
+ * MADV_HARD poisoning of dirty page-cache data records an expected
+ * -EIO in the file mapping. NFS reports this error on close(), so
+ * skip this variant.
+ */
+ if (!fs_type || fs_type == TMPFS_MAGIC ||
+ (fs_type == NFS_SUPER_MAGIC && variant->type == MADV_HARD)) {
+ close(fd);
SKIP(return, "unsupported filesystem :%x\n", fs_type);
+ }
addr = mmap(0, self->page_size, PROT_READ | PROT_WRITE,
MAP_SHARED, fd, 0);
@@ -337,8 +349,8 @@ TEST_F(memory_failure, dirty_pagecache)
prepare(_metadata, self, addr);
ret = sigsetjmp(signal_jmp_buf, 1);
- if (!self->triggered) {
- self->triggered = true;
+ if (!self->injection_attempted) {
+ self->injection_attempted = true;
ASSERT_EQ(variant->inject(self, addr), 0);
FORCE_READ(*addr);
}
@@ -348,8 +360,6 @@ TEST_F(memory_failure, dirty_pagecache)
else
check(_metadata, self, addr, MADV_SOFT_DIRTY_PAGECACHE, ret);
- cleanup(_metadata, self, addr);
-
ASSERT_EQ(munmap(addr, self->page_size), 0);
ASSERT_EQ(close(fd), 0);
diff --git a/tools/testing/selftests/mm/merge.c b/tools/testing/selftests/mm/merge.c
index 519e5ac02db7..52b8727b6628 100644
--- a/tools/testing/selftests/mm/merge.c
+++ b/tools/testing/selftests/mm/merge.c
@@ -1305,6 +1305,63 @@ TEST_F(merge, merge_vmas_with_mseal)
ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 2 * page_size);
}
+TEST_F(merge, anon_and_page_offset_mismatch_memfd)
+{
+ struct procmap_fd *procmap = &self->procmap;
+ unsigned int page_size = self->page_size;
+ char *carveout = self->carveout;
+ char *ptr, *ptr2;
+ int fd;
+
+ /* Create a 10 page memfd descriptor. */
+ fd = memfd_create("anon_page_offset_test", MFD_CLOEXEC);
+ ASSERT_NE(fd, -1);
+ ASSERT_EQ(ftruncate(fd, 10 * page_size), 0);
+
+ /* Map a region using the memfd at page offset 0. */
+ ptr = mmap(carveout, 5 * page_size, PROT_READ | PROT_WRITE,
+ MAP_FIXED | MAP_PRIVATE, fd, 0);
+ ASSERT_NE(ptr, MAP_FAILED);
+
+ /*
+ * Map another separately and trigger a CoW fault at page offset 5:
+ *
+ * |-----------| |---------|
+ * | unfaulted | | faulted |
+ * |-----------| |---------|
+ */
+ ptr2 = mmap(&carveout[10 * page_size], 5 * page_size,
+ PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE,
+ fd, 5 * page_size);
+ ASSERT_NE(ptr2, MAP_FAILED);
+ ptr2[0] = 'x';
+
+ /*
+ * Now move it in place:
+ *
+ * |----------|
+ * | |
+ * v |
+ * |-----------| |---------|
+ * | unfaulted | | faulted |
+ * |-----------| |---------|
+ *
+ * Because the anonymous page offset of the faulted region is now
+ * &carveout[10 * page_size], despite the two regions being mergeable
+ * due to file page offset, they are NOT mergeable due to anonymous
+ * page offset.
+ */
+ ptr2 = sys_mremap(ptr2, 5 * page_size, 5 * page_size,
+ MREMAP_MAYMOVE | MREMAP_FIXED,
+ &carveout[5 * page_size]);
+ ASSERT_NE(ptr2, MAP_FAILED);
+
+ /* Assert that they did not merge. */
+ ASSERT_TRUE(find_vma_procmap(procmap, ptr));
+ ASSERT_EQ(procmap->query.vma_start, (unsigned long)ptr);
+ ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 5 * page_size);
+}
+
TEST_F(merge_with_fork, mremap_faulted_to_unfaulted_prev)
{
struct procmap_fd *procmap = &self->procmap;
diff --git a/tools/testing/selftests/mm/migration.c b/tools/testing/selftests/mm/migration.c
index 29f7492453d4..f19d53c69576 100644
--- a/tools/testing/selftests/mm/migration.c
+++ b/tools/testing/selftests/mm/migration.c
@@ -7,7 +7,7 @@
#include "kselftest_harness.h"
#include "hugepage_settings.h"
-#include <strings.h>
+#include <string.h>
#include <pthread.h>
#include <numa.h>
#include <numaif.h>
@@ -20,7 +20,6 @@
#define TWOMEG (2<<20)
#define RUNTIME (20)
-#define MAX_RETRIES 100
#define ALIGN(x, a) (((x) + (a - 1)) & (~((a) - 1)))
HUGETLB_SETUP_DEFAULT_PAGES(1)
@@ -110,7 +109,7 @@ int migrate(uint64_t *ptr, int n1, int n2)
int ret, tmp;
int status = 0;
struct timespec ts1, ts2;
- int failures = 0;
+ int success = 0;
if (clock_gettime(CLOCK_MONOTONIC, &ts1))
return -1;
@@ -119,29 +118,33 @@ int migrate(uint64_t *ptr, int n1, int n2)
if (clock_gettime(CLOCK_MONOTONIC, &ts2))
return -1;
- if (ts2.tv_sec - ts1.tv_sec >= RUNTIME)
- return 0;
+ if (ts2.tv_sec - ts1.tv_sec >= RUNTIME) {
+ /* Reaching both targets verifies a cross-node move. */
+ if (success >= 2)
+ return 0;
+ else
+ return -2;
+ }
ret = move_pages(0, 1, (void **) &ptr, &n2, &status,
MPOL_MF_MOVE_ALL);
- if (ret) {
- if (ret > 0) {
- /* Migration is best effort; try again */
- if (++failures < MAX_RETRIES)
- continue;
- printf("Didn't migrate %d pages\n", ret);
- }
- else
- perror("Couldn't migrate pages");
+ if (ret < 0) {
+ perror("Couldn't migrate pages");
+ return ret;
+ }
+ /* Migration is best effort. Try again */
+ if (ret > 0 || status < 0)
+ continue;
+ if (status != n2) {
+ printf("Page is on node %d instead of target node %d\n",
+ status, n2);
return -2;
}
- failures = 0;
+ success++;
tmp = n2;
n2 = n1;
n1 = tmp;
}
-
- return 0;
}
void *access_mem(void *ptr)
diff --git a/tools/testing/selftests/mm/mremap_dontunmap.c b/tools/testing/selftests/mm/mremap_dontunmap.c
index a4f75d836733..96ba537facf7 100644
--- a/tools/testing/selftests/mm/mremap_dontunmap.c
+++ b/tools/testing/selftests/mm/mremap_dontunmap.c
@@ -313,7 +313,7 @@ static void mremap_dontunmap_partial_mapping_overwrite(void)
mremap(source_mapping, 5 * page_size,
5 * page_size,
MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED, dest_mapping);
- BUG_ON(dest_mapping == MAP_FAILED, "mremap");
+ BUG_ON(remapped_mapping == MAP_FAILED, "mremap");
BUG_ON(dest_mapping != remapped_mapping, "expected to remap to dest_mapping");
BUG_ON(check_region_contains_byte(source_mapping, 5 * page_size, 0) !=
diff --git a/tools/testing/selftests/mm/mseal_test.c b/tools/testing/selftests/mm/mseal_test.c
index faad4833366a..1a05e6921fed 100644
--- a/tools/testing/selftests/mm/mseal_test.c
+++ b/tools/testing/selftests/mm/mseal_test.c
@@ -234,7 +234,7 @@ static void test_seal_addseal(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
ret = sys_mseal(ptr, size);
FAIL_TEST_IF_FALSE(!ret);
@@ -250,7 +250,7 @@ static void test_seal_unmapped_start(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* munmap 2 pages from ptr. */
ret = sys_munmap(ptr, 2 * page_size);
@@ -278,7 +278,7 @@ static void test_seal_unmapped_middle(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* munmap 2 pages from ptr + page. */
ret = sys_munmap(ptr + page_size, 2 * page_size);
@@ -310,7 +310,7 @@ static void test_seal_unmapped_end(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* unmap last 2 pages. */
ret = sys_munmap(ptr + 2 * page_size, 2 * page_size);
@@ -339,7 +339,7 @@ static void test_seal_multiple_vmas(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split the vma into 3. */
ret = sys_mprotect(ptr + page_size, 2 * page_size,
@@ -370,7 +370,7 @@ static void test_seal_split_start(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split at middle */
ret = sys_mprotect(ptr, 2 * page_size, PROT_READ | PROT_WRITE);
@@ -395,7 +395,7 @@ static void test_seal_split_end(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split at middle */
ret = sys_mprotect(ptr, 2 * page_size, PROT_READ | PROT_WRITE);
@@ -420,7 +420,7 @@ static void test_seal_invalid_input(void)
int ret;
setup_single_address(8 * page_size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
ret = clean_single_address(ptr + 4 * page_size, 4 * page_size);
FAIL_TEST_IF_FALSE(!ret);
@@ -455,7 +455,7 @@ static void test_seal_zero_length(void)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
ret = sys_mprotect(ptr, 0, PROT_READ | PROT_WRITE);
FAIL_TEST_IF_FALSE(!ret);
@@ -505,7 +505,7 @@ static void test_seal_twice(void)
unsigned long size = 4 * page_size;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
ret = sys_mseal(ptr, size);
FAIL_TEST_IF_FALSE(!ret);
@@ -525,7 +525,7 @@ static void test_seal_mprotect(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr, size);
@@ -549,7 +549,7 @@ static void test_seal_start_mprotect(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr, page_size);
@@ -579,7 +579,7 @@ static void test_seal_end_mprotect(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr + page_size, 3 * page_size);
@@ -609,7 +609,7 @@ static void test_seal_mprotect_unalign_len(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr, page_size * 2 - 1);
@@ -638,7 +638,7 @@ static void test_seal_mprotect_unalign_len_variant_2(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr, page_size * 2 + 1);
FAIL_TEST_IF_FALSE(!ret);
@@ -666,7 +666,7 @@ static void test_seal_mprotect_two_vma(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split */
ret = sys_mprotect(ptr, page_size * 2, PROT_READ | PROT_WRITE);
@@ -701,7 +701,7 @@ static void test_seal_mprotect_two_vma_with_split(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split as two vma. */
ret = sys_mprotect(ptr, page_size * 2, PROT_READ | PROT_WRITE);
@@ -748,7 +748,7 @@ static void test_seal_mprotect_partial_mprotect(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* seal one page. */
if (seal) {
@@ -780,7 +780,7 @@ static void test_seal_mprotect_partial_mprotect_tail(bool seal)
*/
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr + page_size, page_size);
@@ -810,7 +810,7 @@ static void test_seal_mprotect_two_vma_with_gap(void)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split. */
ret = sys_mprotect(ptr, page_size, PROT_READ | PROT_WRITE);
@@ -853,7 +853,7 @@ static void test_seal_mprotect_split(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split. */
ret = sys_mprotect(ptr, page_size, PROT_READ | PROT_WRITE);
@@ -890,7 +890,7 @@ static void test_seal_mprotect_merge(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split one page. */
ret = sys_mprotect(ptr, page_size, PROT_READ | PROT_WRITE);
@@ -924,7 +924,7 @@ static void test_seal_munmap(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -955,7 +955,7 @@ static void test_seal_munmap_two_vma(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect to split */
ret = sys_mprotect(ptr, page_size * 2, PROT_READ | PROT_WRITE);
@@ -996,7 +996,7 @@ static void test_seal_munmap_vma_with_gap(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
ret = sys_munmap(ptr + page_size, page_size * 2);
FAIL_TEST_IF_FALSE(!ret);
@@ -1028,7 +1028,7 @@ static void test_seal_munmap_partial_across_vmas(bool seal)
int prot;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr + page_size, page_size);
@@ -1058,7 +1058,7 @@ static void test_munmap_start_freed(bool seal)
int prot;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* unmap the first page. */
ret = sys_munmap(ptr, page_size);
@@ -1097,7 +1097,7 @@ static void test_munmap_end_freed(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* unmap last page. */
ret = sys_munmap(ptr + page_size * 3, page_size);
@@ -1128,7 +1128,7 @@ static void test_munmap_middle_freed(bool seal)
int prot;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* unmap 2 pages in the middle. */
ret = sys_munmap(ptr + page_size, page_size * 2);
@@ -1172,7 +1172,7 @@ static void test_seal_mremap_shrink(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1201,7 +1201,7 @@ static void test_seal_mremap_expand(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* ummap last 2 pages. */
ret = sys_munmap(ptr + 2 * page_size, 2 * page_size);
FAIL_TEST_IF_FALSE(!ret);
@@ -1233,9 +1233,9 @@ static void test_seal_mremap_move(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
setup_single_address(size, &newPtr);
- FAIL_TEST_IF_FALSE(newPtr != (void *)-1);
+ FAIL_TEST_IF_FALSE(newPtr != MAP_FAILED);
ret = clean_single_address(newPtr, size);
FAIL_TEST_IF_FALSE(!ret);
@@ -1266,7 +1266,7 @@ static void test_seal_mmap_overwrite_prot(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1294,7 +1294,7 @@ static void test_seal_mmap_expand(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* ummap last 4 pages. */
ret = sys_munmap(ptr + 8 * page_size, 4 * page_size);
FAIL_TEST_IF_FALSE(!ret);
@@ -1325,7 +1325,7 @@ static void test_seal_mmap_shrink(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1354,9 +1354,9 @@ static void test_seal_mremap_shrink_fixed(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
setup_single_address(size, &newAddr);
- FAIL_TEST_IF_FALSE(newAddr != (void *)-1);
+ FAIL_TEST_IF_FALSE(newAddr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1385,9 +1385,9 @@ static void test_seal_mremap_expand_fixed(bool seal)
void *ret2;
setup_single_address(page_size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
setup_single_address(size, &newAddr);
- FAIL_TEST_IF_FALSE(newAddr != (void *)-1);
+ FAIL_TEST_IF_FALSE(newAddr != MAP_FAILED);
if (seal) {
ret = sys_mseal(newAddr, size);
@@ -1416,9 +1416,9 @@ static void test_seal_mremap_move_fixed(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
setup_single_address(size, &newAddr);
- FAIL_TEST_IF_FALSE(newAddr != (void *)-1);
+ FAIL_TEST_IF_FALSE(newAddr != MAP_FAILED);
if (seal) {
ret = sys_mseal(newAddr, size);
@@ -1445,7 +1445,7 @@ static void test_seal_mremap_move_fixed_zero(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1476,7 +1476,7 @@ static void test_seal_mremap_move_dontunmap(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1505,7 +1505,7 @@ static void test_seal_mremap_move_dontunmap_anyaddr(bool seal)
void *ret2;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1517,7 +1517,7 @@ static void test_seal_mremap_move_dontunmap_anyaddr(bool seal)
* use allocate/free to similate that.
*/
setup_single_address(size, &ptr2);
- FAIL_TEST_IF_FALSE(ptr2 != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr2 != MAP_FAILED);
ret = sys_munmap(ptr2, size);
FAIL_TEST_IF_FALSE(!ret);
@@ -1547,7 +1547,7 @@ static void test_seal_merge_and_split(void)
/* (24 RO) */
setup_single_address(24 * page_size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
/* use mprotect(NONE) to set out boundary */
/* (1 NONE) (22 RO) (1 NONE) */
@@ -1637,7 +1637,7 @@ static void test_seal_discard_ro_anon_on_rw(bool seal)
int ret;
setup_single_address_rw(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1669,7 +1669,7 @@ static void test_seal_discard_ro_anon_on_pkey(bool seal)
SKIP_TEST_IF_FALSE(pkey_supported());
setup_single_address_rw(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
pkey = sys_pkey_alloc(0, PKEY_UNRESTRICTED);
FAIL_TEST_IF_FALSE(pkey > 0);
@@ -1751,7 +1751,7 @@ static void test_seal_discard_ro_anon_on_shared(bool seal)
unsigned long mapflags = MAP_ANONYMOUS | MAP_SHARED;
ptr = mmap(NULL, size, PROT_READ, mapflags, -1, 0);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = sys_mseal(ptr, size);
@@ -1779,7 +1779,7 @@ static void test_seal_discard_ro_anon(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr, size);
@@ -1809,7 +1809,7 @@ static void test_seal_discard_across_vmas(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr + page_size, page_size);
@@ -1840,7 +1840,7 @@ static void test_seal_madvise_nodiscard(bool seal)
int ret;
setup_single_address(size, &ptr);
- FAIL_TEST_IF_FALSE(ptr != (void *)-1);
+ FAIL_TEST_IF_FALSE(ptr != MAP_FAILED);
if (seal) {
ret = seal_single_address(ptr, size);
@@ -1876,7 +1876,7 @@ int main(void)
if (!pkey_supported())
ksft_print_msg("PKEY not supported\n");
- ksft_set_plan(88);
+ ksft_set_plan(87);
test_seal_addseal();
test_seal_unmapped_start();
@@ -1914,7 +1914,6 @@ int main(void)
test_seal_mprotect_partial_mprotect(true);
test_seal_mprotect_two_vma_with_gap();
- test_seal_mprotect_two_vma_with_gap();
test_seal_mprotect_merge(false);
test_seal_mprotect_merge(true);
diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c
index f9bcff8e78fa..eadc7159ca5b 100644
--- a/tools/testing/selftests/mm/pagemap_ioctl.c
+++ b/tools/testing/selftests/mm/pagemap_ioctl.c
@@ -8,6 +8,7 @@
#include <errno.h>
#include <malloc.h>
#include <linux/types.h>
+#include <linux/mman.h>
#include <linux/memfd.h>
#include <linux/userfaultfd.h>
#include <linux/fs.h>
@@ -212,6 +213,8 @@ int userfaultfd_tests(void)
vec_size = mem_size/page_size;
vec = calloc(vec_size, sizeof(struct page_region));
+ if (!vec)
+ ksft_exit_fail_msg("error nomem\n");
written = pagemap_ioctl(mem, mem_size, vec, 1, PM_SCAN_WP_MATCHING | PM_SCAN_CHECK_WPASYNC,
vec_size - 2, PAGE_IS_WRITTEN, 0, 0, PAGE_IS_WRITTEN);
@@ -699,6 +702,8 @@ int base_tests(char *prefix, char *mem, unsigned long long mem_size, int skip)
vec_size = mem_size/page_size;
vec = calloc(vec_size, sizeof(struct page_region));
vec2 = calloc(vec_size, sizeof(struct page_region));
+ if (!vec || !vec2)
+ ksft_exit_fail_msg("error nomem\n");
/* 1. all new pages must be not be written (dirty) */
written = pagemap_ioctl(mem, mem_size, vec, 1, PM_SCAN_WP_MATCHING | PM_SCAN_CHECK_WPASYNC,
@@ -1000,6 +1005,8 @@ int unmapped_region_tests(void)
int written, len = 0x00040000;
long vec_size = len / page_size;
struct page_region *vec = calloc(vec_size, sizeof(struct page_region));
+ if (!vec)
+ ksft_exit_fail_msg("error nomem\n");
/* 1. Get written pages */
written = pagemap_ioctl(start, len, vec, vec_size, 0, 0,
@@ -1058,50 +1065,96 @@ static void test_simple(void)
* the generic path (reached e.g. via category_anyof_mask) must report every
* page written.
*/
-static void unpopulated_scan_test(void)
+/*
+ * Populate @mem (optionally collapsing it into a THP first), drop it with
+ * MADV_DONTNEED, then check PAGEMAP_SCAN reports the whole range written via
+ * both the fast and generic query paths. A dropped THP leaves a pmd_none hole
+ * with no page table, exercising pagemap_scan_pte_hole(); a base-page range
+ * leaves pte_none entries.
+ */
+static void unpopulated_written_test(const char *name, char *mem, long size,
+ bool use_thp)
{
- int npages = 16, i;
- long mem_size = npages * page_size;
+ long npages = size / page_size, fast = 0, slow = 0, ret;
struct page_region regions[16];
- long fast = 0, slow = 0, ret;
- char *mem;
-
- mem = mmap(NULL, mem_size, PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
- if (mem == MAP_FAILED)
- ksft_exit_fail_msg("%s mmap failed\n", __func__);
+ int i;
- wp_init(mem, mem_size);
+ wp_init(mem, size);
- /* Populate, then drop: the ptes become pte_none without a marker. */
- memset(mem, 1, mem_size);
- if (madvise(mem, mem_size, MADV_DONTNEED))
- ksft_exit_fail_msg("%s MADV_DONTNEED failed\n", __func__);
+ /* Populate, optionally collapse to a THP, then drop it. */
+ memset(mem, 1, size);
+ if (use_thp &&
+ (madvise(mem, size, MADV_COLLAPSE) ||
+ !check_huge_anon(mem, size, size / hpage_size, hpage_size))) {
+ ksft_test_result_skip("%s could not form a THP\n", name);
+ goto out;
+ }
+ if (madvise(mem, size, MADV_DONTNEED)) {
+ ksft_test_result_fail("%s MADV_DONTNEED failed\n", name);
+ goto out;
+ }
/* Fast path: category_mask == return_mask == PAGE_IS_WRITTEN. */
- ret = pagemap_ioctl(mem, mem_size, regions, npages, 0, 0,
+ ret = pagemap_ioctl(mem, size, regions, ARRAY_SIZE(regions), 0, 0,
PAGE_IS_WRITTEN, 0, 0, PAGE_IS_WRITTEN);
- if (ret < 0)
- ksft_exit_fail_msg("%s fast scan failed\n", __func__);
- for (i = 0; i < ret; i++)
+ for (i = 0; ret > 0 && i < ret; i++)
fast += LEN(regions[i]);
/* Generic path: same query expressed via category_anyof_mask. */
- ret = pagemap_ioctl(mem, mem_size, regions, npages, 0, 0,
+ ret = pagemap_ioctl(mem, size, regions, ARRAY_SIZE(regions), 0, 0,
0, PAGE_IS_WRITTEN, 0, PAGE_IS_WRITTEN);
- if (ret < 0)
- ksft_exit_fail_msg("%s generic scan failed\n", __func__);
- for (i = 0; i < ret; i++)
+ for (i = 0; ret > 0 && i < ret; i++)
slow += LEN(regions[i]);
ksft_test_result(fast == npages && slow == npages,
- "%s unpopulated ptes reported written by both paths (%ld, %ld of %d)\n",
- __func__, fast, slow, npages);
+ "%s unpopulated range reported written by both paths (%ld, %ld of %ld)\n",
+ name, fast, slow, npages);
+out:
+ wp_free(mem, size);
+}
- wp_free(mem, mem_size);
+static void unpopulated_scan_test(void)
+{
+ long mem_size = 16 * page_size;
+ char *mem;
+
+ mem = mmap(NULL, mem_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (mem == MAP_FAILED) {
+ ksft_test_result_skip("%s mmap failed\n", __func__);
+ return;
+ }
+
+ unpopulated_written_test(__func__, mem, mem_size, false);
munmap(mem, mem_size);
}
+/*
+ * Same as unpopulated_scan_test(), but the range is a THP: a full-PMD
+ * MADV_DONTNEED leaves a pmd_none hole with no page table.
+ */
+static void unpopulated_thp_scan_test(void)
+{
+ char *area, *mem;
+
+ if (!hpage_size) {
+ ksft_test_result_skip("%s THP not supported\n", __func__);
+ return;
+ }
+
+ /* Over-allocate so a PMD-aligned, THP-sized range fits inside. */
+ area = mmap(NULL, 2 * hpage_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (area == MAP_FAILED) {
+ ksft_test_result_skip("%s mmap failed\n", __func__);
+ return;
+ }
+ mem = (char *)(((unsigned long)area + hpage_size - 1) & ~(hpage_size - 1));
+
+ unpopulated_written_test(__func__, mem, hpage_size, true);
+ munmap(area, 2 * hpage_size);
+}
+
int sanity_tests(void)
{
unsigned long long mem_size, vec_size;
@@ -1116,7 +1169,7 @@ int sanity_tests(void)
vec = calloc(vec_size, sizeof(struct page_region));
mem = mmap(NULL, mem_size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANON, -1, 0);
- if (mem == MAP_FAILED || vec == MAP_FAILED)
+ if (mem == MAP_FAILED || !vec)
ksft_exit_fail_msg("error nomem\n");
wp_init(mem, mem_size);
@@ -1279,12 +1332,6 @@ int mprotect_tests(void)
int ret;
char *mem, *mem2;
struct page_region vec;
- int pagemap_fd = open("/proc/self/pagemap", O_RDONLY);
-
- if (pagemap_fd < 0) {
- fprintf(stderr, "open() failed\n");
- exit(1);
- }
/* 1. Map two pages */
mem = mmap(0, 2 * page_size, PROT_READ|PROT_WRITE, MAP_PRIVATE | MAP_ANON, -1, 0);
@@ -1610,7 +1657,7 @@ int main(int __attribute__((unused)) argc, char *argv[])
if (!hugetlb_setup_default(4))
ksft_print_msg("HugeTLB test will be skipped\n");
- ksft_set_plan(118);
+ ksft_set_plan(119);
page_size = getpagesize();
hpage_size = read_pmd_pagesize();
@@ -1790,6 +1837,7 @@ int main(int __attribute__((unused)) argc, char *argv[])
/* 18. Unpopulated pte scan-path consistency */
unpopulated_scan_test();
+ unpopulated_thp_scan_test();
close(pagemap_fd);
ksft_finished();
diff --git a/tools/testing/selftests/mm/pkey-helpers.h b/tools/testing/selftests/mm/pkey-helpers.h
index 2c377f4e9df1..46a8a1878dc1 100644
--- a/tools/testing/selftests/mm/pkey-helpers.h
+++ b/tools/testing/selftests/mm/pkey-helpers.h
@@ -68,7 +68,9 @@ static inline void sigsafe_printf(const char *format, ...)
#define dprintf3(args...) dprintf_level(3, args)
#define dprintf4(args...) dprintf_level(4, args)
-extern void abort_hooks(void);
+void tracing_on(void);
+void tracing_off(void);
+void abort_hooks(void);
#define pkey_assert(condition) do { \
if (!(condition)) { \
dprintf0("# assert() at %s::%d test_nr: %d iteration: %d\n", \
diff --git a/tools/testing/selftests/mm/pkey-powerpc.h b/tools/testing/selftests/mm/pkey-powerpc.h
index 17bf2d1b0192..2ce85580b404 100644
--- a/tools/testing/selftests/mm/pkey-powerpc.h
+++ b/tools/testing/selftests/mm/pkey-powerpc.h
@@ -126,7 +126,7 @@ static inline void *malloc_pkey_with_mprotect_subpage(long size, int prot, u16 p
size, prot, pkey);
pkey_assert(pkey < NR_PKEYS);
ptr = mmap(NULL, size, prot, MAP_ANONYMOUS|MAP_PRIVATE, -1, 0);
- pkey_assert(ptr != (void *)-1);
+ pkey_assert(ptr != MAP_FAILED);
ret = syscall(__NR_subpage_prot, ptr, size, NULL);
if (ret) {
diff --git a/tools/testing/selftests/mm/pkey_sighandler_tests.c b/tools/testing/selftests/mm/pkey_sighandler_tests.c
index 302fef54049c..c218d0510a2a 100644
--- a/tools/testing/selftests/mm/pkey_sighandler_tests.c
+++ b/tools/testing/selftests/mm/pkey_sighandler_tests.c
@@ -19,7 +19,6 @@
#include <stdint.h>
#include <stdbool.h>
#include <signal.h>
-#include <assert.h>
#include <stdlib.h>
#include <sys/mman.h>
#include <sys/types.h>
@@ -207,22 +206,25 @@ static void test_sigsegv_handler_with_pkey0_disabled(void)
struct sigaction sa;
pthread_attr_t attr;
pthread_t thr;
+ int ret;
sa.sa_flags = SA_SIGINFO;
sa.sa_sigaction = sigsegv_handler;
sigemptyset(&sa.sa_mask);
- if (sigaction(SIGSEGV, &sa, NULL) == -1) {
- perror("sigaction");
- exit(EXIT_FAILURE);
- }
+ ret = sigaction(SIGSEGV, &sa, NULL);
+ pkey_assert(ret == 0);
memset(&siginfo, 0, sizeof(siginfo));
pthread_attr_init(&attr);
pthread_attr_setdetachstate(&attr, PTHREAD_CREATE_DETACHED);
- pthread_create(&thr, &attr, thread_segv_with_pkey0_disabled, NULL);
+ ret = pthread_create(&thr, &attr, thread_segv_with_pkey0_disabled, NULL);
+ if (ret) {
+ errno = ret;
+ pkey_assert(0);
+ }
pthread_mutex_lock(&mutex);
while (siginfo.si_signo == 0)
@@ -247,22 +249,25 @@ static void test_sigsegv_handler_cannot_access_stack(void)
struct sigaction sa;
pthread_attr_t attr;
pthread_t thr;
+ int ret;
sa.sa_flags = SA_SIGINFO;
sa.sa_sigaction = sigsegv_handler;
sigemptyset(&sa.sa_mask);
- if (sigaction(SIGSEGV, &sa, NULL) == -1) {
- perror("sigaction");
- exit(EXIT_FAILURE);
- }
+ ret = sigaction(SIGSEGV, &sa, NULL);
+ pkey_assert(ret == 0);
memset(&siginfo, 0, sizeof(siginfo));
pthread_attr_init(&attr);
pthread_attr_setdetachstate(&attr, PTHREAD_CREATE_DETACHED);
- pthread_create(&thr, &attr, thread_segv_pkuerr_stack, NULL);
+ ret = pthread_create(&thr, &attr, thread_segv_pkuerr_stack, NULL);
+ if (ret) {
+ errno = ret;
+ pkey_assert(0);
+ }
pthread_mutex_lock(&mutex);
while (siginfo.si_signo == 0)
@@ -285,24 +290,22 @@ static void test_sigsegv_handler_with_different_pkey_for_stack(void)
static stack_t sigstack;
void *stack;
int pkey;
- int parent_pid = 0;
int child_pid = 0;
u64 pkey_reg;
+ long ret;
sa.sa_flags = SA_SIGINFO | SA_ONSTACK;
sa.sa_sigaction = sigsegv_handler;
sigemptyset(&sa.sa_mask);
- if (sigaction(SIGSEGV, &sa, NULL) == -1) {
- perror("sigaction");
- exit(EXIT_FAILURE);
- }
+ ret = sigaction(SIGSEGV, &sa, NULL);
+ pkey_assert(ret == 0);
stack = mmap(0, STACK_SIZE, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
- assert(stack != MAP_FAILED);
+ pkey_assert(stack != MAP_FAILED);
/* Allow access to MPK 0 and MPK 1 */
pkey_reg = pkey_reg_restrictive_default();
@@ -317,33 +320,41 @@ static void test_sigsegv_handler_with_different_pkey_for_stack(void)
/* Set up alternate signal stack that will use the default MPK */
sigstack.ss_sp = mmap(0, STACK_SIZE, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ pkey_assert(sigstack.ss_sp != MAP_FAILED);
sigstack.ss_flags = 0;
sigstack.ss_size = STACK_SIZE;
memset(&siginfo, 0, sizeof(siginfo));
/* Use clone to avoid newer glibcs using rseq on new threads */
- long ret = clone_raw(CLONE_VM | CLONE_FS | CLONE_FILES |
- CLONE_SIGHAND | CLONE_THREAD | CLONE_SYSVSEM |
- CLONE_PARENT_SETTID | CLONE_CHILD_CLEARTID |
- CLONE_DETACHED,
- stack + STACK_SIZE,
- &parent_pid,
- &child_pid);
+ ret = clone_raw(CLONE_VM | CLONE_FS | CLONE_FILES |
+ CLONE_SIGHAND | CLONE_THREAD | CLONE_SYSVSEM |
+ CLONE_DETACHED,
+ stack + STACK_SIZE,
+ NULL,
+ NULL);
if (ret < 0) {
errno = -ret;
- perror("clone");
+ pkey_assert(0);
} else if (ret == 0) {
thread_segv_maperr_ptr(&sigstack);
syscall_raw(SYS_exit, 0, 0, 0, 0, 0, 0);
}
+ child_pid = ret;
+
pthread_mutex_lock(&mutex);
while (siginfo.si_signo == 0)
pthread_cond_wait(&cond, &mutex);
pthread_mutex_unlock(&mutex);
+ /* Wait for child to exit before returning */
+ do {
+ sched_yield();
+ ret = syscall_raw(SYS_tkill, child_pid, 0, 0, 0, 0, 0);
+ } while (ret != -ESRCH && ret != -EINVAL);
+
ksft_test_result(siginfo.si_signo == SIGSEGV &&
siginfo.si_code == SEGV_MAPERR &&
siginfo.si_addr == NULL,
@@ -358,6 +369,7 @@ static void test_pkru_preserved_after_sigusr1(void)
{
struct sigaction sa;
u64 pkey_reg;
+ int ret;
/* Allow access to MPK 0 and an arbitrary set of keys */
pkey_reg = pkey_reg_restrictive_default();
@@ -369,10 +381,8 @@ static void test_pkru_preserved_after_sigusr1(void)
sa.sa_sigaction = sigusr1_handler;
sigemptyset(&sa.sa_mask);
- if (sigaction(SIGUSR1, &sa, NULL) == -1) {
- perror("sigaction");
- exit(EXIT_FAILURE);
- }
+ ret = sigaction(SIGUSR1, &sa, NULL);
+ pkey_assert(ret == 0);
memset(&siginfo, 0, sizeof(siginfo));
@@ -441,9 +451,15 @@ static void test_pkru_sigreturn(void)
static stack_t sigstack;
void *stack;
int pkey;
- int parent_pid = 0;
int child_pid = 0;
u64 pkey_reg;
+ long ret;
+
+ /*
+ * SIGSEGV handler is reset to SIG_DFL below; turn tracing off first
+ * so a crash does not leave ftrace enabled.
+ */
+ tracing_off();
sa.sa_handler = SIG_DFL;
sa.sa_flags = 0;
@@ -453,24 +469,20 @@ static void test_pkru_sigreturn(void)
* For this testcase, we do not want to handle SIGSEGV. Reset handler
* to default so that the application can crash if it receives SIGSEGV.
*/
- if (sigaction(SIGSEGV, &sa, NULL) == -1) {
- perror("sigaction");
- exit(EXIT_FAILURE);
- }
+ ret = sigaction(SIGSEGV, &sa, NULL);
+ pkey_assert(ret == 0);
sa.sa_flags = SA_SIGINFO | SA_ONSTACK;
sa.sa_sigaction = sigusr2_handler;
sigemptyset(&sa.sa_mask);
- if (sigaction(SIGUSR2, &sa, NULL) == -1) {
- perror("sigaction");
- exit(EXIT_FAILURE);
- }
+ ret = sigaction(SIGUSR2, &sa, NULL);
+ pkey_assert(ret == 0);
stack = mmap(0, STACK_SIZE, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
- assert(stack != MAP_FAILED);
+ pkey_assert(stack != MAP_FAILED);
/*
* Allow access to MPK 0 and MPK 2. The child thread (to be created
@@ -490,21 +502,21 @@ static void test_pkru_sigreturn(void)
/* Set up alternate signal stack that will use the default MPK */
sigstack.ss_sp = mmap(0, STACK_SIZE, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ pkey_assert(sigstack.ss_sp != MAP_FAILED);
sigstack.ss_flags = 0;
sigstack.ss_size = STACK_SIZE;
/* Use clone to avoid newer glibcs using rseq on new threads */
- long ret = clone_raw(CLONE_VM | CLONE_FS | CLONE_FILES |
- CLONE_SIGHAND | CLONE_THREAD | CLONE_SYSVSEM |
- CLONE_PARENT_SETTID | CLONE_CHILD_CLEARTID |
- CLONE_DETACHED,
- stack + STACK_SIZE,
- &parent_pid,
- &child_pid);
+ ret = clone_raw(CLONE_VM | CLONE_FS | CLONE_FILES |
+ CLONE_SIGHAND | CLONE_THREAD | CLONE_SYSVSEM |
+ CLONE_DETACHED,
+ stack + STACK_SIZE,
+ NULL,
+ NULL);
if (ret < 0) {
errno = -ret;
- perror("clone");
+ pkey_assert(0);
} else if (ret == 0) {
thread_sigusr2_self(&sigstack);
syscall_raw(SYS_exit, 0, 0, 0, 0, 0, 0);
@@ -530,16 +542,17 @@ static void (*pkey_tests[])(void) = {
int main(int argc, char *argv[])
{
- int i;
-
ksft_print_header();
ksft_set_plan(ARRAY_SIZE(pkey_tests));
if (!is_pkeys_supported())
ksft_exit_skip("pkeys not supported\n");
- for (i = 0; i < ARRAY_SIZE(pkey_tests); i++)
- (*pkey_tests[i])();
+ for (test_nr = 0; test_nr < ARRAY_SIZE(pkey_tests); test_nr++) {
+ tracing_on();
+ (*pkey_tests[test_nr])();
+ tracing_off();
+ }
ksft_finished();
return 0;
diff --git a/tools/testing/selftests/mm/pkey_util.c b/tools/testing/selftests/mm/pkey_util.c
index 255b332f7a08..fbef3cd45447 100644
--- a/tools/testing/selftests/mm/pkey_util.c
+++ b/tools/testing/selftests/mm/pkey_util.c
@@ -2,9 +2,99 @@
#define __SANE_USERSPACE_TYPES__
#include <sys/syscall.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <stdio.h>
+#include <string.h>
#include "pkey-helpers.h"
+int iteration_nr = 1;
+int test_nr;
+int dprint_in_signal;
+
+#if CONTROL_TRACING > 0
+static void cat_into_file(char *str, char *file)
+{
+ int fd = open(file, O_RDWR);
+ int ret;
+
+ dprintf2("%s(): writing '%s' to '%s'\n", __func__, str, file);
+ /*
+ * these need to be raw because they are called under
+ * pkey_assert()
+ */
+ if (fd < 0) {
+ fprintf(stderr, "error opening '%s'\n", file);
+ perror("error: ");
+ exit(__LINE__);
+ }
+
+ ret = write(fd, str, strlen(str));
+ if (ret != strlen(str)) {
+ perror("write to file failed");
+ fprintf(stderr, "filename: '%s' str: '%s'\n", file, str);
+ exit(__LINE__);
+ }
+ close(fd);
+}
+
+static int warned_tracing;
+static int tracing_root_ok(void)
+{
+ if (geteuid() != 0) {
+ if (!warned_tracing)
+ fprintf(stderr, "WARNING: not run as root, "
+ "can not do tracing control\n");
+ warned_tracing = 1;
+ return 0;
+ }
+ return 1;
+}
+#endif
+
+void tracing_on(void)
+{
+#if CONTROL_TRACING > 0
+#define TRACEDIR "/sys/kernel/tracing"
+ char pidstr[32];
+
+ if (!tracing_root_ok())
+ return;
+
+ sprintf(pidstr, "%d", getpid());
+ cat_into_file("0", TRACEDIR "/tracing_on");
+ cat_into_file("\n", TRACEDIR "/trace");
+ if (1) {
+ cat_into_file("function_graph", TRACEDIR "/current_tracer");
+ cat_into_file("1", TRACEDIR "/options/funcgraph-proc");
+ } else {
+ cat_into_file("nop", TRACEDIR "/current_tracer");
+ }
+ cat_into_file(pidstr, TRACEDIR "/set_ftrace_pid");
+ cat_into_file("1", TRACEDIR "/tracing_on");
+ dprintf1("enabled tracing\n");
+#endif
+}
+
+void tracing_off(void)
+{
+#if CONTROL_TRACING > 0
+ if (!tracing_root_ok())
+ return;
+ cat_into_file("0", "/sys/kernel/tracing/tracing_on");
+#endif
+}
+
+void abort_hooks(void)
+{
+ fflush(stdout);
+ fprintf(stderr, "running %s()...\n", __func__);
+ tracing_off();
+#ifdef SLEEP_ON_ABORT
+ sleep(SLEEP_ON_ABORT);
+#endif
+}
+
int sys_pkey_alloc(unsigned long flags, unsigned long init_val)
{
int ret = syscall(SYS_pkey_alloc, flags, init_val);
diff --git a/tools/testing/selftests/mm/prctl_thp_disable.c b/tools/testing/selftests/mm/prctl_thp_disable.c
index d8d9d1de57b8..82c6e96ea6eb 100644
--- a/tools/testing/selftests/mm/prctl_thp_disable.c
+++ b/tools/testing/selftests/mm/prctl_thp_disable.c
@@ -67,7 +67,7 @@ static int test_mmap_thp(enum thp_collapse_type madvise_buf, size_t pmdsize)
/* HACK: make sure we have a separate VMA that we can check reliably. */
mprotect(mem, pmdsize, PROT_READ);
- ret = check_huge_anon(mem, 1, pmdsize);
+ ret = check_huge_anon(mem, pmdsize, 1, pmdsize);
munmap(mmap_mem, mmap_size);
return ret;
}
diff --git a/tools/testing/selftests/mm/protection_keys.c b/tools/testing/selftests/mm/protection_keys.c
index 9a6d954ee371..ae6e1530b354 100644
--- a/tools/testing/selftests/mm/protection_keys.c
+++ b/tools/testing/selftests/mm/protection_keys.c
@@ -49,11 +49,7 @@
#include "hugepage_settings.h"
#include "pkey-helpers.h"
-int iteration_nr = 1;
-int test_nr;
-
u64 shadow_pkey_reg;
-int dprint_in_signal;
noinline int read_ptr(int *ptr)
{
@@ -62,89 +58,6 @@ noinline int read_ptr(int *ptr)
return *ptr;
}
-#if CONTROL_TRACING > 0
-static void cat_into_file(char *str, char *file)
-{
- int fd = open(file, O_RDWR);
- int ret;
-
- dprintf2("%s(): writing '%s' to '%s'\n", __func__, str, file);
- /*
- * these need to be raw because they are called under
- * pkey_assert()
- */
- if (fd < 0) {
- fprintf(stderr, "error opening '%s'\n", str);
- perror("error: ");
- exit(__LINE__);
- }
-
- ret = write(fd, str, strlen(str));
- if (ret != strlen(str)) {
- perror("write to file failed");
- fprintf(stderr, "filename: '%s' str: '%s'\n", file, str);
- exit(__LINE__);
- }
- close(fd);
-}
-
-static int warned_tracing;
-static int tracing_root_ok(void)
-{
- if (geteuid() != 0) {
- if (!warned_tracing)
- fprintf(stderr, "WARNING: not run as root, "
- "can not do tracing control\n");
- warned_tracing = 1;
- return 0;
- }
- return 1;
-}
-#endif
-
-static void tracing_on(void)
-{
-#if CONTROL_TRACING > 0
-#define TRACEDIR "/sys/kernel/tracing"
- char pidstr[32];
-
- if (!tracing_root_ok())
- return;
-
- sprintf(pidstr, "%d", getpid());
- cat_into_file("0", TRACEDIR "/tracing_on");
- cat_into_file("\n", TRACEDIR "/trace");
- if (1) {
- cat_into_file("function_graph", TRACEDIR "/current_tracer");
- cat_into_file("1", TRACEDIR "/options/funcgraph-proc");
- } else {
- cat_into_file("nop", TRACEDIR "/current_tracer");
- }
- cat_into_file(pidstr, TRACEDIR "/set_ftrace_pid");
- cat_into_file("1", TRACEDIR "/tracing_on");
- dprintf1("enabled tracing\n");
-#endif
-}
-
-static void tracing_off(void)
-{
-#if CONTROL_TRACING > 0
- if (!tracing_root_ok())
- return;
- cat_into_file("0", "/sys/kernel/tracing/tracing_on");
-#endif
-}
-
-void abort_hooks(void)
-{
- fflush(stdout);
- fprintf(stderr, "running %s()...\n", __func__);
- tracing_off();
-#ifdef SLEEP_ON_ABORT
- sleep(SLEEP_ON_ABORT);
-#endif
-}
-
/*
* This attempts to have roughly a page of instructions followed by a few
* instructions that do a write, and another page of instructions. That
@@ -669,7 +582,7 @@ static void *malloc_pkey_with_mprotect(long size, int prot, u16 pkey)
size, prot, pkey);
pkey_assert(pkey < NR_PKEYS);
ptr = mmap(NULL, size, prot, MAP_ANONYMOUS|MAP_PRIVATE, -1, 0);
- pkey_assert(ptr != (void *)-1);
+ pkey_assert(ptr != MAP_FAILED);
ret = mprotect_pkey((void *)ptr, PAGE_SIZE, prot, pkey);
pkey_assert(!ret);
record_pkey_malloc(ptr, size, prot);
@@ -692,7 +605,7 @@ static void *malloc_pkey_anon_huge(long size, int prot, u16 pkey)
*/
size = ALIGN_UP(size, HPAGE_SIZE * 2);
ptr = mmap(NULL, size, PROT_NONE, MAP_ANONYMOUS|MAP_PRIVATE, -1, 0);
- pkey_assert(ptr != (void *)-1);
+ pkey_assert(ptr != MAP_FAILED);
record_pkey_malloc(ptr, size, prot);
mprotect_pkey(ptr, size, prot, pkey);
@@ -750,7 +663,7 @@ static void *malloc_pkey_hugetlb(long size, int prot, u16 pkey)
size = ALIGN_UP(size, HPAGE_SIZE * 2);
pkey_assert(pkey < NR_PKEYS);
ptr = mmap(NULL, size, PROT_NONE, flags, -1, 0);
- pkey_assert(ptr != (void *)-1);
+ pkey_assert(ptr != MAP_FAILED);
mprotect_pkey(ptr, size, prot, pkey);
record_pkey_malloc(ptr, size, prot);
@@ -779,7 +692,7 @@ static void *malloc_pkey(long size, int prot, u16 pkey)
pkey_assert(malloc_type < nr_malloc_types);
ret = pkey_malloc[malloc_type](size, prot, pkey);
- pkey_assert(ret != (void *)-1);
+ pkey_assert(ret != MAP_FAILED);
malloc_type++;
if (malloc_type >= nr_malloc_types)
@@ -1197,6 +1110,7 @@ static void arch_force_pkey_reg_init(void)
* doing the XSAVE size enumeration dance.
*/
buf = mmap(NULL, 1*MB, PROT_READ|PROT_WRITE, MAP_ANONYMOUS|MAP_PRIVATE, -1, 0);
+ pkey_assert(buf != MAP_FAILED);
/* These __builtins require compiling with -mxsave */
@@ -1763,7 +1677,8 @@ int main(void)
ksft_print_msg("running PKEY tests for unsupported CPU/OS\n");
ptr = mmap(NULL, size, PROT_NONE, MAP_ANONYMOUS|MAP_PRIVATE, -1, 0);
- assert(ptr != (void *)-1);
+ if (ptr == MAP_FAILED)
+ ksft_exit_fail_perror("mmap");
test_mprotect_pkey_on_unsupported_cpu(ptr, 1);
ksft_test_result_pass("pkey on unsupported CPU/OS\n");
ksft_finished();
diff --git a/tools/testing/selftests/mm/rmap.c b/tools/testing/selftests/mm/rmap.c
index 53f2058b0ef2..1c293ad3f8b8 100644
--- a/tools/testing/selftests/mm/rmap.c
+++ b/tools/testing/selftests/mm/rmap.c
@@ -430,4 +430,85 @@ TEST_F(migrate, ksm)
propagate_children(_metadata, data);
}
+static bool range_maps_the_same_pfn(int pagemap_fd, void *region, int nr_pages)
+{
+ int i;
+ int retries = 0;
+ unsigned long first_pfn;
+
+retry:
+ if (retries > 10)
+ return false;
+
+ first_pfn = pagemap_get_pfn(pagemap_fd, region);
+ for (i = 0; i < nr_pages; i++) {
+ if (pagemap_get_pfn(pagemap_fd, region + i * getpagesize()) != first_pfn) {
+ /*
+ * Retry up to 10 times at most in case of the low chance of page
+ * compaction migrating the page while we check for pfn.
+ */
+ retries++;
+ goto retry;
+ }
+ }
+
+ return true;
+}
+
+TEST_F(migrate, ksm_and_mremap)
+{
+ unsigned long old_pfn, new_pfn;
+ void *region, *mremap_region;
+ const int nr_pages = 16;
+ size_t mmap_size;
+ int pagemap_fd;
+
+ /* Skip if KSM is not available */
+ if (ksm_stop() < 0)
+ SKIP(return, "accessing \"/sys/kernel/mm/ksm/run\" failed");
+ if (ksm_get_full_scans() < 0)
+ SKIP(return, "accessing \"/sys/kernel/mm/ksm/full_scan\" failed");
+
+ pagemap_fd = open("/proc/self/pagemap", O_RDONLY);
+ if (pagemap_fd < 0)
+ SKIP(return, "opening pagemap failed");
+
+ /* Allocate and populate twice the anon pages initially. */
+ mmap_size = 2 * nr_pages * getpagesize();
+ region = mmap(NULL, mmap_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANON, -1, 0);
+ ASSERT_NE(region, MAP_FAILED);
+ memset(region, 0x77, mmap_size);
+
+ /* mremap the second half over the first half, to stress rmap handling */
+ mmap_size /= 2;
+ mremap_region = mremap(region + mmap_size, mmap_size, mmap_size,
+ MREMAP_MAYMOVE | MREMAP_FIXED, region);
+ ASSERT_EQ(mremap_region, region);
+
+ /* Merge all pages into a single KSM page. */
+ madvise(region, mmap_size, MADV_MERGEABLE);
+ ASSERT_EQ(ksm_start(), 0);
+
+ /* The whole range should map the same KSM page. */
+ old_pfn = pagemap_get_pfn(pagemap_fd, region);
+ if (old_pfn == -1ul)
+ SKIP(return, "Obtaining PFN failed");
+ ksm_start();
+ ASSERT_TRUE(range_maps_the_same_pfn(pagemap_fd, region, nr_pages));
+
+ /*
+ * Migrate the KSM page; the whole range should map the new (migrated)
+ * KSM page.
+ */
+ ASSERT_EQ(try_to_move_page(region), 0);
+
+ new_pfn = pagemap_get_pfn(pagemap_fd, region);
+ if (new_pfn == -1ul)
+ SKIP(return, "Obtaining PFN failed");
+ ASSERT_NE(new_pfn, old_pfn);
+ ASSERT_TRUE(range_maps_the_same_pfn(pagemap_fd, region, nr_pages));
+}
+
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh
index 8c296dedf047..d09f9f6a384e 100755
--- a/tools/testing/selftests/mm/run_vmtests.sh
+++ b/tools/testing/selftests/mm/run_vmtests.sh
@@ -262,7 +262,6 @@ echo "TAP version 13" | tap_output
CATEGORY="hugetlb" run_test ./hugetlb-mmap
CATEGORY="hugetlb" run_test ./hugetlb-shm
CATEGORY="hugetlb" run_test ./hugetlb-mremap
-CATEGORY="hugetlb" run_test ./hugetlb-vmemmap
CATEGORY="hugetlb" run_test ./hugetlb-madvise
CATEGORY="hugetlb" run_test ./hugetlb_dio
CATEGORY="hugetlb" run_test ./hugetlb_fault_after_madv
@@ -303,7 +302,7 @@ CATEGORY="compaction" run_test ./compaction_test
if command -v sudo &> /dev/null && sudo -u nobody ls ./on-fault-limit >/dev/null;
then
- CATEGORY="mlock" run_test sudo -u nobody ./on-fault-limit
+ CATEGORY="mlock" run_test sudo -u nobody sh -c ./on-fault-limit
else
echo "# SKIP ./on-fault-limit"
fi
@@ -411,6 +410,8 @@ CATEGORY="thp" run_test ./khugepaged all:shmem
CATEGORY="thp" run_test ./khugepaged -s 4 all:shmem
+CATEGORY="thp" run_test ./khugepaged -c 4 mthp_khugepaged:anon
+
# Try to create XFS if not provided
if [ -z "${SPLIT_HUGE_PAGE_TEST_XFS_PATH}" ]; then
if test_selected "thp"; then
diff --git a/tools/testing/selftests/mm/soft-dirty.c b/tools/testing/selftests/mm/soft-dirty.c
index fb1864a68e1c..5f278913c4d7 100644
--- a/tools/testing/selftests/mm/soft-dirty.c
+++ b/tools/testing/selftests/mm/soft-dirty.c
@@ -103,7 +103,7 @@ static void test_hugepage(int pagemap_fd, int pagesize)
for (i = 0; i < hpage_len; i++)
map[i] = (char)i;
- if (check_huge_anon(map, 1, hpage_len)) {
+ if (check_huge_anon(map, hpage_len, 1, hpage_len)) {
ksft_test_result_pass("Test %s huge page allocation\n", __func__);
clear_softdirty();
@@ -152,7 +152,8 @@ static void test_mprotect(int pagemap_fd, int pagesize, bool anon)
return;
}
unlink(fname);
- ftruncate(test_fd, pagesize);
+ if (ftruncate(test_fd, pagesize) != 0)
+ ksft_exit_fail_msg("ftruncate failed\n");
map = mmap(NULL, pagesize, PROT_READ|PROT_WRITE,
MAP_SHARED, test_fd, 0);
if (map == MAP_FAILED)
diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c
index 32b991472f74..86a603692826 100644
--- a/tools/testing/selftests/mm/split_huge_page_test.c
+++ b/tools/testing/selftests/mm/split_huge_page_test.c
@@ -104,129 +104,6 @@ fail:
return false;
}
-static int vaddr_pageflags_get(char *vaddr, int pagemap_fd, int kpageflags_fd,
- uint64_t *flags)
-{
- unsigned long pfn;
-
- pfn = pagemap_get_pfn(pagemap_fd, vaddr);
-
- /* non-present PFN */
- if (pfn == -1UL)
- return 1;
-
- if (pageflags_get(pfn, kpageflags_fd, flags))
- return -1;
-
- return 0;
-}
-
-/*
- * gather_after_split_folio_orders - scan through [vaddr_start, len) and record
- * folio orders
- *
- * @vaddr_start: start vaddr
- * @len: range length
- * @pagemap_fd: file descriptor to /proc/<pid>/pagemap
- * @kpageflags_fd: file descriptor to /proc/kpageflags
- * @orders: output folio order array
- * @nr_orders: folio order array size
- *
- * gather_after_split_folio_orders() scan through [vaddr_start, len) and check
- * all folios within the range and record their orders. All order-0 pages will
- * be recorded. Non-present vaddr is skipped.
- *
- * NOTE: the function is used to check folio orders after a split is performed,
- * so it assumes [vaddr_start, len) fully maps to after-split folios within that
- * range.
- *
- * Return: 0 - no error, -1 - unhandled cases
- */
-static int gather_after_split_folio_orders(char *vaddr_start, size_t len,
- int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders)
-{
- uint64_t page_flags = 0;
- int cur_order = -1;
- char *vaddr;
-
- if (pagemap_fd == -1 || kpageflags_fd == -1)
- return -1;
- if (!orders)
- return -1;
- if (nr_orders <= 0)
- return -1;
-
- for (vaddr = vaddr_start; vaddr < vaddr_start + len;) {
- char *next_folio_vaddr;
- int status;
-
- status = vaddr_pageflags_get(vaddr, pagemap_fd, kpageflags_fd,
- &page_flags);
- if (status < 0)
- return -1;
-
- /* skip non present vaddr */
- if (status == 1) {
- vaddr += psize();
- continue;
- }
-
- /* all order-0 pages with possible false postive (non folio) */
- if (!(page_flags & (KPF_COMPOUND_HEAD | KPF_COMPOUND_TAIL))) {
- orders[0]++;
- vaddr += psize();
- continue;
- }
-
- /* skip non thp compound pages */
- if (!(page_flags & KPF_THP)) {
- vaddr += psize();
- continue;
- }
-
- /* vpn points to part of a THP at this point */
- if (page_flags & KPF_COMPOUND_HEAD)
- cur_order = 1;
- else {
- vaddr += psize();
- continue;
- }
-
- next_folio_vaddr = vaddr + (1UL << (cur_order + pshift()));
-
- if (next_folio_vaddr >= vaddr_start + len)
- break;
-
- while ((status = vaddr_pageflags_get(next_folio_vaddr,
- pagemap_fd, kpageflags_fd,
- &page_flags)) >= 0) {
- /*
- * non present vaddr, next compound head page, or
- * order-0 page
- */
- if (status == 1 ||
- (page_flags & KPF_COMPOUND_HEAD) ||
- !(page_flags & (KPF_COMPOUND_HEAD | KPF_COMPOUND_TAIL))) {
- if (cur_order < nr_orders) {
- orders[cur_order]++;
- cur_order = -1;
- vaddr = next_folio_vaddr;
- }
- break;
- }
-
- cur_order++;
- next_folio_vaddr = vaddr + (1UL << (cur_order + pshift()));
- }
-
- if (status < 0)
- return status;
- }
- if (cur_order > 0 && cur_order < nr_orders)
- orders[cur_order]++;
- return 0;
-}
-
static int check_after_split_folio_orders(char *vaddr_start, size_t len,
int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders)
{
@@ -240,7 +117,7 @@ static int check_after_split_folio_orders(char *vaddr_start, size_t len,
ksft_exit_fail_msg("Cannot allocate memory for vaddr_orders");
memset(vaddr_orders, 0, sizeof(int) * nr_orders);
- status = gather_after_split_folio_orders(vaddr_start, len, pagemap_fd,
+ status = gather_folio_orders(vaddr_start, len, pagemap_fd,
kpageflags_fd, vaddr_orders, nr_orders);
if (status)
ksft_exit_fail_msg("gather folio info failed\n");
@@ -296,7 +173,7 @@ static void verify_rss_anon_split_huge_page_all_zeroes(char *one_page, int nr_hp
unsigned long rss_anon_before, rss_anon_after;
size_t i;
- if (!check_huge_anon(one_page, nr_hpages, pmd_pagesize))
+ if (!check_huge_anon(one_page, nr_hpages * pmd_pagesize, nr_hpages, pmd_pagesize))
ksft_exit_fail_msg("No THP is allocated\n");
rss_anon_before = rss_anon();
@@ -311,7 +188,7 @@ static void verify_rss_anon_split_huge_page_all_zeroes(char *one_page, int nr_hp
if (one_page[i] != (char)0)
ksft_exit_fail_msg("%ld byte corrupted\n", i);
- if (!check_huge_anon(one_page, 0, pmd_pagesize))
+ if (!check_huge_anon(one_page, nr_hpages * pmd_pagesize, 0, pmd_pagesize))
ksft_exit_fail_msg("Still AnonHugePages not split\n");
rss_anon_after = rss_anon();
@@ -347,7 +224,7 @@ static void split_pmd_thp_to_order(int order)
for (i = 0; i < len; i++)
one_page[i] = (char)i;
- if (!check_huge_anon(one_page, 4, pmd_pagesize))
+ if (!check_huge_anon(one_page, 4 * pmd_pagesize, 4, pmd_pagesize))
ksft_exit_fail_msg("No THP is allocated\n");
/* split all THPs */
@@ -366,7 +243,7 @@ static void split_pmd_thp_to_order(int order)
(pmd_order + 1)))
ksft_exit_fail_msg("Unexpected THP split\n");
- if (!check_huge_anon(one_page, 0, pmd_pagesize))
+ if (!check_huge_anon(one_page, 4 * pmd_pagesize, 0, pmd_pagesize))
ksft_exit_fail_msg("Still AnonHugePages not split\n");
ksft_test_result_pass("Split huge pages to order %d successful\n", order);
@@ -393,7 +270,7 @@ static void split_pte_mapped_thp(void)
for (i = 0; i < thp_area_size; i++)
thp_area[i] = (char)i;
- if (!check_huge_anon(thp_area, nr_thps, pmd_pagesize)) {
+ if (!check_huge_anon(thp_area, nr_thps * pmd_pagesize, nr_thps, pmd_pagesize)) {
ksft_test_result_skip("Not all THPs allocated\n");
goto out;
}
@@ -657,7 +534,7 @@ static int create_pagecache_thp_and_fd(const char *testfile, size_t fd_size,
force_read_pages(*addr, fd_size / pmd_pagesize, pmd_pagesize);
- if (!check_huge_file(*addr, fd_size / pmd_pagesize, pmd_pagesize)) {
+ if (!check_huge_file(*addr, fd_size, fd_size / pmd_pagesize, pmd_pagesize)) {
ksft_print_msg("No large pagecache folio generated, please provide a filesystem supporting large folio\n");
munmap(*addr, fd_size);
close(*fd);
@@ -735,7 +612,7 @@ static void split_thp_in_pagecache_to_order_at(size_t fd_size,
goto out;
}
- if (!check_huge_file(addr, 0, pmd_pagesize)) {
+ if (!check_huge_file(addr, fd_size, 0, pmd_pagesize)) {
ksft_print_msg("Still FilePmdMapped not split\n");
err = EXIT_FAILURE;
goto out;
diff --git a/tools/testing/selftests/mm/thuge-gen.c b/tools/testing/selftests/mm/thuge-gen.c
index 22b9c2f1c35d..50d0805b65db 100644
--- a/tools/testing/selftests/mm/thuge-gen.c
+++ b/tools/testing/selftests/mm/thuge-gen.c
@@ -71,12 +71,16 @@ void test_mmap(unsigned long size, unsigned flags)
void test_shmget(unsigned long size, unsigned flags)
{
- int id;
- unsigned long before, after;
+ /* values for PAGE_SIZE test */
+ unsigned long before = NUM_PAGES;
+ unsigned long after = 0;
struct shm_info i;
char *map;
+ int id;
+
+ if (size != getpagesize())
+ before = hugetlb_free_pages(size);
- before = hugetlb_free_pages(size);
id = shmget(IPC_PRIVATE, size * NUM_PAGES, IPC_CREAT|0600|flags);
if (id < 0) {
if (errno == EPERM) {
@@ -97,10 +101,11 @@ void test_shmget(unsigned long size, unsigned flags)
shmctl(id, IPC_RMID, NULL);
memset(map, 0xff, size*NUM_PAGES);
- after = hugetlb_free_pages(size);
+ if (size != getpagesize())
+ after = hugetlb_free_pages(size);
show(size);
- ksft_test_result(size == getpagesize() || (before - after) == NUM_PAGES,
+ ksft_test_result((before - after) == NUM_PAGES,
"%s: mmap %lu %x\n", __func__, size, flags);
if (shmdt(map))
ksft_exit_fail_msg("%s: shmdt: %s\n", __func__, strerror(errno));
diff --git a/tools/testing/selftests/mm/uffd-common.c b/tools/testing/selftests/mm/uffd-common.c
index edd02328f77b..1fb967ef4985 100644
--- a/tools/testing/selftests/mm/uffd-common.c
+++ b/tools/testing/selftests/mm/uffd-common.c
@@ -194,7 +194,9 @@ static void shmem_alias_mapping(uffd_global_test_opts_t *gopts, __u64 *start,
static void shmem_check_pmd_mapping(uffd_global_test_opts_t *gopts, void *p, int expect_nr_hpages)
{
- if (!check_huge_shmem(gopts->area_dst_alias, expect_nr_hpages,
+ size_t len = expect_nr_hpages * read_pmd_pagesize();
+
+ if (!check_huge_shmem(gopts->area_dst_alias, len, expect_nr_hpages,
read_pmd_pagesize()))
err("Did not find expected %d number of hugepages",
expect_nr_hpages);
@@ -639,8 +641,13 @@ int __copy_page(uffd_global_test_opts_t *gopts, unsigned long offset, bool retry
uffdio_copy.mode = 0;
uffdio_copy.copy = 0;
if (ioctl(gopts->uffd, UFFDIO_COPY, &uffdio_copy)) {
- /* real retval in ufdio_copy.copy */
- if (uffdio_copy.copy != -EEXIST)
+ /*
+ * real retval in uffdio_copy.copy
+ *
+ * -EEXIST: the page was faulted in concurrently
+ * -ENOENT: the destination range was concurrently removed
+ */
+ if (uffdio_copy.copy != -EEXIST && uffdio_copy.copy != -ENOENT)
err("UFFDIO_COPY error: %"PRId64,
(int64_t)uffdio_copy.copy);
wake_range(gopts->uffd, uffdio_copy.dst, gopts->page_size);
diff --git a/tools/testing/selftests/mm/uffd-unit-tests.c b/tools/testing/selftests/mm/uffd-unit-tests.c
index a6c14109e818..ef9b3956bdcf 100644
--- a/tools/testing/selftests/mm/uffd-unit-tests.c
+++ b/tools/testing/selftests/mm/uffd-unit-tests.c
@@ -7,6 +7,8 @@
#include "uffd-common.h"
+#include <linux/fs.h>
+#include <sys/uio.h>
#include "../../../../mm/gup_test.h"
#ifdef __NR_userfaultfd
@@ -109,6 +111,10 @@ static void uffd_test_skip(const char *message)
static void test_uffd_api(bool use_dev)
{
+ const uint64_t expected_ioctls =
+ BIT_ULL(_UFFDIO_REGISTER) |
+ BIT_ULL(_UFFDIO_UNREGISTER) |
+ BIT_ULL(_UFFDIO_API);
struct uffdio_api uffdio_api;
int uffd;
@@ -148,6 +154,15 @@ static void test_uffd_api(bool use_dev)
goto out;
}
+ /* Verify returned fd-level ioctls bitmask */
+ if ((uffdio_api.ioctls & expected_ioctls) != expected_ioctls) {
+ uffd_test_fail("UFFDIO_API missing expected ioctls: "
+ "got=0x%"PRIx64", expected=0x%"PRIx64,
+ (uint64_t)uffdio_api.ioctls,
+ expected_ioctls);
+ goto out;
+ }
+
/* Test double requests of UFFDIO_API with a random feature set */
uffdio_api.features = BIT_ULL(0);
if (ioctl(uffd, UFFDIO_API, &uffdio_api) == 0) {
@@ -602,6 +617,814 @@ void uffd_minor_collapse_test(uffd_global_test_opts_t *gopts, uffd_test_args_t *
uffd_minor_test_common(gopts, true, false);
}
+static int uffd_register_rwp(int uffd, void *addr, uint64_t len)
+{
+ struct uffdio_register reg = {
+ .range = { .start = (unsigned long)addr, .len = len },
+ .mode = UFFDIO_REGISTER_MODE_RWP,
+ };
+
+ if (ioctl(uffd, UFFDIO_REGISTER, &reg) == -1)
+ return -errno;
+ return 0;
+}
+
+static void rwprotect_range(int uffd, __u64 start, __u64 len, bool protect)
+{
+ struct uffdio_rwprotect rwp = {
+ .range = { .start = start, .len = len },
+ .mode = protect ? UFFDIO_RWPROTECT_MODE_RWP : 0,
+ };
+
+ if (ioctl(uffd, UFFDIO_RWPROTECT, &rwp))
+ err("UFFDIO_RWPROTECT failed");
+}
+
+static void set_async_mode(int uffd, bool enable)
+{
+ struct uffdio_set_mode mode = { };
+
+ if (enable)
+ mode.enable = UFFD_FEATURE_RWP_ASYNC;
+ else
+ mode.disable = UFFD_FEATURE_RWP_ASYNC;
+
+ if (ioctl(uffd, UFFDIO_SET_MODE, &mode))
+ err("UFFDIO_SET_MODE failed");
+}
+
+/*
+ * Test async RWP faults on anonymous memory.
+ * Populate pages, register MODE_RWP with RWP_ASYNC,
+ * RW-protect, re-access, verify content preserved and no faults delivered.
+ */
+static void uffd_rwp_async_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ unsigned long p;
+
+ /* Populate all pages with known content */
+ for (p = 0; p < nr_pages; p++)
+ memset(gopts->area_dst + p * page_size, p % 255 + 1, page_size);
+
+ /* Register MODE_RWP */
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failure");
+
+ /* RW-protect all pages (sets protnone) */
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ /* Access all pages — should auto-resolve, no faults */
+ for (p = 0; p < nr_pages; p++) {
+ unsigned char *page = (unsigned char *)gopts->area_dst +
+ p * page_size;
+ unsigned char expected = p % 255 + 1;
+
+ if (page[0] != expected) {
+ uffd_test_fail("page %lu content mismatch: %u != %u",
+ p, page[0], expected);
+ return;
+ }
+ }
+
+ uffd_test_pass();
+}
+
+/*
+ * Fault handler for RWP — unprotect the page via UFFDIO_RWPROTECT.
+ */
+static void uffd_handle_rwp_fault(uffd_global_test_opts_t *gopts,
+ struct uffd_msg *msg,
+ struct uffd_args *uargs)
+{
+ if (!(msg->arg.pagefault.flags & UFFD_PAGEFAULT_FLAG_RWP))
+ err("expected RWP fault, got 0x%llx",
+ msg->arg.pagefault.flags);
+
+ rwprotect_range(gopts->uffd, msg->arg.pagefault.address,
+ gopts->page_size, false);
+ uargs->minor_faults++;
+}
+
+/*
+ * Test sync RWP faults on anonymous memory.
+ * Populate pages, register MODE_RWP (sync), RW-protect,
+ * access from worker thread, verify fault delivered, UFFDIO_RWPROTECT resolves.
+ */
+static void uffd_rwp_sync_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ pthread_t uffd_mon;
+ struct uffd_args uargs = { };
+ bool failed = false;
+ char c = '\0';
+ unsigned long p;
+
+ uargs.gopts = gopts;
+ uargs.handle_fault = uffd_handle_rwp_fault;
+
+ /* Populate all pages */
+ for (p = 0; p < nr_pages; p++)
+ memset(gopts->area_dst + p * page_size, p % 255 + 1, page_size);
+
+ /* Register MODE_RWP */
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failure");
+
+ /* RW-protect all pages */
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ /* Start fault handler thread */
+ if (pthread_create(&uffd_mon, NULL, uffd_poll_thread, &uargs))
+ err("uffd_poll_thread create");
+
+ /* Access all pages — triggers sync RWP faults, handler unprotects */
+ for (p = 0; p < nr_pages; p++) {
+ unsigned char *page = (unsigned char *)gopts->area_dst +
+ p * page_size;
+
+ if (page[0] != (p % 255 + 1)) {
+ uffd_test_fail("page %lu content mismatch", p);
+ failed = true;
+ goto out;
+ }
+ }
+
+out:
+ /*
+ * Stop the handler before reading minor_faults: the last fault
+ * resolution rwprotect_range()s before incrementing the counter,
+ * so the main thread can race ahead of the increment.
+ */
+ if (write(gopts->pipefd[1], &c, sizeof(c)) != sizeof(c))
+ err("pipe write");
+ if (pthread_join(uffd_mon, NULL))
+ err("join() failed");
+
+ if (failed)
+ return;
+ if (uargs.minor_faults == 0)
+ uffd_test_fail("expected RWP faults, got 0");
+ else
+ uffd_test_pass();
+}
+
+/*
+ * Test PAGEMAP_SCAN working-set discovery via the "hot" (accessed) scan.
+ *
+ * The working-set primitive is to find pages that were accessed: scan for
+ * PAGE_IS_ACCESSED, which is set once an access clears the protnone+uffd
+ * marker. A VMM treats every access fault as "hot" (RWP here; MINOR/MISSING
+ * for non-resident pages) and reclaims the rest from the backing file.
+ *
+ * We deliberately do NOT use an inverted "cold" scan: that only sees
+ * VMA-resident ptes, so for a file mapping it misses cached-but-unmapped (and
+ * never-faulted, pre-populated) pages, which are pte_none and thus invisible.
+ * Hot tracking + file-level reclaim covers them; a cold pte scan cannot.
+ */
+static void uffd_rwp_pagemap_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ unsigned long p;
+ struct page_region regions[16];
+ struct pm_scan_arg pm_arg;
+ int pagemap_fd;
+ long ret;
+
+ /* Need at least 4 pages */
+ if (nr_pages < 4) {
+ uffd_test_skip("need at least 4 pages");
+ return;
+ }
+
+ /* Populate all pages */
+ for (p = 0; p < nr_pages; p++)
+ memset(gopts->area_dst + p * page_size, 0xab, page_size);
+
+ /* Register and RW-protect */
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failure");
+
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ /* Touch first half of pages to re-activate them (async auto-resolve) */
+ for (p = 0; p < nr_pages / 2; p++) {
+ volatile char *page = gopts->area_dst + p * page_size;
+ (void)*page;
+ }
+
+ uint64_t start = (uint64_t)gopts->area_dst;
+ uint64_t boundary = start + (nr_pages / 2) * page_size;
+ uint64_t end = start + nr_pages * page_size;
+
+ pagemap_fd = open("/proc/self/pagemap", O_RDONLY);
+ if (pagemap_fd < 0)
+ err("open pagemap");
+
+ /*
+ * Hot scan: report the pages that were accessed. PAGE_IS_ACCESSED is
+ * set once the protnone+uffd marker is cleared (by the access, async
+ * auto-resolve here). The touched first half must come back as exactly
+ * one hot region [start, boundary); the untouched second half must not
+ * appear.
+ */
+ memset(&pm_arg, 0, sizeof(pm_arg));
+ pm_arg.size = sizeof(pm_arg);
+ pm_arg.start = start;
+ pm_arg.end = end;
+ pm_arg.vec = (uint64_t)regions;
+ pm_arg.vec_len = ARRAY_SIZE(regions);
+ pm_arg.category_mask = PAGE_IS_ACCESSED;
+ pm_arg.return_mask = PAGE_IS_ACCESSED;
+
+ ret = ioctl(pagemap_fd, PAGEMAP_SCAN, &pm_arg);
+ close(pagemap_fd);
+
+ if (ret < 0) {
+ uffd_test_fail("PAGEMAP_SCAN failed: %s", strerror(errno));
+ return;
+ }
+
+ if (ret != 1 || regions[0].start != start ||
+ regions[0].end != boundary) {
+ uffd_test_fail("hot set wrong: got %ld regions [0x%lx,0x%lx), expected 1 [0x%lx,0x%lx)",
+ ret, (unsigned long)regions[0].start,
+ (unsigned long)regions[0].end,
+ (unsigned long)start, (unsigned long)boundary);
+ return;
+ }
+
+ uffd_test_pass();
+}
+
+/*
+ * Test that RWP protection survives a mprotect(PROT_NONE) ->
+ * mprotect(PROT_READ|PROT_WRITE) round-trip. The uffd-wp bit on a
+ * VM_UFFD_RWP VMA must continue to carry PROT_NONE semantics after
+ * mprotect() changes the base protection; otherwise accesses would
+ * silently succeed and the pagemap bit would stick without a fault
+ * ever clearing it.
+ */
+static void uffd_rwp_mprotect_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ unsigned long p;
+ struct page_region regions[16];
+ struct pm_scan_arg pm_arg;
+ int pagemap_fd;
+ uint64_t value;
+ long ret;
+
+ /* Populate all pages */
+ for (p = 0; p < nr_pages; p++)
+ memset(gopts->area_dst + p * page_size, 0xab, page_size);
+
+ /* Register and RW-protect the whole range */
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failure");
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ /* Round-trip mprotect(): PROT_NONE -> PROT_READ|PROT_WRITE */
+ if (mprotect(gopts->area_dst, nr_pages * page_size, PROT_NONE))
+ err("mprotect() PROT_NONE");
+ if (mprotect(gopts->area_dst, nr_pages * page_size,
+ PROT_READ | PROT_WRITE))
+ err("mprotect() PROT_READ|PROT_WRITE");
+
+ /*
+ * The marker must survive the round-trip; if mprotect() dropped it,
+ * the touches below would not fault and the scan would pass
+ * vacuously.
+ */
+ pagemap_fd = pagemap_open();
+ value = pagemap_get_entry(pagemap_fd, gopts->area_dst);
+ close(pagemap_fd);
+ if (!(value & PM_UFFD_WP)) {
+ uffd_test_fail("RWP marker lost across mprotect()");
+ return;
+ }
+
+ /* Touch every page. Async RWP must auto-resolve each fault. */
+ for (p = 0; p < nr_pages; p++) {
+ volatile char *page = gopts->area_dst + p * page_size;
+ (void)*page;
+ }
+
+ /*
+ * After touching, no page should remain RW-protected. A stuck
+ * uffd-wp bit would mean mprotect() silently dropped PROT_NONE and
+ * the access never faulted.
+ */
+ pagemap_fd = open("/proc/self/pagemap", O_RDONLY);
+ if (pagemap_fd < 0)
+ err("open pagemap");
+
+ memset(&pm_arg, 0, sizeof(pm_arg));
+ pm_arg.size = sizeof(pm_arg);
+ pm_arg.start = (uint64_t)gopts->area_dst;
+ pm_arg.end = (uint64_t)gopts->area_dst + nr_pages * page_size;
+ pm_arg.vec = (uint64_t)regions;
+ pm_arg.vec_len = ARRAY_SIZE(regions);
+ pm_arg.category_mask = PAGE_IS_ACCESSED;
+ pm_arg.category_inverted = PAGE_IS_ACCESSED;
+ pm_arg.return_mask = PAGE_IS_ACCESSED;
+
+ ret = ioctl(pagemap_fd, PAGEMAP_SCAN, &pm_arg);
+ close(pagemap_fd);
+
+ if (ret < 0) {
+ uffd_test_fail("PAGEMAP_SCAN failed: %s", strerror(errno));
+ return;
+ }
+ if (ret != 0) {
+ uffd_test_fail("expected no cold pages after mprotect()+touch, got %ld regions",
+ ret);
+ return;
+ }
+
+ uffd_test_pass();
+}
+
+/*
+ * Test that GUP resolves through protnone PTEs (async mode).
+ * vmsplice() into a pipe pins user pages via get_user_pages_fast() --
+ * unlike write(), which goes through copy_from_user() and ordinary
+ * hardware page faults -- so it exercises gup_can_follow_protnone() on
+ * the RW-protected PTE. In async mode the kernel auto-restores
+ * permissions and GUP returns the page.
+ */
+static void uffd_rwp_gup_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ struct iovec iov;
+ char buf;
+ int pipefd[2];
+
+ /* Populate first page with known content */
+ memset(gopts->area_dst, 0xCD, gopts->page_size);
+
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst, gopts->page_size))
+ err("register failure");
+
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ gopts->page_size, true);
+
+ if (pipe(pipefd))
+ err("pipe");
+
+ /*
+ * One byte's worth of iov is enough to GUP the containing page and
+ * keeps the pipe transfer well under any pipe-capacity limit even on
+ * hugetlb-backed runs.
+ */
+ iov.iov_base = gopts->area_dst;
+ iov.iov_len = 1;
+ if (vmsplice(pipefd[1], &iov, 1, 0) != 1) {
+ uffd_test_fail("vmsplice from RW-protected page failed: %s",
+ strerror(errno));
+ goto out;
+ }
+
+ if (read(pipefd[0], &buf, 1) != 1) {
+ uffd_test_fail("read from pipe failed");
+ goto out;
+ }
+
+ if (buf != (char)0xCD) {
+ uffd_test_fail("content mismatch: got 0x%02x, expected 0xCD",
+ (unsigned char)buf);
+ goto out;
+ }
+
+ uffd_test_pass();
+out:
+ close(pipefd[0]);
+ close(pipefd[1]);
+}
+
+/*
+ * Test runtime toggle between async and sync modes.
+ * Start in async mode (detection), flip to sync (eviction), verify faults
+ * block, resolve them, flip back to async.
+ */
+static void uffd_rwp_async_toggle_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ struct uffd_args uargs = { };
+ pthread_t uffd_mon;
+ char c = '\0';
+ unsigned long p;
+
+ uargs.gopts = gopts;
+ uargs.handle_fault = uffd_handle_rwp_fault;
+
+ /* Populate */
+ for (p = 0; p < nr_pages; p++)
+ memset(gopts->area_dst + p * page_size, p % 255 + 1, page_size);
+
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failure");
+
+ /* Phase 1: async detection — RW-protect, access first half */
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ for (p = 0; p < nr_pages / 2; p++) {
+ volatile char *page = gopts->area_dst + p * page_size;
+ (void)*page; /* auto-resolves in async mode */
+ }
+
+ /* Phase 2: flip to sync for eviction */
+ set_async_mode(gopts->uffd, false);
+
+ /* Start handler — will receive faults for cold pages */
+ if (pthread_create(&uffd_mon, NULL, uffd_poll_thread, &uargs))
+ err("uffd_poll_thread create");
+
+ /* Access second half (cold pages) — should trigger sync faults */
+ for (p = nr_pages / 2; p < nr_pages; p++) {
+ unsigned char *page = (unsigned char *)gopts->area_dst +
+ p * page_size;
+ if (page[0] != (p % 255 + 1)) {
+ uffd_test_fail("page %lu content mismatch", p);
+ goto out;
+ }
+ }
+
+ /*
+ * Stop the handler before reading minor_faults: the last fault
+ * resolution rwprotect_range()s before incrementing the counter,
+ * so the main thread can race ahead of the increment. Stopping
+ * here also makes Phase 3 a clean async-only test -- with the
+ * handler still running it would silently resolve any sync fault
+ * the kernel erroneously delivers, masking a regression.
+ */
+ if (write(gopts->pipefd[1], &c, sizeof(c)) != sizeof(c))
+ err("pipe write");
+ if (pthread_join(uffd_mon, NULL))
+ err("join() failed");
+
+ if (uargs.minor_faults == 0) {
+ uffd_test_fail("expected sync faults, got 0");
+ return;
+ }
+
+ /* Phase 3: flip back to async */
+ set_async_mode(gopts->uffd, true);
+
+ /* RW-protect and access again — should auto-resolve */
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ for (p = 0; p < nr_pages; p++) {
+ volatile char *page = gopts->area_dst + p * page_size;
+ (void)*page;
+ }
+
+ uffd_test_pass();
+ return;
+out:
+ if (write(gopts->pipefd[1], &c, sizeof(c)) != sizeof(c))
+ err("pipe write");
+ if (pthread_join(uffd_mon, NULL))
+ err("join() failed");
+}
+
+/*
+ * Test that RW-protected pages become accessible after closing uffd.
+ */
+static void uffd_rwp_close_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ unsigned long p;
+
+ /* Populate */
+ for (p = 0; p < nr_pages; p++)
+ memset(gopts->area_dst + p * page_size, p % 255 + 1, page_size);
+
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failure");
+
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ nr_pages * page_size, true);
+
+ /* Close uffd — should restore protnone PTEs */
+ close(gopts->uffd);
+ gopts->uffd = -1;
+
+ /* All pages should be accessible with original content */
+ for (p = 0; p < nr_pages; p++) {
+ unsigned char *page = (unsigned char *)gopts->area_dst +
+ p * page_size;
+ unsigned char expected = p % 255 + 1;
+
+ if (page[0] != expected) {
+ uffd_test_fail("page %lu not accessible after close", p);
+ return;
+ }
+ }
+
+ uffd_test_pass();
+}
+
+/*
+ * Test that RWP protection is preserved across fork() when
+ * UFFD_FEATURE_EVENT_FORK is enabled. Without preservation, the child's
+ * PTEs would lose the uffd-wp marker and RWP-protected accesses would
+ * silently fall through to do_numa_page().
+ */
+static void uffd_rwp_fork_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ int pagemap_fd;
+ uint64_t value;
+
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst,
+ nr_pages * page_size))
+ err("register failed");
+
+ /* Populate + RWP-protect */
+ *gopts->area_dst = 1;
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst,
+ page_size, true);
+
+ /* Parent: verify uffd-wp bit is set before fork */
+ pagemap_fd = pagemap_open();
+ value = pagemap_get_entry(pagemap_fd, gopts->area_dst);
+ pagemap_check_wp(value, true);
+
+ /*
+ * Fork with EVENT_FORK: child inherits VM_UFFD_RWP. Child reads
+ * its own pagemap and must still see the uffd-wp bit set.
+ */
+ if (pagemap_test_fork(gopts, true, false)) {
+ uffd_test_fail("RWP marker lost in child after fork");
+ goto out;
+ }
+
+ uffd_test_pass();
+out:
+ close(pagemap_fd);
+}
+
+/*
+ * Test that RWP protection on a pinned anon page is preserved across fork().
+ * Pinning forces copy_present_page() in the child path, which must restore
+ * PAGE_NONE on top of the uffd bit. Using async mode, a read in the child
+ * auto-resolves if — and only if — the PTE was actually protnone+uffd; the
+ * cleared uffd bit afterward proves the fault path ran.
+ */
+static void uffd_rwp_fork_pin_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long page_size = gopts->page_size;
+ fork_event_args fevent_args = { .gopts = gopts, .child_uffd = -1 };
+ pin_args pin_args = {};
+ int pagemap_fd, status;
+ pthread_t fevent_thread;
+ uint64_t value;
+ pid_t child;
+
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst, page_size))
+ err("register failed");
+
+ /* Populate. */
+ *gopts->area_dst = 1;
+
+ /* RO-longterm pin so fork() takes copy_present_page() for this PTE. */
+ if (pin_pages(&pin_args, gopts->area_dst, page_size)) {
+ uffd_test_skip("Possibly CONFIG_GUP_TEST missing or unprivileged");
+ uffd_unregister(gopts->uffd, gopts->area_dst, page_size);
+ return;
+ }
+
+ /* RWP-protect: PTE is now PAGE_NONE + uffd bit. */
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst, page_size, true);
+
+ pagemap_fd = pagemap_open();
+ value = pagemap_get_entry(pagemap_fd, gopts->area_dst);
+ pagemap_check_wp(value, true);
+
+ /*
+ * UFFD_FEATURE_EVENT_FORK is required so the child inherits
+ * VM_UFFD_RWP and the marker; without it dup_userfaultfd() resets
+ * the child VMA and the test would pass for the wrong reason.
+ * dup_userfaultfd() blocks until the EVENT_FORK message is consumed,
+ * so spawn a reader before the fork().
+ */
+ gopts->ready_for_fork = false;
+ if (pthread_create(&fevent_thread, NULL, fork_event_consumer,
+ &fevent_args))
+ err("pthread_create() for fork event consumer");
+ while (!gopts->ready_for_fork)
+ ; /* Wait for consumer to start polling. */
+
+ child = fork();
+ if (child < 0)
+ err("fork");
+ if (child == 0) {
+ volatile char c;
+ int cfd;
+
+ /*
+ * Precondition: the child must have inherited the marker.
+ * If copy_present_page() dropped it together with PAGE_NONE,
+ * the read below would succeed without the fault path and
+ * the after-read check would pass for the wrong reason.
+ */
+ cfd = pagemap_open();
+ value = pagemap_get_entry(cfd, gopts->area_dst);
+ if (!(value & PM_UFFD_WP)) {
+ close(cfd);
+ _exit(2);
+ }
+
+ /*
+ * Read the pinned page. Only reaches the fault path if the
+ * child PTE is protnone + uffd; async mode auto-resolves and
+ * clears the uffd bit. If copy_present_page() dropped
+ * PAGE_NONE, the read would silently succeed and the bit
+ * would still be set.
+ */
+ c = *(volatile char *)gopts->area_dst;
+ (void)c;
+
+ value = pagemap_get_entry(cfd, gopts->area_dst);
+ close(cfd);
+ _exit((value & PM_UFFD_WP) ? 1 : 0);
+ }
+ if (waitpid(child, &status, 0) < 0)
+ err("waitpid");
+ if (pthread_join(fevent_thread, NULL))
+ err("pthread_join() for fork event consumer");
+ if (fevent_args.child_uffd >= 0)
+ close(fevent_args.child_uffd);
+
+ unpin_pages(&pin_args);
+ close(pagemap_fd);
+ if (uffd_unregister(gopts->uffd, gopts->area_dst, page_size))
+ err("unregister failed");
+
+ if (WIFEXITED(status) && WEXITSTATUS(status) == 2) {
+ uffd_test_fail("RWP marker not inherited by child");
+ return;
+ }
+ if (!WIFEXITED(status) || WEXITSTATUS(status) != 0) {
+ uffd_test_fail("RWP not enforced in child after pinned fork");
+ return;
+ }
+
+ uffd_test_pass();
+}
+
+/*
+ * A non-exclusive (forked, COW-shared) anon page that is RWP-protected and
+ * then swapped out must keep tracking across swap-in. On the write that
+ * swaps it back in, do_swap_page() restores PAGE_NONE and the access retries
+ * through the RWP fault path, instead of being COWed straight to a fresh
+ * accessible page -- which would silently drop the marker for a non-exclusive
+ * folio. Sync mode lets us observe the fault directly: with the bug, the
+ * write COWs without delivering any RWP fault.
+ *
+ * Needs a swap device; skipped if MADV_PAGEOUT cannot evict the page.
+ */
+static void uffd_rwp_swap_cow_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long page_size = gopts->page_size;
+ struct uffd_args uargs = { };
+ int pagemap_fd, go[2], i;
+ pthread_t uffd_mon;
+ char c = '\0';
+ pid_t child;
+
+ uargs.gopts = gopts;
+ uargs.handle_fault = uffd_handle_rwp_fault;
+
+ if (uffd_register_rwp(gopts->uffd, gopts->area_dst, page_size))
+ err("register failed");
+
+ /* Populate one page (exclusive at this point). */
+ *gopts->area_dst = 0x11;
+
+ /* RWP-protect: PTE becomes PAGE_NONE + uffd bit (still exclusive). */
+ rwprotect_range(gopts->uffd, (uint64_t)gopts->area_dst, page_size, true);
+
+ /*
+ * Swap the page out while it is still exclusive: a shared (forked)
+ * folio does not get reclaimed by MADV_PAGEOUT. Retry, since a hot
+ * page may just be rotated on the first reclaim pass.
+ */
+ pagemap_fd = pagemap_open();
+ for (i = 0; i < 100; i++) {
+ if (madvise(gopts->area_dst, page_size, MADV_PAGEOUT))
+ err("MADV_PAGEOUT");
+ if (pagemap_is_swapped(pagemap_fd, gopts->area_dst))
+ break;
+ usleep(10000);
+ }
+ if (!pagemap_is_swapped(pagemap_fd, gopts->area_dst)) {
+ uffd_test_skip("MADV_PAGEOUT did not swap the page; is swap enabled?");
+ close(pagemap_fd);
+ uffd_unregister(gopts->uffd, gopts->area_dst, page_size);
+ return;
+ }
+
+ /*
+ * fork() now: the child duplicates the swap entry, so the slot becomes
+ * non-exclusive. The child parks (keeping the reference) until the
+ * parent has faulted the page back in.
+ */
+ if (pipe(go))
+ err("pipe");
+ child = fork();
+ if (child < 0)
+ err("fork");
+ if (child == 0) {
+ close(go[1]);
+ read(go[0], &c, 1);
+ _exit(0);
+ }
+ close(go[0]);
+
+ if (pthread_create(&uffd_mon, NULL, uffd_poll_thread, &uargs))
+ err("uffd_poll_thread create");
+
+ /*
+ * Write the page: swaps it back in (do_swap_page) on a non-exclusive
+ * folio with FAULT_FLAG_WRITE. The marker must survive and deliver an
+ * RWP fault rather than COW silently.
+ */
+ *gopts->area_dst = 0x22;
+
+ if (write(gopts->pipefd[1], &c, sizeof(c)) != sizeof(c))
+ err("pipe write");
+ if (pthread_join(uffd_mon, NULL))
+ err("join failed");
+
+ if (uargs.minor_faults == 0)
+ uffd_test_fail("no RWP fault on swapped-in non-exclusive page");
+ else
+ uffd_test_pass();
+
+ close(pagemap_fd);
+ if (write(go[1], &c, 1) != 1)
+ err("child release");
+ close(go[1]);
+ waitpid(child, NULL, 0);
+}
+
+/*
+ * WP and RWP share the uffd-wp PTE bit and cannot coexist in the same VMA.
+ * Registration requesting both modes must be rejected.
+ */
+static void uffd_rwp_wp_exclusive_test(uffd_global_test_opts_t *gopts,
+ uffd_test_args_t *args)
+{
+ unsigned long nr_pages = gopts->nr_pages;
+ unsigned long page_size = gopts->page_size;
+ struct uffdio_register reg = { };
+
+ reg.range.start = (unsigned long)gopts->area_dst;
+ reg.range.len = nr_pages * page_size;
+ reg.mode = UFFDIO_REGISTER_MODE_WP | UFFDIO_REGISTER_MODE_RWP;
+
+ if (ioctl(gopts->uffd, UFFDIO_REGISTER, &reg) == 0) {
+ uffd_test_fail("register with WP|RWP unexpectedly succeeded");
+ return;
+ }
+ if (errno != EINVAL) {
+ uffd_test_fail("register with WP|RWP: expected EINVAL, got %d",
+ errno);
+ return;
+ }
+ uffd_test_pass();
+}
+
static sigjmp_buf jbuf, *sigbuf;
static void sighndl(int sig, siginfo_t *siginfo, void *ptr)
@@ -1605,6 +2428,83 @@ uffd_test_case_t uffd_tests[] = {
.uffd_feature_required = UFFD_FEATURE_MINOR_SHMEM,
},
{
+ .name = "rwp-async",
+ .uffd_fn = uffd_rwp_async_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_RWP_ASYNC,
+ },
+ {
+ .name = "rwp-sync",
+ .uffd_fn = uffd_rwp_sync_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required = UFFD_FEATURE_RWP,
+ },
+ {
+ .name = "rwp-pagemap",
+ .uffd_fn = uffd_rwp_pagemap_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_RWP_ASYNC,
+ },
+ {
+ .name = "rwp-mprotect",
+ .uffd_fn = uffd_rwp_mprotect_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_RWP_ASYNC,
+ },
+ {
+ .name = "rwp-gup",
+ .uffd_fn = uffd_rwp_gup_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_RWP_ASYNC,
+ },
+ {
+ .name = "rwp-async-toggle",
+ .uffd_fn = uffd_rwp_async_toggle_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_RWP_ASYNC,
+ },
+ {
+ .name = "rwp-close",
+ .uffd_fn = uffd_rwp_close_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required = UFFD_FEATURE_RWP,
+ },
+ {
+ .name = "rwp-fork",
+ .uffd_fn = uffd_rwp_fork_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_EVENT_FORK,
+ },
+ {
+ .name = "rwp-fork-pin",
+ .uffd_fn = uffd_rwp_fork_pin_test,
+ .mem_targets = MEM_ANON,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP | UFFD_FEATURE_RWP_ASYNC |
+ UFFD_FEATURE_EVENT_FORK,
+ },
+ {
+ .name = "rwp-swap-cow",
+ .uffd_fn = uffd_rwp_swap_cow_test,
+ .mem_targets = MEM_ANON,
+ .uffd_feature_required = UFFD_FEATURE_RWP,
+ },
+ {
+ .name = "rwp-wp-exclusive",
+ .uffd_fn = uffd_rwp_wp_exclusive_test,
+ .mem_targets = MEM_ALL,
+ .uffd_feature_required =
+ UFFD_FEATURE_RWP |
+ UFFD_FEATURE_PAGEFAULT_FLAG_WP |
+ UFFD_FEATURE_WP_HUGETLBFS_SHMEM,
+ },
+ {
.name = "sigbus",
.uffd_fn = uffd_sigbus_test,
.mem_targets = MEM_ALL,
@@ -1814,6 +2714,25 @@ int main(int argc, char *argv[])
uffd_test_skip(errmsg);
continue;
}
+ /*
+ * RWP tracks protection on ptes; a THP-backed shmem/anon
+ * range (e.g. shmem_enabled=always) would split on
+ * rwprotect and change behaviour under the test. Keep
+ * such ranges off THP. hugetlb is huge by definition and
+ * rejects MADV_NOHUGEPAGE, so skip it.
+ */
+ if ((test->uffd_feature_required & UFFD_FEATURE_RWP) &&
+ !(mem_type->mem_flag & (MEM_HUGETLB | MEM_HUGETLB_PRIVATE))) {
+ unsigned long len = gopts.nr_pages * gopts.page_size;
+
+ /*
+ * EINVAL means CONFIG_TRANSPARENT_HUGEPAGE=n:
+ * nothing to opt out of.
+ */
+ if (madvise(gopts.area_dst, len, MADV_NOHUGEPAGE) &&
+ errno != EINVAL)
+ err("madvise(MADV_NOHUGEPAGE)");
+ }
test->uffd_fn(&gopts, &args);
uffd_test_ctx_clear(&gopts);
}
diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c
index 311fc5b4513e..4821a3563036 100644
--- a/tools/testing/selftests/mm/vm_util.c
+++ b/tools/testing/selftests/mm/vm_util.c
@@ -15,6 +15,9 @@
#define SMAP_FILE_PATH "/proc/self/smaps"
#define STATUS_FILE_PATH "/proc/self/status"
#define MAX_LINE_LENGTH 500
+#define PAGEMAP_PATH "/proc/self/pagemap"
+#define KPAGEFLAGS_PATH "/proc/kpageflags"
+#define MAX_NR_ORDERS 20
unsigned int __page_size;
unsigned int __page_shift;
@@ -31,7 +34,7 @@ uint64_t pagemap_get_entry(int fd, char *start)
return entry;
}
-static uint64_t __pagemap_scan_get_categories(int fd, char *start, struct page_region *r)
+static int __pagemap_scan_get_categories(int fd, char *start, struct page_region *r)
{
struct pm_scan_arg arg;
@@ -55,7 +58,7 @@ static uint64_t __pagemap_scan_get_categories(int fd, char *start, struct page_r
static uint64_t pagemap_scan_get_categories(int fd, char *start)
{
struct page_region r;
- long ret;
+ int ret;
ret = __pagemap_scan_get_categories(fd, start, &r);
if (ret < 0)
@@ -194,6 +197,125 @@ err_out:
return rss_anon;
}
+static int vaddr_pageflags_get(char *vaddr, int pagemap_fd, int kpageflags_fd,
+ uint64_t *flags)
+{
+ unsigned long pfn;
+
+ pfn = pagemap_get_pfn(pagemap_fd, vaddr);
+
+ /* non-present PFN */
+ if (pfn == -1UL)
+ return 1;
+
+ if (pageflags_get(pfn, kpageflags_fd, flags))
+ return -1;
+
+ return 0;
+}
+
+/*
+ * gather_folio_orders - scan through [vaddr_start, len) and record
+ * folio orders
+ *
+ * @vaddr_start: start vaddr
+ * @len: range length
+ * @pagemap_fd: file descriptor to /proc/<pid>/pagemap
+ * @kpageflags_fd: file descriptor to /proc/kpageflags
+ * @orders: output folio order array
+ * @nr_orders: folio order array size
+ *
+ * gather_folio_orders() scan through [vaddr_start, len) and check
+ * all folios within the range and record their orders. All order-0 pages will
+ * be recorded. Non-present vaddr is skipped.
+ *
+ * Return: 0 - no error, -1 - unhandled cases
+ */
+int gather_folio_orders(char *vaddr_start, size_t len,
+ int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders)
+{
+ uint64_t page_flags = 0;
+ int cur_order = -1;
+ char *vaddr;
+
+ if (pagemap_fd == -1 || kpageflags_fd == -1)
+ return -1;
+ if (!orders)
+ return -1;
+ if (nr_orders <= 0)
+ return -1;
+
+ for (vaddr = vaddr_start; vaddr < vaddr_start + len;) {
+ char *next_folio_vaddr;
+ int status;
+
+ status = vaddr_pageflags_get(vaddr, pagemap_fd, kpageflags_fd,
+ &page_flags);
+ if (status < 0)
+ return -1;
+
+ /* skip non present vaddr */
+ if (status == 1) {
+ vaddr += psize();
+ continue;
+ }
+
+ /* all order-0 pages with possible false postive (non folio) */
+ if (!(page_flags & (KPF_COMPOUND_HEAD | KPF_COMPOUND_TAIL))) {
+ orders[0]++;
+ vaddr += psize();
+ continue;
+ }
+
+ /* skip non thp compound pages */
+ if (!(page_flags & KPF_THP)) {
+ vaddr += psize();
+ continue;
+ }
+
+ /* vpn points to part of a THP at this point */
+ if (page_flags & KPF_COMPOUND_HEAD)
+ cur_order = 1;
+ else {
+ vaddr += psize();
+ continue;
+ }
+
+ next_folio_vaddr = vaddr + (1UL << (cur_order + pshift()));
+
+ if (next_folio_vaddr >= vaddr_start + len)
+ break;
+
+ while ((status = vaddr_pageflags_get(next_folio_vaddr,
+ pagemap_fd, kpageflags_fd,
+ &page_flags)) >= 0) {
+ /*
+ * non present vaddr, next compound head page, or
+ * order-0 page
+ */
+ if (status == 1 ||
+ (page_flags & KPF_COMPOUND_HEAD) ||
+ !(page_flags & (KPF_COMPOUND_HEAD | KPF_COMPOUND_TAIL))) {
+ if (cur_order < nr_orders) {
+ orders[cur_order]++;
+ cur_order = -1;
+ vaddr = next_folio_vaddr;
+ }
+ break;
+ }
+
+ cur_order++;
+ next_folio_vaddr = vaddr + (1UL << (cur_order + pshift()));
+ }
+
+ if (status < 0)
+ return status;
+ }
+ if (cur_order > 0 && cur_order < nr_orders)
+ orders[cur_order]++;
+ return 0;
+}
+
char *__get_smap_entry(void *addr, const char *pattern, char *buf, size_t len)
{
int ret;
@@ -229,7 +351,7 @@ err_out:
return entry;
}
-bool __check_huge(void *addr, char *pattern, int nr_hpages,
+static bool __check_pmd_huge(void *addr, char *pattern, int nr_hpages,
uint64_t hpage_size)
{
char buffer[MAX_LINE_LENGTH];
@@ -247,19 +369,84 @@ err_out:
return thp == (nr_hpages * (hpage_size >> 10));
}
-bool check_huge_anon(void *addr, int nr_hpages, uint64_t hpage_size)
+static bool check_large_folios(void *addr, size_t len, int nr_hpages,
+ uint64_t hpage_size)
{
- return __check_huge(addr, "AnonHugePages: ", nr_hpages, hpage_size);
+ int order = 0, pagesize = getpagesize();
+ unsigned int nr_pages = hpage_size / pagesize;
+ int orders[MAX_NR_ORDERS], status;
+ int pagemap_fd, kpageflags_fd;
+ bool ret = false;
+
+ if (!nr_pages)
+ ksft_exit_fail_msg("invalid hugepage size\n");
+
+ order = 31 - __builtin_clz(nr_pages);
+ if (!order || order >= MAX_NR_ORDERS)
+ ksft_exit_fail_msg("invalid order\n");
+
+ memset(orders, 0, sizeof(int) * MAX_NR_ORDERS);
+ pagemap_fd = open(PAGEMAP_PATH, O_RDONLY);
+ if (pagemap_fd == -1)
+ ksft_exit_fail_msg("read pagemap fail\n");
+
+ kpageflags_fd = open(KPAGEFLAGS_PATH, O_RDONLY);
+ if (kpageflags_fd == -1) {
+ close(pagemap_fd);
+ ksft_exit_fail_msg("read kpageflags fail\n");
+ }
+
+ status = gather_folio_orders(addr, len, pagemap_fd,
+ kpageflags_fd, orders, MAX_NR_ORDERS);
+ if (status)
+ goto out;
+
+ if (orders[order] == nr_hpages)
+ ret = true;
+
+out:
+ close(pagemap_fd);
+ close(kpageflags_fd);
+ return ret;
}
-bool check_huge_file(void *addr, int nr_hpages, uint64_t hpage_size)
+bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size)
{
- return __check_huge(addr, "FilePmdMapped:", nr_hpages, hpage_size);
+ uint64_t pmd_pagesize = read_pmd_pagesize();
+
+ if (!pmd_pagesize)
+ ksft_exit_fail_msg("reading PMD pagesize failed\n");
+
+ if (hpage_size == pmd_pagesize)
+ return __check_pmd_huge(addr, "AnonHugePages: ", nr_hpages, hpage_size);
+
+ return check_large_folios(addr, len, nr_hpages, hpage_size);
+}
+
+bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size)
+{
+ uint64_t pmd_pagesize = read_pmd_pagesize();
+
+ if (!pmd_pagesize)
+ ksft_exit_fail_msg("reading PMD pagesize failed\n");
+
+ if (hpage_size == pmd_pagesize)
+ return __check_pmd_huge(addr, "FilePmdMapped:", nr_hpages, hpage_size);
+
+ return check_large_folios(addr, len, nr_hpages, hpage_size);
}
-bool check_huge_shmem(void *addr, int nr_hpages, uint64_t hpage_size)
+bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size)
{
- return __check_huge(addr, "ShmemPmdMapped:", nr_hpages, hpage_size);
+ uint64_t pmd_pagesize = read_pmd_pagesize();
+
+ if (!pmd_pagesize)
+ ksft_exit_fail_msg("reading PMD pagesize failed\n");
+
+ if (hpage_size == pmd_pagesize)
+ return __check_pmd_huge(addr, "ShmemPmdMapped:", nr_hpages, hpage_size);
+
+ return check_large_folios(addr, len, nr_hpages, hpage_size);
}
int64_t allocate_transhuge(void *ptr, int pagemap_fd)
@@ -719,7 +906,7 @@ int read_file(const char *path, char *buf, size_t buflen)
return (unsigned int) numread;
}
-void write_file(const char *path, const char *buf, size_t buflen)
+static void __write_file(const char *path, const char *buf, size_t buflen, bool ignore_einval)
{
int fd, saved_errno;
ssize_t numwritten;
@@ -735,30 +922,48 @@ void write_file(const char *path, const char *buf, size_t buflen)
saved_errno = errno;
close(fd);
errno = saved_errno;
- if (numwritten < 0)
+ if (numwritten < 0) {
+ if (ignore_einval && errno == EINVAL)
+ return;
ksft_exit_fail_msg("%s write(%.*s) failed: %s\n", path, (int)(buflen - 1),
buf, strerror(errno));
+ }
if (numwritten != buflen - 1)
ksft_exit_fail_msg("%s write(%.*s) is truncated, expected %zu bytes, got %zd bytes\n",
path, (int)(buflen - 1), buf, buflen - 1, numwritten);
}
+void write_file(const char *path, const char *buf, size_t buflen)
+{
+ __write_file(path, buf, buflen, /* ignore_einval = */ false);
+}
+
unsigned long read_num(const char *path)
{
char buf[21];
- if (read_file(path, buf, sizeof(buf)) < 0)
+ if (!read_file(path, buf, sizeof(buf)))
ksft_exit_fail_perror("read_file()");
return strtoul(buf, NULL, 10);
}
-void write_num(const char *path, unsigned long num)
+static void __write_num(const char *path, unsigned long num, bool ignore_einval)
{
char buf[21];
sprintf(buf, "%lu", num);
- write_file(path, buf, strlen(buf) + 1);
+ __write_file(path, buf, strlen(buf) + 1, ignore_einval);
+}
+
+void write_num(const char *path, unsigned long num)
+{
+ return __write_num(path, num, /* ignore_einval = */ false);
+}
+
+void write_num_ignore_einval(const char *path, unsigned long num)
+{
+ return __write_num(path, num, /* ignore_einval = */ true);
}
static unsigned long shmall, shmmax;
diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h
index ea8fc8fdf0eb..9a49af88702e 100644
--- a/tools/testing/selftests/mm/vm_util.h
+++ b/tools/testing/selftests/mm/vm_util.h
@@ -90,11 +90,13 @@ void clear_softdirty(void);
bool check_for_pattern(FILE *fp, const char *pattern, char *buf, size_t len);
uint64_t read_pmd_pagesize(void);
unsigned long rss_anon(void);
-bool check_huge_anon(void *addr, int nr_hpages, uint64_t hpage_size);
-bool check_huge_file(void *addr, int nr_hpages, uint64_t hpage_size);
-bool check_huge_shmem(void *addr, int nr_hpages, uint64_t hpage_size);
+bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size);
+bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size);
+bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size);
int64_t allocate_transhuge(void *ptr, int pagemap_fd);
int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags);
+int gather_folio_orders(char *vaddr_start, size_t len,
+ int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders);
int uffd_register(int uffd, void *addr, uint64_t len,
bool miss, bool wp, bool minor);
@@ -168,6 +170,7 @@ void write_file(const char *path, const char *buf, size_t buflen);
int read_file(const char *path, char *buf, size_t buflen);
unsigned long read_num(const char *path);
void write_num(const char *path, unsigned long num);
+void write_num_ignore_einval(const char *path, unsigned long num);
void shm_limits_prepare(unsigned long length);
void __shm_limits_restore(void);
diff --git a/tools/testing/selftests/namespaces/nsid_test.c b/tools/testing/selftests/namespaces/nsid_test.c
index 46dc838cba82..a16f31f41d38 100644
--- a/tools/testing/selftests/namespaces/nsid_test.c
+++ b/tools/testing/selftests/namespaces/nsid_test.c
@@ -649,8 +649,6 @@ TEST_F(nsid, timens_separate)
/* Fork a grandchild to actually enter the new namespace */
pid_t grandchild = fork();
if (grandchild == 0) {
- /* Grandchild is in the new namespace */
- write(pipefd[1], "Y", 1);
close(pipefd[1]);
pause();
_exit(0);
@@ -771,8 +769,6 @@ TEST_F(nsid, pidns_separate)
/* Fork a grandchild to actually enter the new namespace */
pid_t grandchild = fork();
if (grandchild == 0) {
- /* Grandchild is in the new namespace */
- write(pipefd[1], "Y", 1);
close(pipefd[1]);
pause();
_exit(0);
diff --git a/tools/testing/selftests/nci/nci_dev.c b/tools/testing/selftests/nci/nci_dev.c
index 312f84ee0444..07427fa42888 100644
--- a/tools/testing/selftests/nci/nci_dev.c
+++ b/tools/testing/selftests/nci/nci_dev.c
@@ -8,6 +8,7 @@
#include <stdlib.h>
#include <errno.h>
+#include <stdint.h>
#include <string.h>
#include <sys/ioctl.h>
#include <fcntl.h>
@@ -87,6 +88,16 @@ struct msgtemplate {
char buf[MAX_MSG_SIZE];
};
+static int join_thread_status(pthread_t thread)
+{
+ void *thread_ret = NULL;
+
+ if (pthread_join(thread, &thread_ret))
+ return -1;
+
+ return (int)(intptr_t)thread_ret;
+}
+
static int create_nl_socket(void)
{
int fd;
@@ -182,7 +193,7 @@ static int get_family_id(int sd, __u32 pid, __u32 *event_group)
} ans;
struct nlattr *na;
int resp_len;
- __u16 id;
+ __u16 id = 0;
int len;
int rc;
@@ -438,13 +449,13 @@ FIXTURE_SETUP(NCI)
else
rc = pthread_create(&thread_t, NULL, virtual_dev_open,
(void *)&self->virtual_nci_fd);
- ASSERT_GT(rc, -1);
+ ASSERT_EQ(rc, 0);
rc = send_cmd_with_idx(self->sd, self->fid, self->pid,
NFC_CMD_DEV_UP, self->dev_idex);
EXPECT_EQ(rc, 0);
- pthread_join(thread_t, (void **)&status);
+ status = join_thread_status(thread_t);
ASSERT_EQ(status, 0);
self->open_state = true;
}
@@ -509,12 +520,12 @@ FIXTURE_TEARDOWN(NCI)
rc = pthread_create(&thread_t, NULL, virtual_deinit,
(void *)&self->virtual_nci_fd);
- ASSERT_GT(rc, -1);
+ ASSERT_EQ(rc, 0);
rc = send_cmd_with_idx(self->sd, self->fid, self->pid,
NFC_CMD_DEV_DOWN, self->dev_idex);
EXPECT_EQ(rc, 0);
- pthread_join(thread_t, (void **)&status);
+ status = join_thread_status(thread_t);
ASSERT_EQ(status, 0);
}
@@ -585,12 +596,11 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p
void *nla_start_poll_data[2] = {&dev_idx, &proto};
int nla_start_poll_len[2] = {4, 4};
pthread_t thread_t;
- int status;
int rc;
rc = pthread_create(&thread_t, NULL, virtual_poll_start,
(void *)&virtual_fd);
- if (rc < 0)
+ if (rc)
return rc;
rc = send_cmd_mt_nla(sd, fid, pid, NFC_CMD_START_POLL, 2, nla_start_poll_type,
@@ -598,19 +608,17 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p
if (rc != 0)
return rc;
- pthread_join(thread_t, (void **)&status);
- return status;
+ return join_thread_status(thread_t);
}
int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid)
{
pthread_t thread_t;
- int status;
int rc;
rc = pthread_create(&thread_t, NULL, virtual_poll_stop,
(void *)&virtual_fd);
- if (rc < 0)
+ if (rc)
return rc;
rc = send_cmd_with_idx(sd, fid, pid,
@@ -618,8 +626,7 @@ int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid)
if (rc != 0)
return rc;
- pthread_join(thread_t, (void **)&status);
- return status;
+ return join_thread_status(thread_t);
}
TEST_F(NCI, start_poll)
@@ -830,10 +837,14 @@ int disconnect_tag(int nfc_sock, int virtual_fd)
status = pthread_create(&thread_t, NULL, virtual_deactivate_proc,
(void *)&virtual_fd);
+ if (status)
+ return status;
close(nfc_sock);
- pthread_join(thread_t, (void **)&status);
- return status;
+ if (status)
+ return -1;
+
+ return join_thread_status(thread_t);
}
TEST_F(NCI, t4t_tag_read)
@@ -874,13 +885,13 @@ TEST_F(NCI, deinit)
else
rc = pthread_create(&thread_t, NULL, virtual_deinit,
(void *)&self->virtual_nci_fd);
- ASSERT_GT(rc, -1);
+ ASSERT_EQ(rc, 0);
rc = send_cmd_with_idx(self->sd, self->fid, self->pid,
NFC_CMD_DEV_DOWN, self->dev_idex);
EXPECT_EQ(rc, 0);
- pthread_join(thread_t, (void **)&status);
+ status = join_thread_status(thread_t);
self->open_state = 0;
ASSERT_EQ(status, 0);
diff --git a/tools/testing/selftests/net/Makefile b/tools/testing/selftests/net/Makefile
index 708d960ae07d..3ee3378f8b26 100644
--- a/tools/testing/selftests/net/Makefile
+++ b/tools/testing/selftests/net/Makefile
@@ -13,6 +13,7 @@ TEST_PROGS := \
arp_ndisc_untracked_subnets.sh \
bareudp.sh \
big_tcp.sh \
+ big_tcp_tunnels.sh \
bind_bhash.sh \
bpf_offload.py \
bridge_stp_mode.sh \
@@ -24,9 +25,11 @@ TEST_PROGS := \
cmsg_so_mark.sh \
cmsg_so_priority.sh \
cmsg_time.sh \
+ cork_fragsize.py \
double_udp_encap.sh \
drop_monitor_tests.sh \
ecmp_rehash.sh \
+ exception_cache.sh \
fcnal-ipv4.sh \
fcnal-ipv6.sh \
fcnal-other.sh \
@@ -39,6 +42,7 @@ TEST_PROGS := \
fib_rule_tests.sh \
fib_tests.sh \
fin_ack_lat.sh \
+ fou_mcast_encap.sh \
fq_band_pktlimit.sh \
gre_gso.sh \
gre_ipv6_lladdr.sh \
@@ -85,6 +89,7 @@ TEST_PROGS := \
rxtimestamp.sh \
sctp_vrf.sh \
skf_net_off.sh \
+ srv6_encap_lookup_l3vpn_test.sh \
srv6_end_dt46_l3vpn_test.sh \
srv6_end_dt4_l3vpn_test.sh \
srv6_end_dt6_l3vpn_test.sh \
@@ -147,7 +152,6 @@ TEST_GEN_FILES := \
ip_local_port_range \
ipsec \
ipv6_flowlabel \
- ipv6_flowlabel_mgr \
msg_zerocopy \
nettest \
psock_fanout \
@@ -180,6 +184,7 @@ TEST_GEN_PROGS := \
epoll_busy_poll \
getsockopt_iter \
icmp_rfc4884 \
+ ipv6_flowlabel_mgr \
ipv6_fragmentation \
proc_net_pktgen \
reuseaddr_conflict \
diff --git a/tools/testing/selftests/net/af_unix/.gitignore b/tools/testing/selftests/net/af_unix/.gitignore
index 973176644103..954f0958dd03 100644
--- a/tools/testing/selftests/net/af_unix/.gitignore
+++ b/tools/testing/selftests/net/af_unix/.gitignore
@@ -3,6 +3,8 @@ msg_oob
scm_inq
scm_pidfd
scm_rights
+scm_rights_denial_lsm
+scm_rights_denial_lsm.bpf.o
so_peek_off
unix_connect
unix_connreset
diff --git a/tools/testing/selftests/net/af_unix/Makefile b/tools/testing/selftests/net/af_unix/Makefile
index 57d159803a3a..a66f10fb0c23 100644
--- a/tools/testing/selftests/net/af_unix/Makefile
+++ b/tools/testing/selftests/net/af_unix/Makefile
@@ -11,10 +11,18 @@ TEST_GEN_PROGS := \
scm_inq \
scm_pidfd \
scm_rights \
+ scm_rights_denial_lsm \
so_peek_off \
unix_connect \
unix_connreset \
unix_listen \
# end of TEST_GEN_PROGS
+TEST_GEN_FILES := scm_rights_denial_lsm.bpf.o
+
include ../../lib.mk
+include ../bpf.mk
+
+$(OUTPUT)/scm_rights_denial_lsm: $(BPFOBJ)
+$(OUTPUT)/scm_rights_denial_lsm: CFLAGS += -I$(SCRATCH_DIR)/include
+$(OUTPUT)/scm_rights_denial_lsm: LDLIBS += -lelf -lz
diff --git a/tools/testing/selftests/net/af_unix/config b/tools/testing/selftests/net/af_unix/config
index 41dbb03c747e..46450fea8407 100644
--- a/tools/testing/selftests/net/af_unix/config
+++ b/tools/testing/selftests/net/af_unix/config
@@ -1,4 +1,11 @@
CONFIG_AF_UNIX_OOB=y
+CONFIG_BPF=y
+CONFIG_BPF_EVENTS=y
+CONFIG_BPF_JIT=y
+CONFIG_BPF_LSM=y
+CONFIG_BPF_SYSCALL=y
+CONFIG_DEBUG_INFO_BTF=y
+CONFIG_SECURITY=y
CONFIG_UNIX=y
CONFIG_UNIX_DIAG=m
CONFIG_USER_NS=y
diff --git a/tools/testing/selftests/net/af_unix/msg_oob.c b/tools/testing/selftests/net/af_unix/msg_oob.c
index 1b499d56656c..f051d79f7a8e 100644
--- a/tools/testing/selftests/net/af_unix/msg_oob.c
+++ b/tools/testing/selftests/net/af_unix/msg_oob.c
@@ -290,6 +290,25 @@ static void __setinlinepair(struct __test_metadata *_metadata,
}
}
+static void __setblockingpair(struct __test_metadata *_metadata,
+ FIXTURE_DATA(msg_oob) *self)
+{
+ int i;
+
+ for (i = 0; i < 2; i++) {
+ int ret, old_flags, flags;
+
+ old_flags = fcntl(self->fd[i * 2 + 1], F_GETFL, 0);
+ ASSERT_NE(-1, old_flags);
+
+ ret = fcntl(self->fd[i * 2 + 1], F_SETFL, old_flags & ~O_NONBLOCK);
+ ASSERT_EQ(0, ret);
+
+ flags = fcntl(self->fd[i * 2 + 1], F_GETFL, 0);
+ ASSERT_EQ(old_flags & ~O_NONBLOCK, flags);
+ }
+}
+
static void __siocatmarkpair(struct __test_metadata *_metadata,
FIXTURE_DATA(msg_oob) *self,
bool oob_head)
@@ -347,6 +366,9 @@ static void __resetpair(struct __test_metadata *_metadata,
#define setinlinepair() \
__setinlinepair(_metadata, self)
+#define setblockingpair() \
+ __setblockingpair(_metadata, self)
+
#define resetpair(reset) \
__resetpair(_metadata, self, variant, reset)
@@ -888,4 +910,49 @@ TEST_F(msg_oob, inline_ex_oob_siocatmark)
resetpair(true);
}
+TEST_F(msg_oob, zero_buf_oob)
+{
+ sendpair("a", 1, MSG_OOB);
+ recvpair("", 0, 0, 0);
+}
+
+TEST_F(msg_oob, zero_buf_oob_blocking)
+{
+ sendpair("a", 1, MSG_OOB);
+ setblockingpair();
+ recvpair("", 0, 0, 0);
+}
+
+TEST_F(msg_oob, zero_buf_non_oob_oob)
+{
+ sendpair("ab", 2, MSG_OOB);
+ recvpair("", 0, 0, 0);
+}
+
+TEST_F(msg_oob, zero_buf_non_oob_oob_blocking)
+{
+ sendpair("ab", 2, MSG_OOB);
+ setblockingpair();
+ recvpair("", 0, 0, 0);
+}
+
+TEST_F(msg_oob, zero_buf_ex_oob_oob)
+{
+ sendpair("a", 1, MSG_OOB);
+ recvpair("a", 1, 1, MSG_OOB);
+
+ sendpair("b", 1, MSG_OOB);
+ recvpair("", 0, 0, 0);
+}
+
+TEST_F(msg_oob, zero_buf_ex_oob_oob_blocking)
+{
+ sendpair("a", 1, MSG_OOB);
+ recvpair("a", 1, 1, MSG_OOB);
+
+ sendpair("b", 1, MSG_OOB);
+ setblockingpair();
+ recvpair("", 0, 0, 0);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/net/af_unix/scm_rights.c b/tools/testing/selftests/net/af_unix/scm_rights.c
index d82a79c21c17..c165f250220a 100644
--- a/tools/testing/selftests/net/af_unix/scm_rights.c
+++ b/tools/testing/selftests/net/af_unix/scm_rights.c
@@ -378,4 +378,21 @@ TEST_F(scm_rights, backtrack_from_scc)
close_sockets(10);
}
+TEST_F(scm_rights, mixed_lowpoint)
+{
+ create_sockets(6);
+
+ send_fd(0, 1);
+ send_fd(1, 2);
+ send_fd(2, 1);
+ send_fd(1, 0);
+
+ send_fd(3, 4);
+ send_fd(4, 5);
+ send_fd(5, 4);
+ send_fd(4, 3);
+
+ close_sockets(6);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.bpf.c b/tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.bpf.c
new file mode 100644
index 000000000000..4f2414465bfd
--- /dev/null
+++ b/tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.bpf.c
@@ -0,0 +1,36 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/bpf.h>
+#include <linux/errno.h>
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+struct inode {
+ unsigned long i_ino;
+} __attribute__((preserve_access_index));
+
+struct file {
+ struct inode *f_inode;
+} __attribute__((preserve_access_index));
+
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(max_entries, 16);
+ __type(key, __u64); /* inode number */
+ __type(value, __u32); /* tgid of the receiver being tested */
+} denied_inodes SEC(".maps");
+
+SEC("lsm/file_receive")
+int BPF_PROG(scm_rights_deny, struct file *file)
+{
+ __u32 tgid = bpf_get_current_pid_tgid() >> 32;
+ __u64 ino = file->f_inode->i_ino;
+ __u32 *owner;
+
+ owner = bpf_map_lookup_elem(&denied_inodes, &ino);
+ if (owner && *owner == tgid)
+ return -EPERM;
+
+ return 0;
+}
diff --git a/tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.c b/tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.c
new file mode 100644
index 000000000000..55c7ecdbb5fe
--- /dev/null
+++ b/tools/testing/selftests/net/af_unix/scm_rights_denial_lsm.c
@@ -0,0 +1,292 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/socket.h>
+#include <sys/stat.h>
+#include <sys/types.h>
+
+#include <bpf/bpf.h>
+#include <bpf/libbpf.h>
+
+#include "kselftest_harness.h"
+
+#ifndef SO_RIGHTS_NOTRUNC
+#define SO_RIGHTS_NOTRUNC 85
+#endif
+
+#define NR_FILES 2
+
+/* Per-file content, so a received fd can be matched to the file sent */
+#define SECRET(n) "secret %d", (n)
+
+/* Indices into the socketpair */
+#define SK_SENDER 0
+#define SK_RECEIVER 1
+
+FIXTURE(scm_rights_denial_bpf)
+{
+ struct bpf_object *obj;
+ struct bpf_link *link;
+ int map_fd;
+ int sk[2];
+ int files[NR_FILES];
+ __u64 inos[NR_FILES];
+ char paths[NR_FILES][64];
+};
+
+FIXTURE_VARIANT(scm_rights_denial_bpf)
+{
+ int sock_type;
+};
+
+FIXTURE_VARIANT_ADD(scm_rights_denial_bpf, stream)
+{
+ .sock_type = SOCK_STREAM,
+};
+
+FIXTURE_VARIANT_ADD(scm_rights_denial_bpf, dgram)
+{
+ .sock_type = SOCK_DGRAM,
+};
+
+FIXTURE_VARIANT_ADD(scm_rights_denial_bpf, seqpacket)
+{
+ .sock_type = SOCK_SEQPACKET,
+};
+
+FIXTURE_SETUP(scm_rights_denial_bpf)
+{
+ struct bpf_program *prog;
+ char lsms[256] = {};
+ int i, fd;
+
+ if (geteuid() != 0)
+ SKIP(return, "requires root");
+
+ fd = open("/sys/kernel/security/lsm", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_LT(0, read(fd, lsms, sizeof(lsms) - 1));
+ close(fd);
+
+ if (!strstr(lsms, "bpf"))
+ SKIP(return, "BPF LSM not active (boot with lsm=...,bpf)");
+
+ self->obj = bpf_object__open_file("scm_rights_denial_lsm.bpf.o", NULL);
+ ASSERT_NE(NULL, self->obj);
+ ASSERT_EQ(0, bpf_object__load(self->obj));
+
+ prog = bpf_object__find_program_by_name(self->obj, "scm_rights_deny");
+ ASSERT_NE(NULL, prog);
+
+ self->link = bpf_program__attach_lsm(prog);
+ ASSERT_NE(NULL, self->link);
+
+ self->map_fd = bpf_object__find_map_fd_by_name(self->obj,
+ "denied_inodes");
+ ASSERT_GE(self->map_fd, 0);
+
+ ASSERT_EQ(0, socketpair(AF_UNIX, variant->sock_type, 0, self->sk));
+
+ for (i = 0; i < NR_FILES; i++) {
+ struct stat st;
+
+ snprintf(self->paths[i], sizeof(self->paths[i]),
+ "/tmp/scm_rights_denial_bpf.%d.XXXXXX", i);
+ self->files[i] = mkstemp(self->paths[i]);
+ ASSERT_GE(self->files[i], 0);
+
+ ASSERT_LT(0, dprintf(self->files[i], SECRET(i)));
+
+ ASSERT_EQ(0, fstat(self->files[i], &st));
+ self->inos[i] = st.st_ino;
+ }
+}
+
+FIXTURE_TEARDOWN(scm_rights_denial_bpf)
+{
+ bpf_link__destroy(self->link);
+ bpf_object__close(self->obj);
+
+ for (int i = 0; i < NR_FILES; i++) {
+ if (self->files[i] >= 0) {
+ close(self->files[i]);
+ unlink(self->paths[i]);
+ }
+ }
+
+ close(self->sk[SK_SENDER]);
+ close(self->sk[SK_RECEIVER]);
+}
+
+static int deny_inode(int map_fd, __u64 ino)
+{
+ __u32 tgid = getpid();
+
+ return bpf_map_update_elem(map_fd, &ino, &tgid, BPF_ANY);
+}
+
+static int set_notrunc(int sk)
+{
+ int one = 1;
+
+ return setsockopt(sk, SOL_SOCKET, SO_RIGHTS_NOTRUNC,
+ &one, sizeof(one));
+}
+
+static int send_fds(int sk, int *fds, int n)
+{
+ char ctrl[CMSG_SPACE(NR_FILES * sizeof(int))] = {};
+ char data = 'x';
+ struct iovec iov = {
+ .iov_base = &data,
+ .iov_len = sizeof(data),
+ };
+ struct msghdr msg = {
+ .msg_iov = &iov,
+ .msg_iovlen = 1,
+ .msg_control = ctrl,
+ .msg_controllen = CMSG_SPACE(n * sizeof(int)),
+ };
+ struct cmsghdr *cmsg = CMSG_FIRSTHDR(&msg);
+ int ret;
+
+ cmsg->cmsg_level = SOL_SOCKET;
+ cmsg->cmsg_type = SCM_RIGHTS;
+ cmsg->cmsg_len = CMSG_LEN(n * sizeof(int));
+ memcpy(CMSG_DATA(cmsg), fds, n * sizeof(int));
+
+ ret = sendmsg(sk, &msg, 0);
+ if (ret != 1)
+ return -1;
+
+ return 0;
+}
+
+static int recv_fd_slots(int sk, int *slots, int *msg_flags)
+{
+ int nr_slots;
+ char ctrl[CMSG_SPACE(NR_FILES * sizeof(int))];
+ char data;
+ struct iovec iov = {
+ .iov_base = &data,
+ .iov_len = sizeof(data),
+ };
+ struct msghdr msg = {
+ .msg_iov = &iov,
+ .msg_iovlen = 1,
+ .msg_control = ctrl,
+ .msg_controllen = sizeof(ctrl),
+ };
+ struct cmsghdr *cmsg;
+
+ if (recvmsg(sk, &msg, 0) < 0)
+ return -1;
+
+ *msg_flags = msg.msg_flags;
+
+ cmsg = CMSG_FIRSTHDR(&msg);
+ if (!cmsg)
+ return 0;
+
+ nr_slots = (cmsg->cmsg_len - CMSG_LEN(0)) / sizeof(int);
+ memcpy(slots, CMSG_DATA(cmsg), nr_slots * sizeof(int));
+
+ return nr_slots;
+}
+
+/* Prove a received fd works by reading back the file's content. */
+static int check_secret(int fd, int idx)
+{
+ char want[32], got[32] = {};
+
+ snprintf(want, sizeof(want), SECRET(idx));
+ if (pread(fd, got, sizeof(got) - 1, 0) < 0)
+ return -1;
+
+ return strcmp(want, got);
+}
+
+TEST_F(scm_rights_denial_bpf, all_allowed)
+{
+ int slots[NR_FILES], nr_slots, flags;
+
+ ASSERT_EQ(0, set_notrunc(self->sk[SK_RECEIVER]));
+ ASSERT_EQ(0, send_fds(self->sk[SK_SENDER], self->files, NR_FILES));
+ nr_slots = recv_fd_slots(self->sk[SK_RECEIVER], slots, &flags);
+
+ ASSERT_EQ(NR_FILES, nr_slots);
+ EXPECT_EQ(0, flags & MSG_CTRUNC);
+
+ for (int i = 0; i < nr_slots; i++) {
+ ASSERT_GE(slots[i], 0);
+ EXPECT_EQ(0, check_secret(slots[i], i));
+ close(slots[i]);
+ }
+}
+
+TEST_F(scm_rights_denial_bpf, first_denied)
+{
+ int slots[NR_FILES], nr_slots, flags;
+
+ ASSERT_EQ(0, deny_inode(self->map_fd, self->inos[0]));
+
+ ASSERT_EQ(0, set_notrunc(self->sk[SK_RECEIVER]));
+ ASSERT_EQ(0, send_fds(self->sk[SK_SENDER], self->files, NR_FILES));
+ nr_slots = recv_fd_slots(self->sk[SK_RECEIVER], slots, &flags);
+
+ ASSERT_EQ(NR_FILES, nr_slots);
+ EXPECT_EQ(0, flags & MSG_CTRUNC);
+
+ EXPECT_EQ(-EPERM, slots[0]);
+ for (int i = 1; i < nr_slots; i++) {
+ ASSERT_GE(slots[i], 0);
+ EXPECT_EQ(0, check_secret(slots[i], i));
+ close(slots[i]);
+ }
+}
+
+TEST_F(scm_rights_denial_bpf, all_denied)
+{
+ int slots[NR_FILES], nr_slots, flags, i;
+
+ for (i = 0; i < NR_FILES; i++)
+ ASSERT_EQ(0, deny_inode(self->map_fd, self->inos[i]));
+
+ ASSERT_EQ(0, set_notrunc(self->sk[SK_RECEIVER]));
+ ASSERT_EQ(0, send_fds(self->sk[SK_SENDER], self->files, NR_FILES));
+ nr_slots = recv_fd_slots(self->sk[SK_RECEIVER], slots, &flags);
+
+ ASSERT_EQ(NR_FILES, nr_slots);
+ EXPECT_EQ(0, flags & MSG_CTRUNC);
+
+ for (i = 0; i < nr_slots; i++)
+ EXPECT_EQ(-EPERM, slots[i]);
+}
+
+TEST_F(scm_rights_denial_bpf, denied_without_notrunc)
+{
+ int slots[NR_FILES], nr_slots, flags;
+
+ /*
+ * Baseline behaviour without SO_RIGHTS_NOTRUNC: the fd array is
+ * truncated at the first denied fd and MSG_CTRUNC is set.
+ */
+ ASSERT_EQ(0, deny_inode(self->map_fd, self->inos[1]));
+
+ ASSERT_EQ(0, send_fds(self->sk[SK_SENDER], self->files, NR_FILES));
+ nr_slots = recv_fd_slots(self->sk[SK_RECEIVER], slots, &flags);
+
+ ASSERT_EQ(1, nr_slots);
+ EXPECT_NE(0, flags & MSG_CTRUNC);
+
+ ASSERT_GE(slots[0], 0);
+ EXPECT_EQ(0, check_secret(slots[0], 0));
+ close(slots[0]);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/net/big_tcp_tunnels.sh b/tools/testing/selftests/net/big_tcp_tunnels.sh
new file mode 100755
index 000000000000..cc0875e52fb9
--- /dev/null
+++ b/tools/testing/selftests/net/big_tcp_tunnels.sh
@@ -0,0 +1,218 @@
+#!/usr/bin/env bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Testing for IPv4 and IPv6 BIG TCP over VXLAN and GENEVE tunnels.
+
+source "$(dirname "$0")/lib.sh"
+
+SERVER_NS=$(mktemp -u server-XXXXXXXX)
+SERVER_IP4="192.168.1.1"
+SERVER_IP6="2001:db8::1:1"
+SERVER_IP4_TUN="192.168.2.1"
+SERVER_IP6_TUN="2001:db8::2:1"
+
+CLIENT_NS=$(mktemp -u client-XXXXXXXX)
+CLIENT_IP4="192.168.1.2"
+CLIENT_IP6="2001:db8::1:2"
+CLIENT_IP4_TUN="192.168.2.2"
+CLIENT_IP6_TUN="2001:db8::2:2"
+
+# Kselftest framework requirement - SKIP code is 4.
+ksft_skip=4
+
+if [ -z "$PACKETS_THRESHOLD" ]; then
+ if [ "$KSFT_MACHINE_SLOW" = yes ]; then
+ echo 'Debug kernel detected, lowering the default threshold'
+ PACKETS_THRESHOLD=100
+ else
+ PACKETS_THRESHOLD=1000
+ fi
+fi
+
+setup() {
+ ip netns add "$SERVER_NS"
+ ip netns add "$CLIENT_NS"
+ ip -netns "$SERVER_NS" link add link1 type veth peer name link0 netns "$CLIENT_NS"
+
+ ip -netns "$CLIENT_NS" link set link0 up
+ ip -netns "$CLIENT_NS" addr replace "$CLIENT_IP4/24" dev link0
+ ip -netns "$CLIENT_NS" addr replace "$CLIENT_IP6/112" dev link0 nodad
+ ip -netns "$CLIENT_NS" link set link0 \
+ gso_max_size 196608 gso_ipv4_max_size 196608 \
+ gro_max_size 196608 gro_ipv4_max_size 196608
+ ip -netns "$SERVER_NS" link set link1 up
+ ip -netns "$SERVER_NS" addr replace "$SERVER_IP4/24" dev link1
+ ip -netns "$SERVER_NS" addr replace "$SERVER_IP6/112" dev link1 nodad
+ ip -netns "$SERVER_NS" link set link1 \
+ gso_max_size 196608 gso_ipv4_max_size 196608 \
+ gro_max_size 196608 gro_ipv4_max_size 196608
+
+ ip netns exec "$SERVER_NS" netserver >/dev/null
+ wait_local_port_listen "$SERVER_NS" 12865 tcp
+
+ DEFAULT_TCP_MIN_TSO_SEGS=$(ip netns exec "$CLIENT_NS" sysctl -n net.ipv4.tcp_min_tso_segs)
+}
+
+setup_tunnel() {
+ if [ "$2" = 4 ]; then
+ SERVER_IP="$SERVER_IP4"
+ CLIENT_IP="$CLIENT_IP4"
+ echo "Setting up ${1^^} over IPv4, veth tx csum offload $3"
+ else
+ SERVER_IP="$SERVER_IP6"
+ CLIENT_IP="$CLIENT_IP6"
+ echo "Setting up ${1^^} over IPv6, veth tx csum offload $3"
+ fi
+
+ if [ "$1" = vxlan ]; then
+ ip -netns "$CLIENT_NS" link add tun0 type vxlan \
+ id 5001 remote "$SERVER_IP" local "$CLIENT_IP" dev link0 dstport 4789
+ else
+ ip -netns "$CLIENT_NS" link add tun0 type geneve \
+ id 5001 remote "$SERVER_IP"
+ fi
+ ip -netns "$CLIENT_NS" link set tun0 up
+ ip -netns "$CLIENT_NS" addr replace "$CLIENT_IP4_TUN/24" dev tun0
+ ip -netns "$CLIENT_NS" addr replace "$CLIENT_IP6_TUN/112" dev tun0 nodad
+ ip -netns "$CLIENT_NS" link set tun0 \
+ gso_max_size 196608 gso_ipv4_max_size 196608 \
+ gro_max_size 196608 gro_ipv4_max_size 196608
+ if [ "$1" = vxlan ]; then
+ ip -netns "$SERVER_NS" link add tun1 type vxlan \
+ id 5001 remote "$CLIENT_IP" local "$SERVER_IP" dev link1 dstport 4789
+ else
+ ip -netns "$SERVER_NS" link add tun1 type geneve \
+ id 5001 remote "$CLIENT_IP"
+ fi
+ ip -netns "$SERVER_NS" link set tun1 up
+ ip -netns "$SERVER_NS" addr replace "$SERVER_IP4_TUN/24" dev tun1
+ ip -netns "$SERVER_NS" addr replace "$SERVER_IP6_TUN/112" dev tun1 nodad
+ ip -netns "$SERVER_NS" link set tun1 \
+ gso_max_size 196608 gso_ipv4_max_size 196608 \
+ gro_max_size 196608 gro_ipv4_max_size 196608
+
+ ip netns exec "$CLIENT_NS" ethtool -K link0 tx-checksumming "$3" > /dev/null
+ ip netns exec "$SERVER_NS" ethtool -K link1 tx-checksumming "$3" > /dev/null
+}
+
+cleanup_tunnel() {
+ ip -netns "$CLIENT_NS" link del tun0
+ ip -netns "$SERVER_NS" link del tun1
+}
+
+cleanup() {
+ ip netns pids "$SERVER_NS" | xargs -r kill
+ ip netns pids "$CLIENT_NS" | xargs -r kill
+ ip netns del "$SERVER_NS"
+ ip netns del "$CLIENT_NS"
+ rm -rf "$WORKDIR"
+}
+
+do_test() {
+ local packets_threshold="$PACKETS_THRESHOLD"
+
+ # When tx csum offload is off, software GSO is performed before passing the
+ # packet to veth. Check BIG TCP packets inside the VXLAN tunnel to verify
+ # the software checksum path: if the checksum code is broken, these packets
+ # will be dropped.
+ if [ "$3" = on ]; then
+ CAPTURE_IFACE='link'
+ if [ "$1" = 4 ]; then
+ IPTABLES=iptables
+ else
+ IPTABLES=ip6tables
+ fi
+ else
+ CAPTURE_IFACE='tun'
+ if [ "$2" = 4 ]; then
+ IPTABLES=iptables
+ else
+ IPTABLES=ip6tables
+ fi
+ packets_threshold=$(( PACKETS_THRESHOLD / 10 ))
+ fi
+ if [ "$2" = 4 ]; then
+ IPTABLES_SACK=iptables
+ else
+ IPTABLES_SACK=ip6tables
+ fi
+
+ if [ "$3" != 'on' ] && [ "$KSFT_MACHINE_SLOW" = yes ]; then
+ echo 'Slow configuration; increasing net.ipv4.tcp_min_tso_segs and initcwnd'
+ ip netns exec "$CLIENT_NS" sysctl -w net.ipv4.tcp_min_tso_segs=52
+ if [ "$2" = 4 ]; then
+ ip -netns "$CLIENT_NS" \
+ route change 192.168.2.0/24 dev tun0 initcwnd 100
+ else
+ ip -netns "$CLIENT_NS" -6 \
+ route change 2001:db8::2:0/112 dev tun0 initcwnd 100
+ fi
+ else
+ ip netns exec "$CLIENT_NS" \
+ sysctl -w net.ipv4.tcp_min_tso_segs="$DEFAULT_TCP_MIN_TSO_SEGS"
+ fi
+
+ ip netns exec "$SERVER_NS" "$IPTABLES" -w -t raw -I PREROUTING -i "${CAPTURE_IFACE}1" -m length ! --length 0:65535 -m comment --comment "bigtcp"
+ ip netns exec "$CLIENT_NS" "$IPTABLES" -w -t raw -I OUTPUT -o "${CAPTURE_IFACE}0" -m length ! --length 0:65535 -m comment --comment "bigtcp"
+ ip netns exec "$SERVER_NS" "$IPTABLES_SACK" -w -t raw -I OUTPUT -o "tun1" -p tcp -m tcp --tcp-flags ACK ACK --tcp-option 5 -m comment --comment "sack"
+
+ if [ "$2" = 4 ]; then
+ SERVER_IP="$SERVER_IP4_TUN"
+ echo "Running IPv4 traffic in the tunnel"
+ else
+ SERVER_IP="$SERVER_IP6_TUN"
+ echo "Running IPv6 traffic in the tunnel"
+ fi
+
+ ip netns exec "$CLIENT_NS" netperf -t TCP_STREAM -l 5 -H "$SERVER_IP" -- \
+ -m 80000 > /dev/null
+
+ PACKETS_SERVER=$(ip netns exec "$SERVER_NS" "$IPTABLES-save" -c -t raw | sed -rn '/ --comment bigtcp/{s/^\[([0-9]+):.*/\1/p;q}')
+ PACKETS_CLIENT=$(ip netns exec "$CLIENT_NS" "$IPTABLES-save" -c -t raw | sed -rn '/ --comment bigtcp/{s/^\[([0-9]+):.*/\1/p;q}')
+ PACKETS_SACK=$(ip netns exec "$SERVER_NS" "$IPTABLES_SACK-save" -c -t raw | sed -rn '/ --comment sack/{s/^\[([0-9]+):.*/\1/p;q}')
+ ip netns exec "$SERVER_NS" "$IPTABLES" -w -t raw -D PREROUTING -i "${CAPTURE_IFACE}1" -m length ! --length 0:65535 -m comment --comment "bigtcp"
+ ip netns exec "$CLIENT_NS" "$IPTABLES" -w -t raw -D OUTPUT -o "${CAPTURE_IFACE}0" -m length ! --length 0:65535 -m comment --comment "bigtcp"
+ ip netns exec "$SERVER_NS" "$IPTABLES_SACK" -w -t raw -D OUTPUT -o "tun1" -p tcp -m tcp --tcp-flags ACK ACK --tcp-option 5 -m comment --comment "sack"
+
+ echo "Captured BIG TCP RX packets: $PACKETS_SERVER"
+ echo "Captured BIG TCP TX packets: $PACKETS_CLIENT"
+ echo "Captured TCP SACK packets: $PACKETS_SACK"
+ [ "$PACKETS_SERVER" -gt "$packets_threshold" ] || return 1
+ [ "$PACKETS_CLIENT" -gt "$packets_threshold" ] || return 1
+ [ "$PACKETS_SACK" -lt "$(( PACKETS_CLIENT / 2 ))" ] || return 1
+}
+
+if ! netperf -V &> /dev/null; then
+ echo "SKIP: Could not run test without netperf tool"
+ exit "$ksft_skip"
+fi
+
+if ! iptables --version &> /dev/null; then
+ echo "SKIP: Could not run test without iptables tool"
+ exit "$ksft_skip"
+fi
+
+if ! ethtool --version &> /dev/null; then
+ echo "SKIP: Could not run test without ethtool tool"
+ exit "$ksft_skip"
+fi
+
+if ! ip link help 2>&1 | grep gso_ipv4_max_size &> /dev/null; then
+ echo "SKIP: Could not run test without gso/gro_ipv4_max_size supported in ip-link"
+ exit "$ksft_skip"
+fi
+
+WORKDIR=$(mktemp -d)
+trap cleanup EXIT
+setup
+for tunnel in vxlan geneve; do
+ for tun_family in 4 6; do
+ for traffic_family in 4 6; do
+ for csum_offload in on off; do
+ setup_tunnel "$tunnel" "$tun_family" "$csum_offload" || exit "$?"
+ do_test "$tun_family" "$traffic_family" "$csum_offload" || exit "$?"
+ cleanup_tunnel
+ done
+ done
+ done
+done
diff --git a/tools/testing/selftests/net/config b/tools/testing/selftests/net/config
index e1ce35c2abbe..737e7e6327b3 100644
--- a/tools/testing/selftests/net/config
+++ b/tools/testing/selftests/net/config
@@ -38,6 +38,8 @@ CONFIG_IP_NF_TARGET_REJECT=m
CONFIG_IP_NF_TARGET_TTL=m
CONFIG_IP_SCTP=m
CONFIG_IPV6=y
+CONFIG_IPV6_FOU=m
+CONFIG_IPV6_FOU_TUNNEL=m
CONFIG_IPV6_GRE=m
CONFIG_IPV6_ILA=m
CONFIG_IPV6_IOAM6_LWTUNNEL=y
@@ -80,6 +82,7 @@ CONFIG_NETFILTER=y
CONFIG_NETFILTER_ADVANCED=y
CONFIG_NETFILTER_XTABLES_LEGACY=y
CONFIG_NETFILTER_XT_MATCH_BPF=m
+CONFIG_NETFILTER_XT_MATCH_COMMENT=y
CONFIG_NETFILTER_XT_MATCH_LENGTH=m
CONFIG_NETFILTER_XT_MATCH_POLICY=m
CONFIG_NETFILTER_XT_NAT=m
@@ -114,14 +117,11 @@ CONFIG_NFT_COMPAT=m
CONFIG_NFT_NAT=m
CONFIG_NUMA=y
CONFIG_OPENVSWITCH=m
-CONFIG_OPENVSWITCH_GENEVE=m
-CONFIG_OPENVSWITCH_GRE=m
-CONFIG_OPENVSWITCH_VXLAN=m
CONFIG_PAGE_POOL_STATS=y
-CONFIG_PROC_SYSCTL=y
CONFIG_PSAMPLE=m
CONFIG_RPS=y
CONFIG_SYN_COOKIES=y
+CONFIG_SYSCTL=y
CONFIG_SYSFS=y
CONFIG_TAP=m
CONFIG_TCP_CONG_DCTCP=y
diff --git a/tools/testing/selftests/net/cork_fragsize.py b/tools/testing/selftests/net/cork_fragsize.py
new file mode 100755
index 000000000000..7afd643d07ec
--- /dev/null
+++ b/tools/testing/selftests/net/cork_fragsize.py
@@ -0,0 +1,187 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: GPL-2.0
+
+'''Test possible UDP length overflow in udp_send_skb/udp_v6_send_skb.'''
+
+import errno
+import gzip
+import os
+import socket
+import struct
+import subprocess
+from contextlib import contextmanager
+
+from lib.py import (
+ KsftNamedVariant,
+ KsftSkipEx,
+ NetNS,
+ NetNSEnter,
+ defer,
+ ip,
+ ksft_eq,
+ ksft_exit,
+ ksft_pr,
+ ksft_raises,
+ ksft_run,
+ ksft_true,
+ ksft_variants,
+)
+
+IP_MTU_DISCOVER = 10
+IP_PMTUDISC_PROBE = 3
+IPV6_MTU_DISCOVER = 23
+IPV6_PMTUDISC_DO = 2
+IPV6_PMTUDISC_PROBE = 3
+IPV6_TLV_JUMBO = 194
+
+
+def check_kernel_config(option: str) -> bool | None:
+ '''
+ Check whether the option is enabled in the config of the running kernel.
+ Returns None if the config is not found; otherwise returns True/False
+ depending on the option value in the config.
+ '''
+
+ for filename, method in [
+ ('/proc/config.gz', gzip.open),
+ (f'/boot/config-{os.uname().release}', open),
+ ]:
+ try:
+ with method(filename, 'rt') as config:
+ for line in config:
+ if line.rstrip() == f'{option}=y':
+ return True
+ return False
+ except OSError:
+ continue
+ return None
+
+
+def assert_debug_kernel() -> None:
+ '''
+ Skip the test if CONFIG_DEBUG_NET is not set in the kernel config.
+ '''
+
+ res = check_kernel_config('CONFIG_DEBUG_NET')
+ if res is None:
+ ksft_pr("WARN: Can't read kernel config; assuming debug kernel, and running the test")
+ elif not res:
+ raise KsftSkipEx('CONFIG_DEBUG_NET is not set')
+
+
+def check_dmesg_clean(func: str) -> bool:
+ '''
+ Check if the given function produced a WARN in dmesg.
+ '''
+
+ with subprocess.Popen(['dmesg'], stdout=subprocess.PIPE) as dmesg:
+ res = subprocess.run(['grep', '-q', f'WARNING:.*{func}'], stdin=dmesg.stdout, check=False)
+ return res.returncode != 0 and dmesg.returncode == 0
+
+
+@contextmanager
+def dummy_netdev(ns: NetNS, mtu: int, ipv6: bool) -> None:
+ '''
+ Create a dummy netdev inside the given namespace, and tune it for the test.
+ '''
+
+ ip('link add dummy type dummy', ns=ns)
+ with defer(ip, 'link del dummy', ns=ns):
+ ip(f'link set dummy mtu {mtu}', ns=ns)
+ ip('link set dummy up', ns=ns)
+ flag = '-6' if ipv6 else ''
+ nodad = 'nodad' if ipv6 else ''
+ local = 'fd00::1/64' if ipv6 else '10.0.0.1/24'
+ remote = 'fd00::2' if ipv6 else '10.0.0.2'
+ ip(f'{flag} addr add {local} dev dummy {nodad}', ns=ns)
+ ip(f'{flag} neigh add {remote} lladdr 02:00:00:00:00:02 dev dummy nud permanent', ns=ns)
+ yield
+
+
+@ksft_variants([
+ KsftNamedVariant(
+ 'ipv6',
+ True,
+ socket.AF_INET6,
+ (socket.IPPROTO_IPV6, IPV6_MTU_DISCOVER, IPV6_PMTUDISC_DO),
+ 'fd00::2',
+ 'udp_v6_send_skb',
+ ),
+ KsftNamedVariant(
+ 'ipv4',
+ False,
+ socket.AF_INET,
+ (socket.IPPROTO_IP, IP_MTU_DISCOVER, IP_PMTUDISC_PROBE),
+ '10.0.0.2',
+ 'udp_send_skb',
+ ),
+])
+def test_udp(
+ ipv6: bool,
+ af: socket.AddressFamily,
+ sockopts: tuple[int, int, int],
+ destip: str,
+ func: str
+) -> None:
+ '''
+ Test that sending an oversized UDP packet over a UDP socket doesn't overflow
+ the 16-bit length field in the UDP header, which could happen on older
+ kernels in udp_send_skb/udp_v6_send_skb.
+
+ IPv4: The packet will be dropped with EMSGSIZE, but the overflow could
+ happen before it happens. The only way to test this is to check dmesg on
+ CONFIG_DEBUG_NET=y kernels that have udp_set_len_short with the warning.
+
+ IPv6: The packet will be dropped with EMSGSIZE on fixed kernels, and will be
+ sent corrupted on older kernels. Test both: sendto must return EMSGSIZE, and
+ dmesg must be clean of warnings on CONFIG_DEBUG_NET=y kernels.
+ '''
+
+ if not ipv6:
+ assert_debug_kernel()
+
+ with (
+ NetNS() as ns,
+ dummy_netdev(ns, 65556 + 20 * ipv6, ipv6),
+ NetNSEnter(ns),
+ socket.socket(af, socket.SOCK_DGRAM) as fd,
+ ):
+ fd.setsockopt(*sockopts)
+ with ksft_raises(OSError) as e:
+ fd.sendto(b' ' * 65528, (destip, 1234))
+ # IPv6: EMSGSIZE happens on kernels with the fix.
+ # IPv4: EMSGSIZE happens on both fixed and unfixed kernels, after the
+ # WARN is printed - ignore it and rely on the dmesg check.
+ if e.exception is not None:
+ ksft_eq(e.exception.errno, errno.EMSGSIZE)
+
+ ksft_true(check_dmesg_clean(func), 'WARNING detected in dmesg')
+
+
+def test_ipv6_jumbo() -> None:
+ '''
+ Test that sending UDP jumbograms over a raw IPv6 socket works, despite
+ having the fix for oversized UDP packets. sendto must not raise an OSError
+ exception (when raised, the test fails automatically).
+ '''
+
+ with (
+ NetNS() as ns,
+ dummy_netdev(ns, 65584, True),
+ NetNSEnter(ns),
+ socket.socket(socket.AF_INET6, socket.SOCK_RAW, socket.IPPROTO_UDP) as fd,
+ ):
+ hopopts = struct.pack('!BBBBI', 0, 0, IPV6_TLV_JUMBO, 4, 65544)
+ fd.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_HOPOPTS, hopopts)
+ fd.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_CHECKSUM, 6)
+ fd.setsockopt(socket.IPPROTO_IPV6, IPV6_MTU_DISCOVER, IPV6_PMTUDISC_PROBE)
+ udp = struct.pack('!HHHH', 1234, 1234, 0, 0) + b' ' * 65528
+ fd.sendto(udp, ('fd00::2', 0))
+
+
+if __name__ == "__main__":
+ ksft_run([
+ test_udp,
+ test_ipv6_jumbo,
+ ])
+ ksft_exit()
diff --git a/tools/testing/selftests/net/exception_cache.sh b/tools/testing/selftests/net/exception_cache.sh
new file mode 100755
index 000000000000..8d3eed5c532a
--- /dev/null
+++ b/tools/testing/selftests/net/exception_cache.sh
@@ -0,0 +1,521 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Test that the state of the route exception cache after an ICMP error is
+# processed does not depend on whether the quoted packet was matched to a
+# socket. Otherwise, an off-path attacker can probe the cache to discover the
+# ephemeral port used by a connected UDP socket.
+#
+# When the quoted packet is not matched to a socket, the same exception is
+# created as when it is matched, so that neither its presence nor its contents
+# reveal the result of socket matching.
+#
+# +----+
+# +---------| r1 |
+# | +----+
+# +----+ +--------+ | .1
+# | h1 |---| bridge | | 198.51.100.0/30
+# +----+ +--------+ | 2001:db8:2::/64
+# .1 | | .2
+# | +----+ +----+
+# +---------| r2 |-----------| h2 |
+# .2 .3 +----+ .1 .2 +----+
+# 203.0.113.0/24
+# 2001:db8:3::/64
+# 192.0.2.0/24
+# 2001:db8:1::/64
+#
+# Traffic from h1 to h2 is routed via r1, which reaches h2's network via r2
+# over the point-to-point link. The MTU of the r2 - h2 link is lowered so that
+# r2 emits ICMP errors towards h1.
+#
+# For the redirect tests r1's route to h2's network is replaced with one via r2
+# on the shared segment, so that r1 forwards the packet back to the segment it
+# arrived from and emits a redirect towards h1.
+#
+# The packets that provoke the ICMP errors are injected with a packet socket so
+# that no socket is ever associated with them. A socket is created separately,
+# with socat, when a test needs the ICMP error to be matched.
+
+# shellcheck disable=SC1091,SC2034,SC2154,SC2329
+source lib.sh
+
+require_command jq
+require_command mausezahn
+require_command nstat
+require_command socat
+
+ALL_TESTS="
+ pmtu_no_socket_ipv4
+ pmtu_no_socket_ipv6
+ pmtu_socket_ipv4
+ pmtu_socket_ipv6
+ pmtu_omit_ipv4
+ pmtu_omit_ipv6
+ redirect_no_socket_ipv4
+ redirect_no_socket_ipv6
+ redirect_socket_ipv4
+ redirect_socket_ipv6
+"
+
+# Shared segment.
+H1_ADDR4=192.0.2.1
+R1_ADDR4=192.0.2.2
+R2_ADDR4=192.0.2.3
+H1_ADDR6=2001:db8:1::1
+R1_ADDR6=2001:db8:1::2
+R2_ADDR6=2001:db8:1::3
+
+# r1 - r2 link.
+R2_R1_ADDR4=198.51.100.2
+R2_R1_ADDR6=2001:db8:2::2
+
+# r2 - h2 link.
+H2_ADDR4=203.0.113.2
+H2_NET4=203.0.113.0/24
+H2_ADDR6=2001:db8:3::2
+H2_NET6=2001:db8:3::/64
+
+SPORT=12345
+DPORT=54321
+
+# The MTU of the shared segment and of the r1 - r2 link. Large enough for the
+# injected packets to reach r2 intact.
+SEGMENT_MTU=2000
+# Size of the injected packets. The PMTU tests need a size that exceeds every
+# MTU used for the r2 - h2 link, so that r2 responds with an ICMP error. The
+# redirect tests need a size that does not, otherwise r2 would respond with an
+# ICMP error in addition to the redirect emitted by r1.
+PMTU_PACKET_SIZE=1800
+REDIRECT_PACKET_SIZE=100
+
+# The MTUs used for the r2 - h2 link. All of them must be at least
+# IPV6_MIN_MTU, otherwise IPv6 silently ignores the error instead of creating
+# an exception.
+MTU_MID=1400
+MTU_LOW=1300
+
+# Values for the IP{,V6}_MTU_DISCOVER socket option.
+PMTUDISC_DONT=0
+PMTUDISC_OMIT=5
+
+SOCAT_PID=
+
+linklocal_get()
+{
+ local ns=$1; shift
+ local dev=$1; shift
+
+ ip -n "$ns" -j -6 addr show dev "$dev" | \
+ jq -r '.[]["addr_info"][] | select(.scope == "link") | .local'
+}
+
+linklocal_exists()
+{
+ local ns=$1; shift
+ local dev=$1; shift
+
+ [ -n "$(linklocal_get "$ns" "$dev")" ]
+}
+
+family_vars_set()
+{
+ local family=$1; shift
+
+ FAMILY=$family
+
+ if [ "$family" -eq 4 ]; then
+ H1_ADDR=$H1_ADDR4
+ H2_ADDR=$H2_ADDR4
+ MZ_FAMILY_OPT=()
+ # Without the Don't Fragment bit set r2 fragments the packet
+ # instead of reporting the MTU of the next hop.
+ MZ_IP_OPTS="df,"
+ SOCAT_DST="UDP4-CONNECT:$H2_ADDR4:$DPORT"
+ SOCAT_BIND="bind=$H1_ADDR4:$SPORT"
+ SOCAT_PMTUDISC="ip-mtu-discover"
+ else
+ H1_ADDR=$H1_ADDR6
+ H2_ADDR=$H2_ADDR6
+ MZ_FAMILY_OPT=(-6)
+ MZ_IP_OPTS=
+ SOCAT_DST="UDP6-CONNECT:[$H2_ADDR6]:$DPORT"
+ SOCAT_BIND="bind=[$H1_ADDR6]:$SPORT"
+ SOCAT_PMTUDISC="ipv6-mtu-discover"
+ fi
+}
+
+topology_setup()
+{
+ local ns
+
+ setup_ns h1 r1 r2 h2 sw
+ defer cleanup_all_ns
+
+ # Link-local addresses are generated from the MAC address and read
+ # back during setup, so request that generation mode explicitly and
+ # make the addresses available as soon as the devices are brought up.
+ for ns in "$h1" "$r1" "$r2" "$h2" "$sw"; do
+ ip netns exec "$ns" sysctl -qw \
+ net.ipv6.conf.default.addr_gen_mode=0 \
+ net.ipv6.conf.default.accept_dad=0 \
+ net.ipv6.conf.all.accept_dad=0
+ done
+
+ ip -n "$sw" link add name br0 type bridge
+ ip -n "$sw" link set dev br0 mtu "$SEGMENT_MTU" up
+
+ ip -n "$h1" link add name eth0 mtu "$SEGMENT_MTU" type veth \
+ peer name swp1 mtu "$SEGMENT_MTU" netns "$sw"
+ ip -n "$r1" link add name eth0 mtu "$SEGMENT_MTU" type veth \
+ peer name swp2 mtu "$SEGMENT_MTU" netns "$sw"
+ ip -n "$r2" link add name eth0 mtu "$SEGMENT_MTU" type veth \
+ peer name swp3 mtu "$SEGMENT_MTU" netns "$sw"
+ ip -n "$r1" link add name eth1 mtu "$SEGMENT_MTU" type veth \
+ peer name eth1 mtu "$SEGMENT_MTU" netns "$r2"
+ ip -n "$r2" link add name eth2 type veth peer name eth0 netns "$h2"
+
+ ip -n "$sw" link set dev swp1 master br0 up
+ ip -n "$sw" link set dev swp2 master br0 up
+ ip -n "$sw" link set dev swp3 master br0 up
+
+ ip -n "$h1" link set dev eth0 up
+ ip -n "$r1" link set dev eth0 up
+ ip -n "$r1" link set dev eth1 up
+ ip -n "$r2" link set dev eth0 up
+ ip -n "$r2" link set dev eth1 up
+ ip -n "$r2" link set dev eth2 up
+ ip -n "$h2" link set dev eth0 up
+
+ ip -n "$h1" address add "$H1_ADDR4/24" dev eth0
+ ip -n "$r1" address add "$R1_ADDR4/24" dev eth0
+ ip -n "$r2" address add "$R2_ADDR4/24" dev eth0
+ ip -n "$r1" address add 198.51.100.1/30 dev eth1
+ ip -n "$r2" address add "$R2_R1_ADDR4/30" dev eth1
+ ip -n "$r2" address add 203.0.113.1/24 dev eth2
+ ip -n "$h2" address add "$H2_ADDR4/24" dev eth0
+
+ ip -n "$h1" -6 address add "$H1_ADDR6/64" dev eth0 nodad
+ ip -n "$r1" -6 address add "$R1_ADDR6/64" dev eth0 nodad
+ ip -n "$r2" -6 address add "$R2_ADDR6/64" dev eth0 nodad
+ ip -n "$r1" -6 address add 2001:db8:2::1/64 dev eth1 nodad
+ ip -n "$r2" -6 address add "$R2_R1_ADDR6/64" dev eth1 nodad
+ ip -n "$r2" -6 address add 2001:db8:3::1/64 dev eth2 nodad
+ ip -n "$h2" -6 address add "$H2_ADDR6/64" dev eth0 nodad
+
+ ip netns exec "$r1" sysctl -qw net.ipv4.ip_forward=1
+ ip netns exec "$r1" sysctl -qw net.ipv4.conf.all.send_redirects=1
+ ip netns exec "$r1" sysctl -qw net.ipv6.conf.all.forwarding=1
+ ip netns exec "$r2" sysctl -qw net.ipv4.ip_forward=1
+ ip netns exec "$r2" sysctl -qw net.ipv6.conf.all.forwarding=1
+
+ ip netns exec "$h1" sysctl -qw net.ipv4.conf.all.accept_redirects=1
+ ip netns exec "$h1" sysctl -qw net.ipv4.conf.eth0.accept_redirects=1
+ ip netns exec "$h1" sysctl -qw net.ipv6.conf.all.accept_redirects=1
+ ip netns exec "$h1" sysctl -qw net.ipv6.conf.eth0.accept_redirects=1
+
+ slowwait 5 linklocal_exists "$r1" eth0
+ check_err $? "r1: link-local address was not generated"
+ slowwait 5 linklocal_exists "$r2" eth0
+ check_err $? "r2: link-local address was not generated"
+
+ R1_LLADDR=$(linklocal_get "$r1" eth0)
+ R2_LLADDR=$(linklocal_get "$r2" eth0)
+ R1_MAC=$(ip -n "$r1" -j link show dev eth0 | jq -r '.[]["address"]')
+ R2_MAC=$(ip -n "$r2" -j link show dev eth0 | jq -r '.[]["address"]')
+
+ ip -n "$h1" route add "$H2_NET4" via "$R1_ADDR4" dev eth0
+ ip -n "$h1" -6 route add "$H2_NET6" via "$R1_LLADDR" dev eth0
+ ip -n "$r1" route add "$H2_NET4" via "$R2_R1_ADDR4" dev eth1
+ ip -n "$r1" -6 route add "$H2_NET6" via "$R2_R1_ADDR6" dev eth1
+ ip -n "$h2" route add default via 203.0.113.1 dev eth0
+ ip -n "$h2" -6 route add default via 2001:db8:3::1 dev eth0
+
+ far_mtu_set "$MTU_MID"
+}
+
+# Make r1 forward towards h2's network over the segment it receives the packet
+# from, so that it emits a redirect towards h1.
+redirect_route_set()
+{
+ ip -n "$r1" route replace "$H2_NET4" via "$R2_ADDR4" dev eth0
+ ip -n "$r1" -6 route replace "$H2_NET6" via "$R2_LLADDR" dev eth0
+
+ # __ip_do_redirect() only creates an exception if the new gateway is
+ # already a valid neighbour. Otherwise it merely triggers address
+ # resolution. IPv6 resolves the target itself, in rt6_do_redirect().
+ ip -n "$h1" neigh replace "$R2_ADDR4" lladdr "$R2_MAC" dev eth0 \
+ nud permanent
+}
+
+far_mtu_set()
+{
+ local mtu=$1; shift
+
+ ip -n "$r2" link set dev eth2 mtu "$mtu"
+ ip -n "$h2" link set dev eth0 mtu "$mtu"
+}
+
+socket_is_open()
+{
+ ip netns exec "$h1" ss -uHn "sport = :$SPORT" | grep -q .
+}
+
+socket_start()
+{
+ # Disable PMTU discovery by default so that ICMP errors are not
+ # reported to the socket. Otherwise socat would exit when the first one
+ # arrives and later packets in the same test would not be matched to a
+ # socket. The exception is still created, as ip{,6}_sk_accept_pmtu()
+ # only rejects IP{,V6}_PMTUDISC_{INTERFACE,OMIT}.
+ local pmtudisc=${1:-$PMTUDISC_DONT}
+
+ # Send socat's diagnostics to /dev/null. It reports the ICMP errors
+ # that reach the socket, which is exactly what the tests provoke.
+ ip netns exec "$h1" socat -u -lf/dev/null \
+ "$SOCAT_DST,$SOCAT_BIND,$SOCAT_PMTUDISC=$pmtudisc" \
+ OPEN:/dev/null,wronly=1 &
+ SOCAT_PID=$!
+ defer socket_stop
+
+ slowwait 5 socket_is_open
+ check_err $? "socket did not open"
+}
+
+socket_stop()
+{
+ [ -z "$SOCAT_PID" ] && return 0
+
+ kill "$SOCAT_PID" &> /dev/null
+ wait "$SOCAT_PID" 2> /dev/null
+ SOCAT_PID=
+}
+
+# Inject a packet towards h2 with a packet socket. No socket is associated with
+# it, so an ICMP error quoting it is matched to a socket only if one was
+# created separately with the same source port.
+packet_send()
+{
+ local size=$1; shift
+
+ ip netns exec "$h1" mausezahn "${MZ_FAMILY_OPT[@]}" eth0 \
+ -a own -b "$R1_MAC" -A "$H1_ADDR" -B "$H2_ADDR" \
+ -t udp "${MZ_IP_OPTS}sp=$SPORT,dp=$DPORT" \
+ -p "$size" -c 1 -q
+}
+
+exception_show()
+{
+ if [ "$FAMILY" -eq 4 ]; then
+ # IPv4 exceptions without a bound route are not dumped, but
+ # "route get" reports the exception and binds a route to it.
+ ip -n "$h1" route get "$H2_ADDR"
+ else
+ # IPv6 does not report a cache indication in "route get"
+ # output, so dump the exceptions instead.
+ ip -n "$h1" -6 route show cache | grep -F "$H2_ADDR" || true
+ fi
+}
+
+exception_mtu_get()
+{
+ exception_show | grep -o "mtu [0-9]*" | cut -d ' ' -f 2
+}
+
+exception_gw_get()
+{
+ exception_show | grep -o "via [0-9a-f.:]*" | cut -d ' ' -f 2
+}
+
+exception_mtu_check()
+{
+ local expected=$1; shift
+
+ [ "$(exception_mtu_get)" = "$expected" ]
+}
+
+icmp_errors_get()
+{
+ local ctr=IcmpInDestUnreachs
+
+ [ "$FAMILY" -eq 6 ] && ctr=Icmp6InPktTooBigs
+
+ ip netns exec "$h1" nstat -asz "$ctr" | \
+ awk -v ctr="$ctr" '$1 == ctr { print $2 }'
+}
+
+exception_pmtu_check()
+{
+ local mtu=$1; shift
+ local desc=$1; shift
+
+ busywait "$BUSYWAIT_TIMEOUT" exception_mtu_check "$mtu"
+ check_err $? "$desc: exception does not carry an MTU of $mtu"
+}
+
+pmtu_no_socket()
+{
+ local family=$1; shift
+
+ RET=0
+ family_vars_set "$family"
+ topology_setup
+
+ packet_send "$PMTU_PACKET_SIZE"
+ exception_pmtu_check "$MTU_MID" "No socket"
+
+ log_test "IPv$family: PMTU: exception without a matching socket"
+}
+
+pmtu_no_socket_ipv4()
+{
+ pmtu_no_socket 4
+}
+
+pmtu_no_socket_ipv6()
+{
+ pmtu_no_socket 6
+}
+
+pmtu_socket()
+{
+ local family=$1; shift
+ local t0
+
+ RET=0
+ family_vars_set "$family"
+ topology_setup
+ socket_start
+
+ packet_send "$PMTU_PACKET_SIZE"
+ exception_pmtu_check "$MTU_MID" "Matching socket"
+
+ # A lower PMTU replaces the one currently stored in the exception.
+ far_mtu_set "$MTU_LOW"
+ packet_send "$PMTU_PACKET_SIZE"
+ exception_pmtu_check "$MTU_LOW" "Lower PMTU"
+
+ # A higher PMTU is ignored, so the exception is left as it is. Wait
+ # for the error to be received, as otherwise the check below would
+ # pass even if it never was.
+ far_mtu_set "$MTU_MID"
+ t0=$(icmp_errors_get)
+ packet_send "$PMTU_PACKET_SIZE"
+ busywait "$BUSYWAIT_TIMEOUT" until_counter_is ">= $((t0 + 1))" \
+ icmp_errors_get > /dev/null
+ check_err $? "Higher PMTU: ICMP error was not received"
+
+ exception_mtu_check "$MTU_LOW"
+ check_err $? "Higher PMTU: exception does not carry an MTU of $MTU_LOW"
+
+ log_test "IPv$family: PMTU: exception with a matching socket"
+}
+
+pmtu_socket_ipv4()
+{
+ pmtu_socket 4
+}
+
+pmtu_socket_ipv6()
+{
+ pmtu_socket 6
+}
+
+pmtu_omit()
+{
+ local family=$1; shift
+
+ RET=0
+ family_vars_set "$family"
+ topology_setup
+ socket_start "$PMTUDISC_OMIT"
+
+ packet_send "$PMTU_PACKET_SIZE"
+ exception_pmtu_check "$MTU_MID" "PMTU discovery disabled"
+
+ log_test "IPv$family: PMTU: exception with a socket ignoring it"
+}
+
+pmtu_omit_ipv4()
+{
+ pmtu_omit 4
+}
+
+pmtu_omit_ipv6()
+{
+ pmtu_omit 6
+}
+
+exception_gw_check()
+{
+ local expected=$1; shift
+
+ [ -n "$expected" ] && [ "$(exception_gw_get)" = "$expected" ]
+}
+
+redirect_gw_new()
+{
+ if [ "$FAMILY" -eq 4 ]; then
+ echo "$R2_ADDR4"
+ else
+ echo "$R2_LLADDR"
+ fi
+}
+
+redirect_no_socket()
+{
+ local family=$1; shift
+
+ RET=0
+ family_vars_set "$family"
+ topology_setup
+ redirect_route_set
+
+ packet_send "$REDIRECT_PACKET_SIZE"
+ busywait "$BUSYWAIT_TIMEOUT" exception_gw_check "$(redirect_gw_new)"
+ check_err $? "No socket: exception does not carry the new gateway"
+
+ log_test "IPv$family: Redirect: exception without a matching socket"
+}
+
+redirect_no_socket_ipv4()
+{
+ redirect_no_socket 4
+}
+
+redirect_no_socket_ipv6()
+{
+ redirect_no_socket 6
+}
+
+redirect_socket()
+{
+ local family=$1; shift
+
+ RET=0
+ family_vars_set "$family"
+ topology_setup
+ redirect_route_set
+ socket_start
+
+ packet_send "$REDIRECT_PACKET_SIZE"
+ busywait "$BUSYWAIT_TIMEOUT" exception_gw_check "$(redirect_gw_new)"
+ check_err $? "Matching socket: exception does not carry the new gateway"
+
+ log_test "IPv$family: Redirect: exception with a matching socket"
+}
+
+redirect_socket_ipv4()
+{
+ redirect_socket 4
+}
+
+redirect_socket_ipv6()
+{
+ redirect_socket 6
+}
+
+trap defer_scopes_cleanup EXIT
+tests_run
+
+exit "$EXIT_STATUS"
diff --git a/tools/testing/selftests/net/fib_nexthops.sh b/tools/testing/selftests/net/fib_nexthops.sh
index ac868a731694..431d7bed7622 100755
--- a/tools/testing/selftests/net/fib_nexthops.sh
+++ b/tools/testing/selftests/net/fib_nexthops.sh
@@ -30,6 +30,7 @@ IPV4_TESTS="
ipv4_large_res_grp
ipv4_compat_mode
ipv4_fdb_grp_fcnal
+ ipv4_fdb_port_fcnal
ipv4_mpath_select
ipv4_torture
ipv4_res_torture
@@ -44,6 +45,7 @@ IPV6_TESTS="
ipv6_large_res_grp
ipv6_compat_mode
ipv6_fdb_grp_fcnal
+ ipv6_fdb_port_fcnal
ipv6_mpath_select
ipv6_torture
ipv6_res_torture
@@ -432,6 +434,15 @@ check_nexthop_fdb_support()
fi
}
+check_nexthop_fdb_port_support()
+{
+ $IP nexthop help 2>&1 | grep -q "dst_port"
+ if [ $? -ne 0 ]; then
+ echo "SKIP: iproute2 too old, missing nexthop dst_port support"
+ return $ksft_skip
+ fi
+}
+
check_nexthop_res_support()
{
$IP nexthop help 2>&1 | grep -q resilient
@@ -522,6 +533,20 @@ ipv6_fdb_grp_fcnal()
run_cmd "$BRIDGE fdb add 02:02:00:00:00:14 dev vx10 nhid 61 self"
log_test $? 255 "Fdb mac add with nexthop"
+ # fdb entries with a nexthop group cannot be aged out
+ run_cmd "$BRIDGE fdb add 02:02:00:00:00:15 dev vx10 nhid 102 self static"
+ log_test $? 0 "Fdb mac add with nexthop group and static state"
+
+ run_cmd "$BRIDGE fdb add 02:02:00:00:00:16 dev vx10 nhid 102 self dynamic"
+ log_test $? 255 "Fdb mac add with nexthop group and dynamic state"
+
+ run_cmd "$BRIDGE fdb add 02:02:00:00:00:17 dev vx10 nhid 102 self"
+ run_cmd "$BRIDGE fdb replace 02:02:00:00:00:17 dev vx10 dst 2001:db8:91::11 self dynamic"
+ log_test $? 255 "Fdb mac replace with nexthop group and dynamic state"
+
+ run_cmd "$BRIDGE fdb append 02:02:00:00:00:17 dev vx10 dst 2001:db8:91::11 self dynamic"
+ log_test $? 255 "Fdb mac append with nexthop group and dynamic state"
+
run_cmd "$IP -6 ro add 2001:db8:101::1/128 nhid 66"
log_test $? 2 "Route add with fdb nexthop"
@@ -541,6 +566,42 @@ ipv6_fdb_grp_fcnal()
$IP link del dev vx10
}
+ipv6_fdb_port_fcnal()
+{
+ echo
+ echo "IPv6 fdb nexthop dst_port functional"
+ echo "------------------------------------"
+
+ check_nexthop_fdb_port_support
+ if [ $? -eq $ksft_skip ]; then
+ return $ksft_skip
+ fi
+
+ # NHA_DST_PORT: optional per-nexthop VXLAN destination UDP port,
+ # letting an fdb nexthop group balance a flow across legs that share
+ # an underlay IP but listen on different UDP ports.
+ run_cmd "$IP nexthop add id 80 via 2001:db8:91::2 fdb dst_port 4790"
+ check_nexthop "id 80" \
+ "id 80 via 2001:db8:91::2 scope link fdb dst_port 4790"
+ log_test $? 0 "Fdb nexthop with dst_port"
+
+ run_cmd "$IP nexthop add id 81 fdb dst_port 4790"
+ log_test $? 2 "Fdb nexthop with dst_port but no gateway"
+
+ run_cmd "$IP nexthop add id 81 via 2001:db8:91::2 fdb dst_port 0"
+ log_test $? 2 "Fdb nexthop with dst_port 0"
+
+ run_cmd "$IP nexthop add id 82 via 2001:db8:91::2 fdb dst_port 4789"
+ run_cmd "$IP nexthop add id 83 via 2001:db8:91::3 fdb dst_port 5789"
+ run_cmd "$IP nexthop add id 106 group 82/83 fdb"
+ check_nexthop "id 106" "id 106 group 82/83 fdb"
+ log_test $? 0 "Fdb nexthop group with legs differing in dst_port"
+
+ run_cmd "$IP nexthop add id 84 via 2001:db8:91::2 fdb"
+ check_nexthop "id 84" "id 84 via 2001:db8:91::2 scope link fdb"
+ log_test $? 0 "Fdb nexthop without dst_port omits dst_port"
+}
+
ipv4_fdb_grp_fcnal()
{
local rc
@@ -622,6 +683,20 @@ ipv4_fdb_grp_fcnal()
run_cmd "$BRIDGE fdb add 02:02:00:00:00:14 dev vx10 nhid 12 self"
log_test $? 255 "Fdb mac add with nexthop"
+ # fdb entries with a nexthop group cannot be aged out
+ run_cmd "$BRIDGE fdb add 02:02:00:00:00:15 dev vx10 nhid 102 self static"
+ log_test $? 0 "Fdb mac add with nexthop group and static state"
+
+ run_cmd "$BRIDGE fdb add 02:02:00:00:00:16 dev vx10 nhid 102 self dynamic"
+ log_test $? 255 "Fdb mac add with nexthop group and dynamic state"
+
+ run_cmd "$BRIDGE fdb add 02:02:00:00:00:17 dev vx10 nhid 102 self"
+ run_cmd "$BRIDGE fdb replace 02:02:00:00:00:17 dev vx10 dst 10.0.0.3 self dynamic"
+ log_test $? 255 "Fdb mac replace with nexthop group and dynamic state"
+
+ run_cmd "$BRIDGE fdb append 02:02:00:00:00:17 dev vx10 dst 10.0.0.3 self dynamic"
+ log_test $? 255 "Fdb mac append with nexthop group and dynamic state"
+
run_cmd "$IP ro add 172.16.0.0/22 nhid 16"
log_test $? 2 "Route add with fdb nexthop"
@@ -641,6 +716,42 @@ ipv4_fdb_grp_fcnal()
$IP link del dev vx10
}
+ipv4_fdb_port_fcnal()
+{
+ echo
+ echo "IPv4 fdb nexthop dst_port functional"
+ echo "------------------------------------"
+
+ check_nexthop_fdb_port_support
+ if [ $? -eq $ksft_skip ]; then
+ return $ksft_skip
+ fi
+
+ # NHA_DST_PORT: optional per-nexthop VXLAN destination UDP port,
+ # letting an fdb nexthop group balance a flow across legs that share
+ # an underlay IP but listen on different UDP ports.
+ run_cmd "$IP nexthop add id 30 via 172.16.1.2 fdb dst_port 4790"
+ check_nexthop "id 30" \
+ "id 30 via 172.16.1.2 scope link fdb dst_port 4790"
+ log_test $? 0 "Fdb nexthop with dst_port"
+
+ run_cmd "$IP nexthop add id 31 fdb dst_port 4790"
+ log_test $? 2 "Fdb nexthop with dst_port but no gateway"
+
+ run_cmd "$IP nexthop add id 31 via 172.16.1.2 fdb dst_port 0"
+ log_test $? 2 "Fdb nexthop with dst_port 0"
+
+ run_cmd "$IP nexthop add id 32 via 172.16.1.2 fdb dst_port 4789"
+ run_cmd "$IP nexthop add id 33 via 172.16.1.3 fdb dst_port 5789"
+ run_cmd "$IP nexthop add id 105 group 32/33 fdb"
+ check_nexthop "id 105" "id 105 group 32/33 fdb"
+ log_test $? 0 "Fdb nexthop group with legs differing in dst_port"
+
+ run_cmd "$IP nexthop add id 34 via 172.16.1.2 fdb"
+ check_nexthop "id 34" "id 34 via 172.16.1.2 scope link fdb"
+ log_test $? 0 "Fdb nexthop without dst_port omits dst_port"
+}
+
ipv4_mpath_select()
{
local rc dev match h addr
diff --git a/tools/testing/selftests/net/fin_ack_lat.c b/tools/testing/selftests/net/fin_ack_lat.c
index 70187494b57a..4068f8e227cf 100644
--- a/tools/testing/selftests/net/fin_ack_lat.c
+++ b/tools/testing/selftests/net/fin_ack_lat.c
@@ -69,7 +69,7 @@ static void client(int port)
lat = timediff(start, end);
sum_lat += lat;
nr_lat++;
- if (lat < 100000)
+ if (lat < 1000000)
goto close;
if (getsockname(sock, (struct sockaddr *)&laddr, &len) == -1)
@@ -103,7 +103,8 @@ static void server(int sock, struct sockaddr_in address)
static void sig_handler(int signum)
{
- kill(SIGTERM, child_pid);
+ if (child_pid > 0)
+ kill(child_pid, SIGTERM);
exit(0);
}
@@ -142,6 +143,8 @@ int main(int argc, char const *argv[])
fprintf(stderr, "server port: %d\n", ntohs(laddr.sin_port));
child_pid = fork();
+ if (child_pid < 0)
+ error(-1, errno, "fork");
if (!child_pid)
client(ntohs(laddr.sin_port));
else
diff --git a/tools/testing/selftests/net/fin_ack_lat.sh b/tools/testing/selftests/net/fin_ack_lat.sh
index a3ff6e0b2c7a..a8aa2238ab5c 100755
--- a/tools/testing/selftests/net/fin_ack_lat.sh
+++ b/tools/testing/selftests/net/fin_ack_lat.sh
@@ -9,7 +9,7 @@ set -e
tmpfile=$(mktemp /tmp/fin_ack_latency.XXXX.log)
cleanup() {
- kill $(pidof fin_ack_lat)
+ kill $(pidof fin_ack_lat) 2>/dev/null || true
rm -f $tmpfile
}
diff --git a/tools/testing/selftests/net/forwarding/lib.sh b/tools/testing/selftests/net/forwarding/lib.sh
index ac8358bcb22c..05acd4011456 100644
--- a/tools/testing/selftests/net/forwarding/lib.sh
+++ b/tools/testing/selftests/net/forwarding/lib.sh
@@ -406,6 +406,21 @@ get_ifname_by_ip()
__run_on "$target" ip -j addr show to "$ip_addr" | jq -r '.[].ifname'
}
+# Wait for the device to refresh its HW statistics. Devices latch the stats
+# reported via ethtool only every stats-block-usecs, so sample after that.
+hw_stats_settle()
+{
+ local iface=$1; shift
+ local usecs
+
+ # Match only a non-zero integer; 0 or "n/a" use default (20msec)
+ usecs=$(run_on "$iface" ethtool -c "$iface" 2>/dev/null | \
+ sed -n 's/^stats-block-usecs:[[:space:]]*\([1-9][0-9]*\)$/\1/p')
+ usecs=${usecs:-20000}
+
+ sleep "$(echo "$usecs * 1.25 / 1000 / 1000" | bc -l)"
+}
+
# Whether the test is conforming to the requirements and usage described in
# drivers/net/README.rst.
: "${DRIVER_TEST_CONFORMANT:=no}"
diff --git a/tools/testing/selftests/net/fou_mcast_encap.sh b/tools/testing/selftests/net/fou_mcast_encap.sh
new file mode 100755
index 000000000000..70210d39fba3
--- /dev/null
+++ b/tools/testing/selftests/net/fou_mcast_encap.sh
@@ -0,0 +1,172 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Test that UDP encapsulation (FOU) correctly handles packet resubmit
+# when packets are delivered via the multicast UDP delivery path.
+#
+# When a FOU-encapsulated packet arrives with a multicast destination IP,
+# __udp4_lib_mcast_deliver() / __udp6_lib_mcast_deliver() must resubmit
+# it to the inner protocol handler (e.g., GRE) rather than consuming it.
+# This test verifies both IPv4 and IPv6 paths by creating a FOU/GRETAP
+# tunnel with a multicast remote address and sending ping through it.
+#
+# The early demux optimization can mask this issue by routing packets via
+# the unicast path (udp[6]_unicast_rcv_skb), so we disable it to force
+# packets through the multicast delivery function.
+
+source lib.sh
+
+NSENDER=""
+NRECV=""
+
+FOU_PORT4=4797
+FOU_PORT6=4798
+MCAST4=239.0.0.1
+MCAST6=ff0e::1
+
+TUN4_S=192.168.99.1
+TUN4_R=192.168.99.2
+TUN6_S=2001:db8:99::1
+TUN6_R=2001:db8:99::2
+
+cleanup() {
+ cleanup_all_ns
+}
+
+trap cleanup EXIT
+
+setup_common() {
+ setup_ns NSENDER NRECV
+
+ # Create veth pair directly inside namespaces to avoid name
+ # collisions with devices in the root namespace.
+ ip link add veth_s netns "$NSENDER" type veth \
+ peer name veth_r netns "$NRECV"
+
+ ip -n "$NSENDER" link set veth_s up
+ ip -n "$NRECV" link set veth_r up
+
+ # Same sysctl controls early demux for both IPv4 and IPv6.
+ ip netns exec "$NRECV" sysctl -wq net.ipv4.ip_early_demux=0
+}
+
+setup_ipv4() {
+ # IPv4 FOU (CONFIG_NET_FOU) is built in on kernels configured for
+ # these tests, so no module load is needed here.
+ ip -n "$NSENDER" addr add 10.0.0.1/24 dev veth_s
+ ip -n "$NRECV" addr add 10.0.0.2/24 dev veth_r
+
+ # Join multicast group on receiver
+ ip -n "$NRECV" addr add "$MCAST4/32" dev veth_r autojoin
+
+ ip -n "$NSENDER" route add 239.0.0.0/8 dev veth_s
+ ip -n "$NRECV" route add 239.0.0.0/8 dev veth_r
+
+ # Sender: GRETAP with FOU encap (no FOU listener needed on TX side)
+ ip -n "$NSENDER" link add eoudp4 type gretap \
+ remote "$MCAST4" local 10.0.0.1 \
+ encap fou encap-sport "$FOU_PORT4" encap-dport "$FOU_PORT4" \
+ key "$MCAST4"
+ ip -n "$NSENDER" link set eoudp4 up
+ ip -n "$NSENDER" addr add "$TUN4_S/24" dev eoudp4
+
+ # Receiver: FOU listener + GRETAP
+ ip netns exec "$NRECV" ip fou add port "$FOU_PORT4" ipproto 47
+ ip -n "$NRECV" link add eoudp4 type gretap \
+ remote "$MCAST4" local 10.0.0.2 \
+ encap fou encap-sport "$FOU_PORT4" encap-dport "$FOU_PORT4" \
+ key "$MCAST4"
+ ip -n "$NRECV" link set eoudp4 up
+ ip -n "$NRECV" addr add "$TUN4_R/24" dev eoudp4
+
+ # Static neigh on sender: ARP replies cannot traverse the
+ # unidirectional multicast tunnel.
+ local recv_mac
+ recv_mac=$(ip -n "$NRECV" link show eoudp4 | awk '/ether/{print $2}')
+ ip -n "$NSENDER" neigh add "$TUN4_R" lladdr "$recv_mac" dev eoudp4
+}
+
+setup_ipv6() {
+ # Skip cleanly if IPv6 or the fou6 module is not available.
+ [ -e /proc/sys/net/ipv6 ] || return "$ksft_skip"
+ modprobe -q fou6 || return "$ksft_skip"
+
+ ip -n "$NSENDER" addr add 2001:db8::1/64 dev veth_s nodad
+ ip -n "$NRECV" addr add 2001:db8::2/64 dev veth_r nodad
+
+ # Join multicast group on receiver
+ ip -n "$NRECV" addr add "$MCAST6/128" dev veth_r autojoin
+
+ ip -n "$NSENDER" -6 route add ff00::/8 dev veth_s
+ ip -n "$NRECV" -6 route add ff00::/8 dev veth_r
+
+ # Sender: ip6gretap with FOU encap
+ ip -n "$NSENDER" link add eoudp6 type ip6gretap \
+ remote "$MCAST6" local 2001:db8::1 \
+ encap fou encap-sport "$FOU_PORT6" encap-dport "$FOU_PORT6" \
+ key 42
+ ip -n "$NSENDER" link set eoudp6 up
+ ip -n "$NSENDER" addr add "$TUN6_S/64" dev eoudp6 nodad
+
+ # Receiver: FOU listener (IPv6) + ip6gretap
+ ip netns exec "$NRECV" ip fou add port "$FOU_PORT6" ipproto 47 -6
+ ip -n "$NRECV" link add eoudp6 type ip6gretap \
+ remote "$MCAST6" local 2001:db8::2 \
+ encap fou encap-sport "$FOU_PORT6" encap-dport "$FOU_PORT6" \
+ key 42
+ ip -n "$NRECV" link set eoudp6 up
+ ip -n "$NRECV" addr add "$TUN6_R/64" dev eoudp6 nodad
+
+ # Static neigh on sender: neighbor discovery cannot traverse the
+ # unidirectional multicast tunnel.
+ local recv_mac
+ recv_mac=$(ip -n "$NRECV" link show eoudp6 | awk '/ether/{print $2}')
+ ip -n "$NSENDER" neigh add "$TUN6_R" lladdr "$recv_mac" dev eoudp6
+}
+
+get_rx_packets() {
+ local dev="$1"
+
+ ip -n "$NRECV" -s link show "$dev" | awk '/RX:/{getline; print $2}'
+}
+
+run_ping_test() {
+ local family="$1"
+ local dev="$2"
+ local dst="$3"
+ local name="$4"
+ local count=100
+ local rx_before rx_after rx_delta
+
+ # Warmup: let any initial broadcast/ND traffic settle
+ ip netns exec "$NSENDER" ping "$family" -c 1 -W 1 "$dst" \
+ >/dev/null 2>&1
+ sleep 1
+
+ rx_before=$(get_rx_packets "$dev")
+ ip netns exec "$NSENDER" ping "$family" -i 0.01 -c $count -W 1 "$dst" \
+ >/dev/null 2>&1
+ sleep 1
+ rx_after=$(get_rx_packets "$dev")
+
+ rx_delta=$((rx_after - rx_before))
+
+ if [ "$rx_delta" -ge "$count" ]; then
+ RET=$ksft_pass
+ else
+ RET=$ksft_fail
+ fi
+ log_test "$name (received $rx_delta/$count)"
+}
+
+setup_common
+setup_ipv4
+run_ping_test -4 eoudp4 "$TUN4_R" "FOU/GRETAP IPv4 multicast encap resubmit"
+
+if setup_ipv6; then
+ run_ping_test -6 eoudp6 "$TUN6_R" "FOU/ip6gretap IPv6 multicast encap resubmit"
+else
+ log_test_skip "FOU/ip6gretap IPv6 multicast encap resubmit"
+fi
+
+exit "$EXIT_STATUS"
diff --git a/tools/testing/selftests/net/getsockopt_iter.c b/tools/testing/selftests/net/getsockopt_iter.c
index 209569354d0e..6c2408df4612 100644
--- a/tools/testing/selftests/net/getsockopt_iter.c
+++ b/tools/testing/selftests/net/getsockopt_iter.c
@@ -11,6 +11,8 @@
* that always reports the required buffer length back via optlen,
* even when the user buffer is too small to receive any group bits.
* - vsock: SO_VM_SOCKETS_BUFFER_SIZE covers the u64 path.
+ * - raw: ICMP_FILTER covers a fixed-size struct payload that clamps
+ * the length down on a short buffer instead of failing.
*
* Author: Breno Leitao <leitao@debian.org>
*/
@@ -24,12 +26,35 @@
#include <linux/rtnetlink.h>
#include <linux/time_types.h>
#include <linux/vm_sockets.h>
+#include <linux/icmp.h>
+#include <netinet/in.h>
+#include <netinet/tcp.h>
+#include <arpa/inet.h>
#include <sys/socket.h>
+#include <linux/tls.h>
#include "kselftest_harness.h"
#ifndef AF_VSOCK
#define AF_VSOCK 40
#endif
+#ifndef SOL_RAW
+#define SOL_RAW 255
+#endif
+#ifndef ICMP_FILTER
+#define ICMP_FILTER 1
+#endif
+#ifndef IPV6_HDRINCL
+#define IPV6_HDRINCL 36
+#endif
+#ifndef IPV6_CHECKSUM
+#define IPV6_CHECKSUM 7
+#endif
+#ifndef SOL_TLS
+#define SOL_TLS 282
+#endif
+#ifndef TCP_ULP
+#define TCP_ULP 31
+#endif
/* ---------- netlink ---------- */
@@ -297,4 +322,315 @@ TEST_F(vsock, connect_timeout_old_exact)
ASSERT_EQ(sizeof(tv), optlen);
}
+/* ---------- raw (ipv4) ---------- */
+
+FIXTURE(raw)
+{
+ int fd;
+};
+
+FIXTURE_SETUP(raw)
+{
+ struct icmp_filter filt = { .data = 0xdeadbeef };
+
+ self->fd = socket(AF_INET, SOCK_RAW, IPPROTO_ICMP);
+ if (self->fd < 0)
+ SKIP(return, "SOCK_RAW/ICMP socket: %s", strerror(errno));
+
+ if (setsockopt(self->fd, SOL_RAW, ICMP_FILTER, &filt, sizeof(filt)) < 0)
+ SKIP(return, "set ICMP_FILTER: %s", strerror(errno));
+}
+
+FIXTURE_TEARDOWN(raw)
+{
+ if (self->fd >= 0)
+ close(self->fd);
+}
+
+TEST_F(raw, icmpfilter_exact)
+{
+ struct icmp_filter filt = {};
+ socklen_t optlen = sizeof(filt);
+
+ ASSERT_EQ(0, getsockopt(self->fd, SOL_RAW, ICMP_FILTER,
+ &filt, &optlen));
+ ASSERT_EQ(sizeof(filt), optlen);
+ ASSERT_EQ(0xdeadbeef, filt.data);
+}
+
+TEST_F(raw, icmpfilter_oversize_clamped)
+{
+ char buf[16] = {};
+ socklen_t optlen = sizeof(buf);
+
+ ASSERT_EQ(0, getsockopt(self->fd, SOL_RAW, ICMP_FILTER,
+ buf, &optlen));
+ ASSERT_EQ(sizeof(struct icmp_filter), optlen);
+}
+
+/* Unlike the int/u64 options above, ICMP_FILTER clamps the length down
+ * to the user buffer instead of returning EINVAL: a short buffer
+ * succeeds and reports the truncated length back via optlen.
+ */
+TEST_F(raw, icmpfilter_undersize_clamped)
+{
+ char buf[2] = {};
+ socklen_t optlen = sizeof(buf);
+
+ ASSERT_EQ(0, getsockopt(self->fd, SOL_RAW, ICMP_FILTER,
+ buf, &optlen));
+ ASSERT_EQ(sizeof(buf), optlen);
+}
+
+TEST_F(raw, icmpfilter_wrong_proto)
+{
+ struct icmp_filter filt;
+ socklen_t optlen = sizeof(filt);
+ int fd;
+
+ fd = socket(AF_INET, SOCK_RAW, IPPROTO_UDP);
+ if (fd < 0)
+ SKIP(return, "SOCK_RAW/UDP socket: %s", strerror(errno));
+
+ ASSERT_EQ(-1, getsockopt(fd, SOL_RAW, ICMP_FILTER, &filt, &optlen));
+ ASSERT_EQ(EOPNOTSUPP, errno);
+ close(fd);
+}
+
+TEST_F(raw, bad_optname)
+{
+ socklen_t optlen;
+ int val;
+
+ optlen = sizeof(val);
+
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_RAW, 0x7fff, &val, &optlen));
+ ASSERT_EQ(ENOPROTOOPT, errno);
+ ASSERT_EQ(sizeof(val), optlen);
+}
+
+/* ---------- raw (ipv6) ---------- */
+
+FIXTURE(rawv6)
+{
+ int fd;
+};
+
+FIXTURE_SETUP(rawv6)
+{
+ self->fd = socket(AF_INET6, SOCK_RAW, IPPROTO_UDP);
+ if (self->fd < 0)
+ SKIP(return, "SOCK_RAW/IPv6 socket: %s", strerror(errno));
+}
+
+FIXTURE_TEARDOWN(rawv6)
+{
+ if (self->fd >= 0)
+ close(self->fd);
+}
+
+TEST_F(rawv6, hdrincl_exact)
+{
+ socklen_t optlen;
+ int val = -1;
+
+ optlen = sizeof(val);
+
+ ASSERT_EQ(0, getsockopt(self->fd, IPPROTO_IPV6, IPV6_HDRINCL,
+ &val, &optlen));
+ ASSERT_EQ(sizeof(int), optlen);
+ ASSERT_TRUE(val == 0 || val == 1);
+}
+
+TEST_F(rawv6, hdrincl_oversize_clamped)
+{
+ char buf[16] = {};
+ socklen_t optlen = sizeof(buf);
+
+ ASSERT_EQ(0, getsockopt(self->fd, IPPROTO_IPV6, IPV6_HDRINCL,
+ buf, &optlen));
+ ASSERT_EQ(sizeof(int), optlen);
+}
+
+/* Raw int options clamp the reported length down to the user buffer
+ * instead of returning EINVAL on a short buffer.
+ */
+TEST_F(rawv6, hdrincl_undersize_clamped)
+{
+ socklen_t optlen = 2;
+ int val = 0;
+
+ ASSERT_EQ(0, getsockopt(self->fd, IPPROTO_IPV6, IPV6_HDRINCL,
+ &val, &optlen));
+ ASSERT_EQ(2, optlen);
+}
+
+TEST_F(rawv6, checksum_default)
+{
+ socklen_t optlen;
+ int val = 0;
+
+ optlen = sizeof(val);
+
+ /* A non-ICMPv6 raw socket has the checksum disabled, reported as -1. */
+ ASSERT_EQ(0, getsockopt(self->fd, IPPROTO_IPV6, IPV6_CHECKSUM,
+ &val, &optlen));
+ ASSERT_EQ(sizeof(int), optlen);
+ ASSERT_EQ(-1, val);
+}
+
+TEST_F(rawv6, bad_optname)
+{
+ socklen_t optlen;
+ int val;
+
+ optlen = sizeof(val);
+
+ /* SOL_RAW reaches do_rawv6_getsockopt() directly. */
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_RAW, 0x7fff, &val, &optlen));
+ ASSERT_EQ(ENOPROTOOPT, errno);
+ ASSERT_EQ(sizeof(val), optlen);
+}
+
+/* ---------- tls ---------- */
+
+FIXTURE(tls)
+{
+ int fd;
+ int sfd;
+};
+
+FIXTURE_SETUP(tls)
+{
+ struct sockaddr_in a = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ };
+ socklen_t alen = sizeof(a);
+ int lfd;
+
+ self->fd = -1;
+ self->sfd = -1;
+
+ lfd = socket(AF_INET, SOCK_STREAM, 0);
+ if (lfd < 0)
+ SKIP(return, "TCP socket: %s", strerror(errno));
+ if (bind(lfd, (struct sockaddr *)&a, sizeof(a)) || listen(lfd, 1) ||
+ getsockname(lfd, (struct sockaddr *)&a, &alen)) {
+ close(lfd);
+ SKIP(return, "listener setup: %s", strerror(errno));
+ }
+ self->fd = socket(AF_INET, SOCK_STREAM, 0);
+ if (self->fd < 0) {
+ close(lfd);
+ SKIP(return, "TCP socket: %s", strerror(errno));
+ }
+ if (connect(self->fd, (struct sockaddr *)&a, sizeof(a))) {
+ close(lfd);
+ SKIP(return, "connect: %s", strerror(errno));
+ }
+ self->sfd = accept(lfd, NULL, NULL);
+ close(lfd);
+ if (setsockopt(self->fd, IPPROTO_TCP, TCP_ULP, "tls", sizeof("tls")))
+ SKIP(return, "TCP_ULP=tls: %s (built without TLS?)",
+ strerror(errno));
+}
+
+FIXTURE_TEARDOWN(tls)
+{
+ if (self->fd >= 0)
+ close(self->fd);
+ if (self->sfd >= 0)
+ close(self->sfd);
+}
+
+/* do_tls_getsockopt_tx_zc(): fixed-size int, exact length required. */
+TEST_F(tls, tx_zerocopy_exact)
+{
+ socklen_t optlen = sizeof(int);
+ int val = -1;
+
+ ASSERT_EQ(0, getsockopt(self->fd, SOL_TLS, TLS_TX_ZEROCOPY_RO,
+ &val, &optlen));
+ ASSERT_EQ(sizeof(int), optlen);
+ ASSERT_TRUE(val == 0 || val == 1);
+}
+
+TEST_F(tls, tx_zerocopy_wrong_len)
+{
+ socklen_t optlen = 2;
+ int val;
+
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_TLS, TLS_TX_ZEROCOPY_RO,
+ &val, &optlen));
+ ASSERT_EQ(EINVAL, errno);
+}
+
+/* do_tls_getsockopt_conf(): NULL optval still yields EINVAL -- the
+ * converted code tests opt->iter_out.ubuf in place of optval.
+ */
+TEST_F(tls, conf_null_optval)
+{
+ socklen_t optlen = 64;
+
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_TLS, TLS_TX, NULL, &optlen));
+ ASSERT_EQ(EINVAL, errno);
+}
+
+TEST_F(tls, conf_short)
+{
+ socklen_t optlen = 2;
+ char buf[2];
+
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_TLS, TLS_TX, buf, &optlen));
+ ASSERT_EQ(EINVAL, errno);
+}
+
+/* TLS_TX before crypto is set reports not-ready. */
+TEST_F(tls, conf_not_ready)
+{
+ struct tls_crypto_info info;
+ socklen_t optlen = sizeof(info);
+
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_TLS, TLS_TX, &info, &optlen));
+ ASSERT_EQ(EBUSY, errno);
+}
+
+/* Set TX crypto, then read it back at the base and full sizes, exercising
+ * both copy_to_iter() branches. SKIP if AES-GCM is unavailable.
+ */
+TEST_F(tls, conf_crypto_roundtrip)
+{
+ struct tls12_crypto_info_aes_gcm_128 tx = {
+ .info.version = TLS_1_2_VERSION,
+ .info.cipher_type = TLS_CIPHER_AES_GCM_128,
+ };
+ struct tls12_crypto_info_aes_gcm_128 full;
+ struct tls_crypto_info base;
+ socklen_t optlen;
+
+ if (setsockopt(self->fd, SOL_TLS, TLS_TX, &tx, sizeof(tx)))
+ SKIP(return, "set TLS_TX aes_gcm_128: %s", strerror(errno));
+
+ optlen = sizeof(base);
+ ASSERT_EQ(0, getsockopt(self->fd, SOL_TLS, TLS_TX, &base, &optlen));
+ ASSERT_EQ(sizeof(base), optlen);
+ ASSERT_EQ(TLS_1_2_VERSION, base.version);
+ ASSERT_EQ(TLS_CIPHER_AES_GCM_128, base.cipher_type);
+
+ optlen = sizeof(full);
+ ASSERT_EQ(0, getsockopt(self->fd, SOL_TLS, TLS_TX, &full, &optlen));
+ ASSERT_EQ(sizeof(full), optlen);
+ ASSERT_EQ(TLS_CIPHER_AES_GCM_128, full.info.cipher_type);
+}
+
+TEST_F(tls, bad_optname)
+{
+ socklen_t optlen = sizeof(int);
+ int val;
+
+ ASSERT_EQ(-1, getsockopt(self->fd, SOL_TLS, 0x7fff, &val, &optlen));
+ ASSERT_EQ(ENOPROTOOPT, errno);
+}
+
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/net/hsr/Makefile b/tools/testing/selftests/net/hsr/Makefile
index 31fb9326cf53..2150e487ac7d 100644
--- a/tools/testing/selftests/net/hsr/Makefile
+++ b/tools/testing/selftests/net/hsr/Makefile
@@ -4,6 +4,7 @@ top_srcdir = ../../../../..
TEST_PROGS := \
hsr_ping.sh \
+ hsr_prp_redbox.sh \
hsr_redbox.sh \
link_faults.sh \
prp_ping.sh \
diff --git a/tools/testing/selftests/net/hsr/hsr_prp_redbox.sh b/tools/testing/selftests/net/hsr/hsr_prp_redbox.sh
new file mode 100755
index 000000000000..479c892225b1
--- /dev/null
+++ b/tools/testing/selftests/net/hsr/hsr_prp_redbox.sh
@@ -0,0 +1,99 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Test a PRP RedBox (PRP-SAN): a SAN that sits behind the interlink port must
+# reach, and be reached by, a peer DANP on the PRP network with its own MAC
+# preserved on the wire, and the RedBox must announce the SAN with a RedBox-MAC
+# TLV (terminated by an EOT marker) in its PRP supervision frames.
+#
+# RB PRP RedBox: prp0 over rb_a/rb_b (LAN A/B) + interlink rb_il
+# PEER peer DANP : prp0 over pe_a/pe_b, 100.64.0.2
+# SAN SAN : san_il, own MAC, 100.64.0.51 (behind the interlink)
+
+ipv6=false
+
+source ./hsr_common.sh
+
+check_prerequisites
+
+if ! command -v tcpdump >/dev/null 2>&1; then
+ echo "SKIP: This test requires tcpdump"
+ exit $ksft_skip
+fi
+
+if ! ip link help hsr 2>&1 | grep -q interlink; then
+ echo "SKIP: iproute2 too old (no hsr interlink support)"
+ exit $ksft_skip
+fi
+
+setup_ns RB PEER SAN
+trap 'cleanup_ns "$RB" "$PEER" "$SAN"' EXIT
+
+ip link add rb_a netns "$RB" type veth peer name pe_a netns "$PEER"
+ip link add rb_b netns "$RB" type veth peer name pe_b netns "$PEER"
+ip link add rb_il netns "$RB" type veth peer name san_il netns "$SAN"
+
+ip -n "$RB" link set rb_a up
+ip -n "$RB" link set rb_b up
+ip -n "$RB" link set rb_il up
+ip -n "$PEER" link set pe_a up
+ip -n "$PEER" link set pe_b up
+ip -n "$SAN" link set san_il up
+ip -n "$SAN" addr add 100.64.0.51/24 dev san_il
+
+# Feature gate: PRP interlink (RedBox) creation. A kernel without PRP RedBox
+# support rejects this with -EINVAL, so SKIP rather than FAIL.
+if ! ip -n "$RB" link add name prp0 type hsr slave1 rb_a slave2 rb_b \
+ interlink rb_il proto 1 2>/dev/null; then
+ echo "SKIP: kernel without PRP RedBox (interlink) support"
+ exit $ksft_skip
+fi
+ip -n "$RB" link set prp0 up
+ip -n "$PEER" link add name prp0 type hsr slave1 pe_a slave2 pe_b proto 1
+ip -n "$PEER" link set prp0 up
+ip -n "$PEER" addr add 100.64.0.2/24 dev prp0
+sleep 1
+
+san_mac=$(ip -n "$SAN" -br link show san_il | awk '{print $3}')
+rb_mac=$(ip -n "$RB" -br link show rb_il | awk '{print $3}')
+
+# Bidirectional unicast across the interlink.
+do_ping "$PEER" 100.64.0.51
+do_ping "$SAN" 100.64.0.2
+stop_if_error "PRP RedBox bidirectional unicast failed"
+
+# The SAN source MAC must be preserved on the PRP network, not laundered to the
+# RedBox MAC: the peer resolves the SAN IP to the SAN's own MAC.
+neigh=$(ip -n "$PEER" neigh show 100.64.0.51 | awk '{print $5}')
+if [ "$neigh" != "$san_mac" ]; then
+ echo "SAN MAC preservation [ FAIL ]: peer resolved 100.64.0.51 to" \
+ "'$neigh', expected $san_mac" 1>&2
+ ret=1
+fi
+stop_if_error "SAN MAC not preserved on the PRP network"
+
+# The proxy-announce supervision frame must carry, in order, the life-check TLV
+# (type 0x14, len 6) + MacAddressA == SAN MAC + the RedBox-MAC TLV (type 0x1e,
+# len 6) + MacAddressRedBox == RedBox MAC + the EOT marker (0x0000).
+ip netns exec "$SAN" ping -i 0.2 -q 100.64.0.2 >/dev/null 2>&1 &
+ping_pid=$!
+cap=$(ip netns exec "$PEER" timeout 5 tcpdump -i pe_a -nn -x \
+ "ether proto 0x88fb and ether src $rb_mac" 2>/dev/null || true)
+kill "$ping_pid" 2>/dev/null || true
+wait "$ping_pid" 2>/dev/null || true
+
+san_hex=$(echo "$san_mac" | tr -d ':')
+rb_hex=$(echo "$rb_mac" | tr -d ':')
+# Reassemble contiguous frame hex: drop the "0x0010:" offset labels and spaces.
+frame_hex=$(echo "$cap" | awk '/^[[:space:]]*0x[0-9a-f]+:/ {
+ sub(/^[[:space:]]*0x[0-9a-f]+:[[:space:]]*/, "");
+ gsub(/ /, ""); printf "%s", $0 }')
+if ! echo "$frame_hex" | grep -q "1406${san_hex}1e06${rb_hex}0000"; then
+ echo "supervision RedBox-MAC TLV [ FAIL ]: missing SAN MAC, Type-30" \
+ "payload, or EOT" 1>&2
+ ret=1
+fi
+stop_if_error "PRP RedBox supervision RedBox-MAC TLV/EOT check failed"
+
+echo "INFO: PRP RedBox (PRP-SAN) conformance checks passed"
+exit $ret
diff --git a/tools/testing/selftests/net/ipv6_flowlabel.sh b/tools/testing/selftests/net/ipv6_flowlabel.sh
index cee95e252bee..5d1b5464c54c 100755
--- a/tools/testing/selftests/net/ipv6_flowlabel.sh
+++ b/tools/testing/selftests/net/ipv6_flowlabel.sh
@@ -7,9 +7,6 @@
set -e
-echo "TEST management"
-./in_netns.sh ./ipv6_flowlabel_mgr
-
echo "TEST datapath"
./in_netns.sh \
sh -c 'sysctl -q -w net.ipv6.auto_flowlabels=0 && ./ipv6_flowlabel -l 1'
diff --git a/tools/testing/selftests/net/ipv6_flowlabel_mgr.c b/tools/testing/selftests/net/ipv6_flowlabel_mgr.c
index af95b48acea9..072fb3a9b121 100644
--- a/tools/testing/selftests/net/ipv6_flowlabel_mgr.c
+++ b/tools/testing/selftests/net/ipv6_flowlabel_mgr.c
@@ -6,45 +6,36 @@
#include <arpa/inet.h>
#include <error.h>
#include <errno.h>
+#include <fcntl.h>
#include <limits.h>
#include <linux/in6.h>
+#include <net/if.h>
+#include <sched.h>
#include <stdbool.h>
#include <stdio.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
+#include <sys/ioctl.h>
#include <sys/socket.h>
#include <sys/stat.h>
#include <sys/time.h>
#include <sys/types.h>
#include <sys/wait.h>
#include <unistd.h>
+#include "kselftest_harness.h"
/* uapi/glibc weirdness may leave this undefined */
#ifndef IPV6_FLOWLABEL_MGR
#define IPV6_FLOWLABEL_MGR 32
#endif
+#ifndef IPV6_FLOWINFO_SEND
+#define IPV6_FLOWINFO_SEND 33
+#endif
/* from net/ipv6/ip6_flowlabel.c */
#define FL_MIN_LINGER 6
-#define explain(x) \
- do { if (cfg_verbose) fprintf(stderr, " " x "\n"); } while (0)
-
-#define __expect(x) \
- do { \
- if (!(x)) \
- fprintf(stderr, "[OK] " #x "\n"); \
- else \
- error(1, 0, "[ERR] " #x " (line %d)", __LINE__); \
- } while (0)
-
-#define expect_pass(x) __expect(x)
-#define expect_fail(x) __expect(!(x))
-
-static bool cfg_long_running;
-static bool cfg_verbose;
-
static int flowlabel_get(int fd, uint32_t label, uint8_t share, uint16_t flags)
{
struct in6_flowlabel_req req = {
@@ -71,129 +62,484 @@ static int flowlabel_put(int fd, uint32_t label)
return setsockopt(fd, SOL_IPV6, IPV6_FLOWLABEL_MGR, &req, sizeof(req));
}
-static void run_tests(int fd)
+static int flowlabel_renew(int fd, uint32_t label, uint8_t share,
+ uint16_t linger)
{
- int wstatus;
- pid_t pid;
+ struct in6_flowlabel_req req = {
+ .flr_action = IPV6_FL_A_RENEW,
+ .flr_label = htonl(label),
+ .flr_share = share,
+ .flr_linger = linger,
+ };
+
+ return setsockopt(fd, SOL_IPV6, IPV6_FLOWLABEL_MGR, &req, sizeof(req));
+}
+
+static struct sockaddr_in6 loopback_addr(void)
+{
+ struct sockaddr_in6 addr = {
+ .sin6_family = AF_INET6,
+ .sin6_addr = IN6ADDR_LOOPBACK_INIT,
+ .sin6_port = htons(8888),
+ };
+
+ return addr;
+}
+
+static int tcp_listen(void)
+{
+ struct sockaddr_in6 addr = loopback_addr();
+ const int one = 1;
+ int fd;
+
+ fd = socket(PF_INET6, SOCK_STREAM, 0);
+ if (fd == -1)
+ error(1, errno, "socket listener");
+ if (setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &one, sizeof(one)))
+ error(1, errno, "setsockopt SO_REUSEADDR");
+ if (bind(fd, (void *)&addr, sizeof(addr)))
+ error(1, errno, "bind");
+ if (listen(fd, 1))
+ error(1, errno, "listen");
+
+ return fd;
+}
+
+static void tcp_connect(int listener, uint32_t flowlabel,
+ int *client, int *accepted)
+{
+ struct sockaddr_in6 addr = loopback_addr();
+ const int one = 1;
+ int cfd, afd;
+
+ cfd = socket(PF_INET6, SOCK_STREAM, 0);
+ if (cfd == -1)
+ error(1, errno, "socket client");
+
+ if (flowlabel_get(cfd, flowlabel, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE))
+ error(1, errno, "flowlabel_get");
+ if (setsockopt(cfd, SOL_IPV6, IPV6_FLOWINFO_SEND, &one, sizeof(one)))
+ error(1, errno, "setsockopt flowinfo_send");
+ addr.sin6_flowinfo = htonl(flowlabel);
+
+ if (connect(cfd, (void *)&addr, sizeof(addr)))
+ error(1, errno, "connect");
+
+ afd = accept(listener, NULL, NULL);
+ if (afd == -1)
+ error(1, errno, "accept");
+
+ if (flowlabel_put(cfd, flowlabel))
+ error(1, errno, "flowlabel_put");
+
+ *client = cfd;
+ *accepted = afd;
+}
+
+static int bringup_loopback(void)
+{
+ struct ifreq ifr = {
+ .ifr_name = "lo"
+ };
+ int fd;
+
+ fd = socket(AF_LOCAL, SOCK_STREAM, 0);
+ if (fd < 0)
+ return -1;
+
+ if (ioctl(fd, SIOCGIFFLAGS, &ifr) < 0)
+ goto err;
+
+ ifr.ifr_flags = ifr.ifr_flags | IFF_UP;
+
+ if (ioctl(fd, SIOCSIFFLAGS, &ifr) < 0)
+ goto err;
+
+ close(fd);
+ return 0;
- explain("cannot get non-existent label");
- expect_fail(flowlabel_get(fd, 1, IPV6_FL_S_ANY, 0));
-
- explain("cannot put non-existent label");
- expect_fail(flowlabel_put(fd, 1));
-
- explain("cannot create label greater than 20 bits");
- expect_fail(flowlabel_get(fd, 0x1FFFFF, IPV6_FL_S_ANY,
- IPV6_FL_F_CREATE));
-
- explain("create a new label (FL_F_CREATE)");
- expect_pass(flowlabel_get(fd, 1, IPV6_FL_S_ANY, IPV6_FL_F_CREATE));
- explain("can get the label (without FL_F_CREATE)");
- expect_pass(flowlabel_get(fd, 1, IPV6_FL_S_ANY, 0));
- explain("can get it again with create flag set, too");
- expect_pass(flowlabel_get(fd, 1, IPV6_FL_S_ANY, IPV6_FL_F_CREATE));
- explain("cannot get it again with the exclusive (FL_FL_EXCL) flag");
- expect_fail(flowlabel_get(fd, 1, IPV6_FL_S_ANY,
- IPV6_FL_F_CREATE | IPV6_FL_F_EXCL));
- explain("can now put exactly three references");
- expect_pass(flowlabel_put(fd, 1));
- expect_pass(flowlabel_put(fd, 1));
- expect_pass(flowlabel_put(fd, 1));
- expect_fail(flowlabel_put(fd, 1));
-
- explain("create a new exclusive label (FL_S_EXCL)");
- expect_pass(flowlabel_get(fd, 2, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE));
- explain("cannot get it again in non-exclusive mode");
- expect_fail(flowlabel_get(fd, 2, IPV6_FL_S_ANY, IPV6_FL_F_CREATE));
- explain("cannot get it again in exclusive mode either");
- expect_fail(flowlabel_get(fd, 2, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE));
- expect_pass(flowlabel_put(fd, 2));
-
- if (cfg_long_running) {
- explain("cannot reuse the label, due to linger");
- expect_fail(flowlabel_get(fd, 2, IPV6_FL_S_ANY,
- IPV6_FL_F_CREATE));
- explain("after sleep, can reuse");
- sleep(FL_MIN_LINGER * 2 + 1);
- expect_pass(flowlabel_get(fd, 2, IPV6_FL_S_ANY,
- IPV6_FL_F_CREATE));
+err:
+ close(fd);
+ return -1;
+}
+
+FIXTURE(flowlabel) {};
+
+FIXTURE_SETUP(flowlabel)
+{
+ int ret;
+
+ ret = unshare(CLONE_NEWNET);
+ ASSERT_EQ(ret, 0) {
+ TH_LOG("unshare(CLONE_NEWNET) failed: %s", strerror(errno));
}
- explain("create a new user-private label (FL_S_USER)");
- expect_pass(flowlabel_get(fd, 3, IPV6_FL_S_USER, IPV6_FL_F_CREATE));
- explain("cannot get it again in non-exclusive mode");
- expect_fail(flowlabel_get(fd, 3, IPV6_FL_S_ANY, 0));
- explain("cannot get it again in exclusive mode");
- expect_fail(flowlabel_get(fd, 3, IPV6_FL_S_EXCL, 0));
- explain("can get it again in user mode");
- expect_pass(flowlabel_get(fd, 3, IPV6_FL_S_USER, 0));
- explain("child process can get it too, but not after setuid(nobody)");
+ ret = bringup_loopback();
+ ASSERT_EQ(ret, 0) TH_LOG("Failed to bring up loopback interface");
+}
+
+FIXTURE_TEARDOWN(flowlabel)
+{
+}
+
+TEST_F(flowlabel, cannot_get_non_existent_label)
+{
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 9, IPV6_FL_S_ANY, 0);
+ EXPECT_TRUE(err) TH_LOG("expected get of a non-existent label to fail");
+ EXPECT_EQ(ENOENT, errno) TH_LOG("expected ENOENT, got %d", errno);
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, cannot_put_non_existent_label)
+{
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_put(fd, 10);
+ EXPECT_TRUE(err) TH_LOG("expected put of a non-existent label to fail");
+ EXPECT_EQ(ESRCH, errno) TH_LOG("expected ESRCH, got %d", errno);
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, cannot_create_label_greater_than_20_bits)
+{
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 0x1FFFFF, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(err) TH_LOG("expected label > 20 bits to be rejected");
+ EXPECT_EQ(EINVAL, errno) TH_LOG("expected EINVAL, got %d", errno);
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, can_create_and_get_and_put_labels)
+{
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 1, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err) TH_LOG("failed to create label (FL_F_CREATE)");
+
+ err = flowlabel_get(fd, 1, IPV6_FL_S_ANY, 0);
+ EXPECT_TRUE(!err) TH_LOG("failed to get the label without FL_F_CREATE");
+
+ err = flowlabel_get(fd, 1, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to get it again with create flag set, too");
+
+ err = flowlabel_get(fd, 1, IPV6_FL_S_ANY,
+ IPV6_FL_F_CREATE | IPV6_FL_F_EXCL);
+ EXPECT_TRUE(err)
+ TH_LOG("expected FL_F_EXCL to reject existing label");
+ EXPECT_EQ(EEXIST, errno) TH_LOG("expected EEXIST, got %d", errno);
+
+ err = flowlabel_put(fd, 1);
+ EXPECT_TRUE(!err) TH_LOG("failed to put first reference");
+ err = flowlabel_put(fd, 1);
+ EXPECT_TRUE(!err) TH_LOG("failed to put second reference");
+ err = flowlabel_put(fd, 1);
+ EXPECT_TRUE(!err) TH_LOG("failed to put third reference");
+ err = flowlabel_put(fd, 1);
+ EXPECT_TRUE(err)
+ TH_LOG("expected fourth put to fail, no references left");
+ EXPECT_EQ(ESRCH, errno) TH_LOG("expected ESRCH, got %d", errno);
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, exclusive_label_share)
+{
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 2, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to create a new exclusive label (FL_S_EXCL)");
+
+ err = flowlabel_get(fd, 2, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(err) TH_LOG("expected reuse in non-exclusive mode to fail");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
+
+ err = flowlabel_get(fd, 2, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(err) TH_LOG("expected reuse in exclusive mode to fail too");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
+
+ err = flowlabel_put(fd, 2);
+ EXPECT_TRUE(!err) TH_LOG("failed to put the exclusive label");
+
+ err = flowlabel_get(fd, 2, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(err) TH_LOG("expected reuse to fail, due to linger");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
+
+ sleep(FL_MIN_LINGER * 2 + 1);
+
+ err = flowlabel_get(fd, 2, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err) TH_LOG("expected reuse to succeed after linger");
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, user_private_label_share)
+{
+ int fd, err, wstatus;
+ pid_t pid;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 3, IPV6_FL_S_USER, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to create a new user-private label (FL_S_USER)");
+
+ err = flowlabel_get(fd, 3, IPV6_FL_S_ANY, 0);
+ EXPECT_TRUE(err) TH_LOG("expected get in non-exclusive mode to fail");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
+
+ err = flowlabel_get(fd, 3, IPV6_FL_S_EXCL, 0);
+ EXPECT_TRUE(err) TH_LOG("expected get in exclusive mode to fail");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
+
+ err = flowlabel_get(fd, 3, IPV6_FL_S_USER, 0);
+ EXPECT_TRUE(!err) TH_LOG("failed to get it again in user mode");
+
pid = fork();
- if (pid == -1)
- error(1, errno, "fork");
+ ASSERT_NE(-1, pid) TH_LOG("fork failed");
if (!pid) {
- expect_pass(flowlabel_get(fd, 3, IPV6_FL_S_USER, 0));
+ err = flowlabel_get(fd, 3, IPV6_FL_S_USER, 0);
+ EXPECT_TRUE(!err)
+ TH_LOG("child failed to get the user-private label");
+
if (setuid(USHRT_MAX))
- fprintf(stderr, "[INFO] skip setuid child test\n");
- else
- expect_fail(flowlabel_get(fd, 3, IPV6_FL_S_USER, 0));
+ exit(KSFT_SKIP);
+
+ err = flowlabel_get(fd, 3, IPV6_FL_S_USER, 0);
+ EXPECT_TRUE(err)
+ TH_LOG("child unexpectedly got label after setuid");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
exit(0);
}
- if (wait(&wstatus) == -1)
- error(1, errno, "wait");
- if (!WIFEXITED(wstatus) || WEXITSTATUS(wstatus) != 0)
- error(1, errno, "wait: unexpected child result");
-
- explain("create a new process-private label (FL_S_PROCESS)");
- expect_pass(flowlabel_get(fd, 4, IPV6_FL_S_PROCESS, IPV6_FL_F_CREATE));
- explain("can get it again");
- expect_pass(flowlabel_get(fd, 4, IPV6_FL_S_PROCESS, 0));
- explain("child process cannot can get it");
+ ASSERT_EQ(pid, wait(&wstatus)) TH_LOG("wait failed");
+ ASSERT_TRUE(WIFEXITED(wstatus)) TH_LOG("child did not exit normally");
+ if (WEXITSTATUS(wstatus) == KSFT_SKIP)
+ SKIP(return,
+ "setuid(USHRT_MAX) unavailable (no CAP_SETUID or uid unmapped)");
+ EXPECT_EQ(0, WEXITSTATUS(wstatus))
+ TH_LOG("child reported unexpected result");
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, process_private_label_share)
+{
+ int fd, err, wstatus;
+ pid_t pid;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 4, IPV6_FL_S_PROCESS, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to create a new process-private label");
+
+ err = flowlabel_get(fd, 4, IPV6_FL_S_PROCESS, 0);
+ EXPECT_TRUE(!err) TH_LOG("failed to get it again");
+
pid = fork();
- if (pid == -1)
- error(1, errno, "fork");
+ ASSERT_NE(-1, pid) TH_LOG("fork failed");
if (!pid) {
- expect_fail(flowlabel_get(fd, 4, IPV6_FL_S_PROCESS, 0));
+ err = flowlabel_get(fd, 4, IPV6_FL_S_PROCESS, 0);
+ EXPECT_TRUE(err)
+ TH_LOG("child unexpectedly got process-private label");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
exit(0);
}
- if (wait(&wstatus) == -1)
- error(1, errno, "wait");
- if (!WIFEXITED(wstatus) || WEXITSTATUS(wstatus) != 0)
- error(1, errno, "wait: unexpected child result");
-}
-
-static void parse_opts(int argc, char **argv)
-{
- int c;
-
- while ((c = getopt(argc, argv, "lv")) != -1) {
- switch (c) {
- case 'l':
- cfg_long_running = true;
- break;
- case 'v':
- cfg_verbose = true;
- break;
- default:
- error(1, 0, "%s: parse error", argv[0]);
- }
- }
+ ASSERT_EQ(pid, wait(&wstatus)) TH_LOG("wait failed");
+ ASSERT_TRUE(WIFEXITED(wstatus)) TH_LOG("child did not exit normally");
+ EXPECT_EQ(0, WEXITSTATUS(wstatus))
+ TH_LOG("child reported unexpected result");
+
+ EXPECT_EQ(0, close(fd));
}
-int main(int argc, char **argv)
+TEST_F(flowlabel, cannot_renew_non_existent_label)
{
- int fd;
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_renew(fd, 5, IPV6_FL_S_EXCL,
+ 2 * (FL_MIN_LINGER * 2 + 1));
+ EXPECT_TRUE(err)
+ TH_LOG("expected renew of a non-existent label to fail");
+ EXPECT_EQ(ESRCH, errno) TH_LOG("expected ESRCH, got %d", errno);
- parse_opts(argc, argv);
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, can_renew_existing_label)
+{
+ int fd, err;
fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 5, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to create a new label for renew validation");
+
+ err = flowlabel_renew(fd, 5, IPV6_FL_S_EXCL,
+ 2 * (FL_MIN_LINGER * 2 + 1));
+ EXPECT_TRUE(!err) TH_LOG("failed to renew an existing valid label");
+
+ err = flowlabel_put(fd, 5);
+ EXPECT_TRUE(!err) TH_LOG("failed to put the label");
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, renew_label_linger)
+{
+ /* RENEW must extend a label's linger period: putting a renewed
+ * label and waiting out its original linger time must not be
+ * enough to allow the label to be recreated.
+ */
+ int fd, err;
+
+ fd = socket(PF_INET6, SOCK_DGRAM, 0);
+ ASSERT_GE(fd, 0) TH_LOG("socket failed");
+
+ err = flowlabel_get(fd, 6, IPV6_FL_S_EXCL, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to create label with FL_MIN_LINGER linger time");
+
+ err = flowlabel_renew(fd, 6, IPV6_FL_S_EXCL,
+ 2 * (FL_MIN_LINGER * 2 + 1));
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to renew the label to increase its linger time");
+
+ err = flowlabel_put(fd, 6);
+ EXPECT_TRUE(!err) TH_LOG("failed to put the label");
+
+ sleep(FL_MIN_LINGER * 2 + 1);
+
+ err = flowlabel_get(fd, 6, IPV6_FL_S_ANY, IPV6_FL_F_CREATE);
+ EXPECT_TRUE(err)
+ TH_LOG("expected reuse to fail, new linger time not over yet");
+ EXPECT_EQ(EPERM, errno) TH_LOG("expected EPERM, got %d", errno);
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_F(flowlabel, remote_flag)
+{
+ /* The REMOTE flag, used for getsockopt, is expected to retrieve the
+ * label from the latest received header.
+ */
+ struct in6_flowlabel_req freq = {
+ .flr_action = IPV6_FL_A_GET,
+ .flr_flags = IPV6_FL_F_REMOTE,
+ };
+ socklen_t freq_len = sizeof(freq);
+ int listener, cfd, afd, err;
+
+ listener = tcp_listen();
+ tcp_connect(listener, 7, &cfd, &afd);
+
+ err = getsockopt(afd, SOL_IPV6, IPV6_FLOWLABEL_MGR, &freq, &freq_len);
+ EXPECT_TRUE(!err) TH_LOG("getsockopt with IPV6_FL_F_REMOTE failed");
+ EXPECT_EQ(7, ntohl(freq.flr_label))
+ TH_LOG("unexpected remote flow label");
+
+ EXPECT_EQ(0, close(afd));
+ EXPECT_EQ(0, close(cfd));
+ EXPECT_EQ(0, close(listener));
+}
+
+static bool disable_flowlabel_consistency(void)
+{
+ int fd;
+
+ fd = open("/proc/sys/net/ipv6/flowlabel_consistency", O_WRONLY);
if (fd == -1)
- error(1, errno, "socket");
+ return false;
- run_tests(fd);
+ if (write(fd, "0", 1) != 1) {
+ close(fd);
+ return false;
+ }
+ close(fd);
- if (close(fd))
- error(1, errno, "close");
+ return true;
+}
- return 0;
+TEST_F(flowlabel, reflect_flag)
+{
+ /* The REFLECT flag acts as a trigger to the REPFLOW bit. When REPFLOW
+ * is triggered for a socket, it adopts the label received from the
+ * connected socket.
+ */
+ struct in6_flowlabel_req reflect_on = {
+ .flr_action = IPV6_FL_A_GET,
+ .flr_flags = IPV6_FL_F_REFLECT,
+ };
+ struct in6_flowlabel_req reflect_query = {
+ .flr_action = IPV6_FL_A_GET,
+ };
+ struct in6_flowlabel_req reflect_off = {
+ .flr_action = IPV6_FL_A_PUT,
+ .flr_flags = IPV6_FL_F_REFLECT,
+ };
+ socklen_t reflect_query_len = sizeof(reflect_query);
+ int listener, cfd, afd, err;
+
+ if (!disable_flowlabel_consistency())
+ SKIP(return,
+ "cannot disable net.ipv6.flowlabel_consistency");
+
+ listener = tcp_listen();
+ err = setsockopt(listener, SOL_IPV6, IPV6_FLOWLABEL_MGR,
+ &reflect_on, sizeof(reflect_on));
+ EXPECT_TRUE(!err) TH_LOG("failed to enable REFLECT on the listener");
+
+ tcp_connect(listener, 8, &cfd, &afd);
+
+ err = getsockopt(afd, SOL_IPV6, IPV6_FLOWLABEL_MGR,
+ &reflect_query, &reflect_query_len);
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to query the accepted socket's outgoing label");
+ EXPECT_EQ(8, ntohl(reflect_query.flr_label))
+ TH_LOG("accepted socket did not reflect client's label");
+
+ err = setsockopt(afd, SOL_IPV6, IPV6_FLOWLABEL_MGR,
+ &reflect_off, sizeof(reflect_off));
+ EXPECT_TRUE(!err)
+ TH_LOG("failed to disable REFLECT on the accepted socket");
+
+ err = setsockopt(afd, SOL_IPV6, IPV6_FLOWLABEL_MGR,
+ &reflect_off, sizeof(reflect_off));
+ EXPECT_TRUE(err) TH_LOG("expected disabling REFLECT twice to fail");
+ EXPECT_EQ(ESRCH, errno) TH_LOG("expected ESRCH, got %d", errno);
+
+ EXPECT_EQ(0, close(afd));
+ EXPECT_EQ(0, close(cfd));
+ EXPECT_EQ(0, close(listener));
}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/net/lib/py/__init__.py b/tools/testing/selftests/net/lib/py/__init__.py
index e58bdbdc58ee..34935886b6ad 100644
--- a/tools/testing/selftests/net/lib/py/__init__.py
+++ b/tools/testing/selftests/net/lib/py/__init__.py
@@ -17,7 +17,7 @@ from .utils import CmdExitFailure, fd_read_timeout, cmd, bkg, defer, \
wait_file, tool, tc
from .bpf import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from .ynl import NlError, NlctrlFamily, YnlFamily, \
- EthtoolFamily, NetdevFamily, RtnlFamily, RtnlAddrFamily
+ EthtoolFamily, NetdevFamily, RtnlFamily, RtnlAddrFamily, RtnlRouteFamily
from .ynl import NetshaperFamily, DevlinkFamily, PSPFamily, Netlink
__all__ = ["KSRC",
@@ -34,4 +34,4 @@ __all__ = ["KSRC",
"NetdevSim", "NetdevSimDev",
"NetshaperFamily", "DevlinkFamily", "PSPFamily", "NlError",
"YnlFamily", "EthtoolFamily", "NetdevFamily", "RtnlFamily",
- "NlctrlFamily", "RtnlAddrFamily", "Netlink"]
+ "NlctrlFamily", "RtnlAddrFamily", "RtnlRouteFamily", "Netlink"]
diff --git a/tools/testing/selftests/net/lib/py/ynl.py b/tools/testing/selftests/net/lib/py/ynl.py
index 2e567062aa6c..08deff756f29 100644
--- a/tools/testing/selftests/net/lib/py/ynl.py
+++ b/tools/testing/selftests/net/lib/py/ynl.py
@@ -29,7 +29,7 @@ except ModuleNotFoundError as e:
__all__ = [
"NlError", "NlPolicy", "Netlink", "YnlFamily", "SPEC_PATH",
- "EthtoolFamily", "RtnlFamily", "RtnlAddrFamily",
+ "EthtoolFamily", "RtnlFamily", "RtnlAddrFamily", "RtnlRouteFamily",
"NetdevFamily", "NetshaperFamily", "NlctrlFamily", "DevlinkFamily",
"PSPFamily",
]
@@ -54,6 +54,11 @@ class RtnlAddrFamily(YnlFamily):
super().__init__((SPEC_PATH / Path('rt-addr.yaml')).as_posix(),
schema='', recv_size=recv_size)
+class RtnlRouteFamily(YnlFamily):
+ def __init__(self, recv_size=0):
+ super().__init__((SPEC_PATH / Path('rt-route.yaml')).as_posix(),
+ schema='', recv_size=recv_size)
+
class NetdevFamily(YnlFamily):
def __init__(self, recv_size=0):
super().__init__((SPEC_PATH / Path('netdev.yaml')).as_posix(),
diff --git a/tools/testing/selftests/net/mptcp/mptcp_connect.c b/tools/testing/selftests/net/mptcp/mptcp_connect.c
index cbe573c4ab3a..178d98d91fea 100644
--- a/tools/testing/selftests/net/mptcp/mptcp_connect.c
+++ b/tools/testing/selftests/net/mptcp/mptcp_connect.c
@@ -381,6 +381,9 @@ static int sock_connect_mptcp(const char * const remoteaddr,
hints.ai_family = pf;
+ /* Keep the resolved address alive for the whole execution: it is
+ * used again when reconnecting, and will be released at exit time.
+ */
xgetaddrinfo(remoteaddr, port, &hints, &addr);
for (a = addr; a; a = a->ai_next) {
sock = socket(a->ai_family, a->ai_socktype, proto);
@@ -421,7 +424,6 @@ static int sock_connect_mptcp(const char * const remoteaddr,
sock = -1;
}
- freeaddrinfo(addr);
if (sock != -1)
SOCK_TEST_TCPULP(sock, proto);
return sock;
@@ -1264,7 +1266,7 @@ static void apply_cmsg_types(int fd, const struct cfg_cmsg_types *cmsg)
static void parse_cmsg_types(const char *type)
{
- char *next = strchr(type, ',');
+ const char *next = strchr(type, ',');
unsigned int len = 0;
cfg_cmsg_types.cmsg_enabled = 1;
@@ -1292,7 +1294,7 @@ static void parse_cmsg_types(const char *type)
static void parse_setsock_options(const char *name)
{
- char *next = strchr(name, ',');
+ const char *next = strchr(name, ',');
unsigned int len = 0;
if (next) {
diff --git a/tools/testing/selftests/net/mptcp/mptcp_connect.sh b/tools/testing/selftests/net/mptcp/mptcp_connect.sh
index d158678fa6ab..7a2a851fa0ad 100755
--- a/tools/testing/selftests/net/mptcp/mptcp_connect.sh
+++ b/tools/testing/selftests/net/mptcp/mptcp_connect.sh
@@ -212,8 +212,8 @@ if $checksum; then
fi
if $capture; then
- rndh="${ns1:4}"
- mptcp_lib_pr_info "Packet capture files will have this prefix: ${rndh}-"
+ capprefix="mptcp_connect-${ns1:4}"
+ mptcp_lib_pr_info "pcap will have this prefix: ${capprefix}-"
fi
set_ethtool_flags() {
@@ -372,8 +372,8 @@ do_transfer()
capuser="-Z $SUDO_USER"
fi
- local capfile="${rndh}-${connector_ns:0:3}-${listener_ns:0:3}-${cl_proto}-${srv_proto}-${connect_addr}-${port}"
- local capopt="-i any -s 65535 -B 32768 ${capuser}"
+ local capfile="${capprefix}-${connector_ns:0:3}-${listener_ns:0:3}-${cl_proto}-${srv_proto}-${connect_addr}-${port}"
+ local capopt="-i any -s 128 -B 32768 ${capuser}"
ip netns exec ${listener_ns} tcpdump ${capopt} \
-w "${capfile}-listener.pcap" >> "${capout}" 2>&1 &
diff --git a/tools/testing/selftests/net/mptcp/mptcp_diag.c b/tools/testing/selftests/net/mptcp/mptcp_diag.c
index 5e222ba977e4..3b8d2c8a6216 100644
--- a/tools/testing/selftests/net/mptcp/mptcp_diag.c
+++ b/tools/testing/selftests/net/mptcp/mptcp_diag.c
@@ -377,7 +377,8 @@ static void get_subflow_info(char *subflow_addrs)
int ret;
int fd;
- ret = sscanf(subflow_addrs, "%[^:]:%d %[^:]:%d", saddr, &sport, daddr, &dport);
+ ret = sscanf(subflow_addrs, "%63[^:]:%d %63[^:]:%d",
+ saddr, &sport, daddr, &dport);
if (ret != 4)
die_perror("IP PORT Pairs has style problems!");
diff --git a/tools/testing/selftests/net/mptcp/mptcp_join.sh b/tools/testing/selftests/net/mptcp/mptcp_join.sh
index 7dc91fac4917..18ce7136a2b0 100755
--- a/tools/testing/selftests/net/mptcp/mptcp_join.sh
+++ b/tools/testing/selftests/net/mptcp/mptcp_join.sh
@@ -75,6 +75,14 @@ unset join_syn_tx
unset join_create_err
unset join_bind_err
unset join_connect_err
+unset join_synack_no_mpjoin
+unset join_ack_no_mpjoin
+unset join_ack_no_ctx
+unset join_not_established
+unset join_no_id_found
+
+unset rst_md5sig
+unset rst_dss
unset fb_ns1
unset fb_ns2
@@ -971,7 +979,7 @@ cond_start_capture()
capfile=$(printf "mp_join-%02u-%s.pcap" "$MPTCP_LIB_TEST_COUNTER" "$ns")
echo "Capturing traffic for test $MPTCP_LIB_TEST_COUNTER into $capfile"
- ip netns exec "$ns" tcpdump -i any -s 65535 -B 32768 $capuser -w "$capfile" > "$capout" 2>&1 &
+ ip netns exec "$ns" tcpdump -i any -s 128 -B 32768 $capuser -w "$capfile" > "$capout" 2>&1 &
cappid=$!
sleep 1
@@ -1353,6 +1361,8 @@ chk_rst_nr()
local rst_tx=$1
local rst_rx=$2
local ns_invert=${3:-""}
+ local md5sig=${rst_md5sig:-0}
+ local dss=${rst_dss:-0}
local count
local ns_tx=$ns1
local ns_rx=$ns2
@@ -1389,6 +1399,21 @@ chk_rst_nr()
else
print_ok
fi
+
+ # MPTCP_RST_EMPTCP reset-event counters; default 0, gated on
+ # availability. Fixed namespaces: MD5SigReset fires on the listener
+ # (server), DssReset on the data receiver (client).
+ count=$(mptcp_lib_get_counter ${ns1} "MPTcpExtMD5SigReset")
+ if [ -n "$count" ] && [ "$count" != "$md5sig" ]; then
+ print_check "MD5SigReset"
+ fail_test "got $count MD5SigReset expected $md5sig"
+ fi
+
+ count=$(mptcp_lib_get_counter ${ns2} "MPTcpExtDssReset")
+ if [ -n "$count" ] && [ "$count" != "$dss" ]; then
+ print_check "DssReset"
+ fail_test "got $count DssReset expected $dss"
+ fi
}
chk_infi_nr()
@@ -1587,6 +1612,11 @@ chk_join_nr()
local rst_nr=${join_rst_nr:-0}
local infi_nr=${join_infi_nr:-0}
local corrupted_pkts=${join_corrupted_pkts:-0}
+ local synack_no_mpjoin=${join_synack_no_mpjoin:-0}
+ local ack_no_mpjoin=${join_ack_no_mpjoin:-0}
+ local ack_no_ctx=${join_ack_no_ctx:-0}
+ local not_established=${join_not_established:-0}
+ local no_id_found=${join_no_id_found:-0}
local rc=${KSFT_PASS}
local count
local with_cookie
@@ -1655,6 +1685,44 @@ chk_join_nr()
fail_test "got $count JOIN[s] syn rejected expected $syn_rej"
fi
+ # Per-event MPTCP_RST_EMPTCP JOIN counters; default 0, gated on
+ # availability. Fixed namespaces: the *SynAck* one fires on the
+ # client receiving the SYN/ACK, the others on the server.
+ count=$(mptcp_lib_get_counter ${ns2} "MPTcpExtMPJoinSynAckNoMPJoin")
+ if [ -n "$count" ] && [ "$count" != "$synack_no_mpjoin" ]; then
+ rc=${KSFT_FAIL}
+ print_check "synack no mpjoin"
+ fail_test "got $count JOIN[s] synack no mpjoin expected $synack_no_mpjoin"
+ fi
+
+ count=$(mptcp_lib_get_counter ${ns1} "MPTcpExtMPJoinAckNoMPJoin")
+ if [ -n "$count" ] && [ "$count" != "$ack_no_mpjoin" ]; then
+ rc=${KSFT_FAIL}
+ print_check "ack no mpjoin"
+ fail_test "got $count JOIN[s] ack no mpjoin expected $ack_no_mpjoin"
+ fi
+
+ count=$(mptcp_lib_get_counter ${ns1} "MPTcpExtMPJoinAckNoCtx")
+ if [ -n "$count" ] && [ "$count" != "$ack_no_ctx" ]; then
+ rc=${KSFT_FAIL}
+ print_check "ack no ctx"
+ fail_test "got $count JOIN[s] ack no ctx expected $ack_no_ctx"
+ fi
+
+ count=$(mptcp_lib_get_counter ${ns1} "MPTcpExtMPJoinNotEstablished")
+ if [ -n "$count" ] && [ "$count" != "$not_established" ]; then
+ rc=${KSFT_FAIL}
+ print_check "join not established"
+ fail_test "got $count JOIN[s] not established expected $not_established"
+ fi
+
+ count=$(mptcp_lib_get_counter ${ns1} "MPTcpExtMPJoinNoIdFound")
+ if [ -n "$count" ] && [ "$count" != "$no_id_found" ]; then
+ rc=${KSFT_FAIL}
+ print_check "join no id found"
+ fail_test "got $count JOIN[s] no id found expected $no_id_found"
+ fi
+
print_results "join Rx" ${rc}
join_syn_tx="${join_syn_tx:-${syn_nr}}" \
@@ -2359,6 +2427,31 @@ signal_address_tests()
chk_add_nr 4 4
fi
fi
+
+ # signalled address belongs to the client, where a TCP-only
+ # listener is bound at it: the client's MP_JOIN routes locally
+ # to the listener and receives a SYN/ACK without MP_JOIN.
+ # MPJoinSynAckNoMPJoin increments on the client side.
+ if reset "signal address, TCP-only listener on client"; then
+ local extra_bind
+ local port
+
+ pm_nl_set_limits $ns1 0 1
+ pm_nl_set_limits $ns2 1 1
+ pm_nl_add_endpoint $ns1 10.0.2.2 flags signal
+
+ port=$(get_port)
+ ip netns exec ${ns2} ./mptcp_connect -l -t -1 -p "$port" \
+ -s TCP 10.0.2.2 &
+ extra_bind=$!
+ mptcp_lib_wait_local_port_listen "$ns2" "$port"
+
+ run_tests $ns1 $ns2 10.0.1.1
+ join_synack_no_mpjoin=1 join_syn_tx=1 \
+ chk_join_nr 0 0 0
+
+ kill ${extra_bind} 2>/dev/null
+ fi
}
laminar_endp_tests()
diff --git a/tools/testing/selftests/net/mptcp/mptcp_lib.sh b/tools/testing/selftests/net/mptcp/mptcp_lib.sh
index 5ef6033775c8..b9d14647f401 100644
--- a/tools/testing/selftests/net/mptcp/mptcp_lib.sh
+++ b/tools/testing/selftests/net/mptcp/mptcp_lib.sh
@@ -108,12 +108,14 @@ mptcp_lib_pr_info() {
mptcp_lib_pr_nstat() {
local ns="${1}"
- local hist="/tmp/${ns}.out"
+ local cache="/tmp/${ns}.out"
+ local hist="/tmp/${ns}.nstat"
- if [ -f "${hist}" ]; then
- awk '$2 != 0 { print " "$0 }' "${hist}"
+ if [ -f "${cache}" ]; then
+ awk '$2 != 0 { print " "$0 }' "${cache}"
else
- ip netns exec "${ns}" nstat -as | grep Tcp
+ NSTAT_HISTORY="${hist}" ip netns exec "${ns}" nstat -s |
+ grep Tcp
fi
}
@@ -414,19 +416,21 @@ mptcp_lib_nstat_get() {
}
# $1: ns, $2: MIB counter
-# Get the counter from the history (mptcp_lib_nstat_{init,get}()) if available.
-# If not, get the counter from nstat ignoring any history.
+# Get the counter from the cache (mptcp_lib_nstat_{init,get}()) if available.
+# If not, get the counter from nstat ignoring any cache, but using the history.
mptcp_lib_get_counter() {
local ns="${1}"
local counter="${2}"
- local hist="/tmp/${ns}.out"
+ local cache="/tmp/${ns}.out"
+ local hist="/tmp/${ns}.nstat"
local count
- if [[ -s "${hist}" && "${counter}" == *"Tcp"* ]]; then
- count=$(awk "/^${counter} / {print \$2; exit}" "${hist}")
+ if [[ -s "${cache}" && "${counter}" == *"Tcp"* ]]; then
+ count=$(awk "/^${counter} / {print \$2; exit}" "${cache}")
else
- count=$(ip netns exec "${ns}" nstat -asz "${counter}" |
- awk 'NR==1 {next} {print $2}')
+ count=$(NSTAT_HISTORY="${hist}" ip netns exec "${ns}" \
+ nstat -sz "${counter}" |
+ awk 'NR==1 {next} {print $2}')
fi
if [ -z "${count}" ]; then
mptcp_lib_fail_if_expected_feature "${counter} counter"
diff --git a/tools/testing/selftests/net/mptcp/simult_flows.sh b/tools/testing/selftests/net/mptcp/simult_flows.sh
index 7b9aabe10170..3ea3d1efe32e 100755
--- a/tools/testing/selftests/net/mptcp/simult_flows.sh
+++ b/tools/testing/selftests/net/mptcp/simult_flows.sh
@@ -24,6 +24,7 @@ small=""
sout=""
cout=""
capout=""
+capprefix=""
size=0
usage() {
@@ -70,6 +71,11 @@ setup()
mptcp_lib_ns_init ns1 ns2 ns3
+ if $capture; then
+ capprefix="simult_flows-${ns1:4}"
+ mptcp_lib_pr_info "pcap will have this prefix: ${capprefix}-"
+ fi
+
ip link add ns1eth1 netns "$ns1" type veth peer name ns2eth1 netns "$ns2"
ip link add ns1eth2 netns "$ns1" type veth peer name ns2eth2 netns "$ns2"
ip link add ns2eth3 netns "$ns2" type veth peer name ns3eth1 netns "$ns3"
@@ -136,15 +142,14 @@ do_transfer()
if $capture; then
local capuser
- local rndh="${ns1:4}"
if [ -z $SUDO_USER ] ; then
capuser=""
else
capuser="-Z $SUDO_USER"
fi
- local capfile="${rndh}-${port}"
- local capopt="-i any -s 65535 -B 32768 ${capuser}"
+ local capfile="${capprefix}-${port}"
+ local capopt="-i any -s 108 -B 32768 ${capuser}"
ip netns exec ${ns3} tcpdump ${capopt} -w "${capfile}-listener.pcap" >> "${capout}" 2>&1 &
local cappid_listener=$!
diff --git a/tools/testing/selftests/net/netfilter/conntrack_dump_flush.c b/tools/testing/selftests/net/netfilter/conntrack_dump_flush.c
index 5cecb8a1bc94..31b8250ddc53 100644
--- a/tools/testing/selftests/net/netfilter/conntrack_dump_flush.c
+++ b/tools/testing/selftests/net/netfilter/conntrack_dump_flush.c
@@ -102,7 +102,6 @@ static int conntrack_data_insert(struct mnl_socket *sock, struct nlmsghdr *nlh,
uint16_t zone)
{
char buf[MNL_SOCKET_BUFFER_SIZE];
- struct nlmsghdr *rplnlh;
unsigned int portid;
int ret;
@@ -216,12 +215,11 @@ static int count_entries(const struct nlmsghdr *nlh, void *data)
return MNL_CB_OK;
}
-static int conntracK_count_zone(struct mnl_socket *sock, uint16_t zone)
+static int conntrack_count_zone(struct mnl_socket *sock, uint16_t zone)
{
char buf[MNL_SOCKET_BUFFER_SIZE];
- struct nlmsghdr *nlh, *rplnlh;
+ struct nlmsghdr *nlh;
struct nfgenmsg *nfh;
- struct nlattr *nest;
unsigned int portid;
int ret;
@@ -266,9 +264,8 @@ static int conntracK_count_zone(struct mnl_socket *sock, uint16_t zone)
static int conntrack_flush_zone(struct mnl_socket *sock, uint16_t zone)
{
char buf[MNL_SOCKET_BUFFER_SIZE];
- struct nlmsghdr *nlh, *rplnlh;
+ struct nlmsghdr *nlh;
struct nfgenmsg *nfh;
- struct nlattr *nest;
unsigned int portid;
int ret;
@@ -326,7 +323,7 @@ FIXTURE_SETUP(conntrack_dump_flush)
ret = mnl_socket_bind(self->sock, 0, MNL_SOCKET_AUTOPID);
EXPECT_EQ(ret, 0);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID);
if (ret < 0 && errno == EPERM)
SKIP(return, "Needs to be run as root");
else if (ret < 0 && errno == EOPNOTSUPP)
@@ -423,7 +420,7 @@ FIXTURE_SETUP(conntrack_dump_flush)
NF_CT_DEFAULT_ZONE_ID);
EXPECT_EQ(ret, 0);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID);
EXPECT_GE(ret, 2);
if (ret > 2)
SKIP(return, "kernel does not support filtering by zone");
@@ -437,7 +434,7 @@ TEST_F(conntrack_dump_flush, test_dump_by_zone)
{
int ret;
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID);
EXPECT_EQ(ret, 2);
}
@@ -447,13 +444,13 @@ TEST_F(conntrack_dump_flush, test_flush_by_zone)
ret = conntrack_flush_zone(self->sock, TEST_ZONE_ID);
EXPECT_EQ(ret, 0);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID);
EXPECT_EQ(ret, 0);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID + 1);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID + 1);
EXPECT_EQ(ret, 2);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID + 2);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID + 2);
EXPECT_EQ(ret, 2);
- ret = conntracK_count_zone(self->sock, NF_CT_DEFAULT_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, NF_CT_DEFAULT_ZONE_ID);
EXPECT_EQ(ret, 2);
}
@@ -463,13 +460,13 @@ TEST_F(conntrack_dump_flush, test_flush_by_zone_default)
ret = conntrack_flush_zone(self->sock, NF_CT_DEFAULT_ZONE_ID);
EXPECT_EQ(ret, 0);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID);
EXPECT_EQ(ret, 2);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID + 1);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID + 1);
EXPECT_EQ(ret, 2);
- ret = conntracK_count_zone(self->sock, TEST_ZONE_ID + 2);
+ ret = conntrack_count_zone(self->sock, TEST_ZONE_ID + 2);
EXPECT_EQ(ret, 2);
- ret = conntracK_count_zone(self->sock, NF_CT_DEFAULT_ZONE_ID);
+ ret = conntrack_count_zone(self->sock, NF_CT_DEFAULT_ZONE_ID);
EXPECT_EQ(ret, 0);
}
diff --git a/tools/testing/selftests/net/nl_nlctrl.py b/tools/testing/selftests/net/nl_nlctrl.py
index fe1f66dc9435..237b3d273260 100755
--- a/tools/testing/selftests/net/nl_nlctrl.py
+++ b/tools/testing/selftests/net/nl_nlctrl.py
@@ -9,40 +9,86 @@ from lib.py import ksft_run, ksft_exit
from lib.py import ksft_eq, ksft_ge, ksft_true, ksft_in, ksft_not_in
from lib.py import NetdevFamily, EthtoolFamily, NlctrlFamily
+# Families we can expect to always be around, and which between them
+# cover ops with a do, with a dump, and with both.
+FAMILIES = ('nlctrl', 'netdev')
-def getfamily_do(ctrl) -> None:
- """Query a single family by name and validate its ops."""
- fam = ctrl.getfamily({'family-name': 'netdev'})
- ksft_eq(fam['family-name'], 'netdev')
+
+def _get_ops(ctrl, name):
+ """Get the ops of a family, keyed by command id."""
+ fam = ctrl.getfamily({'family-name': name})
+ ksft_eq(fam['family-name'], name)
ksft_true(fam['family-id'] > 0)
# The format of ops is quite odd, [{$idx: {"id"...}}, {$idx: {"id"...}}]
# Discard the indices and re-key by command id.
ops_by_id = {v['id']: v for op in fam['ops'] for v in op.values()}
- ksft_eq(len(ops_by_id), len(fam['ops']))
+ ksft_eq(len(ops_by_id), len(fam['ops']),
+ comment=f"{name} lists a command twice")
+ return ops_by_id
+
+
+def _get_policy_map(ctrl, req):
+ """
+ The policy map in the Netlink replies looks like this:
+
+ [{'family-id': 16, 'op-policy': {'do': 0, 'dump': 0, 'op-id': 3}},
+ {'family-id': 16, 'op-policy': {'dump': 1, 'op-id': 4}}, ...]
+
+ Return the mapping:
+
+ {3:{'do','dump'}, 4:{'dump'}}
+
+ The policy itself is discarded here, only return which command has policy.
+ """
+ pol_map = {}
+ for msg in ctrl.getpolicy(req, dump=True):
+ if 'op-policy' not in msg:
+ continue
+ modes = dict(msg['op-policy'])
+ cmd = modes.pop('op-id')
+ ksft_not_in(cmd, pol_map, comment=f"command {cmd} reported twice")
+ pol_map[cmd] = set(modes.keys())
+ return pol_map
+
+
+def getfamily_do(ctrl) -> None:
+ """Query single families by name and validate their ops."""
+ ops = {name: _get_ops(ctrl, name) for name in FAMILIES}
+
+ for name, ops_by_id in ops.items():
+ for op in ops_by_id.values():
+ # All ops in nlctrl and netdev have a policy
+ ksft_in('cmd-cap-haspol', op['flags'],
+ comment=f"{name} op {op['id']} missing haspol")
+ ksft_true(op['flags'] & {'cmd-cap-do', 'cmd-cap-dump'},
+ comment=f"{name} op {op['id']} has no handler")
- # All ops should have a policy (either do or dump has one)
- for op in ops_by_id.values():
- ksft_in('cmd-cap-haspol', op['flags'],
- comment=f"op {op['id']} missing haspol")
+ # nlctrl getfamily (id 3) does both, getpolicy (id 10) is dump-only
+ ksft_in('cmd-cap-do', ops['nlctrl'][3]['flags'])
+ ksft_in('cmd-cap-dump', ops['nlctrl'][3]['flags'])
+ ksft_not_in('cmd-cap-do', ops['nlctrl'][10]['flags'])
+ ksft_in('cmd-cap-dump', ops['nlctrl'][10]['flags'])
+
+ netdev = ops['netdev']
# dev-get (id 1) should support both do and dump
- ksft_in('cmd-cap-do', ops_by_id[1]['flags'])
- ksft_in('cmd-cap-dump', ops_by_id[1]['flags'])
+ ksft_in('cmd-cap-do', netdev[1]['flags'])
+ ksft_in('cmd-cap-dump', netdev[1]['flags'])
# qstats-get (id 12) is dump-only
- ksft_not_in('cmd-cap-do', ops_by_id[12]['flags'])
- ksft_in('cmd-cap-dump', ops_by_id[12]['flags'])
+ ksft_not_in('cmd-cap-do', netdev[12]['flags'])
+ ksft_in('cmd-cap-dump', netdev[12]['flags'])
# napi-set (id 14) is do-only and requires admin
- ksft_in('cmd-cap-do', ops_by_id[14]['flags'])
- ksft_not_in('cmd-cap-dump', ops_by_id[14]['flags'])
- ksft_in('admin-perm', ops_by_id[14]['flags'])
+ ksft_in('cmd-cap-do', netdev[14]['flags'])
+ ksft_not_in('cmd-cap-dump', netdev[14]['flags'])
+ ksft_in('admin-perm', netdev[14]['flags'])
# Notification-only commands (dev-add/del/change-ntf etc.) must
# not appear in the ops list since they have no do/dump handlers.
for ntf_id in [2, 3, 4, 6, 7, 8]:
- ksft_not_in(ntf_id, ops_by_id,
+ ksft_not_in(ntf_id, netdev,
comment=f"ntf-only cmd {ntf_id} should not be in ops")
@@ -103,6 +149,41 @@ def getpolicy_dump(_ctrl) -> None:
comment="linkinfo-set should not have a dump policy")
+def getpolicy_op_map(ctrl) -> None:
+ """Check the op-to-policy map consistency. Each op with 'haspol' flag
+ has to have a policy. The policy back-references must name only
+ real ops that exist, have given modes (do vs dump) and have 'haspol'.
+ """
+ for name in FAMILIES:
+ ops_by_id = _get_ops(ctrl, name)
+ haspol = {cmd for cmd, op in ops_by_id.items()
+ if 'cmd-cap-haspol' in op['flags']}
+
+ pol_map = _get_policy_map(ctrl, {'family-name': name})
+ ksft_eq(set(pol_map), haspol,
+ comment=f"{name} policy map does not match the op list")
+
+ # Walk the op list rather than the map, the map may be missing
+ # the very op we are after. Asking for a command the family does
+ # not have is an error, so it must not come from the map either.
+ for cmd in sorted(haspol):
+ modes = pol_map.get(cmd, set())
+
+ # The kernel only reports a mode the op actually has.
+ if 'do' in modes:
+ ksft_in('cmd-cap-do', ops_by_id[cmd]['flags'],
+ comment=f"{name} cmd {cmd} has no do")
+ if 'dump' in modes:
+ ksft_in('cmd-cap-dump', ops_by_id[cmd]['flags'],
+ comment=f"{name} cmd {cmd} has no dump")
+
+ # Asking for one op builds the map in a different place in
+ # the kernel, it has to report what the full dump did.
+ single = _get_policy_map(ctrl, {'family-name': name, 'op': cmd})
+ ksft_eq(single, {cmd: modes},
+ comment=f"{name} cmd {cmd} policy differs from the dump")
+
+
def getpolicy_by_op(_ctrl) -> None:
"""Query policy for specific ops, check attr names are resolved."""
ndev = NetdevFamily()
@@ -122,6 +203,7 @@ def main() -> None:
ksft_run([getfamily_do,
getfamily_dump,
getpolicy_dump,
+ getpolicy_op_map,
getpolicy_by_op],
args=(ctrl, ))
ksft_exit()
diff --git a/tools/testing/selftests/net/openvswitch/config b/tools/testing/selftests/net/openvswitch/config
index c659749cd086..a825e0b5c88e 100644
--- a/tools/testing/selftests/net/openvswitch/config
+++ b/tools/testing/selftests/net/openvswitch/config
@@ -1,5 +1,6 @@
CONFIG_GENEVE=m
CONFIG_INET_DIAG=y
+CONFIG_IP_SCTP=y
CONFIG_IPV6=y
CONFIG_NETFILTER=y
CONFIG_NET_IPGRE=m
@@ -7,9 +8,6 @@ CONFIG_NET_IPGRE_DEMUX=m
CONFIG_NF_CONNTRACK=m
CONFIG_NF_CONNTRACK_OVS=y
CONFIG_OPENVSWITCH=m
-CONFIG_OPENVSWITCH_GENEVE=m
-CONFIG_OPENVSWITCH_GRE=m
-CONFIG_OPENVSWITCH_VXLAN=m
CONFIG_PSAMPLE=m
CONFIG_VETH=y
CONFIG_VLAN_8021Q=y
diff --git a/tools/testing/selftests/net/openvswitch/openvswitch.sh b/tools/testing/selftests/net/openvswitch/openvswitch.sh
index 2954245129a2..a31f7fb6882d 100755
--- a/tools/testing/selftests/net/openvswitch/openvswitch.sh
+++ b/tools/testing/selftests/net/openvswitch/openvswitch.sh
@@ -26,12 +26,14 @@ tests="
netlink_checks ovsnl: validate netlink attrs and settings
upcall_interfaces ovs: test the upcall interfaces
tunnel_metadata ovs: test extraction of tunnel metadata
- tunnel_refcount ovs: test tunnel vport reference cleanup
drop_reason drop: test drop reasons are emitted
pop_vlan vlan: POP_VLAN action strips tag
dec_ttl ttl: dec_ttl decrements IP TTL
flow_set flow-set: Flow modify
action_set set: SET action rewrites fields
+ trunc trunc: output truncation
+ icmpv6 icmpv6: ICMPv6 echo type match
+ sctp_connect_v4 sctp: SCTP flow key matching
psample psample: Sampling packets with psample"
info() {
@@ -443,6 +445,261 @@ test_action_set() {
return 0
}
+# trunc test
+# - trunc(14): truncate to ETH_HLEN, strips IP payload, ping fails
+# - trunc(1) and trunc(13): kernel rejects below ETH_HLEN (EINVAL)
+# - restore normal forwarding and verify recovery
+test_trunc() {
+ sbx_add "test_trunc" || return $?
+ ovs_add_dp "test_trunc" trunctest || return 1
+
+ info "create namespaces"
+ for ns in client server; do
+ ovs_add_netns_and_veths "test_trunc" "trunctest" \
+ "$ns" "${ns:0:1}0" "${ns:0:1}1" || return 1
+ done
+
+ ip netns exec client ip addr add 10.0.0.1/24 dev c1
+ ip netns exec client ip link set c1 up
+ ip netns exec server ip addr add 10.0.0.2/24 dev s1
+ ip netns exec server ip link set s1 up
+
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0806),arp()' '2' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(2),eth(),eth_type(0x0806),arp()' '1' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4()' \
+ '2' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(2),eth(),eth_type(0x0800),ipv4()' \
+ '1' || return 1
+
+ info "verify connectivity without truncation"
+ ovs_sbx "test_trunc" ip netns exec client \
+ ping -c 1 -W 2 10.0.0.2 || return 1
+
+ # trunc below ETH_HLEN must be rejected by the kernel
+ info "verify trunc(1) is rejected"
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4()' \
+ 'trunc(1),2' &> /dev/null \
+ && { info "trunc(1) should be rejected"; return 1; }
+
+ info "verify trunc(13) is rejected"
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4()' \
+ 'trunc(13),2' &> /dev/null \
+ && { info "trunc(13) should be rejected"; return 1; }
+
+ ovs_del_flows "test_trunc" trunctest
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0806),arp()' '2' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(2),eth(),eth_type(0x0806),arp()' '1' || return 1
+
+ info "add trunc(14) forwarding flow"
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4()' \
+ 'trunc(14),2' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(2),eth(),eth_type(0x0800),ipv4()' \
+ '1' || return 1
+
+ info "verify ping fails with trunc(14)"
+ ovs_sbx "test_trunc" ip netns exec client \
+ ping -c 1 -W 2 10.0.0.2 >/dev/null 2>&1 \
+ && { info "ping should fail with trunc(14)"
+ return 1; }
+
+ ovs_del_flows "test_trunc" trunctest
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0806),arp()' '2' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(2),eth(),eth_type(0x0806),arp()' '1' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4()' \
+ '2' || return 1
+ ovs_add_flow "test_trunc" trunctest \
+ 'in_port(2),eth(),eth_type(0x0800),ipv4()' \
+ '1' || return 1
+
+ info "verify connectivity restored"
+ ovs_sbx "test_trunc" ip netns exec client \
+ ping -c 1 -W 2 10.0.0.2 || return 1
+
+ return 0
+}
+
+# icmpv6 test
+# - static neighbours to bypass NDP (nud permanent)
+# - icmpv6(type=128) echo request, icmpv6(type=129) echo reply
+# - remove flows and verify ping fails, reinstall and recover
+test_icmpv6() {
+ local t="test_icmpv6"
+ local v6="eth_type(0x86dd),ipv6(proto=58)"
+
+ sbx_add "$t" || return $?
+ ovs_add_dp "$t" icmpv6 || return 1
+
+ info "create namespaces"
+ for ns in client server; do
+ ovs_add_netns_and_veths "$t" "icmpv6" \
+ "$ns" "${ns:0:1}0" "${ns:0:1}1" || return 1
+ done
+
+ ip netns exec client ip addr add fd00::1/64 dev c1 nodad
+ ip netns exec client ip link set c1 up
+ ip netns exec server ip addr add fd00::2/64 dev s1 nodad
+ ip netns exec server ip link set s1 up
+
+ local cl_mac sl_mac
+ cl_mac=$(ip netns exec client ip link show c1 \
+ | awk '/link\/ether/ {print $2}')
+ [ -z "$cl_mac" ] && \
+ { info "failed to get c1 hwaddr"; return 1; }
+ sl_mac=$(ip netns exec server ip link show s1 \
+ | awk '/link\/ether/ {print $2}')
+ [ -z "$sl_mac" ] && \
+ { info "failed to get s1 hwaddr"; return 1; }
+ ip netns exec client ip -6 neigh add fd00::2 \
+ lladdr "$sl_mac" nud permanent dev c1 || return 1
+ ip netns exec server ip -6 neigh add fd00::1 \
+ lladdr "$cl_mac" nud permanent dev s1 || return 1
+
+ # Probe: check if kernel supports icmpv6 flow key.
+ ovs_add_flow "$t" icmpv6 \
+ "in_port(1),eth(),$v6,icmpv6(type=128)" \
+ '2' &>/dev/null
+ if [ $? -ne 0 ]; then
+ info "no support for icmpv6 key - skipping"
+ ovs_exit_sig
+ return $ksft_skip
+ fi
+ ovs_del_flows "$t" icmpv6
+
+ ovs_add_flow "$t" icmpv6 \
+ "in_port(1),eth(),$v6,icmpv6(type=128)" \
+ '2' || return 1
+ ovs_add_flow "$t" icmpv6 \
+ "in_port(2),eth(),$v6,icmpv6(type=129)" \
+ '1' || return 1
+
+ info "verify ICMPv6 echo with type-specific flows"
+ ovs_sbx "$t" ip netns exec client \
+ ping -6 -c 1 -W 2 fd00::2 || return 1
+
+ ovs_del_flows "$t" icmpv6
+
+ info "verify ping fails without echo flows"
+ ovs_sbx "$t" ip netns exec client \
+ ping -6 -c 1 -W 2 fd00::2 >/dev/null 2>&1 \
+ && { info "ping should fail without flows"
+ return 1; }
+
+ ovs_add_flow "$t" icmpv6 \
+ "in_port(1),eth(),$v6,icmpv6(type=128)" \
+ '2' || return 1
+ ovs_add_flow "$t" icmpv6 \
+ "in_port(2),eth(),$v6,icmpv6(type=129)" \
+ '1' || return 1
+
+ info "verify connectivity restored"
+ ovs_sbx "$t" ip netns exec client \
+ ping -6 -c 1 -W 2 fd00::2 || return 1
+
+ return 0
+}
+
+# Check for an SCTP endpoint via /proc, which works without sctp_diag.
+sctp_eps_has() {
+ ip netns exec "$1" awk -v p="$2" '$6==p' /proc/net/sctp/eps | grep -q .
+}
+
+# sctp_connect_v4 test
+# - sctp(dst=4443) matches client-to-server INIT
+# - sctp(src=4443) matches server-to-client INIT-ACK
+# - remove flows and verify connection fails, reinstall and recover
+test_sctp_connect_v4() {
+ local t="test_sctp_connect_v4"
+ local srv_ip=172.31.110.20
+
+ modprobe -q sctp 2>/dev/null || return "$ksft_skip"
+ socat -V 2>&1 | grep -q "define WITH_SCTP" || return "$ksft_skip"
+
+ sbx_add "$t" || return $?
+ ovs_add_dp "$t" sctp4 || return 1
+
+ info "create namespaces"
+ for ns in client server; do
+ ovs_add_netns_and_veths "$t" "sctp4" "$ns" \
+ "${ns:0:1}0" "${ns:0:1}1" || return 1
+ done
+
+ ip netns exec client ip addr add 172.31.110.10/24 dev c1
+ ip netns exec client ip link set c1 up
+ ip netns exec server ip addr add "${srv_ip}/24" dev s1
+ ip netns exec server ip link set s1 up
+
+ # ARP forwarding
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(1),eth(),eth_type(0x0806),arp()' \
+ '2' || return 1
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(2),eth(),eth_type(0x0806),arp()' \
+ '1' || return 1
+
+ # SCTP port matching: dst for request, src for reply
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4(proto=132),sctp(dst=4443)' \
+ '2' || return 1
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(2),eth(),eth_type(0x0800),ipv4(proto=132),sctp(src=4443)' \
+ '1' || return 1
+
+ # The listener forks a child per association, so one instance serves
+ # the whole test and the flows stay the only variable. -t 1 bounds
+ # how long a child lingers after its association closes.
+ ovs_netns_spawn_daemon "$t" "server" \
+ socat -u -t 1 SCTP4-LISTEN:4443,fork STDOUT
+ ovs_wait sctp_eps_has server 4443 || return 1
+
+ info "verify SCTP association with port-keyed flows"
+ ovs_sbx "$t" ip netns exec client \
+ timeout 3 socat -u STDIN "SCTP4-CONNECT:${srv_ip}:4443" </dev/null \
+ || return 1
+
+ ovs_del_flows "$t" sctp4
+
+ info "verify connection fails without flows"
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(1),eth(),eth_type(0x0806),arp()' \
+ '2' || return 1
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(2),eth(),eth_type(0x0806),arp()' \
+ '1' || return 1
+
+ ovs_sbx "$t" ip netns exec client \
+ timeout 3 socat -u STDIN "SCTP4-CONNECT:${srv_ip}:4443" </dev/null \
+ >/dev/null 2>&1 \
+ && { info "connection should fail without flows"
+ return 1; }
+
+ info "reinstall flows and verify recovery"
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(1),eth(),eth_type(0x0800),ipv4(proto=132),sctp(dst=4443)' \
+ '2' || return 1
+ ovs_add_flow "$t" sctp4 \
+ 'in_port(2),eth(),eth_type(0x0800),ipv4(proto=132),sctp(src=4443)' \
+ '1' || return 1
+
+ ovs_sbx "$t" ip netns exec client \
+ timeout 3 socat -u STDIN "SCTP4-CONNECT:${srv_ip}:4443" </dev/null \
+ || return 1
+
+ return 0
+}
+
# psample test
# - use psample to observe packets
test_psample() {
@@ -1042,43 +1299,6 @@ test_tunnel_metadata() {
return 0
}
-test_tunnel_refcount() {
- sbxname="test_tunnel_refcount"
- sbx_add "${sbxname}" || return 1
-
- ovs_sbx "${sbxname}" ip netns add trefns || return 1
- on_exit "ovs_sbx ${sbxname} ip netns del trefns"
-
- for tun_type in gre vxlan geneve; do
- info "testing ${tun_type} tunnel vport refcount"
-
- ovs_sbx "${sbxname}" ip netns exec trefns \
- python3 $ovs_base/ovs-dpctl.py \
- add-dp dp-${tun_type} || return 1
-
- ovs_sbx "${sbxname}" ip netns exec trefns \
- python3 $ovs_base/ovs-dpctl.py \
- add-if --no-lwt -t ${tun_type} \
- dp-${tun_type} ovs-${tun_type}0 || return 1
-
- ovs_wait ip -netns trefns link show \
- ovs-${tun_type}0 >/dev/null 2>&1 || return 1
-
- info "deleting dp - may hang if reference counting is broken"
- ovs_sbx "${sbxname}" ip netns exec trefns \
- python3 $ovs_base/ovs-dpctl.py \
- del-dp dp-${tun_type} &
-
- dev_removed() {
- ! ip -netns trefns link show "$1" >/dev/null 2>&1
- }
- ovs_wait dev_removed dp-${tun_type} || return 1
- ovs_wait dev_removed ovs-${tun_type}0 || return 1
- done
-
- return 0
-}
-
test_pop_vlan() {
local sbx="test_pop_vlan"
sbx_add "$sbx" || return $?
diff --git a/tools/testing/selftests/net/openvswitch/ovs-dpctl.py b/tools/testing/selftests/net/openvswitch/ovs-dpctl.py
index e1ecfad2c03e..1615843c225e 100644
--- a/tools/testing/selftests/net/openvswitch/ovs-dpctl.py
+++ b/tools/testing/selftests/net/openvswitch/ovs-dpctl.py
@@ -1255,11 +1255,16 @@ class ovskey(nla):
lambda x: ipaddress.IPv6Address(x).packed if x else 0,
convert_ipv6,
),
- ("label", "label", "%d", lambda x: int(x) if x else 0),
- ("proto", "proto", "%d", lambda x: int(x) if x else 0),
- ("tclass", "tclass", "%d", lambda x: int(x) if x else 0),
- ("hlimit", "hlimit", "%d", lambda x: int(x) if x else 0),
- ("frag", "frag", "%d", lambda x: int(x) if x else 0),
+ ("label", "label", "%d", lambda x: int(x) if x else 0,
+ convert_int(20)),
+ ("proto", "proto", "%d", lambda x: int(x) if x else 0,
+ convert_int(8)),
+ ("tclass", "tclass", "%d", lambda x: int(x) if x else 0,
+ convert_int(8)),
+ ("hlimit", "hlimit", "%d", lambda x: int(x) if x else 0,
+ convert_int(8)),
+ ("frag", "frag", "%d", lambda x: int(x) if x else 0,
+ convert_int(8)),
)
def __init__(
@@ -1344,8 +1349,10 @@ class ovskey(nla):
)
fields_map = (
- ("type", "type", "%d", lambda x: int(x) if x else 0),
- ("code", "code", "%d", lambda x: int(x) if x else 0),
+ ("type", "type", "%d", lambda x: int(x) if x else 0,
+ convert_int(8)),
+ ("code", "code", "%d", lambda x: int(x) if x else 0,
+ convert_int(8)),
)
def __init__(
@@ -1978,11 +1985,21 @@ class ovskey(nla):
ovskey.ovs_key_udp,
),
(
+ "OVS_KEY_ATTR_SCTP",
+ "sctp",
+ ovskey.ovs_key_sctp,
+ ),
+ (
"OVS_KEY_ATTR_ICMP",
"icmp",
ovskey.ovs_key_icmp,
),
(
+ "OVS_KEY_ATTR_ICMPV6",
+ "icmpv6",
+ ovskey.ovs_key_icmpv6,
+ ),
+ (
"OVS_KEY_ATTR_TCP_FLAGS",
"tcp_flags",
lambda x: parse_flags(x, None),
@@ -2352,9 +2369,6 @@ class OvsDatapath(GenericNetlinkSocket):
class OvsVport(GenericNetlinkSocket):
OVS_VPORT_TYPE_NETDEV = 1
OVS_VPORT_TYPE_INTERNAL = 2
- OVS_VPORT_TYPE_GRE = 3
- OVS_VPORT_TYPE_VXLAN = 4
- OVS_VPORT_TYPE_GENEVE = 5
class ovs_vport_msg(ovs_dp_msg):
nla_map = (
@@ -2362,7 +2376,7 @@ class OvsVport(GenericNetlinkSocket):
("OVS_VPORT_ATTR_PORT_NO", "uint32"),
("OVS_VPORT_ATTR_TYPE", "uint32"),
("OVS_VPORT_ATTR_NAME", "asciiz"),
- ("OVS_VPORT_ATTR_OPTIONS", "vportopts"),
+ ("OVS_VPORT_ATTR_OPTIONS", "none"),
("OVS_VPORT_ATTR_UPCALL_PID", "array(uint32)"),
("OVS_VPORT_ATTR_STATS", "vportstats"),
("OVS_VPORT_ATTR_PAD", "none"),
@@ -2370,13 +2384,6 @@ class OvsVport(GenericNetlinkSocket):
("OVS_VPORT_ATTR_NETNSID", "uint32"),
)
- class vportopts(nla):
- nla_map = (
- ("OVS_TUNNEL_ATTR_UNSPEC", "none"),
- ("OVS_TUNNEL_ATTR_DST_PORT", "uint16"),
- ("OVS_TUNNEL_ATTR_EXTENSION", "none"),
- )
-
class vportstats(nla):
fields = (
("rx_packets", "=Q"),
@@ -2394,25 +2401,13 @@ class OvsVport(GenericNetlinkSocket):
return "netdev"
elif vport_type == OvsVport.OVS_VPORT_TYPE_INTERNAL:
return "internal"
- elif vport_type == OvsVport.OVS_VPORT_TYPE_GRE:
- return "gre"
- elif vport_type == OvsVport.OVS_VPORT_TYPE_VXLAN:
- return "vxlan"
- elif vport_type == OvsVport.OVS_VPORT_TYPE_GENEVE:
- return "geneve"
raise ValueError("Unknown vport type:%d" % vport_type)
def str_to_type(vport_type):
- if vport_type == "netdev":
+ if vport_type in ["netdev", "gre", "vxlan", "geneve"]:
return OvsVport.OVS_VPORT_TYPE_NETDEV
elif vport_type == "internal":
return OvsVport.OVS_VPORT_TYPE_INTERNAL
- elif vport_type == "gre":
- return OvsVport.OVS_VPORT_TYPE_GRE
- elif vport_type == "vxlan":
- return OvsVport.OVS_VPORT_TYPE_VXLAN
- elif vport_type == "geneve":
- return OvsVport.OVS_VPORT_TYPE_GENEVE
raise ValueError("Unknown vport type: '%s'" % vport_type)
def __init__(self, packet=OvsPacket()):
@@ -2445,17 +2440,19 @@ class OvsVport(GenericNetlinkSocket):
raise ne
return reply
- def attach(self, dpindex, vport_ifname, ptype, dport, lwt):
+ def attach(self, dpindex, vport_ifname, ptype, dport):
msg = OvsVport.ovs_vport_msg()
msg["cmd"] = OVS_VPORT_CMD_NEW
msg["version"] = OVS_DATAPATH_VERSION
msg["reserved"] = 0
msg["dpifindex"] = dpindex
- port_type = OvsVport.str_to_type(ptype)
msg["attrs"].append(["OVS_VPORT_ATTR_NAME", vport_ifname])
msg["attrs"].append(
+ ["OVS_VPORT_ATTR_TYPE", OvsVport.str_to_type(ptype)]
+ )
+ msg["attrs"].append(
["OVS_VPORT_ATTR_UPCALL_PID", [self.upcall_packet.epid]]
)
@@ -2468,36 +2465,21 @@ class OvsVport(GenericNetlinkSocket):
if not dport:
dport = tnl[1]
- if not lwt:
- if tnl[0] == "gre":
- # GRE tunnels have no options.
- break
-
- vportopt = OvsVport.ovs_vport_msg.vportopts()
- vportopt["attrs"].append(
- ["OVS_TUNNEL_ATTR_DST_PORT", dport]
- )
- msg["attrs"].append(
- ["OVS_VPORT_ATTR_OPTIONS", vportopt]
- )
- else:
- port_type = OvsVport.OVS_VPORT_TYPE_NETDEV
- ipr = pyroute2.iproute.IPRoute()
-
- if tnl[0] == "geneve":
- ipr.link("add", ifname=vport_ifname, kind=tnl[0],
- geneve_port=dport,
- geneve_collect_metadata=True,
- geneve_udp_zero_csum6_rx=1)
- elif tnl[0] == "gre":
- ipr.link("add", ifname=vport_ifname, kind="gretap",
- gre_collect_metadata=True)
- elif tnl[0] == "vxlan":
- ipr.link("add", ifname=vport_ifname, kind=tnl[0],
- vxlan_learning=0, vxlan_collect_metadata=1,
- vxlan_udp_zero_csum6_rx=1, vxlan_port=dport)
+ ipr = pyroute2.iproute.IPRoute()
+
+ if tnl[0] == "geneve":
+ ipr.link("add", ifname=vport_ifname, kind=tnl[0],
+ geneve_port=dport,
+ geneve_collect_metadata=True,
+ geneve_udp_zero_csum6_rx=1)
+ elif tnl[0] == "gre":
+ ipr.link("add", ifname=vport_ifname, kind="gretap",
+ gre_collect_metadata=True)
+ elif tnl[0] == "vxlan":
+ ipr.link("add", ifname=vport_ifname, kind=tnl[0],
+ vxlan_learning=0, vxlan_collect_metadata=1,
+ vxlan_udp_zero_csum6_rx=1, vxlan_port=dport)
break
- msg["attrs"].append(["OVS_VPORT_ATTR_TYPE", port_type])
try:
reply = self.nlm_request(
@@ -2925,19 +2907,12 @@ def print_ovsdp_full(dp_lookup_rep, ifindex, ndb=NDB(), vpl=OvsVport()):
for iface in ndb.interfaces:
rep = vpl.info(iface.ifname, ifindex)
if rep is not None:
- opts = ""
- vpo = rep.get_attr("OVS_VPORT_ATTR_OPTIONS")
- if vpo:
- dpo = vpo.get_attr("OVS_TUNNEL_ATTR_DST_PORT")
- if dpo:
- opts += " tnl-dport:%s" % dpo
print(
- " port %d: %s (%s%s)"
+ " port %d: %s (%s)"
% (
rep.get_attr("OVS_VPORT_ATTR_PORT_NO"),
rep.get_attr("OVS_VPORT_ATTR_NAME"),
OvsVport.type_to_str(rep.get_attr("OVS_VPORT_ATTR_TYPE")),
- opts,
)
)
@@ -3010,13 +2985,6 @@ def main(argv):
default=0,
help="Destination port (0 for default)"
)
- addifcmd.add_argument(
- "-l",
- "--lwt",
- action=argparse.BooleanOptionalAction,
- default=True,
- help="Use LWT infrastructure instead of vport (default true)."
- )
delifcmd = subparsers.add_parser("del-if")
delifcmd.add_argument("dpname", help="Datapath Name")
delifcmd.add_argument("delif", help="Interface name for adding")
@@ -3096,7 +3064,7 @@ def main(argv):
return 1
dpindex = rep["dpifindex"]
rep = ovsvp.attach(rep["dpifindex"], args.addif, args.ptype,
- args.dport, args.lwt)
+ args.dport)
msg = "vport '%s'" % args.addif
if rep and rep["header"]["error"] is None:
msg += " added."
diff --git a/tools/testing/selftests/net/ovpn/common.sh b/tools/testing/selftests/net/ovpn/common.sh
index 2d844eb3aa6e..5e9c81e885e6 100644
--- a/tools/testing/selftests/net/ovpn/common.sh
+++ b/tools/testing/selftests/net/ovpn/common.sh
@@ -136,6 +136,19 @@ ovpn_create_ns() {
ip netns add "ovpn_peer${1}"
}
+ovpn_peer_vpn_addr() {
+ local peer="$1"
+ local file
+
+ if [ "${OVPN_PROTO}" == "UDP" ]; then
+ file="${OVPN_UDP_PEERS_FILE}"
+ else
+ file="${OVPN_TCP_PEERS_FILE}"
+ fi
+
+ awk -v peer="${peer}" '$1 == peer {print $NF; exit}' "${file}"
+}
+
ovpn_setup_ns() {
local peer="ovpn_peer${1}"
local server_ns="ovpn_peer0"
diff --git a/tools/testing/selftests/net/ovpn/ovpn-cli.c b/tools/testing/selftests/net/ovpn/ovpn-cli.c
index f4effa7580c0..3b612a8a18fe 100644
--- a/tools/testing/selftests/net/ovpn/ovpn-cli.c
+++ b/tools/testing/selftests/net/ovpn/ovpn-cli.c
@@ -650,6 +650,26 @@ err:
return ret;
}
+static int ovpn_nl_put_vpn_addr(struct nl_msg *msg,
+ const struct ovpn_ctx *ovpn)
+{
+ if (!ovpn->peer_ip_set)
+ return 0;
+
+ switch (ovpn->peer_ip.in4.sin_family) {
+ case AF_INET:
+ return nla_put_u32(msg, OVPN_A_PEER_VPN_IPV4,
+ ovpn->peer_ip.in4.sin_addr.s_addr);
+ case AF_INET6:
+ return nla_put(msg, OVPN_A_PEER_VPN_IPV6,
+ sizeof(struct in6_addr),
+ &ovpn->peer_ip.in6.sin6_addr);
+ default:
+ fprintf(stderr, "Invalid family for peer address\n");
+ return -EAFNOSUPPORT;
+ }
+}
+
static int ovpn_new_peer(struct ovpn_ctx *ovpn, bool is_tcp)
{
struct nlattr *attr;
@@ -691,22 +711,9 @@ static int ovpn_new_peer(struct ovpn_ctx *ovpn, bool is_tcp)
}
}
- if (ovpn->peer_ip_set) {
- switch (ovpn->peer_ip.in4.sin_family) {
- case AF_INET:
- NLA_PUT_U32(ctx->nl_msg, OVPN_A_PEER_VPN_IPV4,
- ovpn->peer_ip.in4.sin_addr.s_addr);
- break;
- case AF_INET6:
- NLA_PUT(ctx->nl_msg, OVPN_A_PEER_VPN_IPV6,
- sizeof(struct in6_addr),
- &ovpn->peer_ip.in6.sin6_addr);
- break;
- default:
- fprintf(stderr, "Invalid family for peer address\n");
- goto nla_put_failure;
- }
- }
+ ret = ovpn_nl_put_vpn_addr(ctx->nl_msg, ovpn);
+ if (ret)
+ goto nla_put_failure;
nla_nest_end(ctx->nl_msg, attr);
@@ -732,6 +739,10 @@ static int ovpn_set_peer(struct ovpn_ctx *ovpn)
ovpn->keepalive_interval);
NLA_PUT_U32(ctx->nl_msg, OVPN_A_PEER_KEEPALIVE_TIMEOUT,
ovpn->keepalive_timeout);
+
+ ret = ovpn_nl_put_vpn_addr(ctx->nl_msg, ovpn);
+ if (ret)
+ goto nla_put_failure;
nla_nest_end(ctx->nl_msg, attr);
ret = ovpn_nl_msg_send(ctx, NULL);
@@ -1730,13 +1741,14 @@ static void usage(const char *cmd)
fprintf(stderr, "\tmark: socket FW mark value\n");
fprintf(stderr,
- "* set_peer <iface> <peer_id> <keepalive_interval> <keepalive_timeout>: set peer attributes\n");
+ "* set_peer <iface> <peer_id> <keepalive_interval> <keepalive_timeout> [vpnaddr]: set peer attributes\n");
fprintf(stderr, "\tiface: ovpn interface name\n");
fprintf(stderr, "\tpeer_id: peer ID of the peer to modify\n");
fprintf(stderr,
"\tkeepalive_interval: interval for sending ping messages\n");
fprintf(stderr,
"\tkeepalive_timeout: time after which a peer is timed out\n");
+ fprintf(stderr, "\tvpnaddr: peer VPN IP\n");
fprintf(stderr, "* del_peer <iface> <peer_id>: delete peer\n");
fprintf(stderr, "\tiface: ovpn interface name\n");
@@ -2090,6 +2102,8 @@ static int ovpn_run_cmd(struct ovpn_ctx *ovpn)
return ret;
ret = ovpn_new_peer(ovpn, false);
+ if (ret < 0)
+ return ret;
ovpn_waitbg();
break;
case CMD_NEW_MULTI_PEER:
@@ -2331,6 +2345,12 @@ static int ovpn_parse_cmd_args(struct ovpn_ctx *ovpn, int argc, char *argv[])
"keepalive interval value out of range\n");
return -1;
}
+
+ if (argc > 6) {
+ ret = ovpn_parse_remote(ovpn, NULL, NULL, argv[6]);
+ if (ret < 0)
+ return -1;
+ }
break;
case CMD_DEL_PEER:
if (argc < 4)
diff --git a/tools/testing/selftests/net/ovpn/test.sh b/tools/testing/selftests/net/ovpn/test.sh
index 9b5610837032..392109d5e14e 100755
--- a/tools/testing/selftests/net/ovpn/test.sh
+++ b/tools/testing/selftests/net/ovpn/test.sh
@@ -56,6 +56,76 @@ ovpn_prepare_network() {
done
}
+ovpn_new_test_peer() {
+ local peer_id="$1"
+
+ shift
+ ip netns exec ovpn_peer0 "${OVPN_CLI}" new_peer tun0 \
+ "${peer_id}" none 65000 10.10.1.2 1 "$@"
+}
+
+ovpn_set_peer_vpn_addr() {
+ ip netns exec ovpn_peer0 "${OVPN_CLI}" set_peer tun0 \
+ "$1" 60 120 "$2"
+}
+
+ovpn_run_vpn_addr_validation() {
+ local addr
+ local peer1_addr4
+ local test_peer_id=$((OVPN_NUM_PEERS + 1))
+ local test_peer_addr6="2001:db8::2"
+ # Do not include 0.0.0.0 or :: here. They are invalid on creation, but
+ # clear one address family on update and are valid if the other remains.
+ local -a invalid_addrs=(
+ "127.0.0.1"
+ "224.0.0.1"
+ "255.255.255.255"
+ "::1"
+ "::192.0.2.1"
+ "::ffff:192.0.2.1"
+ "ff02::1"
+ )
+
+ peer1_addr4=$(ovpn_peer_vpn_addr 1)
+
+ ovpn_cmd_fail "reject peer without VPN address" \
+ ovpn_new_test_peer "${test_peer_id}"
+
+ for addr in "0.0.0.0" "::" "${invalid_addrs[@]}"; do
+ ovpn_cmd_fail "reject new peer VPN address ${addr}" \
+ ovpn_new_test_peer "${test_peer_id}" "${addr}"
+ done
+
+ ovpn_cmd_fail "reject duplicate IPv4 address on peer creation" \
+ ovpn_new_test_peer "${test_peer_id}" "${peer1_addr4}"
+ ovpn_cmd_fail "reject clearing the last peer VPN address" \
+ ovpn_set_peer_vpn_addr 1 0.0.0.0
+
+ for addr in "${invalid_addrs[@]}"; do
+ ovpn_cmd_fail "reject updated peer VPN address ${addr}" \
+ ovpn_set_peer_vpn_addr 1 "${addr}"
+ done
+
+ ovpn_cmd_fail "reject duplicate IPv4 address on peer update" \
+ ovpn_set_peer_vpn_addr 2 "${peer1_addr4}"
+
+ ovpn_cmd_ok "add peer IPv6 address" \
+ ovpn_set_peer_vpn_addr 1 "${test_peer_addr6}"
+ ovpn_cmd_fail "reject duplicate IPv6 address on peer creation" \
+ ovpn_new_test_peer "${test_peer_id}" "${test_peer_addr6}"
+ ovpn_cmd_fail "reject duplicate IPv6 address on peer update" \
+ ovpn_set_peer_vpn_addr 2 "${test_peer_addr6}"
+
+ ovpn_cmd_ok "clear peer IPv4 address" \
+ ovpn_set_peer_vpn_addr 1 0.0.0.0
+ ovpn_cmd_fail "reject clearing the remaining peer IPv6 address" \
+ ovpn_set_peer_vpn_addr 1 ::
+ ovpn_cmd_ok "restore peer IPv4 address" \
+ ovpn_set_peer_vpn_addr 1 "${peer1_addr4}"
+ ovpn_cmd_ok "clear peer IPv6 address" \
+ ovpn_set_peer_vpn_addr 1 ::
+}
+
ovpn_run_basic_traffic() {
local p
local header1
@@ -293,15 +363,16 @@ trap ovpn_stage_err ERR
ktap_print_header
if [ "${OVPN_FLOAT}" == "1" ]; then
- ktap_set_plan 13
+ ktap_set_plan 14
else
- ktap_set_plan 12
+ ktap_set_plan 13
fi
ovpn_cleanup
modprobe -q ovpn || true
ovpn_run_stage "setup network topology" ovpn_prepare_network
+ovpn_run_stage "validate peer VPN addresses" ovpn_run_vpn_addr_validation
ovpn_run_stage "run baseline data traffic" ovpn_run_basic_traffic
ovpn_run_stage "run LAN traffic behind peer1" ovpn_run_lan_traffic
[ "${OVPN_FLOAT}" == "1" ] && ovpn_run_stage "run floating peer checks" \
diff --git a/tools/testing/selftests/net/packetdrill/config b/tools/testing/selftests/net/packetdrill/config
index c4a19a785521..b7df8bf30920 100644
--- a/tools/testing/selftests/net/packetdrill/config
+++ b/tools/testing/selftests/net/packetdrill/config
@@ -4,8 +4,8 @@ CONFIG_IPV6=y
CONFIG_NET_NS=y
CONFIG_NET_SCH_FIFO=y
CONFIG_NET_SCH_FQ=y
-CONFIG_PROC_SYSCTL=y
CONFIG_SYN_COOKIES=y
+CONFIG_SYSCTL=y
CONFIG_TCP_CONG_CUBIC=y
CONFIG_TCP_MD5SIG=y
CONFIG_TUN=y
diff --git a/tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv4.pkt b/tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv4.pkt
new file mode 100644
index 000000000000..f2ef931b77a1
--- /dev/null
+++ b/tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv4.pkt
@@ -0,0 +1,67 @@
+// SPDX-License-Identifier: GPL-2.0
+//
+// Test that IPv4 advertised MSS in SYN-ACK is derived from the configured
+// interface MTU (1500 -> MSS 1460), not the ICMP-learned Path MTU.
+
+--ip_version=ipv4
+
+`./defaults.sh
+ethtool -K tun0 tso off
+`
+
+//
+// Connection 1: Learn PMTU exception (MTU 1200 -> MSS 1160)
+//
+ 0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3
+ +0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0
+ +0 bind(3, ..., ...) = 0
+ +0 listen(3, 1) = 0
+
+ +0 < S 0:0(0) win 65535 <mss 1460,sackOK,nop,nop,nop,wscale 8>
+ +0 > S. 0:0(0) ack 1 <mss 1460,nop,nop,sackOK,nop,wscale 8>
+ +.1 < . 1:1(0) ack 1 win 257
+ +0 accept(3, ..., ...) = 4
+
+// Send a full 1460-byte segment
+ +0 write(4, ..., 1460) = 1460
+ +0 > P. 1:1461(1460) ack 1
+
+// ICMP Fragmentation Needed arrives indicating next-hop MTU 1200
+ +0 < icmp unreachable frag_needed mtu 1200 [1:1461(1460)]
+
+// Local host retransmits using the learned MTU 1200 (MSS = 1200 - 40 = 1160)
+ +0 > . 1:1161(1160) ack 1
+ +0 > P. 1161:1461(300) ack 1
+ +0 < R 1:1(0) ack 1461 win 0
+
+// Close connection 1 and listener
+ +0 close(4) = 0
+ +0 close(3) = 0
+
+//
+// Connection 2: New connection from the same peer
+//
+ +0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3
+ +0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0
+ +0 bind(3, ..., ...) = 0
+ +0 listen(3, 1) = 0
+
+ +0 < S 0:0(0) win 65535 <mss 1460,sackOK,nop,nop,nop,wscale 8>
+
+// Verify: SYN-ACK MUST advertise configured MSS 1460, NOT the learned PMTU MSS 1160
+ +0 > S. 0:0(0) ack 1 <mss 1460,nop,nop,sackOK,nop,wscale 8>
+ +0 < . 1:1(0) ack 1 win 257
+ +0 accept(3, ..., ...) = 4
+
+// Verify: Outgoing transmit MSS is still constrained by the learned PMTU 1200
+ +0 write(4, ..., 1460) = 1460
+ +0 > . 1:1161(1160) ack 1
+ +0 > P. 1161:1461(300) ack 1
+ +0 < . 1:1(0) ack 1461 win 257
+
+// Clean up
+ +0 close(4) = 0
+ +0 > F. 1461:1461(0) ack 1
+ +0 < F. 1:1(0) ack 1462 win 257
+ +0 > . 1462:1462(0) ack 2
+ +0 close(3) = 0
diff --git a/tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv6.pkt b/tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv6.pkt
new file mode 100644
index 000000000000..c7638b11a815
--- /dev/null
+++ b/tools/testing/selftests/net/packetdrill/tcp_advmss_pmtu_ipv6.pkt
@@ -0,0 +1,67 @@
+// SPDX-License-Identifier: GPL-2.0
+//
+// Test that IPv6 advertised MSS in SYN-ACK is derived from the configured
+// interface MTU (1520 -> MSS 1460), not the ICMPv6-learned Path MTU.
+
+--ip_version=ipv6
+
+`./defaults.sh
+ethtool -K tun0 tso off
+`
+
+//
+// Connection 1: Learn PMTU exception (MTU 1280 -> MSS 1220)
+//
+ 0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3
+ +0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0
+ +0 bind(3, ..., ...) = 0
+ +0 listen(3, 1) = 0
+
+ +0 < S 0:0(0) win 65535 <mss 1460,sackOK,nop,nop,nop,wscale 8>
+ +0 > S. 0:0(0) ack 1 <mss 1460,nop,nop,sackOK,nop,wscale 8>
+ +.1 < . 1:1(0) ack 1 win 257
+ +0 accept(3, ..., ...) = 4
+
+// Send a full 1460-byte segment
+ +0 write(4, ..., 1460) = 1460
+ +0 > P. 1:1461(1460) ack 1
+
+// ICMPv6 Packet Too Big arrives indicating next-hop MTU 1280
+ +0 < icmp packet_too_big mtu 1280 [1:1461(1460)]
+
+// Local host retransmits using the learned MTU 1280 (MSS = 1280 - 40 - 20 = 1220)
+ +0 > . 1:1221(1220) ack 1
+ +0 > P. 1221:1461(240) ack 1
+ +0 < R 1:1(0) ack 1461 win 0
+
+// Close connection 1 and listener
+ +0 close(4) = 0
+ +0 close(3) = 0
+
+//
+// Connection 2: New connection from the same peer
+//
+ +0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3
+ +0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0
+ +0 bind(3, ..., ...) = 0
+ +0 listen(3, 1) = 0
+
+ +0 < S 0:0(0) win 65535 <mss 1460,sackOK,nop,nop,nop,wscale 8>
+
+// Verify: SYN-ACK MUST advertise configured MSS 1460, NOT the learned PMTU MSS 1220
+ +0 > S. 0:0(0) ack 1 <mss 1460,nop,nop,sackOK,nop,wscale 8>
+ +0 < . 1:1(0) ack 1 win 257
+ +0 accept(3, ..., ...) = 4
+
+// Verify: Outgoing transmit MSS is still constrained by the learned PMTU 1280
+ +0 write(4, ..., 1460) = 1460
+ +0 > . 1:1221(1220) ack 1
+ +0 > P. 1221:1461(240) ack 1
+ +0 < . 1:1(0) ack 1461 win 257
+
+// Clean up
+ +0 close(4) = 0
+ +0 > F. 1461:1461(0) ack 1
+ +0 < F. 1:1(0) ack 1462 win 257
+ +0 > . 1462:1462(0) ack 2
+ +0 close(3) = 0
diff --git a/tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt b/tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt
new file mode 100644
index 000000000000..32dd9de1d366
--- /dev/null
+++ b/tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt
@@ -0,0 +1,29 @@
+// SPDX-License-Identifier: GPL-2.0
+
+`./defaults.sh
+sysctl -q net.ipv4.tcp_invalid_ratelimit=0
+`
+
+// Test rejection of data segments carrying excessively old ACKs
+
+0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3
++0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0
++0 bind(3, ..., ...) = 0
++0 listen(3, 1024) = 0
+
+// ---------------- Handshake ------------------- //
++0 < S 0:0(0) win 65535
++0 > S. 0:0(0) ack 1 <...>
++0 < . 1:1(0) ack 1 win 65535
++0 accept(3, ..., ...) = 4
+
+// Populate receive memory so the following segment can use
+// header prediction.
++0 < P. 1:501(500) ack 1 win 65535
++0 > . 1:1(0) ack 501
+
+// Send an in-sequence data segment carrying an excessively old ACK.
++0 < P. 501:1501(1000) ack 2794967397 win 65535
+
+// Challenge ACK; RCV.NXT must remain 501.
++0 > . 1:1(0) ack 501
diff --git a/tools/testing/selftests/net/packetdrill/tcp_urg_ptr_retransmit.pkt b/tools/testing/selftests/net/packetdrill/tcp_urg_ptr_retransmit.pkt
new file mode 100644
index 000000000000..22f750ce09c1
--- /dev/null
+++ b/tools/testing/selftests/net/packetdrill/tcp_urg_ptr_retransmit.pkt
@@ -0,0 +1,65 @@
+// SPDX-License-Identifier: GPL-2.0
+--ip_version=ipv4
+//
+// Reproduce urg_ptr being copied across segments on a multi-segment retransmit
+// in urgent mode (regression since 10d3be569243).
+//
+// server (kernel, under test) client (packetdrill)
+// | write(5000): 1:1001 .. 4001:5001 | mss 1000 from
+// | -------------------------------------------> | the client SYN
+// | send(MSG_OOB): 5001:5002 urg 1 | snd_up = 5002
+// | -------------------------------------------> |
+// | SACK 2001:5002, leaving hole 1:2001|
+// | <------------------------------------------- |
+// | retransmit hole 1:2001 as ONE skb: |
+// | seq=1, 2 segments, urg_ptr = 5002-1 = 5001|
+// | tun tso off -> software GSO splits it: |
+// | seg A 1:1001 urg_ptr 5001 (correct) |
+// | seg B 1001:2001 urg_ptr ? |
+// | want 5002-1001 = 4001 |
+// | bug inherits 5001 <- caught here |
+// | -------------------------------------------> |
+//
+
+`./defaults.sh`
+
+ 0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3
+ +0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0
+ +0 bind(3, ..., ...) = 0
+ +0 listen(3, 1) = 0
+
+// 1. client force mss=1000
+ +.1 < S 0:0(0) win 32792 <mss 1000,sackOK,nop,nop,nop,wscale 7>
+ +0 > S. 0:0(0) ack 1 <mss 1460,nop,nop,sackOK,nop,wscale 8>
+ +.1 < . 1:1(0) ack 1 win 320
+ +0 accept(3, ..., ...) = 4
+
+// 2. server sends 5000 bytes; TSO on, so packetdrill sees whole super-skbs
+ +0 write(4, ..., 5000) = 5000
+ +0 > P. 1:5001(5000) ack 1
+
+// 3. server send OOB
+ +0 send(4, ..., 1, MSG_OOB) = 1
+ +0 > PU. 5001:5002(1) ack 1 urg 1
+
+// We could disable GSO at the start of the script, but then the PSH flag on
+// the 5 initial server segments is not deterministic and hard to match. Keep
+// TSO on for the initial send (one super-skb, stable PSH) and disable it only
+// here, so software GSO splits the retransmit and each segment's urg_ptr is
+// checked on the wire.
+ +0 `ethtool -K tun0 tso off gso off gro off lro off 2>/dev/null`
+
+// 4. SACKed blocks reach dupthresh -> fast retransmit of the 1:2001 hole.
+ +.05 < . 1:1(0) ack 1 win 320 <sack 2001:3001,nop,nop>
+ +0 < . 1:1(0) ack 1 win 320 <sack 2001:4001,nop,nop>
+ +0 < . 1:1(0) ack 1 win 320 <sack 2001:5002,nop,nop>
+
+// Retransmit must keep a per-segment urg_ptr (5002 - seg.seq): seg A 5001,
+// seg B 4001. The fix sends the hole as two independent skbs, so seg B has
+// no PSH. Unpatched it goes out as one super-skb whose GSO split copies
+// urg_ptr onto seg B and also adds PSH there, so on an unpatched kernel the
+// mismatch shows up on the PSH bit before the urg_ptr.
+ +0 > U. 1:1001(1000) ack 1 urg 5001
+ +0 > U. 1001:2001(1000) ack 1 urg 4001
+
+ +.1 < . 1:1(0) ack 5002 win 320
diff --git a/tools/testing/selftests/net/pmtu.sh b/tools/testing/selftests/net/pmtu.sh
index a3323c21f001..c7cd271714ef 100755
--- a/tools/testing/selftests/net/pmtu.sh
+++ b/tools/testing/selftests/net/pmtu.sh
@@ -1457,7 +1457,7 @@ test_pmtu_ipvX_over_bridged_vxlanY_or_geneveY_exception() {
mtu "${ns_b}" ${type}_b $((${ll_mtu} + 1000))
run_cmd ${ns_c} ${ping} -q -M want -i 0.1 -c 10 -s $((${ll_mtu} + 500)) ${dst} || return 1
- run_cmd ${ns_a} ${ping} -q -M want -i 0.1 -w 1 -s $((${ll_mtu} + 500)) ${dst} || return 1
+ run_cmd ${ns_a} ${ping} -q -M want -i 0.1 -c 10 -s $((${ll_mtu} + 500)) ${dst} || return 1
# Check that exceptions were created
pmtu="$(route_get_dst_pmtu_from_exception "${ns_c}" ${dst})"
diff --git a/tools/testing/selftests/net/psock_snd.c b/tools/testing/selftests/net/psock_snd.c
index edf1e6f80d41..0f6a30b26912 100644
--- a/tools/testing/selftests/net/psock_snd.c
+++ b/tools/testing/selftests/net/psock_snd.c
@@ -39,6 +39,9 @@ static bool cfg_use_gso;
static bool cfg_use_qdisc_bypass;
static bool cfg_use_vlan;
static bool cfg_use_vnet;
+static bool cfg_drop;
+static bool cfg_aux_data;
+static bool cfg_ignore_outgoing;
static char *cfg_ifname = "lo";
static int cfg_mtu = 1500;
@@ -49,6 +52,8 @@ static uint16_t cfg_port = 8000;
/* test sending up to max mtu + 1 */
#define TEST_SZ (sizeof(struct virtio_net_hdr) + ETH_HLEN + ETH_MAX_MTU + 1)
+#define BURST_CNT (1000)
+
static char tbuf[TEST_SZ], rbuf[TEST_SZ];
static unsigned long add_csum_hword(const uint16_t *start, int num_u16)
@@ -167,12 +172,12 @@ static int build_packet(int payload_len)
return off + payload_len;
}
-static void do_bind(int fd)
+static void do_bind_proto(int fd, uint16_t proto)
{
struct sockaddr_ll laddr = {0};
laddr.sll_family = AF_PACKET;
- laddr.sll_protocol = htons(ETH_P_IP);
+ laddr.sll_protocol = htons(proto);
laddr.sll_ifindex = if_nametoindex(cfg_ifname);
if (!laddr.sll_ifindex)
error(1, errno, "if_nametoindex");
@@ -181,6 +186,11 @@ static void do_bind(int fd)
error(1, errno, "bind");
}
+static void do_bind(int fd)
+{
+ do_bind_proto(fd, ETH_P_IP);
+}
+
static void do_send(int fd, char *buf, int len)
{
int ret;
@@ -212,13 +222,14 @@ static void do_send(int fd, char *buf, int len)
if (ret != len)
error(1, 0, "write: %u %u", ret, len);
- fprintf(stderr, "tx: %u\n", ret);
+ if (!cfg_drop)
+ fprintf(stderr, "tx: %u\n", ret);
}
static int do_tx(void)
{
const int one = 1;
- int fd, len;
+ int i, fd, len;
fd = socket(PF_PACKET, cfg_use_dgram ? SOCK_DGRAM : SOCK_RAW, 0);
if (fd == -1)
@@ -242,6 +253,10 @@ static int do_tx(void)
do_send(fd, tbuf, len);
+ if (cfg_drop)
+ for (i = 0; i < BURST_CNT; i++)
+ do_send(fd, tbuf, len);
+
if (close(fd))
error(1, errno, "close t");
@@ -271,11 +286,58 @@ static int setup_rx(void)
return fd;
}
-static void do_rx(int fd, int expected_len, char *expected)
+static void check_aux_data(struct cmsghdr *cmsg, int expected_len)
+{
+ struct tpacket_auxdata *adata;
+
+ if (!cmsg)
+ error(1, 0, "auxdata null");
+
+ if (cmsg->cmsg_level != SOL_PACKET)
+ error(1, 0, "cmsg_level != SOL_PACKET");
+
+ if (cmsg->cmsg_type != PACKET_AUXDATA)
+ error(1, 0, "cmsg_type != PACKET_AUXDATA");
+
+ adata = (struct tpacket_auxdata *)CMSG_DATA(cmsg);
+
+ if (adata->tp_net != ETH_HLEN)
+ error(1, 0, "cmsg tp_net != ETH_HLEN");
+
+ if (adata->tp_len != expected_len)
+ error(1, 0, "cmsg tp_len != %u", expected_len);
+
+ if (adata->tp_snaplen != expected_len)
+ error(1, 0, "cmsg tp_snaplen != %u", expected_len);
+}
+
+/* expected_pkttype < 0 skips the sll_pkttype check. */
+static void do_rx(int fd, int expected_len, char *expected, bool is_psock,
+ int expected_pkttype)
{
+ char cmsg_buf[1024] __attribute__((aligned(8))) = {};
+ bool aux = is_psock && cfg_aux_data;
+ struct sockaddr_ll saddr = {};
+ struct iovec iov = {
+ .iov_base = rbuf,
+ .iov_len = sizeof(rbuf),
+ };
+ struct msghdr msg = {
+ .msg_iov = &iov,
+ .msg_iovlen = 1,
+ };
int ret;
- ret = recv(fd, rbuf, sizeof(rbuf), 0);
+ if (aux) {
+ msg.msg_control = cmsg_buf;
+ msg.msg_controllen = sizeof(cmsg_buf);
+ }
+ if (is_psock) {
+ msg.msg_name = &saddr;
+ msg.msg_namelen = sizeof(saddr);
+ }
+
+ ret = recvmsg(fd, &msg, 0);
if (ret == -1)
error(1, errno, "recv");
if (ret != expected_len)
@@ -284,12 +346,20 @@ static void do_rx(int fd, int expected_len, char *expected)
if (memcmp(rbuf, expected, ret))
error(1, 0, "recv: data mismatch");
+ if (aux)
+ check_aux_data(CMSG_FIRSTHDR(&msg), expected_len);
+
+ if (expected_pkttype >= 0 && saddr.sll_pkttype != expected_pkttype)
+ error(1, 0, "recv: sll_pkttype %d != %d",
+ saddr.sll_pkttype, expected_pkttype);
+
fprintf(stderr, "rx: %u\n", ret);
}
static int setup_sniffer(void)
{
struct timeval tv = { .tv_usec = 100 * 1000 };
+ const int one = 1;
int fd;
fd = socket(PF_PACKET, SOCK_RAW, 0);
@@ -299,8 +369,23 @@ static int setup_sniffer(void)
if (setsockopt(fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)))
error(1, errno, "setsockopt rcv timeout");
+ if (cfg_drop)
+ if (setsockopt(fd, SOL_SOCKET, SO_RCVBUF, &one, sizeof(one)))
+ error(1, errno, "setsockopt SO_RCVBUF");
+
+ if (cfg_aux_data)
+ if (setsockopt(fd, SOL_PACKET, PACKET_AUXDATA, &one, sizeof(one)))
+ error(1, errno, "setsockopt PACKET_AUXDATA");
+
pair_udp_setfilter(fd);
- do_bind(fd);
+
+ /* binding to ETH_P_ALL adds the sniffer to ptype_all, which will see
+ * the dev_queue_xmit_nit copy. ignore_outgoing should suppress this.
+ */
+ if (cfg_ignore_outgoing)
+ do_bind_proto(fd, ETH_P_ALL);
+ else
+ do_bind(fd);
return fd;
}
@@ -309,8 +394,11 @@ static void parse_opts(int argc, char **argv)
{
int c;
- while ((c = getopt(argc, argv, "bcCdgl:qt:vV")) != -1) {
+ while ((c = getopt(argc, argv, "abcCdDgil:qt:vV")) != -1) {
switch (c) {
+ case 'a':
+ cfg_aux_data = true;
+ break;
case 'b':
cfg_use_bind = true;
break;
@@ -323,9 +411,15 @@ static void parse_opts(int argc, char **argv)
case 'd':
cfg_use_dgram = true;
break;
+ case 'D':
+ cfg_drop = true;
+ break;
case 'g':
cfg_use_gso = true;
break;
+ case 'i':
+ cfg_ignore_outgoing = true;
+ break;
case 'l':
cfg_payload_len = strtoul(optarg, NULL, 0);
break;
@@ -357,6 +451,114 @@ static void parse_opts(int argc, char **argv)
if (cfg_use_gso && !cfg_use_csum_off)
error(1, 0, "option gso (-g) requires csum offload (-c)");
+
+ if (cfg_aux_data && cfg_drop)
+ error(1, 0, "option aux data (-a) conflicts with drop (-D)");
+
+ if (cfg_ignore_outgoing && (cfg_drop || cfg_aux_data))
+ error(1, 0,
+ "option ignore outgoing (-i) conflicts with -D and -a");
+}
+
+static void check_packet_stats(int fd, unsigned int expected_packets)
+{
+ struct tpacket_stats st = {};
+ socklen_t len = sizeof(st);
+
+ if (getsockopt(fd, SOL_PACKET, PACKET_STATISTICS, &st, &len))
+ error(1, errno, "getsockopt packet statistics");
+
+ if (cfg_drop) {
+ /* PACKET_STATISTICS reports all packets seen (including
+ * drops) in tp_packets
+ */
+ if (st.tp_packets < st.tp_drops)
+ error(1, 0, "stats: tp_packets %u < tp_drops %u",
+ st.tp_packets, st.tp_drops);
+
+ if (st.tp_drops == 0)
+ error(1, 0, "stats: expected drops but tp_drops == 0");
+ } else {
+ if (st.tp_packets != expected_packets)
+ error(1, 0, "stats: tp_packets %u != %u",
+ st.tp_packets, expected_packets);
+
+ if (st.tp_drops != 0)
+ error(1, 0, "stats: tp_drops %u != 0", st.tp_drops);
+ }
+
+ /* verify clear on read */
+ memset(&st, 0xff, sizeof(st));
+ len = sizeof(st);
+
+ if (getsockopt(fd, SOL_PACKET, PACKET_STATISTICS, &st, &len))
+ error(1, errno, "getsockopt packet statistics");
+
+ if (st.tp_packets != 0)
+ error(1, 0, "stats: tp_packets %u != 0 after clear", st.tp_packets);
+
+ if (st.tp_drops != 0)
+ error(1, 0, "stats: tp_drops %u != 0 after clear", st.tp_drops);
+}
+
+static void set_ignore_outgoing(int fd, int val)
+{
+ socklen_t len = sizeof(int);
+ int got = -1;
+
+ if (setsockopt(fd, SOL_PACKET, PACKET_IGNORE_OUTGOING,
+ &val, sizeof(val)))
+ error(1, errno, "setsockopt PACKET_IGNORE_OUTGOING %d", val);
+
+ if (getsockopt(fd, SOL_PACKET, PACKET_IGNORE_OUTGOING, &got, &len))
+ error(1, errno, "getsockopt PACKET_IGNORE_OUTGOING");
+ if (got != val)
+ error(1, 0, "getsockopt: expected %d got %d", val, got);
+}
+
+static void check_ignore_outgoing_range(int fd)
+{
+ int val;
+
+ /* Values outside [0, 1] must be rejected with -EINVAL. */
+ val = 2;
+ if (setsockopt(fd, SOL_PACKET, PACKET_IGNORE_OUTGOING,
+ &val, sizeof(val)) != -1 || errno != EINVAL)
+ error(1, errno,
+ "setsockopt PACKET_IGNORE_OUTGOING val=2: expected EINVAL");
+
+ val = -1;
+ if (setsockopt(fd, SOL_PACKET, PACKET_IGNORE_OUTGOING,
+ &val, sizeof(val)) != -1 || errno != EINVAL)
+ error(1, errno,
+ "setsockopt PACKET_IGNORE_OUTGOING val=-1: expected EINVAL");
+}
+
+static void test_ignore_outgoing(int fds)
+{
+ char *expected = tbuf + sizeof(struct virtio_net_hdr);
+ int expected_len;
+
+ /* ptype_all sniffer on loopback should produce two copies per packet
+ * (RX and TX).
+ */
+ expected_len = do_tx();
+ expected_len -= sizeof(struct virtio_net_hdr);
+ do_rx(fds, expected_len, expected, true, PACKET_OUTGOING);
+ do_rx(fds, expected_len, expected, true, PACKET_HOST);
+ check_packet_stats(fds, 2);
+
+ /* 0 and 1 accepted; anything else rejected. */
+ set_ignore_outgoing(fds, 0);
+ set_ignore_outgoing(fds, 1);
+ check_ignore_outgoing_range(fds);
+
+ /* With PACKET_IGNORE_OUTGOING set, only the rx copy survives. */
+ do_tx();
+ do_rx(fds, expected_len, expected, true, PACKET_HOST);
+ if (recv(fds, rbuf, sizeof(rbuf), 0) != -1 || errno != EAGAIN)
+ error(1, errno, "expected EAGAIN, got extra packet");
+ check_packet_stats(fds, 1);
}
static void run_test(void)
@@ -366,15 +568,28 @@ static void run_test(void)
fdr = setup_rx();
fds = setup_sniffer();
+ if (cfg_ignore_outgoing) {
+ test_ignore_outgoing(fds);
+ goto out;
+ }
+
total_len = do_tx();
+ if (cfg_drop) {
+ check_packet_stats(fds, 0);
+ goto out;
+ }
+
/* BPF filter accepts only this length, vlan changes MAC */
- if (cfg_payload_len == DATA_LEN && !cfg_use_vlan)
+ if (cfg_payload_len == DATA_LEN && !cfg_use_vlan) {
do_rx(fds, total_len - sizeof(struct virtio_net_hdr),
- tbuf + sizeof(struct virtio_net_hdr));
+ tbuf + sizeof(struct virtio_net_hdr), true, -1);
+ check_packet_stats(fds, 1);
+ }
- do_rx(fdr, cfg_payload_len, tbuf + total_len - cfg_payload_len);
+ do_rx(fdr, cfg_payload_len, tbuf + total_len - cfg_payload_len, false, -1);
+out:
if (close(fds))
error(1, errno, "close s");
if (close(fdr))
diff --git a/tools/testing/selftests/net/psock_snd.sh b/tools/testing/selftests/net/psock_snd.sh
index 1cbfeb5052ec..7fa0a3297988 100755
--- a/tools/testing/selftests/net/psock_snd.sh
+++ b/tools/testing/selftests/net/psock_snd.sh
@@ -92,4 +92,19 @@ echo "raw gso max size"
echo "raw gso max size + 1 (expected to fail)"
(! ./in_netns.sh ./psock_snd -v -c -g -l "${max_mss_exceeds}")
+# test drops statistics
+
+echo "test drops statistics"
+./in_netns.sh ./psock_snd -D
+
+# test aux data
+
+echo "test aux data"
+./in_netns.sh ./psock_snd -a
+
+# test ignore outgoing
+
+echo "test ignore outgoing"
+./in_netns.sh ./psock_snd -i
+
echo "OK. All tests passed"
diff --git a/tools/testing/selftests/net/reuseport_bpf_numa.c b/tools/testing/selftests/net/reuseport_bpf_numa.c
index 8ec52fc5ef41..6e4817ef57c5 100644
--- a/tools/testing/selftests/net/reuseport_bpf_numa.c
+++ b/tools/testing/selftests/net/reuseport_bpf_numa.c
@@ -104,6 +104,26 @@ static void attach_bpf(int fd)
close(bpf_fd);
}
+/*
+ * Return true if it is a cpuless node. Return false if it isn't or any
+ * error (very unlikely) happens during the libnuma calls.
+ */
+static bool is_cpuless_node(int node_id)
+{
+ struct bitmask *cpumask;
+ bool ret = false;
+
+ cpumask = numa_allocate_cpumask();
+ if (!cpumask)
+ return ret;
+
+ if (!numa_node_to_cpus(node_id, cpumask) && !numa_bitmask_weight(cpumask))
+ ret = true;
+
+ numa_bitmask_free(cpumask);
+ return ret;
+}
+
static void send_from_node(int node_id, int family, int proto)
{
struct sockaddr_storage saddr, daddr;
@@ -213,6 +233,8 @@ static void test(int *rcv_fd, int len, int family, int proto)
for (node = 0; node < len; ++node) {
if (!numa_bitmask_isbitset(numa_nodes_ptr, node))
continue;
+ if (is_cpuless_node(node))
+ continue;
send_from_node(node, family, proto);
receive_on_node(rcv_fd, len, epfd, node, proto);
}
@@ -221,6 +243,8 @@ static void test(int *rcv_fd, int len, int family, int proto)
for (node = len - 1; node >= 0; --node) {
if (!numa_bitmask_isbitset(numa_nodes_ptr, node))
continue;
+ if (is_cpuless_node(node))
+ continue;
send_from_node(node, family, proto);
receive_on_node(rcv_fd, len, epfd, node, proto);
}
diff --git a/tools/testing/selftests/net/rtnetlink.py b/tools/testing/selftests/net/rtnetlink.py
index 3622413d793d..5cc3ebdcf08d 100755
--- a/tools/testing/selftests/net/rtnetlink.py
+++ b/tools/testing/selftests/net/rtnetlink.py
@@ -2,27 +2,108 @@
# SPDX-License-Identifier: GPL-2.0
import socket
+import struct
import time
-from lib.py import bkg, ip, ksft_exit, ksft_run, ksft_ge, ksft_true, KsftSkipEx
-from lib.py import CmdExitFailure, NetNS, NetNSEnter, RtnlAddrFamily
+from lib.py import bkg, ip, ksft_exit, ksft_run, ksft_eq, ksft_ge, ksft_true, KsftSkipEx
+from lib.py import ksft_not_in, ksft_not_none
+from lib.py import CmdExitFailure, NetNS, NetNSEnter, RtnlAddrFamily, RtnlRouteFamily
+from lib.py import defer
IPV4_ALL_HOSTS_MULTICAST = b'\xe0\x00\x00\x01'
+IPV4_TEST_MULTICAST = b'\xef\x01\x01\x01'
+IPV6_TEST_MULTICAST = bytes.fromhex('ff020000000000000000000000000123')
+
+
+def _users_for(rtnl: RtnlAddrFamily, family: int, grp: bytes, ifindex: int):
+ """Return mc-users for grp on ifindex, or 0 if absent."""
+
+ addrs = rtnl.getmulticast({"ifa-family": family}, dump=True)
+ matches = [addr for addr in addrs
+ if addr['multicast'] == grp and addr['ifa-index'] == ifindex]
+ if not matches:
+ return 0
+ if 'mc-users' not in matches[0]:
+ return None
+
+ return matches[0]['mc-users']
+
def dump_mcaddr_check() -> None:
"""
- Verify that at least one interface has the IPv4 all-hosts multicast address.
- At least the loopback interface should have this address.
+ Verify IPv4 multicast addresses and their user counts in RTM_GETMULTICAST.
"""
- rtnl = RtnlAddrFamily()
- addresses = rtnl.getmulticast({"ifa-family": socket.AF_INET}, dump=True)
+ with NetNS() as ns:
+ with NetNSEnter(str(ns)):
+ ip("link set lo up")
+ rtnl = RtnlAddrFamily()
+ lo_idx = socket.if_nametoindex('lo')
+ addresses = rtnl.getmulticast({"ifa-family": socket.AF_INET}, dump=True)
+
+ all_host_multicasts = [
+ addr for addr in addresses
+ if addr['multicast'] == IPV4_ALL_HOSTS_MULTICAST
+ ]
- all_host_multicasts = [
- addr for addr in addresses if addr['multicast'] == IPV4_ALL_HOSTS_MULTICAST
- ]
+ ksft_ge(len(all_host_multicasts), 1,
+ "No interface found with the IPv4 all-hosts multicast address")
+
+ mreq = IPV4_TEST_MULTICAST + socket.inet_aton('127.0.0.1')
+ before = _users_for(rtnl, socket.AF_INET, IPV4_TEST_MULTICAST, lo_idx)
+ if before is None:
+ raise KsftSkipEx("kernel does not expose IFA_MC_USERS")
+
+ s1 = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
+ s2 = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
+ try:
+ s1.setsockopt(socket.IPPROTO_IP, socket.IP_ADD_MEMBERSHIP, mreq)
+ s2.setsockopt(socket.IPPROTO_IP, socket.IP_ADD_MEMBERSHIP, mreq)
+
+ after_join = _users_for(rtnl, socket.AF_INET,
+ IPV4_TEST_MULTICAST, lo_idx)
+ if after_join is None:
+ raise KsftSkipEx("kernel does not expose IFA_MC_USERS")
+ ksft_eq(after_join - before, 2,
+ f"users delta != 2 after two joins "
+ f"(before={before}, after={after_join})")
+ finally:
+ s1.close()
+ s2.close()
+
+
+def dump_mcaddr6_check() -> None:
+ """
+ Verify IPv6 multicast addresses and their user counts in RTM_GETMULTICAST.
+ """
+
+ with NetNS() as ns:
+ with NetNSEnter(str(ns)):
+ ip("link set lo up")
+ rtnl = RtnlAddrFamily()
+ lo_idx = socket.if_nametoindex('lo')
+ before = _users_for(rtnl, socket.AF_INET6,
+ IPV6_TEST_MULTICAST, lo_idx)
+ if before is None:
+ raise KsftSkipEx("kernel does not expose IFA_MC_USERS for IPv6")
+
+ mreq = IPV6_TEST_MULTICAST + struct.pack('=I', lo_idx)
+ s1 = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM)
+ s2 = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM)
+ try:
+ s1.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_JOIN_GROUP, mreq)
+ s2.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_JOIN_GROUP, mreq)
+
+ after_join = _users_for(rtnl, socket.AF_INET6,
+ IPV6_TEST_MULTICAST, lo_idx)
+ if after_join is None:
+ raise KsftSkipEx("kernel does not expose IFA_MC_USERS for IPv6")
+ ksft_eq(after_join - before, 2,
+ f"IPv6 users delta != 2 after two joins "
+ f"(before={before}, after={after_join})")
+ finally:
+ s1.close()
+ s2.close()
- ksft_ge(len(all_host_multicasts), 1,
- "No interface found with the IPv4 all-hosts multicast address")
def ipv4_devconf_notify() -> None:
"""
@@ -55,8 +136,189 @@ def ipv4_devconf_notify() -> None:
ksft_true(f"inet {ifname} forwarding on" in cmd_obj.stdout,
f"No 'forwarding on' notificiation found for interface {ifname}")
+def _rtnl_route_subscribe(ns):
+ with NetNSEnter(str(ns)):
+ rtnl = RtnlRouteFamily()
+ defer(rtnl.close)
+ rtnl.ntf_subscribe("rtnlgrp-ipv6-route")
+ return rtnl
+
+
+def _wait_route_ntf(rtnl, name, dst_len, dst=None, deadline=10):
+ """Return the attrs of the first matching notification, None on timeout."""
+
+ for msg in rtnl.poll_ntf(duration=deadline):
+ if msg['name'] != name:
+ continue
+ attrs = msg['msg']
+ if attrs['rtm-dst-len'] != dst_len:
+ continue
+ if dst is not None and attrs.get('dst') != dst:
+ continue
+ return attrs
+ return None
+
+
+def _collect_route_ntfs(rtnl, name, want, deadline=10):
+ """Gather attrs of matching notifications, keyed by (dst_len, dst)."""
+
+ seen = {}
+ for msg in rtnl.poll_ntf(duration=deadline):
+ if msg['name'] != name:
+ continue
+ attrs = msg['msg']
+ key = (attrs['rtm-dst-len'], attrs.get('dst'))
+ if key in want:
+ seen[key] = attrs
+ if len(seen) == len(want):
+ break
+ return seen
+
+
+def _write_ipv6_sysctl(name, value):
+ with open(f"/proc/sys/net/ipv6/{name}", "w") as f:
+ f.write(f"{value}\n")
+
+
+def ipv6_route_del_reason_expired() -> None:
+ """An expired route reports RTA_DEL_REASON == expired."""
+
+ with NetNS() as ns:
+ rtnl = _rtnl_route_subscribe(ns)
+ with NetNSEnter(str(ns)):
+ _write_ipv6_sysctl("route/gc_interval", 2)
+ ip("link add name dummy1 type dummy", ns=str(ns))
+ ip("link set dev dummy1 up", ns=str(ns))
+ ip("-6 route add 2001:db8:2::/64 dev dummy1 expires 2", ns=str(ns))
+
+ attrs = _wait_route_ntf(rtnl, 'delroute-ntf', 64, '2001:db8:2::',
+ deadline=15)
+ ksft_not_none(attrs, "no RTM_DELROUTE for the expired route")
+ if attrs is not None:
+ ksft_eq(attrs.get('del-reason'), 'expired')
+
+
+def _send_ra(sock, ifindex, lifetime, rio=None, pio=None):
+ """The kernel fills in the ICMPv6 checksum on raw ICMPv6 sockets."""
+
+ # type, code, cksum, hop limit, flags, router lifetime,
+ # reachable time, retrans timer
+ ra = struct.pack('!BBHBBHII', 134, 0, 0, 64, 0, lifetime, 0, 0)
+ if rio is not None:
+ prefix, plen, rio_lifetime = rio
+ # RFC 4191 route information option, /64 prefix (8 bytes)
+ ra += struct.pack('!BBBBI', 24, 2, plen, 0, rio_lifetime)
+ ra += socket.inet_pton(socket.AF_INET6, prefix)[:8]
+ if pio is not None:
+ prefix, plen, valid_lft = pio
+ # RFC 4861 prefix information option, on-link only (L set, A clear)
+ ra += struct.pack('!BBBBIII', 3, 4, plen, 0x80, valid_lft, 0, 0)
+ ra += socket.inet_pton(socket.AF_INET6, prefix)
+ sock.sendto(ra, ('ff02::1', 0, 0, ifindex))
+
+
+def _ra_router_sock(ns_r, ifname):
+ with NetNSEnter(str(ns_r)):
+ sock = socket.socket(socket.AF_INET6, socket.SOCK_RAW,
+ socket.IPPROTO_ICMPV6)
+ sock.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_MULTICAST_HOPS, 255)
+ defer(sock.close)
+ return sock, socket.if_nametoindex(ifname)
+
+
+def _ra_advertise_routes(rtnl, sock, ifindex, want, **ra_opts):
+ """
+ Sending fails with EADDRNOTAVAIL while the router's link-local
+ address is still tentative. addrconf_dad_start() only queues
+ addrconf_dad_work(), and IFA_F_TENTATIVE is cleared when that work
+ item runs, so retry until it does.
+ """
+
+ seen = {}
+ for _ in range(10):
+ try:
+ _send_ra(sock, ifindex, **ra_opts)
+ except OSError:
+ time.sleep(0.2)
+ continue
+ seen.update(_collect_route_ntfs(rtnl, 'newroute-ntf',
+ want - set(seen.keys()), deadline=2))
+ if len(seen) == len(want):
+ break
+ return seen
+
+
+def ipv6_route_del_reason_ra_withdrawn() -> None:
+ """
+ Routes withdrawn by a zero-lifetime RA (router lifetime, RFC 4861
+ PIO, RFC 4191 RIO) report RTA_DEL_REASON == ra-withdrawn.
+ """
+
+ # (rtm-dst-len, dst); the default route carries no RTA_DST
+ routes = {(0, None), (64, '2001:db8:6::'), (64, '2001:db8:5::')}
+
+ with NetNS() as ns_h, NetNS() as ns_r:
+ ip(f"link add veth0 netns {ns_h} type veth peer name veth1 netns {ns_r}")
+ with NetNSEnter(str(ns_h)):
+ _write_ipv6_sysctl("conf/veth0/accept_ra", 2)
+ _write_ipv6_sysctl("conf/veth0/forwarding", 0)
+ try:
+ _write_ipv6_sysctl("conf/veth0/accept_ra_rt_info_max_plen", 64)
+ except FileNotFoundError:
+ raise KsftSkipEx("no CONFIG_IPV6_ROUTE_INFO")
+ with NetNSEnter(str(ns_r)):
+ # skip the DAD probe so the router's link-local source only
+ # has to wait for addrconf_dad_work() to clear IFA_F_TENTATIVE
+ _write_ipv6_sysctl("conf/veth1/accept_dad", 0)
+ ip("link set dev veth0 up", ns=str(ns_h))
+ ip("link set dev veth1 up", ns=str(ns_r))
+
+ rtnl = _rtnl_route_subscribe(ns_h)
+ sock, ifindex = _ra_router_sock(ns_r, "veth1")
+
+ seen = _ra_advertise_routes(rtnl, sock, ifindex, routes,
+ lifetime=1800,
+ rio=('2001:db8:5::', 64, 600),
+ pio=('2001:db8:6::', 64, 600))
+ ksft_eq(set(seen), routes, "not all RA routes were installed")
+ if set(seen) != routes:
+ return
+
+ _send_ra(sock, ifindex, 0, rio=('2001:db8:5::', 64, 0),
+ pio=('2001:db8:6::', 64, 0))
+ seen = _collect_route_ntfs(rtnl, 'delroute-ntf', routes)
+ for key in routes:
+ attrs = seen.get(key)
+ ksft_not_none(attrs, f"no RTM_DELROUTE for {key}")
+ if attrs is not None:
+ ksft_eq(attrs.get('del-reason'), 'ra-withdrawn')
+
+
+def ipv6_route_del_reason_absent() -> None:
+ """
+ A deletion path that records no cause (here a userspace request)
+ must not carry RTA_DEL_REASON at all.
+ """
+
+ with NetNS() as ns:
+ rtnl = _rtnl_route_subscribe(ns)
+ ip("link add name dummy1 type dummy", ns=str(ns))
+ ip("link set dev dummy1 up", ns=str(ns))
+ ip("-6 route add 2001:db8:1::/64 dev dummy1", ns=str(ns))
+ ip("-6 route del 2001:db8:1::/64 dev dummy1", ns=str(ns))
+
+ attrs = _wait_route_ntf(rtnl, 'delroute-ntf', 64, '2001:db8:1::')
+ ksft_not_none(attrs, "no RTM_DELROUTE for 2001:db8:1::/64")
+ if attrs is not None:
+ ksft_not_in('del-reason', attrs,
+ "user deletion must not carry del-reason")
+
+
def main() -> None:
- ksft_run([dump_mcaddr_check, ipv4_devconf_notify])
+ ksft_run([dump_mcaddr_check, dump_mcaddr6_check, ipv4_devconf_notify,
+ ipv6_route_del_reason_expired,
+ ipv6_route_del_reason_ra_withdrawn,
+ ipv6_route_del_reason_absent])
ksft_exit()
if __name__ == "__main__":
diff --git a/tools/testing/selftests/net/srv6_encap_lookup_l3vpn_test.sh b/tools/testing/selftests/net/srv6_encap_lookup_l3vpn_test.sh
new file mode 100755
index 000000000000..d6249303b7ea
--- /dev/null
+++ b/tools/testing/selftests/net/srv6_encap_lookup_l3vpn_test.sh
@@ -0,0 +1,1027 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# author: Andrea Mayer <andrea.mayer@uniroma2.it>
+
+# This test evaluates the SRv6 encap "lookup" attribute. After encapsulation
+# the router looks up the route for the first SID, that is the outer IPv6
+# destination of the encapsulated packet. The attribute selects the FIB table
+# used for this post-encap SID route lookup.
+#
+# Two routers (rt-1, rt-2) provide L3 VPN services over an IPv6 underlay
+# (fd00::/64). Each router uses a separate VRF per tenant, with default
+# blackhole routes (IPv4 and IPv6) to prevent traffic from leaking out of
+# the VRF. Tenant traffic is encapsulated, then decapsulated with an
+# End.DT46. Each router proxies both NDP and ARP.
+#
+# The routes that match the first SIDs are installed in a dedicated underlay
+# table (500) rather than the main table (254). The encap routes use
+# "lookup 500" to select this table for the post-encap SID route lookup.
+#
+# Without the "lookup" attribute, the route for the first SID cannot be found:
+# - on the input path (forwarded traffic), the lookup stays in the VRF
+# and hits the blackhole;
+# - on the output path (locally originated traffic), the lookup falls
+# through to the main table, with no route to the first SID.
+#
+#
+# Legend (specific per-tenant addresses are in the instantiation tables below):
+# X = tenant id and VRF table id; two tenants: 100 and 200
+# ("tX" means tenant X, e.g. t100, t200)
+# a, b = the two host ids of the tenant
+# HA, HB = addresses of host a, host b
+# RLO1, RLO2 = rlo-X router loopback address on rt-1, rt-2 (tenant gateway
+# for the output path, dual-stack)
+# vrf-X = per-tenant VRF on each router (table X)
+#
+# Constants (same for every tenant):
+# underlay = table 500; post-encap SID route lookup (via "lookup 500")
+# localsid = table 90; holds the decap SIDs (End.DT46)
+# fd00::/64 = underlay link between rt-1 and rt-2
+# veth-tX = cafe::254/10.0.0.254 (tenant gateway on veth, both routers)
+#
+#
+# +-------------------+ +-------------------+
+# | | | |
+# | hs-tX-a netns | | hs-tX-b netns |
+# | | | |
+# | +-------------+ | | +-------------+ |
+# | | veth0 | | | | veth0 | |
+# | | HA | | | | HB | |
+# | +-------------+ | | +-------------+ |
+# | . | | . |
+# +-------------------+ +-------------------+
+# . .
+# . .
+# +-----------------------------------+ +-----------------------------------+
+# | . | | . |
+# | +---------------+ | | +---------------+ |
+# | | veth-tX | +----------+ | | +----------+ | veth-tX | |
+# | | ::254/.254 | | localsid | | | | localsid | | ::254/.254 | |
+# | +-------+-------+ +----------+ | | +----------+ +-------+-------+ |
+# | | +----------+ | | +----------+ | |
+# | +----+----+ | underlay | | | | underlay | +----+----+ |
+# | | vrf-X | +----------+ | | +----------+ | vrf-X | |
+# | +----+----+ | | +----+----+ |
+# | | | | | |
+# | +-----+----+ +------------+ | | +------------+ +----+-----+ |
+# | | rlo-X | | veth0 | | | | veth0 | | rlo-X | |
+# | | RLO1 | | fd00::1/64 |..|...|..| fd00::2/64 | | RLO2 | |
+# | +----------+ +------------+ | | +------------+ +----------+ |
+# | rt-1 netns | | rt-2 netns |
+# +-----------------------------------+ +-----------------------------------+
+#
+#
+# Per-tenant instantiation:
+# +-----+------+-------------------+-------------------+
+# | X | a, b | HA | HB |
+# +-----+------+-------------------+-------------------+
+# | 100 | 1, 2 | cafe::1, 10.0.0.1 | cafe::2, 10.0.0.2 |
+# | 200 | 3, 4 | cafe::3, 10.0.0.3 | cafe::4, 10.0.0.4 |
+# +-----+------+-------------------+-------------------+
+#
+# Router loopback (rlo-X) addresses, per tenant:
+# +-----+-----------------------+-----------------------+
+# | X | RLO1 (rt-1) | RLO2 (rt-2) |
+# +-----+-----------------------+-----------------------+
+# | 100 | cafe::101, 10.0.0.101 | cafe::102, 10.0.0.102 |
+# | 200 | cafe::201, 10.0.0.201 | cafe::202, 10.0.0.202 |
+# +-----+-----------------------+-----------------------+
+#
+#
+# Network configuration
+# =====================
+#
+# rt-1: localsid table (table 90)
+# +--------+--------------------+----------------------------------+
+# | tenant | SID | Action |
+# +--------+--------------------+----------------------------------+
+# | 100 | fc00:2:1:100::0d46 | apply SRv6 End.DT46 vrftable 100 |
+# | 200 | fc00:2:1:200::0d46 | apply SRv6 End.DT46 vrftable 200 |
+# +--------+--------------------+----------------------------------+
+#
+# rt-1: underlay table (table 500) - post-encap SID route lookup
+# +--------+--------------------+-------------------------------+
+# | tenant | SID | Action |
+# +--------+--------------------+-------------------------------+
+# | 100 | fc00:1:2:100::0d46 | forward via fd00::2 dev veth0 |
+# | 200 | fc00:1:2:200::0d46 | forward via fd00::2 dev veth0 |
+# +--------+--------------------+-------------------------------+
+#
+# rt-1: VRF tables (per tenant: vrf-X = table X)
+# +--------+------------+------------------------------------------+
+# | tenant | dst | encap action |
+# +--------+------------+------------------------------------------+
+# | 100 | cafe::2 | encap segs fc00:1:2:100::0d46 lookup 500 |
+# | | 10.0.0.2 | |
+# | | cafe::102 | |
+# | | 10.0.0.102 | |
+# | 200 | cafe::4 | encap segs fc00:1:2:200::0d46 lookup 500 |
+# | | 10.0.0.4 | |
+# | | cafe::202 | |
+# | | 10.0.0.202 | |
+# +--------+------------+------------------------------------------+
+#
+#
+# rt-2: localsid table (table 90)
+# +--------+--------------------+----------------------------------+
+# | tenant | SID | Action |
+# +--------+--------------------+----------------------------------+
+# | 100 | fc00:1:2:100::0d46 | apply SRv6 End.DT46 vrftable 100 |
+# | 200 | fc00:1:2:200::0d46 | apply SRv6 End.DT46 vrftable 200 |
+# +--------+--------------------+----------------------------------+
+#
+# rt-2: underlay table (table 500) - post-encap SID route lookup
+# +--------+--------------------+-------------------------------+
+# | tenant | SID | Action |
+# +--------+--------------------+-------------------------------+
+# | 100 | fc00:2:1:100::0d46 | forward via fd00::1 dev veth0 |
+# | 200 | fc00:2:1:200::0d46 | forward via fd00::1 dev veth0 |
+# +--------+--------------------+-------------------------------+
+#
+# rt-2: VRF tables (per tenant: vrf-X = table X)
+# +--------+------------+------------------------------------------+
+# | tenant | dst | encap action |
+# +--------+------------+------------------------------------------+
+# | 100 | cafe::1 | encap segs fc00:2:1:100::0d46 lookup 500 |
+# | | 10.0.0.1 | |
+# | | cafe::101 | |
+# | | 10.0.0.101 | |
+# | 200 | cafe::3 | encap segs fc00:2:1:200::0d46 lookup 500 |
+# | | 10.0.0.3 | |
+# | | cafe::201 | |
+# | | 10.0.0.201 | |
+# +--------+------------+------------------------------------------+
+# Within a tenant, a single SID reaches the adjacent router (its loopback)
+# and the remote host connected to it, in both IPv4 and IPv6.
+#
+# For both rt-1 and rt-2, each VRF also has the connected host prefix (cafe::/64
+# or 10.0.0.0/24) and a default blackhole (IPv4 and IPv6).
+#
+#
+# Locally originated traffic (output path)
+# ========================================
+#
+# The configuration above covers forwarded traffic, where packets arrive from
+# a host and are encapsulated by the router. To also test router-originated
+# traffic, each router pings the other router's loopback address through
+# the VPN.
+#
+# Example (tenant 100), rt-1 pings cafe::102 (rt-2's loopback):
+# 1. rt-1 looks up cafe::102 in vrf-100 and encapsulates it (SID
+# fc00:1:2:100::0d46), then "lookup 500" finds the route for the SID in the
+# underlay table (next hop fd00::2) and forwards it;
+# 2. rt-2 decapsulates it (localsid, End.DT46) and delivers it locally
+# (cafe::102 is on the rlo-100 interface);
+# 3. rt-2 replies with destination cafe::101 (rt-1's loopback). rt-2 looks up
+# cafe::101 in vrf-100 and encapsulates it back to rt-1 (again via "lookup
+# 500"). rt-1 decapsulates it and delivers it.
+
+# shellcheck source=lib.sh
+source lib.sh
+
+readonly LOCALSID_TABLE_ID=90
+readonly UNDERLAY_TABLE_ID=500
+readonly IPv6_RT_NETWORK=fd00
+readonly IPv6_HS_NETWORK=cafe
+readonly IPv4_HS_NETWORK=10.0.0
+readonly VPN_LOCATOR_SERVICE=fc00
+readonly DT46_FUNC=0d46
+readonly DUMMY_DEVNAME=dum0
+readonly IPv6_TESTS_ADDR=2001:db8::1
+readonly TESTS_TABLE_ID=54321
+PING_TIMEOUT_SEC=4
+
+SETUP_ERR=1
+
+ret=${ksft_skip}
+nsuccess=0
+nfail=0
+
+PAUSE_ON_FAIL=${PAUSE_ON_FAIL:=no}
+
+log_test()
+{
+ local rc="$1"
+ local expected="$2"
+ local msg="$3"
+
+ if [ "${rc}" -eq "${expected}" ]; then
+ nsuccess=$((nsuccess+1))
+ printf "\n TEST: %-60s [ OK ]\n" "${msg}"
+ else
+ ret=1
+ nfail=$((nfail+1))
+ printf "\n TEST: %-60s [FAIL]\n" "${msg}"
+ if [ "${PAUSE_ON_FAIL}" = "yes" ]; then
+ echo
+ echo "hit enter to continue, 'q' to quit"
+ read -r a
+ [ "$a" = "q" ] && exit 1
+ fi
+ fi
+}
+
+print_log_test_results()
+{
+ printf "\nTests passed: %3d\n" "${nsuccess}"
+ printf "Tests failed: %3d\n" "${nfail}"
+
+ # when a test fails, the value of 'ret' is set to 1 (error code).
+ # Conversely, when all tests are passed successfully, the 'ret' value
+ # is set to 0 (success code).
+ if [ "${ret}" -ne 1 ]; then
+ ret=0
+ fi
+}
+
+log_section()
+{
+ echo
+ echo "################################################################################"
+ echo "TEST SECTION: $*"
+ echo "################################################################################"
+}
+
+get_rtname()
+{
+ local rtid="$1"
+
+ echo "rt_${rtid}"
+}
+
+get_rt_nsname()
+{
+ local rtid="$1"
+ local varname
+
+ varname="$(get_rtname "${rtid}")"
+ echo "${!varname}"
+}
+
+get_hsname()
+{
+ local tid="$1"
+ local hsid="$2"
+
+ echo "hs_t${tid}_${hsid}"
+}
+
+get_hs_nsname()
+{
+ local tid="$1"
+ local hsid="$2"
+ local varname
+
+ varname="$(get_hsname "${tid}" "${hsid}")"
+ echo "${!varname}"
+}
+
+cleanup()
+{
+ ip link del veth-rt-1 2>/dev/null || true
+ ip link del veth-rt-2 2>/dev/null || true
+
+ cleanup_all_ns
+
+ # check whether the setup phase was completed successfully or not. In
+ # case of an error during the setup phase of the testing environment,
+ # the selftest is considered as "skipped".
+ if [ "${SETUP_ERR}" -ne 0 ]; then
+ echo "SKIP: Setting up the testing environment failed"
+ exit "${ksft_skip}"
+ fi
+
+ exit "${ret}"
+}
+
+# Host id of the router loopback (rlo) for a (router, tenant) pair.
+# E.g. rt-1/tenant 100 -> 101, rt-2/tenant 200 -> 202.
+get_rlo_hostid()
+{
+ local rtid="$1"
+ local tid="$2"
+
+ echo "$((tid + rtid))"
+}
+
+build_vpn_sid()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tid="$3"
+
+ echo "${VPN_LOCATOR_SERVICE}:${rtsrc}:${rtdst}:${tid}::${DT46_FUNC}"
+}
+
+# Install a dual-stack (IPv6 and IPv4) encap route in a VRF on the given
+# router.
+# args:
+# $1 - router id
+# $2 - host part of the IPv6 destination
+# $3 - host part of the IPv4 destination
+# $4 - SRv6 SID used as the encap destination
+# $5 - tenant id
+# $6 - if "true", add the "lookup" attribute to the encap route
+__set_encap_route()
+{
+ local rt="$1"
+ local dst6="$2"
+ local dst4="$3"
+ local sid="$4"
+ local tid="$5"
+ local use_lookup="$6"
+ local lookup=''
+ local rtname
+
+ rtname="$(get_rt_nsname "${rt}")"
+
+ if [ "${use_lookup}" = "true" ]; then
+ lookup="lookup ${UNDERLAY_TABLE_ID}"
+ fi
+
+ # shellcheck disable=SC2086
+ ip -netns "${rtname}" -6 route replace \
+ "${IPv6_HS_NETWORK}::${dst6}/128" vrf "vrf-${tid}" \
+ encap seg6 mode encap segs "${sid}" ${lookup} dev veth0
+
+ # shellcheck disable=SC2086
+ ip -netns "${rtname}" -4 route replace \
+ "${IPv4_HS_NETWORK}.${dst4}/32" vrf "vrf-${tid}" \
+ encap seg6 mode encap segs "${sid}" ${lookup} dev veth0
+}
+
+# Install the dual-stack encap route for a tenant host on rt, with the
+# "lookup" attribute so the first SID is looked up in the underlay table.
+# args:
+# $1 - router id where the encap route is installed
+# $2 - host destination id (host part of cafe::<id>/128 and 10.0.0.<id>/32)
+# $3 - SRv6 SID used as the encap destination
+# $4 - tenant id
+set_host_encap_route()
+{
+ local rt="$1"
+ local hsdst="$2"
+ local sid="$3"
+ local tid="$4"
+
+ __set_encap_route "${rt}" "${hsdst}" "${hsdst}" "${sid}" "${tid}" true
+}
+
+set_host_encap_route_nolookup()
+{
+ local rt="$1"
+ local hsdst="$2"
+ local sid="$3"
+ local tid="$4"
+
+ __set_encap_route "${rt}" "${hsdst}" "${hsdst}" "${sid}" "${tid}" false
+}
+
+# Install the dual-stack encap route on rtsrc toward rtdst's rlo loopback
+# (RLO1 or RLO2, see header), with the "lookup" attribute so the first
+# SID is looked up in the underlay table.
+# args:
+# $1 - router id where the encap route is installed
+# $2 - router id whose loopback address is the route destination
+# $3 - SRv6 SID used as the encap destination
+# $4 - tenant id
+set_gw_encap_route()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local sid="$3"
+ local tid="$4"
+ local dst
+
+ dst="$(get_rlo_hostid "${rtdst}" "${tid}")"
+
+ __set_encap_route "${rtsrc}" "${dst}" "${dst}" "${sid}" "${tid}" true
+}
+
+set_gw_encap_route_nolookup()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local sid="$3"
+ local tid="$4"
+ local dst
+
+ dst="$(get_rlo_hostid "${rtdst}" "${tid}")"
+
+ __set_encap_route "${rtsrc}" "${dst}" "${dst}" "${sid}" "${tid}" false
+}
+
+# Setup the basic networking for a router
+setup_rt_networking()
+{
+ local id="$1"
+ local nsname
+
+ nsname="$(get_rt_nsname "${id}")"
+
+ ip link set "veth-rt-${id}" netns "${nsname}"
+ ip -netns "${nsname}" link set "veth-rt-${id}" name veth0
+
+ ip netns exec "${nsname}" sysctl -wq net.ipv6.conf.all.accept_dad=0
+ ip netns exec "${nsname}" sysctl -wq net.ipv6.conf.default.accept_dad=0
+
+ ip -netns "${nsname}" addr add "${IPv6_RT_NETWORK}::${id}/64" dev veth0 nodad
+ ip -netns "${nsname}" link set veth0 up
+
+ ip netns exec "${nsname}" sysctl -wq net.ipv4.ip_forward=1
+ ip netns exec "${nsname}" sysctl -wq net.ipv6.conf.all.forwarding=1
+}
+
+# Setup a host namespace and attach it to its gateway
+setup_hs()
+{
+ local hid="$1"
+ local rid="$2"
+ local tid="$3"
+ local rtveth="veth-t${tid}"
+ local hsname
+ local rtname
+
+ hsname="$(get_hs_nsname "${tid}" "${hid}")"
+ rtname="$(get_rt_nsname "${rid}")"
+
+ ip netns exec "${hsname}" sysctl -wq net.ipv6.conf.all.accept_dad=0
+ ip netns exec "${hsname}" sysctl -wq net.ipv6.conf.default.accept_dad=0
+
+ ip -netns "${hsname}" link add veth0 type veth peer name "${rtveth}"
+ ip -netns "${hsname}" link set "${rtveth}" netns "${rtname}"
+
+ ip -netns "${hsname}" addr add \
+ "${IPv6_HS_NETWORK}::${hid}/64" dev veth0 nodad
+ ip -netns "${hsname}" addr add \
+ "${IPv4_HS_NETWORK}.${hid}/24" dev veth0
+
+ ip -netns "${hsname}" link set veth0 up
+}
+
+# Setup the per-tenant VRF on a router (gateway, loopback, blackhole)
+setup_rt()
+{
+ local rid="$1"
+ local tid="$2"
+ local rtveth="veth-t${tid}"
+ local rlo_dev="rlo-${tid}"
+ local rtname
+ local gw_addr_v6
+ local gw_addr_v4
+
+ rtname="$(get_rt_nsname "${rid}")"
+
+ gw_addr_v6="${IPv6_HS_NETWORK}::$(get_rlo_hostid "${rid}" "${tid}")"
+ gw_addr_v4="${IPv4_HS_NETWORK}.$(get_rlo_hostid "${rid}" "${tid}")"
+
+ ip -netns "${rtname}" link add "vrf-${tid}" type vrf table "${tid}"
+ ip -netns "${rtname}" link set "vrf-${tid}" up
+
+ ip -netns "${rtname}" link set "${rtveth}" master "vrf-${tid}"
+
+ ip -netns "${rtname}" addr add \
+ "${IPv6_HS_NETWORK}::254/64" dev "${rtveth}" nodad
+ ip -netns "${rtname}" addr add \
+ "${IPv4_HS_NETWORK}.254/24" dev "${rtveth}"
+
+ ip -netns "${rtname}" link set "${rtveth}" up
+
+ ip netns exec "${rtname}" \
+ sysctl -wq "net.ipv6.conf.${rtveth}.proxy_ndp=1"
+ ip netns exec "${rtname}" \
+ sysctl -wq "net.ipv4.conf.${rtveth}.proxy_arp=1"
+
+ ip netns exec "${rtname}" sh -c "echo 1 > /proc/sys/net/vrf/strict_mode"
+
+ # router loopback interface for locally originated traffic
+ ip -netns "${rtname}" link add "${rlo_dev}" type dummy
+ ip -netns "${rtname}" link set "${rlo_dev}" master "vrf-${tid}"
+
+ ip -netns "${rtname}" addr add "${gw_addr_v6}/128" \
+ dev "${rlo_dev}" nodad
+ ip -netns "${rtname}" addr add "${gw_addr_v4}/32" \
+ dev "${rlo_dev}"
+
+ ip -netns "${rtname}" link set "${rlo_dev}" up
+
+ # default blackhole routes in the VRF: any traffic that does not match
+ # a specific route is dropped. Without the "lookup" attribute on the
+ # encap route, the route for the first SID cannot be found from within
+ # the VRF.
+ ip -netns "${rtname}" -6 route add blackhole default metric 4278198272 \
+ vrf "vrf-${tid}"
+ ip -netns "${rtname}" -4 route add blackhole default metric 4278198272 \
+ vrf "vrf-${tid}"
+}
+
+# Configure a one-way VPN path towards hsdst (on rtdst) for tenant tid.
+# The encap side is set up on rtsrc and the decap side on rtdst.
+# args:
+# $1 - router id where the encap side is set up
+# $2 - host id of the destination host
+# $3 - router id of the destination router (connected to the destination host)
+# $4 - tenant id
+setup_vpn_config()
+{
+ local rtsrc="$1"
+ local hsdst="$2"
+ local rtdst="$3"
+ local tid="$4"
+ local rtveth="veth-t${tid}"
+ local rtsrc_name
+ local rtdst_name
+ local vpn_sid
+
+ rtsrc_name="$(get_rt_nsname "${rtsrc}")"
+ rtdst_name="$(get_rt_nsname "${rtdst}")"
+ vpn_sid="$(build_vpn_sid "${rtsrc}" "${rtdst}" "${tid}")"
+
+ ip -netns "${rtsrc_name}" -6 neigh add proxy \
+ "${IPv6_HS_NETWORK}::${hsdst}" dev "${rtveth}"
+ set_host_encap_route "${rtsrc}" "${hsdst}" "${vpn_sid}" "${tid}"
+
+ ip -netns "${rtsrc_name}" -6 route add "${vpn_sid}/128" \
+ table "${UNDERLAY_TABLE_ID}" \
+ via "fd00::${rtdst}" dev veth0
+
+ # set the decap route for decapsulating packets arriving from rtsrc
+ # and destined to hsdst
+ ip -netns "${rtdst_name}" -6 route add "${vpn_sid}/128" \
+ table "${LOCALSID_TABLE_ID}" \
+ encap seg6local action End.DT46 \
+ vrftable "${tid}" dev "vrf-${tid}"
+
+ # all SIDs for VPNs start with a common locator which is fc00::/16.
+ # Routes for handling the SRv6 End.DT* behavior instances are grouped
+ # together in the 'localsid' table.
+ #
+ # NOTE: added only once
+ if ! ip -netns "${rtdst_name}" -6 rule show | \
+ grep -q "to ${VPN_LOCATOR_SERVICE}::/16 lookup ${LOCALSID_TABLE_ID}"; then
+ ip -netns "${rtdst_name}" -6 rule add \
+ to "${VPN_LOCATOR_SERVICE}::/16" \
+ lookup "${LOCALSID_TABLE_ID}" prio 999
+ fi
+}
+
+# Configure rtsrc to reach rtdst's loopback address through the VPN.
+# args:
+# $1 - router id where the encap route is installed
+# $2 - router id whose loopback is the destination
+# $3 - tenant id
+setup_vpn_gw_encap()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tid="$3"
+ local sid
+
+ sid="$(build_vpn_sid "${rtsrc}" "${rtdst}" "${tid}")"
+
+ set_gw_encap_route "${rtsrc}" "${rtdst}" "${sid}" "${tid}"
+}
+
+setup()
+{
+ ip link add veth-rt-1 type veth peer name veth-rt-2
+ setup_ns rt_1 rt_2
+ setup_rt_networking 1
+ setup_rt_networking 2
+
+ # setup two hosts for the tenant 100.
+ # - host hs-t100-1 is directly connected to the router rt-1;
+ # - host hs-t100-2 is directly connected to the router rt-2.
+ setup_ns hs_t100_1 hs_t100_2
+ setup_hs 1 1 100
+ setup_hs 2 2 100
+
+ # setup two hosts for the tenant 200.
+ # - host hs-t200-3 is directly connected to the router rt-1;
+ # - host hs-t200-4 is directly connected to the router rt-2.
+ setup_ns hs_t200_3 hs_t200_4
+ setup_hs 3 1 200
+ setup_hs 4 2 200
+
+ # configure each router for each tenant: VRF, blackhole routes,
+ # router loopback interface
+ setup_rt 1 100
+ setup_rt 2 100
+ setup_rt 1 200
+ setup_rt 2 200
+
+ # setup the L3 VPN which connects the host hs-t100-1 and host hs-t100-2
+ # within the same tenant 100.
+ setup_vpn_config 1 2 2 100
+ setup_vpn_config 2 1 1 100
+
+ # setup the L3 VPN which connects the host hs-t200-3 and host hs-t200-4
+ # within the same tenant 200.
+ setup_vpn_config 1 4 2 200
+ setup_vpn_config 2 3 1 200
+
+ # allow each router to reach the other's loopback through the VPN
+ setup_vpn_gw_encap 2 1 100
+ setup_vpn_gw_encap 1 2 100
+ setup_vpn_gw_encap 2 1 200
+ setup_vpn_gw_encap 1 2 200
+
+ # testing environment was set up successfully
+ SETUP_ERR=0
+}
+
+check_rt_connectivity()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local nsname
+
+ nsname="$(get_rt_nsname "${rtsrc}")"
+
+ ip netns exec "${nsname}" ping -c 1 -W 1 "${IPv6_RT_NETWORK}::${rtdst}" \
+ >/dev/null 2>&1
+}
+
+check_and_log_rt_connectivity()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+
+ check_rt_connectivity "${rtsrc}" "${rtdst}"
+ log_test $? 0 "Routers connectivity: rt-${rtsrc} -> rt-${rtdst}"
+}
+
+check_hs_ipv6_connectivity()
+{
+ local hssrc="$1"
+ local hsdst="$2"
+ local tid="$3"
+ local nsname
+
+ nsname="$(get_hs_nsname "${tid}" "${hssrc}")"
+
+ ip netns exec "${nsname}" ping -c 1 -W "${PING_TIMEOUT_SEC}" \
+ "${IPv6_HS_NETWORK}::${hsdst}" >/dev/null 2>&1
+}
+
+check_hs_ipv4_connectivity()
+{
+ local hssrc="$1"
+ local hsdst="$2"
+ local tid="$3"
+ local nsname
+
+ nsname="$(get_hs_nsname "${tid}" "${hssrc}")"
+
+ ip netns exec "${nsname}" ping -c 1 -W "${PING_TIMEOUT_SEC}" \
+ "${IPv4_HS_NETWORK}.${hsdst}" >/dev/null 2>&1
+}
+
+check_and_log_hs_connectivity()
+{
+ local hssrc="$1"
+ local hsdst="$2"
+ local tid="$3"
+
+ check_hs_ipv6_connectivity "${hssrc}" "${hsdst}" "${tid}"
+ log_test $? 0 "IPv6 connectivity: hs-t${tid}-${hssrc} -> hs-t${tid}-${hsdst} (tenant ${tid})"
+
+ check_hs_ipv4_connectivity "${hssrc}" "${hsdst}" "${tid}"
+ log_test $? 0 "IPv4 connectivity: hs-t${tid}-${hssrc} -> hs-t${tid}-${hsdst} (tenant ${tid})"
+}
+
+check_and_log_hs_isolation()
+{
+ local hssrc="$1"
+ local tidsrc="$2"
+ local hsdst="$3"
+ local tiddst="$4"
+
+ check_hs_ipv6_connectivity "${hssrc}" "${hsdst}" "${tidsrc}"
+ log_test $? 1 "IPv6 isolation: hs-t${tidsrc}-${hssrc} -X-> hs-t${tiddst}-${hsdst}"
+
+ check_hs_ipv4_connectivity "${hssrc}" "${hsdst}" "${tidsrc}"
+ log_test $? 1 "IPv4 isolation: hs-t${tidsrc}-${hssrc} -X-> hs-t${tiddst}-${hsdst}"
+}
+
+check_and_log_hs2gw_connectivity()
+{
+ local hssrc="$1"
+ local tid="$2"
+
+ check_hs_ipv6_connectivity "${hssrc}" 254 "${tid}"
+ log_test $? 0 "IPv6 connectivity: hs-t${tid}-${hssrc} -> gw (tenant ${tid})"
+
+ check_hs_ipv4_connectivity "${hssrc}" 254 "${tid}"
+ log_test $? 0 "IPv4 connectivity: hs-t${tid}-${hssrc} -> gw (tenant ${tid})"
+}
+
+router_tests()
+{
+ log_section "IPv6 routers connectivity test"
+
+ check_and_log_rt_connectivity 1 2
+ check_and_log_rt_connectivity 2 1
+}
+
+host2gateway_tests()
+{
+ log_section "Connectivity test among hosts and gateway"
+
+ check_and_log_hs2gw_connectivity 1 100
+ check_and_log_hs2gw_connectivity 2 100
+
+ check_and_log_hs2gw_connectivity 3 200
+ check_and_log_hs2gw_connectivity 4 200
+}
+
+host_vpn_tests()
+{
+ log_section "SRv6 VPN connectivity test among hosts in the same tenant"
+
+ check_and_log_hs_connectivity 1 2 100
+ check_and_log_hs_connectivity 2 1 100
+
+ check_and_log_hs_connectivity 3 4 200
+ check_and_log_hs_connectivity 4 3 200
+}
+
+host_vpn_isolation_tests()
+{
+ local l1="1 2"
+ local l2="3 4"
+ local t1=100
+ local t2=200
+ local i
+ local j
+ local tmp
+
+ log_section "SRv6 VPN isolation test among hosts in different tenants"
+
+ for _ in 0 1; do
+ for i in ${l1}; do
+ for j in ${l2}; do
+ check_and_log_hs_isolation "${i}" "${t1}" "${j}" "${t2}"
+ done
+ done
+
+ # let us test the reverse path
+ tmp="${l1}"; l1="${l2}"; l2="${tmp}"
+ tmp=${t1}; t1=${t2}; t2=${tmp}
+ done
+}
+
+__test_nolookup()
+{
+ local hssrc="$1"
+ local hsdst="$2"
+ local rtsrc="$3"
+ local rtdst="$4"
+ local tid="$5"
+ local vpn_sid
+
+ vpn_sid="$(build_vpn_sid "${rtsrc}" "${rtdst}" "${tid}")"
+
+ # replace encap route(s) without "lookup" attribute
+ set_host_encap_route_nolookup "${rtsrc}" "${hsdst}" "${vpn_sid}" "${tid}"
+
+ check_hs_ipv6_connectivity "${hssrc}" "${hsdst}" "${tid}"
+ log_test $? 1 "IPv6 w/o lookup: hs-t${tid}-${hssrc} -X-> hs-t${tid}-${hsdst} (tenant ${tid})"
+
+ check_hs_ipv4_connectivity "${hssrc}" "${hsdst}" "${tid}"
+ log_test $? 1 "IPv4 w/o lookup: hs-t${tid}-${hssrc} -X-> hs-t${tid}-${hsdst} (tenant ${tid})"
+
+ # restore encap route(s) with "lookup" for subsequent tests
+ set_host_encap_route "${rtsrc}" "${hsdst}" "${vpn_sid}" "${tid}"
+}
+
+host_vpn_nolookup_tests()
+{
+ log_section "SRv6 VPN connectivity test among hosts w/o lookup"
+
+ __test_nolookup 1 2 1 2 100
+ __test_nolookup 2 1 2 1 100
+
+ __test_nolookup 3 4 1 2 200
+ __test_nolookup 4 3 2 1 200
+}
+
+check_gw_ipv6_connectivity()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tidsrc="$3"
+ local tiddst="$4"
+ local rtname
+ local src_v6
+ local dst_v6
+
+ rtname="$(get_rt_nsname "${rtsrc}")"
+ src_v6="${IPv6_HS_NETWORK}::$(get_rlo_hostid "${rtsrc}" "${tidsrc}")"
+ dst_v6="${IPv6_HS_NETWORK}::$(get_rlo_hostid "${rtdst}" "${tiddst}")"
+
+ ip netns exec "${rtname}" ip vrf exec "vrf-${tidsrc}" \
+ ping -c 1 -W "${PING_TIMEOUT_SEC}" \
+ -I "${src_v6}" "${dst_v6}" >/dev/null 2>&1
+}
+
+check_gw_ipv4_connectivity()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tidsrc="$3"
+ local tiddst="$4"
+ local rtname
+ local src_v4
+ local dst_v4
+
+ rtname="$(get_rt_nsname "${rtsrc}")"
+ src_v4="${IPv4_HS_NETWORK}.$(get_rlo_hostid "${rtsrc}" "${tidsrc}")"
+ dst_v4="${IPv4_HS_NETWORK}.$(get_rlo_hostid "${rtdst}" "${tiddst}")"
+
+ ip netns exec "${rtname}" ip vrf exec "vrf-${tidsrc}" \
+ ping -c 1 -W "${PING_TIMEOUT_SEC}" \
+ -I "${src_v4}" "${dst_v4}" >/dev/null 2>&1
+}
+
+check_and_log_gw_connectivity()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tid="$3"
+
+ check_gw_ipv6_connectivity "${rtsrc}" "${rtdst}" "${tid}" "${tid}"
+ log_test $? 0 "IPv6 connectivity: rt-${rtsrc} -> rt-${rtdst} (tenant ${tid})"
+
+ check_gw_ipv4_connectivity "${rtsrc}" "${rtdst}" "${tid}" "${tid}"
+ log_test $? 0 "IPv4 connectivity: rt-${rtsrc} -> rt-${rtdst} (tenant ${tid})"
+}
+
+check_and_log_gw_isolation()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tidsrc="$3"
+ local tiddst="$4"
+
+ check_gw_ipv6_connectivity "${rtsrc}" "${rtdst}" "${tidsrc}" "${tiddst}"
+ log_test $? 1 "IPv6 isolation: rt-${rtsrc} -X-> rt-${rtdst} (tenants ${tidsrc}/${tiddst})"
+
+ check_gw_ipv4_connectivity "${rtsrc}" "${rtdst}" "${tidsrc}" "${tiddst}"
+ log_test $? 1 "IPv4 isolation: rt-${rtsrc} -X-> rt-${rtdst} (tenants ${tidsrc}/${tiddst})"
+}
+
+gw_vpn_isolation_tests()
+{
+ log_section "SRv6 VPN isolation test among routers in different tenants"
+
+ check_and_log_gw_isolation 1 2 100 200
+ check_and_log_gw_isolation 2 1 100 200
+
+ check_and_log_gw_isolation 1 2 200 100
+ check_and_log_gw_isolation 2 1 200 100
+}
+
+gw_vpn_tests()
+{
+ log_section "SRv6 VPN connectivity test among routers in the same tenant"
+
+ check_and_log_gw_connectivity 1 2 100
+ check_and_log_gw_connectivity 2 1 100
+
+ check_and_log_gw_connectivity 1 2 200
+ check_and_log_gw_connectivity 2 1 200
+}
+
+__test_gw_nolookup()
+{
+ local rtsrc="$1"
+ local rtdst="$2"
+ local tid="$3"
+ local sid
+
+ sid="$(build_vpn_sid "${rtsrc}" "${rtdst}" "${tid}")"
+
+ # replace gw encap route without "lookup" attribute
+ set_gw_encap_route_nolookup "${rtsrc}" "${rtdst}" "${sid}" "${tid}"
+
+ check_gw_ipv6_connectivity "${rtsrc}" "${rtdst}" "${tid}" "${tid}"
+ log_test $? 1 "IPv6 w/o lookup: rt-${rtsrc} -X-> rt-${rtdst} (tenant ${tid})"
+
+ check_gw_ipv4_connectivity "${rtsrc}" "${rtdst}" "${tid}" "${tid}"
+ log_test $? 1 "IPv4 w/o lookup: rt-${rtsrc} -X-> rt-${rtdst} (tenant ${tid})"
+
+ # restore gw encap route with "lookup" for subsequent tests
+ set_gw_encap_route "${rtsrc}" "${rtdst}" "${sid}" "${tid}"
+}
+
+gw_vpn_nolookup_tests()
+{
+ log_section "SRv6 VPN connectivity test among routers w/o lookup"
+
+ __test_gw_nolookup 1 2 100
+ __test_gw_nolookup 2 1 100
+
+ __test_gw_nolookup 1 2 200
+ __test_gw_nolookup 2 1 200
+}
+
+test_command_or_ksft_skip()
+{
+ local cmd="$1"
+
+ if [ ! -x "$(command -v "${cmd}")" ]; then
+ echo "SKIP: Could not run test without \"${cmd}\" tool"
+ exit "${ksft_skip}"
+ fi
+}
+
+test_vrf_or_ksft_skip()
+{
+ modprobe vrf &>/dev/null || true
+ if [ ! -e /proc/sys/net/vrf/strict_mode ]; then
+ echo "SKIP: vrf sysctl does not exist"
+ exit "${ksft_skip}"
+ fi
+}
+
+test_dummy_dev_or_ksft_skip()
+{
+ local test_netns
+
+ setup_ns test_netns
+
+ modprobe dummy &>/dev/null || true
+ if ! ip -netns "${test_netns}" link add "${DUMMY_DEVNAME}" \
+ type dummy; then
+ cleanup_ns "${test_netns}"
+ echo "SKIP: dummy dev not supported"
+ exit "${ksft_skip}"
+ fi
+
+ cleanup_ns "${test_netns}"
+}
+
+test_encap_lookup_supp_or_ksft_skip()
+{
+ local nsname
+
+ setup_ns nsname
+
+ ip -netns "${nsname}" link add "${DUMMY_DEVNAME}" type dummy
+ ip -netns "${nsname}" link set "${DUMMY_DEVNAME}" up
+
+ if ! ip -netns "${nsname}" -6 route add "${IPv6_TESTS_ADDR}/128" \
+ encap seg6 mode encap segs fc00::1 \
+ lookup "${TESTS_TABLE_ID}" \
+ dev "${DUMMY_DEVNAME}" 2>/dev/null; then
+ cleanup_ns "${nsname}"
+ echo "SKIP: seg6 encap lookup attribute not supported"
+ exit "${ksft_skip}"
+ fi
+
+ # An old kernel with a recent iproute2 accepts the route but
+ # silently ignores the lookup attribute. Dump the route and check
+ # the attribute is really there, otherwise the test falsely passes.
+ if ! ip -netns "${nsname}" -6 route show "${IPv6_TESTS_ADDR}/128" | \
+ grep -q "lookup ${TESTS_TABLE_ID}"; then
+ cleanup_ns "${nsname}"
+ echo "SKIP: seg6 encap lookup attribute not supported"
+ exit "${ksft_skip}"
+ fi
+
+ cleanup_ns "${nsname}"
+}
+
+if [ "$(id -u)" -ne 0 ]; then
+ echo "SKIP: Need root privileges"
+ exit "${ksft_skip}"
+fi
+
+# required programs to carry out this selftest
+test_command_or_ksft_skip ip
+test_command_or_ksft_skip ping
+test_command_or_ksft_skip sysctl
+test_command_or_ksft_skip grep
+
+test_dummy_dev_or_ksft_skip
+test_vrf_or_ksft_skip
+test_encap_lookup_supp_or_ksft_skip
+
+set -e
+trap cleanup EXIT
+
+setup
+set +e
+
+router_tests
+host2gateway_tests
+host_vpn_tests
+host_vpn_isolation_tests
+host_vpn_nolookup_tests
+gw_vpn_tests
+gw_vpn_isolation_tests
+gw_vpn_nolookup_tests
+
+print_log_test_results
diff --git a/tools/testing/selftests/net/srv6_end_dt46_l3vpn_test.sh b/tools/testing/selftests/net/srv6_end_dt46_l3vpn_test.sh
index a5e959a080bb..50e37d3217ea 100755
--- a/tools/testing/selftests/net/srv6_end_dt46_l3vpn_test.sh
+++ b/tools/testing/selftests/net/srv6_end_dt46_l3vpn_test.sh
@@ -536,6 +536,14 @@ host_vpn_isolation_tests()
done
}
+test_iproute2_supp_or_ksft_skip()
+{
+ if ! ip route add help 2>&1 | grep -qo "End.DT46"; then
+ echo "SKIP: Missing SRv6 End.DT46 support in iproute2"
+ exit "${ksft_skip}"
+ fi
+}
+
if [ "$(id -u)" -ne 0 ];then
echo "SKIP: Need root privileges"
exit $ksft_skip
@@ -546,6 +554,8 @@ if [ ! -x "$(command -v ip)" ]; then
exit $ksft_skip
fi
+test_iproute2_supp_or_ksft_skip
+
modprobe vrf &>/dev/null
if [ ! -e /proc/sys/net/vrf/strict_mode ]; then
echo "SKIP: vrf sysctl does not exist"
diff --git a/tools/testing/selftests/net/tcp_ao/key-management.c b/tools/testing/selftests/net/tcp_ao/key-management.c
index d86bb380b79f..0451f92f4645 100644
--- a/tools/testing/selftests/net/tcp_ao/key-management.c
+++ b/tools/testing/selftests/net/tcp_ao/key-management.c
@@ -63,8 +63,8 @@ static int prepare_lsk(union tcp_addr *addr, uint8_t sndid, uint8_t rcvid)
return sk;
}
-static int test_del_key(int sk, uint8_t sndid, uint8_t rcvid, bool async,
- int current_key, int rnext_key)
+static int test_del_key(int sk, uint8_t sndid, uint8_t rcvid, int ifindex,
+ bool async, int current_key, int rnext_key)
{
struct tcp_ao_info_opt ao_info = {};
struct tcp_ao_getsockopt key = {};
@@ -76,6 +76,10 @@ static int test_del_key(int sk, uint8_t sndid, uint8_t rcvid, bool async,
del.prefix = DEFAULT_TEST_PREFIX;
del.sndid = sndid;
del.rcvid = rcvid;
+ if (ifindex) {
+ del.keyflags = TCP_AO_KEYF_IFINDEX;
+ del.ifindex = ifindex;
+ }
if (current_key >= 0) {
del.set_current = 1;
@@ -95,7 +99,8 @@ static int test_del_key(int sk, uint8_t sndid, uint8_t rcvid, bool async,
tcp_addr_to_sockaddr_in(&sockaddr, &this_ip_dest, 0);
err = test_get_one_ao(sk, &key, &sockaddr, sizeof(sockaddr),
- DEFAULT_TEST_PREFIX, sndid, rcvid);
+ DEFAULT_TEST_PREFIX, sndid, rcvid,
+ del.keyflags, del.ifindex);
if (!err)
return -EEXIST;
if (err != -E2BIG)
@@ -112,12 +117,12 @@ static int test_del_key(int sk, uint8_t sndid, uint8_t rcvid, bool async,
}
static void try_delete_key(char *tst_name, int sk, uint8_t sndid, uint8_t rcvid,
- bool async, int current_key, int rnext_key,
+ int ifindex, bool async, int current_key, int rnext_key,
fault_t inj)
{
int err;
- err = test_del_key(sk, sndid, rcvid, async, current_key, rnext_key);
+ err = test_del_key(sk, sndid, rcvid, ifindex, async, current_key, rnext_key);
if ((err == -EBUSY && fault(BUSY)) || (err == -EINVAL && fault(CURRNEXT))) {
test_ok("%s: key deletion was prevented", tst_name);
return;
@@ -236,15 +241,15 @@ static void check_closed_socket(void)
int sk;
sk = prepare_sk(&this_ip_dest, 200, 200);
- try_delete_key("closed socket, delete a key", sk, 200, 200, 0, -1, -1, 0);
- try_delete_key("closed socket, delete all keys", sk, 100, 100, 0, -1, -1, 0);
+ try_delete_key("closed socket, delete a key", sk, 200, 200, 0, 0, -1, -1, 0);
+ try_delete_key("closed socket, delete all keys", sk, 100, 100, 0, 0, -1, -1, 0);
close(sk);
sk = prepare_sk(&this_ip_dest, 200, 200);
if (test_set_key(sk, 100, 200))
test_error("failed to set current/rnext keys");
- try_delete_key("closed socket, delete current key", sk, 100, 100, 0, -1, -1, FAULT_BUSY);
- try_delete_key("closed socket, delete rnext key", sk, 200, 200, 0, -1, -1, FAULT_BUSY);
+ try_delete_key("closed socket, delete current key", sk, 100, 100, 0, 0, -1, -1, FAULT_BUSY);
+ try_delete_key("closed socket, delete rnext key", sk, 200, 200, 0, 0, -1, -1, FAULT_BUSY);
close(sk);
sk = prepare_sk(&this_ip_dest, 200, 200);
@@ -254,10 +259,12 @@ static void check_closed_socket(void)
if (test_add_key(sk, "Glory to Ukraine!", this_ip_dest,
DEFAULT_TEST_PREFIX, 12, 13))
test_error("test_add_key()");
- try_delete_key("closed socket, delete a key + set current/rnext", sk, 100, 100, 0, 10, 13, 0);
- try_delete_key("closed socket, force-delete current key", sk, 10, 11, 0, 200, -1, 0);
- try_delete_key("closed socket, force-delete rnext key", sk, 12, 13, 0, -1, 200, 0);
- try_delete_key("closed socket, delete current+rnext key", sk, 200, 200, 0, -1, -1, FAULT_BUSY);
+ try_delete_key("closed socket, delete a key + set current/rnext", sk,
+ 100, 100, 0, 0, 10, 13, 0);
+ try_delete_key("closed socket, force-delete current key", sk, 10, 11, 0, 0, 200, -1, 0);
+ try_delete_key("closed socket, force-delete rnext key", sk, 12, 13, 0, 0, -1, 200, 0);
+ try_delete_key("closed socket, delete current+rnext key", sk,
+ 200, 200, 0, 0, -1, -1, FAULT_BUSY);
close(sk);
sk = prepare_sk(&this_ip_dest, 200, 200);
@@ -272,6 +279,18 @@ static void check_closed_socket(void)
this_ip_dest, DEFAULT_TEST_PREFIX,
false, true, 20, 10, 0);
close(sk);
+
+ if (!should_skip_test("closed socket, add + delete VRF-scoped key",
+ KCONFIG_NET_VRF)) {
+ sk = prepare_sk(&this_ip_dest, 200, 200);
+ if (test_add_key_vrf(sk, SECOND_PASSWORD, TCP_AO_KEYF_IFINDEX,
+ this_ip_dest, DEFAULT_TEST_PREFIX,
+ test_vrf_ifindex, 201, 201))
+ test_error("test_add_key_vrf()");
+ try_delete_key("closed socket, add + delete VRF-scoped key", sk, 201, 201,
+ test_vrf_ifindex, 0, -1, -1, 0);
+ close(sk);
+ }
}
static void assert_no_current_rnext(const char *tst_msg, int sk)
@@ -322,8 +341,8 @@ static void check_listen_socket(void)
int sk, err;
sk = prepare_lsk(&this_ip_dest, 200, 200);
- try_delete_key("listen socket, delete a key", sk, 200, 200, 0, -1, -1, 0);
- try_delete_key("listen socket, delete all keys", sk, 100, 100, 0, -1, -1, 0);
+ try_delete_key("listen socket, delete a key", sk, 200, 200, 0, 0, -1, -1, 0);
+ try_delete_key("listen socket, delete all keys", sk, 100, 100, 0, 0, -1, -1, 0);
close(sk);
sk = prepare_lsk(&this_ip_dest, 200, 200);
@@ -345,8 +364,10 @@ static void check_listen_socket(void)
if (listen(sk, 10))
test_error("listen()");
assert_no_current_rnext("listen() after current/rnext keys set", sk);
- try_delete_key("listen socket, delete current key from before listen()", sk, 100, 100, 0, -1, -1, FAULT_FIXME);
- try_delete_key("listen socket, delete rnext key from before listen()", sk, 200, 200, 0, -1, -1, FAULT_FIXME);
+ try_delete_key("listen socket, delete current key from before listen()", sk,
+ 100, 100, 0, 0, -1, -1, FAULT_FIXME);
+ try_delete_key("listen socket, delete rnext key from before listen()", sk,
+ 200, 200, 0, 0, -1, -1, FAULT_FIXME);
close(sk);
assert_no_tcp_repair();
@@ -359,13 +380,13 @@ static void check_listen_socket(void)
DEFAULT_TEST_PREFIX, 12, 13))
test_error("test_add_key()");
try_delete_key("listen socket, delete a key + set current/rnext", sk,
- 100, 100, 0, 10, 13, FAULT_CURRNEXT);
+ 100, 100, 0, 0, 10, 13, FAULT_CURRNEXT);
try_delete_key("listen socket, force-delete current key", sk,
- 10, 11, 0, 200, -1, FAULT_CURRNEXT);
+ 10, 11, 0, 0, 200, -1, FAULT_CURRNEXT);
try_delete_key("listen socket, force-delete rnext key", sk,
- 12, 13, 0, -1, 200, FAULT_CURRNEXT);
+ 12, 13, 0, 0, -1, 200, FAULT_CURRNEXT);
try_delete_key("listen socket, delete a key", sk,
- 200, 200, 0, -1, -1, 0);
+ 200, 200, 0, 0, -1, -1, 0);
close(sk);
sk = prepare_lsk(&this_ip_dest, 200, 200);
@@ -1131,7 +1152,6 @@ static void check_established_socket(void)
{
unsigned int port = test_server_port;
- setup_vrfs();
try_client_run("client: Check current/rnext keys unset before connect()",
port++, 20, -1, -1);
try_client_run("client: Check current/rnext keys set before connect()",
@@ -1150,6 +1170,7 @@ static void *client_fn(void *arg)
{
if (inet_pton(TEST_FAMILY, TEST_WRONG_IP, &wrong_addr) != 1)
test_error("Can't convert ip address %s", TEST_WRONG_IP);
+ setup_vrfs();
check_closed_socket();
check_listen_socket();
check_established_socket();
@@ -1158,6 +1179,6 @@ static void *client_fn(void *arg)
int main(int argc, char *argv[])
{
- test_init(121, server_fn, client_fn);
+ test_init(122, server_fn, client_fn);
return 0;
}
diff --git a/tools/testing/selftests/net/tcp_ao/lib/aolib.h b/tools/testing/selftests/net/tcp_ao/lib/aolib.h
index ebb2899c12fe..53be1744237e 100644
--- a/tools/testing/selftests/net/tcp_ao/lib/aolib.h
+++ b/tools/testing/selftests/net/tcp_ao/lib/aolib.h
@@ -404,7 +404,8 @@ static inline int test_prepare_def_key(struct tcp_ao_add *ao,
extern int test_get_one_ao(int sk, struct tcp_ao_getsockopt *out,
void *addr, size_t addr_sz,
- uint8_t prefix, uint8_t sndid, uint8_t rcvid);
+ uint8_t prefix, uint8_t sndid, uint8_t rcvid,
+ uint8_t keyflags, int ifindex);
extern int test_get_ao_info(int sk, struct tcp_ao_info_opt *out);
extern int test_set_ao_info(int sk, struct tcp_ao_info_opt *in);
extern int test_cmp_getsockopt_setsockopt(const struct tcp_ao_add *a,
@@ -418,7 +419,8 @@ static inline int test_verify_socket_key(int sk, struct tcp_ao_add *key)
int err;
err = test_get_one_ao(sk, &key2, &key->addr, sizeof(key->addr),
- key->prefix, key->sndid, key->rcvid);
+ key->prefix, key->sndid, key->rcvid,
+ key->keyflags, key->ifindex);
if (err)
return err;
diff --git a/tools/testing/selftests/net/tcp_ao/lib/sock.c b/tools/testing/selftests/net/tcp_ao/lib/sock.c
index ef8e9031d47a..2e7b06a1a156 100644
--- a/tools/testing/selftests/net/tcp_ao/lib/sock.c
+++ b/tools/testing/selftests/net/tcp_ao/lib/sock.c
@@ -252,7 +252,7 @@ static int test_get_ao_keys_nr(int sk)
int test_get_one_ao(int sk, struct tcp_ao_getsockopt *out,
void *addr, size_t addr_sz, uint8_t prefix,
- uint8_t sndid, uint8_t rcvid)
+ uint8_t sndid, uint8_t rcvid, uint8_t keyflags, int ifindex)
{
struct tcp_ao_getsockopt tmp = {};
socklen_t tmp_sz = sizeof(tmp);
@@ -262,6 +262,8 @@ int test_get_one_ao(int sk, struct tcp_ao_getsockopt *out,
tmp.prefix = prefix;
tmp.sndid = sndid;
tmp.rcvid = rcvid;
+ tmp.keyflags = keyflags;
+ tmp.ifindex = ifindex;
tmp.nkeys = 1;
ret = getsockopt(sk, IPPROTO_TCP, TCP_AO_GET_KEYS, &tmp, &tmp_sz);
diff --git a/tools/testing/selftests/net/tcp_mmap.c b/tools/testing/selftests/net/tcp_mmap.c
index 2544ae35d07a..487ae659a1f1 100644
--- a/tools/testing/selftests/net/tcp_mmap.c
+++ b/tools/testing/selftests/net/tcp_mmap.c
@@ -141,12 +141,12 @@ static void *mmap_large_buffer(size_t need, size_t *allocated)
buffer = mmap(NULL, sz, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB, -1, 0);
- if (buffer == (void *)-1) {
+ if (buffer == MAP_FAILED) {
sz = need;
buffer = mmap(NULL, sz, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS | MAP_POPULATE,
-1, 0);
- if (buffer != (void *)-1)
+ if (buffer != MAP_FAILED)
fprintf(stderr, "MAP_HUGETLB attempt failed, look at /sys/kernel/mm/hugepages for optimal performance\n");
}
*allocated = sz;
@@ -189,13 +189,13 @@ void *child_thread(void *arg)
fcntl(fd, F_SETFL, O_NDELAY);
buffer = mmap_large_buffer(chunk_size, &buffer_sz);
- if (buffer == (void *)-1) {
+ if (buffer == MAP_FAILED) {
perror("mmap");
goto error;
}
if (zflg) {
raddr = mmap(NULL, chunk_size + map_align, PROT_READ, flags, fd, 0);
- if (raddr == (void *)-1) {
+ if (raddr == MAP_FAILED) {
perror("mmap");
zflg = 0;
} else {
@@ -547,7 +547,7 @@ int main(int argc, char *argv[])
}
buffer = mmap_large_buffer(chunk_size, &buffer_sz);
- if (buffer == (unsigned char *)-1) {
+ if (buffer == MAP_FAILED) {
perror("mmap");
exit(1);
}
diff --git a/tools/testing/selftests/net/test_vxlan_mdb.sh b/tools/testing/selftests/net/test_vxlan_mdb.sh
index 58da5de99ac4..f9600aabd4a2 100755
--- a/tools/testing/selftests/net/test_vxlan_mdb.sh
+++ b/tools/testing/selftests/net/test_vxlan_mdb.sh
@@ -685,6 +685,9 @@ star_g_common()
run_cmd "bridge -n $ns1 mdb add dev vx0 port vx0 grp $grp permanent filter_mode exclude source_list $grp dst $vtep_ip src_vni 10010"
log_test $? 255 "Invalid source in source list"
+ run_cmd "bridge -n $ns1 mdb add dev vx0 port vx0 grp $grp permanent filter_mode exclude source_list $all_zeros_grp dst $vtep_ip src_vni 10010"
+ log_test $? 255 "All-zeros source in source list"
+
run_cmd "bridge -n $ns1 mdb add dev vx0 port vx0 grp $grp permanent source_list $src1 dst $vtep_ip src_vni 10010"
log_test $? 255 "Source list without filter mode"
}
@@ -784,6 +787,9 @@ sg_common()
run_cmd "bridge -n $ns1 mdb add dev vx0 port vx0 grp $grp src $grp permanent dst $vtep_ip src_vni 10010"
log_test $? 255 "(S, G) with an invalid source list"
+ run_cmd "bridge -n $ns1 mdb add dev vx0 port vx0 grp $grp src $all_zeros_grp permanent dst $vtep_ip src_vni 10010"
+ log_test $? 255 "(S, G) with an all-zeros source"
+
run_cmd "bridge -n $ns1 mdb add dev vx0 port vx0 grp $all_zeros_grp src $src permanent dst $vtep_ip src_vni 10010"
log_test $? 255 "All-zeros group with source"
}
diff --git a/tools/testing/selftests/net/test_vxlan_nh.sh b/tools/testing/selftests/net/test_vxlan_nh.sh
index 20f3369f776b..5ce6f27f6cf4 100755
--- a/tools/testing/selftests/net/test_vxlan_nh.sh
+++ b/tools/testing/selftests/net/test_vxlan_nh.sh
@@ -56,6 +56,17 @@ tc_stats_get()
tc_rule_handle_stats_get "dev dummy1 egress" 101 ".packets" "-n $ns1"
}
+nh_stats_get_port()
+{
+ ip -n "$ns1" -s -j nexthop show id 20 | \
+ jq ".[][\"group_stats\"][][\"packets\"]"
+}
+
+tc_stats_get_port()
+{
+ tc_rule_handle_stats_get "dev dummy1 egress" 102 ".packets" "-n $ns1"
+}
+
basic_tx_common()
{
local af_str=$1; shift
@@ -90,6 +101,31 @@ basic_tx_common()
busywait "$BUSYWAIT_TIMEOUT" until_counter_is "== 1" tc_stats_get > /dev/null
check_err $? "tc filter stats did not increase"
+ # Add a second FDB nexthop group whose nexthop carries a per-nexthop
+ # destination port (NHA_DST_PORT) that differs from the VXLAN device
+ # default. Matching outer traffic must egress with that port, so a
+ # separate flower filter keyed on the new port catches it.
+ run_cmd "tc -n $ns1 filter add dev dummy1 egress proto $proto \
+ pref 1 handle 102 flower ip_proto udp dst_ip $remote_addr \
+ dst_port 4790 action pass"
+
+ run_cmd "ip -n $ns1 nexthop add id 2 via $remote_addr fdb dst_port 4790"
+ run_cmd "ip -n $ns1 nexthop add id 20 group 2 fdb"
+
+ run_cmd "bridge -n $ns1 fdb add 00:11:22:33:44:66 dev vx0 \
+ self static nhid 20"
+
+ run_cmd "ip netns exec $ns1 mausezahn vx0 -a own \
+ -b 00:11:22:33:44:66 -c 1 -q"
+
+ busywait "$BUSYWAIT_TIMEOUT" until_counter_is "== 1" \
+ nh_stats_get_port > /dev/null
+ check_err $? "FDB nexthop group stats did not increase (with port)"
+
+ busywait "$BUSYWAIT_TIMEOUT" until_counter_is "== 1" \
+ tc_stats_get_port > /dev/null
+ check_err $? "tc filter stats did not increase (with port)"
+
log_test "VXLAN FDB nexthop: $af_str basic Tx"
}
@@ -210,8 +246,8 @@ require_command arping
require_command ndisc6
require_command jq
-if ! ip nexthop help 2>&1 | grep -q "stats"; then
- echo "SKIP: iproute2 ip too old, missing nexthop stats support"
+if ! ip nexthop help 2>&1 | grep -q "dst_port"; then
+ echo "SKIP: iproute2 ip too old, missing nexthop dst_port support"
exit "$ksft_skip"
fi
diff --git a/tools/testing/selftests/net/tls.c b/tools/testing/selftests/net/tls.c
index a0e72e402748..9d3cd4fff062 100644
--- a/tools/testing/selftests/net/tls.c
+++ b/tools/testing/selftests/net/tls.c
@@ -24,6 +24,7 @@
#include "kselftest_harness.h"
#define TLS_PAYLOAD_MAX_LEN 16384
+#define TLS_HDR_LEN 5
#define SOL_TLS 282
static int fips_enabled;
@@ -2732,28 +2733,83 @@ TEST_F(tls_err, bad_rec)
EXPECT_EQ(errno, EAGAIN);
}
+/* cfd carries a byte stream, so one recv() can return part of a
+ * record. Take the fragment length from the record header and wait
+ * for the remainder.
+ */
+static void tls_send_bad_auth(struct __test_metadata *_metadata,
+ int fd, int cfd, int fd2)
+{
+ char buf[128];
+ int len;
+
+ memrnd(buf, sizeof(buf) / 2);
+ ASSERT_EQ(send(fd, buf, sizeof(buf) / 2, 0), sizeof(buf) / 2);
+
+ ASSERT_EQ(recv(cfd, buf, TLS_HDR_LEN, MSG_WAITALL), TLS_HDR_LEN);
+
+ len = ((unsigned char)buf[3] << 8) | (unsigned char)buf[4];
+ ASSERT_GT(len, 0);
+ ASSERT_LE(len, (int)sizeof(buf) - TLS_HDR_LEN);
+
+ ASSERT_EQ(recv(cfd, buf + TLS_HDR_LEN, len, MSG_WAITALL), len);
+
+ buf[TLS_HDR_LEN + len - 1]++;
+
+ ASSERT_EQ(send(fd2, buf, TLS_HDR_LEN + len, 0), TLS_HDR_LEN + len);
+}
+
TEST_F(tls_err, bad_auth)
{
char buf[128];
- int n;
if (self->notls)
SKIP(return, "no TLS support");
- memrnd(buf, sizeof(buf) / 2);
- EXPECT_EQ(send(self->fd, buf, sizeof(buf) / 2, 0), sizeof(buf) / 2);
- n = recv(self->cfd, buf, sizeof(buf), 0);
- EXPECT_GT(n, sizeof(buf) / 2);
+ tls_send_bad_auth(_metadata, self->fd, self->cfd, self->fd2);
- buf[n - 1]++;
-
- EXPECT_EQ(send(self->fd2, buf, n, 0), n);
EXPECT_EQ(recv(self->cfd2, buf, sizeof(buf), 0), -1);
EXPECT_EQ(errno, EBADMSG);
EXPECT_EQ(recv(self->cfd2, buf, sizeof(buf), 0), -1);
EXPECT_EQ(errno, EBADMSG);
}
+/* A record that did not authenticate breaks the connection for every
+ * reader, splice included.
+ *
+ * The two decrypt paths reach that result differently. A synchronous
+ * decrypt leaves the record parsed, so the splice re-runs the decrypt
+ * and fails on the record itself; the ctx->async_wait.err check in
+ * tls_sw_splice_read() is not what stops it. Only an asynchronous
+ * decrypt, which needs a TLS 1.2 socket and an AEAD advertising
+ * CRYPTO_ALG_ASYNC, consumes the record before the failure is
+ * recorded, leaving that check the sole reason the splice fails.
+ */
+TEST_F(tls_err, bad_auth_splice)
+{
+ char buf[128];
+ ssize_t ret;
+ int p[2];
+
+ if (self->notls)
+ SKIP(return, "no TLS support");
+
+ tls_send_bad_auth(_metadata, self->fd, self->cfd, self->fd2);
+
+ EXPECT_EQ(recv(self->cfd2, buf, sizeof(buf), 0), -1);
+ EXPECT_EQ(errno, EBADMSG);
+
+ ASSERT_GE(pipe(p), 0);
+
+ ret = splice(self->cfd2, NULL, p[1], NULL, sizeof(buf),
+ SPLICE_F_NONBLOCK);
+ EXPECT_EQ(ret, -1);
+ EXPECT_EQ(errno, EBADMSG);
+
+ close(p[0]);
+ close(p[1]);
+}
+
TEST_F(tls_err, bad_in_large_read)
{
char txt[3][64];
@@ -3009,7 +3065,6 @@ static size_t parse_tls_records(struct __test_metadata *_metadata,
{
const __u8 *rec = rx_buf;
size_t total_plaintext_rx = 0;
- const __u8 rec_header_len = 5;
while (rec < rx_buf + rx_len) {
__u16 record_payload_len;
@@ -3029,7 +3084,7 @@ static size_t parse_tls_records(struct __test_metadata *_metadata,
/* Plaintext must not exceed the specified limit */
ASSERT_LE(plaintext_len, max_payload_len);
- rec += rec_header_len + record_payload_len;
+ rec += TLS_HDR_LEN + record_payload_len;
}
return total_plaintext_rx;
diff --git a/tools/testing/selftests/nolibc/Makefile.include b/tools/testing/selftests/nolibc/Makefile.include
index c30ca3a9ef14..ea520eac64a6 100644
--- a/tools/testing/selftests/nolibc/Makefile.include
+++ b/tools/testing/selftests/nolibc/Makefile.include
@@ -5,7 +5,7 @@ _CFLAGS_STACKPROTECTOR ?= $(call try-run, \
echo 'void foo(void) {}' | $(CC) -x c - -o - -S $(CLANG_CROSS_FLAGS) $(__CFLAGS_STACKPROTECTOR) | grep -q __stack_chk_guard, \
$(__CFLAGS_STACKPROTECTOR))
_CFLAGS_SANITIZER ?= $(call cc-option,-fsanitize=undefined -fsanitize-trap=all)
-CFLAGS_NOLIBC_TEST ?= -Os -fno-ident -fno-asynchronous-unwind-tables -std=c89 \
+CFLAGS_NOLIBC_TEST ?= -Os -fno-ident -fno-asynchronous-unwind-tables -std=c89 -ggdb \
-W -Wall -Wextra -Wundef -Wwrite-strings \
$(call cc-option,-fno-stack-protector) $(call cc-option,-Wmissing-prototypes) \
$(_CFLAGS_STACKPROTECTOR) $(_CFLAGS_SANITIZER)
diff --git a/tools/testing/selftests/nolibc/Makefile.nolibc b/tools/testing/selftests/nolibc/Makefile.nolibc
index 06f881e2e90c..f70c8dfca018 100644
--- a/tools/testing/selftests/nolibc/Makefile.nolibc
+++ b/tools/testing/selftests/nolibc/Makefile.nolibc
@@ -112,6 +112,7 @@ EXTRACONFIG_armthumb = -e CONFIG_NAMESPACES
EXTRACONFIG_sparc32 = -e CONFIG_TMPFS
EXTRACONFIG_m68k = -e CONFIG_BLK_DEV_INITRD
EXTRACONFIG_sh4 = -e CONFIG_BLK_DEV_INITRD -e CONFIG_CMDLINE_FROM_BOOTLOADER
+EXTRACONFIG_alpha = -e CONFIG_BLK_DEV_INITRD
EXTRACONFIG = $(EXTRACONFIG_$(XARCH))
# optional tests to run (default = all)
@@ -174,6 +175,7 @@ QEMU_ARGS_m68k = -M virt -append "console=ttyGF0,115200 panic=-1 $(TEST:%=
QEMU_ARGS_sh4 = -M r2d -serial file:/dev/stdout -append "console=ttySC1,115200 panic=-1 $(TEST:%=NOLIBC_TEST=%)"
QEMU_ARGS_openrisc = -M virt -m 512M -append "console=ttyS0 panic=-1 $(TEST:%=NOLIBC_TEST=%)"
QEMU_ARGS_parisc32 = -M B160L -append "console=ttyS0 panic=-1 $(TEST:%=NOLIBC_TEST=%)"
+QEMU_ARGS_alpha = -M clipper -append "console=ttyS0 panic=-1 $(TEST:%=NOLIBC_TEST=%)"
QEMU_ARGS = -m 1G $(QEMU_ARGS_$(XARCH)) $(QEMU_ARGS_BIOS) $(QEMU_ARGS_EXTRA)
# OUTPUT is only set when run from the main makefile, otherwise
diff --git a/tools/testing/selftests/nolibc/nolibc-test.c b/tools/testing/selftests/nolibc/nolibc-test.c
index c1c1ce43a047..ed860b0a15a1 100644
--- a/tools/testing/selftests/nolibc/nolibc-test.c
+++ b/tools/testing/selftests/nolibc/nolibc-test.c
@@ -756,6 +756,10 @@ int run_startup(int min, int max)
/* checking NULL for argv/argv0, environ and _auxv is not enough, let's compare with sbrk(0) or &end */
extern char end;
char *brk = sbrk(0) != (void *)-1 ? sbrk(0) : &end;
+#if defined(__alpha__)
+ /* the ordering above does not work on an alpha kernel due to STACK_TOP != TASK_SIZE */
+ brk = NULL;
+#endif
/* differ from nolibc, both glibc and musl have no global _auxv */
const unsigned long *test_auxv = (void *)-1;
#ifdef NOLIBC
@@ -854,6 +858,58 @@ static int test_dirent(void)
return 0;
}
+int test_getcwd(void)
+{
+ char cwd_syscall[PATH_MAX];
+ char cwd_proc[PATH_MAX];
+ ssize_t len;
+
+ /* Read where the link /proc/self/cwd points */
+ len = readlink("/proc/self/cwd", cwd_proc, sizeof(cwd_proc) - 1);
+ if (len <= 0)
+ return __LINE__;
+
+ /* Terminate the string from readlink() */
+ cwd_proc[len] = '\0';
+
+ /* Get the cwd via syscall */
+ if (getcwd(cwd_syscall, sizeof(cwd_syscall)) == NULL)
+ return __LINE__;
+
+ /* Fail if they aren't the same */
+ if (strcmp(cwd_proc, cwd_syscall) != 0)
+ return __LINE__;
+
+ /* Try getcwd() with NULL for the buffer,
+ * should return NULL and an error in errno.
+ * Other libc's allow this by allocating a buffer
+ * internally.
+ */
+ if (is_nolibc) {
+ errno = 0;
+ if (getcwd(NULL, 0) != NULL || !errno)
+ return __LINE__;
+ }
+
+ /* Try getcwd() with a buffer but make the size 0,
+ * should return NULL and an error in errno.
+ */
+ errno = 0;
+ if (getcwd(cwd_syscall, 0) != NULL || !errno)
+ return __LINE__;
+
+ /* Try getcwd() with a buffer but make the size 1,
+ * should return NULL and an error in errno because
+ * the string written to the buffer is terminated
+ * so you need at least 2 bytes even for "/".
+ */
+ errno = 0;
+ if (getcwd(cwd_syscall, 1) != NULL || !errno)
+ return __LINE__;
+
+ return 0;
+}
+
int test_getrandom(void)
{
uint64_t rng = 0;
@@ -1555,6 +1611,7 @@ int run_syscall(int min, int max)
CASE_TEST(clock_getres); EXPECT_SYSZR(1, clock_getres(CLOCK_MONOTONIC, &ts)); break;
CASE_TEST(clock_gettime); EXPECT_SYSZR(1, clock_gettime(CLOCK_MONOTONIC, &ts)); break;
CASE_TEST(clock_settime); EXPECT_SYSER(1, clock_settime(CLOCK_MONOTONIC, &ts), -1, EINVAL); break;
+ CASE_TEST(getcwd); EXPECT_SYSZR(proc, test_getcwd()); break;
CASE_TEST(getpid); EXPECT_SYSNE(1, getpid(), -1); break;
CASE_TEST(getppid); EXPECT_SYSNE(1, getppid(), -1); break;
CASE_TEST(gettid); EXPECT_SYSNE(has_gettid, gettid(), -1); break;
diff --git a/tools/testing/selftests/nolibc/run-tests.sh b/tools/testing/selftests/nolibc/run-tests.sh
index 6460e25001de..dc0b1649c641 100755
--- a/tools/testing/selftests/nolibc/run-tests.sh
+++ b/tools/testing/selftests/nolibc/run-tests.sh
@@ -30,6 +30,7 @@ all_archs=(
m68k
sh4
parisc32
+ alpha
)
archs="${all_archs[@]}"
@@ -193,7 +194,7 @@ test_arch() {
exit 1
esac
printf '%-15s' "$arch:"
- if [ "$arch" = "m68k" -o "$arch" = "sh4" -o "$arch" = "openrisc" -o "$arch" = "parisc32" ] && [ "$llvm" = "1" ]; then
+ if [ "$arch" = "m68k" -o "$arch" = "sh4" -o "$arch" = "openrisc" -o "$arch" = "parisc32" -o "$arch" = "alpha" ] && [ "$llvm" = "1" ]; then
echo "Unsupported configuration"
return
fi
diff --git a/tools/testing/selftests/power_supply/helpers.sh b/tools/testing/selftests/power_supply/helpers.sh
index 1ec90d7c9108..ff1175aeed72 100644
--- a/tools/testing/selftests/power_supply/helpers.sh
+++ b/tools/testing/selftests/power_supply/helpers.sh
@@ -108,7 +108,7 @@ test_sysfs_prop_optional_range() {
test_sysfs_prop_optional_list() {
PROP=$1
- LIST=$2
+ shift
TEST_NAME="$DEVNAME".sysfs."$PROP"
@@ -118,9 +118,9 @@ test_sysfs_prop_optional_list() {
valid=0
OLDIFS=$IFS
- IFS=","
- for item in $LIST; do
- if [ "$DATA" = "$item" ]; then
+ IFS=",$IFS"
+ for item in $*; do
+ if [ "$item" ] && [ "$DATA" = "$item" ]; then
valid=1
break
fi
diff --git a/tools/testing/selftests/power_supply/test_power_supply_properties.sh b/tools/testing/selftests/power_supply/test_power_supply_properties.sh
index a66b1313ed88..1ebac6fe5d23 100755
--- a/tools/testing/selftests/power_supply/test_power_supply_properties.sh
+++ b/tools/testing/selftests/power_supply/test_power_supply_properties.sh
@@ -74,7 +74,8 @@ for DEVNAME in $supplies; do
test_sysfs_prop_optional model_name
test_sysfs_prop_optional manufacturer
test_sysfs_prop_optional serial_number
- test_sysfs_prop_optional_list technology "Unknown","NiMH","Li-ion","Li-poly","LiFe","NiCd","LiMn"
+ test_sysfs_prop_optional_list technology "Unknown","NiMH","Li-ion","Li-poly","LiFe","NiCd"\
+ ,"LiMn","PbAc","NiZn","RAM","ZnAr"
test_sysfs_prop_optional cycle_count
diff --git a/tools/testing/selftests/powerpc/pmu/ebb/lost_exception_test.c b/tools/testing/selftests/powerpc/pmu/ebb/lost_exception_test.c
index ba2681a12cc7..9be5945f3b1f 100644
--- a/tools/testing/selftests/powerpc/pmu/ebb/lost_exception_test.c
+++ b/tools/testing/selftests/powerpc/pmu/ebb/lost_exception_test.c
@@ -8,6 +8,7 @@
#include <stdio.h>
#include <stdlib.h>
#include <sys/mman.h>
+#include <time.h>
#include "ebb.h"
@@ -22,6 +23,7 @@ static int test_body(void)
{
int i, orig_period, max_period;
struct event event;
+ struct timespec ts = { .tv_sec = 0, .tv_nsec = 1 };
SKIP_IF(!ebb_is_supported());
@@ -57,10 +59,15 @@ static int test_body(void)
* kernel to decide our timeslice is up and context switch to
* the other thread. When we come back our EBB will have been
* lost and we'll spin in this while loop forever.
+ *
+ * Use nanosleep(0) instead of sched_yield() to guarantee a
+ * context switch to the eat_cpu child regardless of the
+ * eligibility state. sched_yield() via yield_task_fair() may
+ * become a no-op when the task is ineligible (vruntime ahead
+ * of avg_vruntime), preventing the required context switch.
*/
-
for (i = 0; i < 100000; i++)
- sched_yield();
+ nanosleep(&ts, NULL);
/* Change the sample period slightly to try and hit the race */
if (sample_period >= (orig_period + 200))
diff --git a/tools/testing/selftests/powerpc/tm/tm-signal-context-force-tm.c b/tools/testing/selftests/powerpc/tm/tm-signal-context-force-tm.c
index 0a4bc479ae39..5dc0f12f467d 100644
--- a/tools/testing/selftests/powerpc/tm/tm-signal-context-force-tm.c
+++ b/tools/testing/selftests/powerpc/tm/tm-signal-context-force-tm.c
@@ -60,7 +60,7 @@ void usr_signal_handler(int signo, siginfo_t *si, void *uc)
ucp->uc_link = mmap(NULL, sizeof(ucontext_t),
PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, 0, 0);
- if (ucp->uc_link == (void *)-1) {
+ if (ucp->uc_link == MAP_FAILED) {
perror("Mmap failed");
exit(-1);
}
@@ -129,7 +129,7 @@ void tm_trap_test(void)
ss.ss_size = SIGSTKSZ;
ss.ss_flags = 0;
- if (ss.ss_sp == (void *)-1) {
+ if (ss.ss_sp == MAP_FAILED) {
perror("mmap error\n");
exit(-1);
}
diff --git a/tools/testing/selftests/powerpc/tm/tm.h b/tools/testing/selftests/powerpc/tm/tm.h
index c03c6e778876..6024ce4ba6ff 100644
--- a/tools/testing/selftests/powerpc/tm/tm.h
+++ b/tools/testing/selftests/powerpc/tm/tm.h
@@ -105,8 +105,12 @@ static inline bool failure_is_nesting(void)
static inline int tcheck(void)
{
long cr;
- asm volatile ("tcheck 0" : "=r"(cr) : : "cr0");
- return (cr >> 28) & 4;
+ asm volatile("tcheck 0;"
+ "mfcr %0;"
+ : "=r"(cr)
+ :
+ : "cr0");
+ return (cr >> 28) & 0xf;
}
static inline bool tcheck_doomed(void)
diff --git a/tools/testing/selftests/prctl/set-anon-vma-name-test.c b/tools/testing/selftests/prctl/set-anon-vma-name-test.c
index ac6721b184a6..5f9589534da8 100644
--- a/tools/testing/selftests/prctl/set-anon-vma-name-test.c
+++ b/tools/testing/selftests/prctl/set-anon-vma-name-test.c
@@ -4,11 +4,13 @@
*/
#include <errno.h>
+#include <fcntl.h>
#include <sys/prctl.h>
#include <stdio.h>
#include <stdlib.h>
#include <sys/mman.h>
#include <string.h>
+#include <unistd.h>
#include "kselftest_harness.h"
@@ -73,15 +75,24 @@ int was_renaming_successful(char *target_name, unsigned long ptr)
FIXTURE(vma) {
void *ptr_anon, *ptr_not_anon;
+ int fd_not_anon;
};
FIXTURE_SETUP(vma) {
+ char template[] = "./set-anon-vma-test-XXXXXX";
+
self->ptr_anon = mmap(NULL, AREA_SIZE, PROT_READ | PROT_WRITE,
- MAP_PRIVATE | MAP_ANONYMOUS, 0, 0);
- ASSERT_NE(self->ptr_anon, NULL);
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ ASSERT_NE(self->ptr_anon, MAP_FAILED);
+
+ self->fd_not_anon = mkstemp(template);
+ ASSERT_NE(self->fd_not_anon, -1);
+ unlink(template);
+ ASSERT_EQ(ftruncate(self->fd_not_anon, AREA_SIZE), 0);
self->ptr_not_anon = mmap(NULL, AREA_SIZE, PROT_READ | PROT_WRITE,
- MAP_PRIVATE, 0, 0);
- ASSERT_NE(self->ptr_not_anon, NULL);
+ MAP_PRIVATE, self->fd_not_anon, 0);
+ ASSERT_NE(self->ptr_not_anon, MAP_FAILED);
+ close(self->fd_not_anon);
}
FIXTURE_TEARDOWN(vma) {
@@ -98,7 +109,7 @@ TEST_F(vma, renaming) {
EXPECT_EQ(rename_vma((unsigned long)self->ptr_anon, AREA_SIZE, BAD_NAME), -EINVAL);
TH_LOG("Try to rename non-anonymous VMA");
- EXPECT_EQ(rename_vma((unsigned long) self->ptr_not_anon, AREA_SIZE, GOOD_NAME), -EINVAL);
+ EXPECT_EQ(rename_vma((unsigned long) self->ptr_not_anon, AREA_SIZE, GOOD_NAME), -EBADF);
}
TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/proc/proc-maps-race.c b/tools/testing/selftests/proc/proc-maps-race.c
index 1026d8c400e1..415eccb70468 100644
--- a/tools/testing/selftests/proc/proc-maps-race.c
+++ b/tools/testing/selftests/proc/proc-maps-race.c
@@ -490,7 +490,8 @@ static bool query_addr_at(int maps_fd, void *addr,
static inline bool split_vma(FIXTURE_DATA(proc_maps_race) *self)
{
- return mmap(self->mod_info->addr, self->page_size, self->mod_info->prot | PROT_EXEC,
+ /* PROT_NONE differs from both readable neighbors. */
+ return mmap(self->mod_info->addr, self->page_size, PROT_NONE,
MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0) != MAP_FAILED;
}
diff --git a/tools/testing/selftests/proc/proc-pidns.c b/tools/testing/selftests/proc/proc-pidns.c
index 25b9a2933c45..6f7c10fe97b3 100644
--- a/tools/testing/selftests/proc/proc-pidns.c
+++ b/tools/testing/selftests/proc/proc-pidns.c
@@ -6,6 +6,7 @@
#include <assert.h>
#include <errno.h>
+#include <fcntl.h>
#include <sched.h>
#include <stdbool.h>
#include <stdlib.h>
diff --git a/tools/testing/selftests/rdma/Makefile b/tools/testing/selftests/rdma/Makefile
index 07af7f15c1bf..a91c14c45006 100644
--- a/tools/testing/selftests/rdma/Makefile
+++ b/tools/testing/selftests/rdma/Makefile
@@ -3,6 +3,7 @@ TEST_PROGS := rxe_rping_between_netns.sh \
rxe_ipv6.sh \
rxe_socket_with_netns.sh \
rxe_test_NETDEV_UNREGISTER.sh \
- rxe_sent_rcvd_bytes.sh
+ rxe_sent_rcvd_bytes.sh \
+ rxe_netns_names.sh
include ../lib.mk
diff --git a/tools/testing/selftests/rdma/config b/tools/testing/selftests/rdma/config
index 4ffb814e253b..e1ff54ec0f57 100644
--- a/tools/testing/selftests/rdma/config
+++ b/tools/testing/selftests/rdma/config
@@ -1,3 +1,5 @@
CONFIG_TUN
CONFIG_VETH
+CONFIG_DUMMY
+CONFIG_NET_NS
CONFIG_RDMA_RXE
diff --git a/tools/testing/selftests/rdma/rxe_netns_names.sh b/tools/testing/selftests/rdma/rxe_netns_names.sh
new file mode 100755
index 000000000000..f40118407f4c
--- /dev/null
+++ b/tools/testing/selftests/rdma/rxe_netns_names.sh
@@ -0,0 +1,334 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Exercise RDMA device name handling across network namespaces.
+
+source "$(dirname "$0")/../kselftest/ktap_helpers.sh"
+
+NAME_PREFIX="rxe_netns_names_$$"
+NETDEV_PREFIX="rxn$$"
+NS1="${NAME_PREFIX}ns1"
+NS2="${NAME_PREFIX}ns2"
+RXE_A="${NAME_PREFIX}rxe_a"
+RXE_B="${NAME_PREFIX}rxe_b"
+RXE_SAME="${NAME_PREFIX}rxe_same"
+RXE_NEW="${NAME_PREFIX}rxe_new"
+DUMMY_A="${NETDEV_PREFIX}a"
+DUMMY_B="${NETDEV_PREFIX}b"
+OLD_MODE=""
+MODE_CHANGED=0
+MODS=("dummy" "rdma_rxe")
+TEST_SAME_NAMES="same RDMA device name can exist in two net namespaces"
+TEST_MOVE_CONFLICT="move without rename fails on destination name conflict"
+TEST_MOVE_RENAME="move then rename succeeds"
+TEST_COMBINED_MOVE_RENAME="move with requested destination name succeeds"
+TEST_SAME_NETNS_DUP_RENAME="same-netns rename rejects duplicate name"
+TEST_TEARDOWN_RETURN="netns delete returns device to init_net and renames on conflict"
+
+ksft_skip()
+{
+ ktap_skip_all "$*"
+ exit "$KSFT_SKIP"
+}
+
+fail()
+{
+ ktap_exit_fail_msg "$*"
+}
+
+need_cmd()
+{
+ command -v "$1" >/dev/null 2>&1 || ksft_skip "missing command: $1"
+}
+
+rdma_ns()
+{
+ local ns=$1
+
+ shift
+ ip netns exec "$ns" rdma "$@"
+}
+
+rdma_dev_exists()
+{
+ local ns=$1
+ local dev=$2
+
+ if [ -n "$ns" ]; then
+ rdma_ns "$ns" dev show "$dev" >/dev/null 2>&1
+ else
+ rdma dev show "$dev" >/dev/null 2>&1
+ fi
+}
+
+add_dummy()
+{
+ local netdev=$1
+
+ ip link add "$netdev" type dummy || return 1
+ ip link set "$netdev" up || return 1
+}
+
+add_rxe()
+{
+ local dev=$1
+ local netdev=$2
+
+ rdma link add "$dev" type rxe netdev "$netdev"
+}
+
+rdma_dev_on_netdev()
+{
+ local netdev=$1
+
+ rdma link show 2>/dev/null | awk -v want="$netdev" '
+ {
+ for (i = 1; i < NF; i++)
+ if ($i == "netdev" && $(i + 1) == want) {
+ dev = $2
+ sub(/\/.*/, "", dev)
+ print dev
+ exit
+ }
+ }'
+}
+
+wait_rdma_dev_on_netdev()
+{
+ local netdev=$1
+ local dev
+ local i
+
+ for i in $(seq 1 50); do
+ dev=$(rdma_dev_on_netdev "$netdev")
+ if [ -n "$dev" ]; then
+ echo "$dev"
+ return 0
+ fi
+ sleep 0.1
+ done
+
+ return 1
+}
+
+# ip link del returns after NETDEV_UNREGISTER, but rxe tears the RDMA device
+# down asynchronously via ib_unregister_device_queued(). Wait until our names
+# are gone.
+wait_rdma_devs_gone()
+{
+ local i name ns
+ local names=("$RXE_A" "$RXE_B" "$RXE_SAME" "$RXE_NEW")
+
+ for i in $(seq 1 50); do
+ local found=0
+
+ for name in "${names[@]}"; do
+ if rdma_dev_exists "" "$name"; then
+ found=1
+ break
+ fi
+ for ns in "$NS1" "$NS2"; do
+ ip netns exec "$ns" true 2>/dev/null || continue
+ if rdma_dev_exists "$ns" "$name"; then
+ found=1
+ break 2
+ fi
+ done
+ done
+
+ [ "$found" -eq 0 ] && return 0
+ sleep 0.1
+ done
+
+ return 1
+}
+
+setup_devs()
+{
+ cleanup_devs || return 1
+
+ add_dummy "$DUMMY_A" || return 1
+ add_dummy "$DUMMY_B" || return 1
+
+ add_rxe "$RXE_A" "$DUMMY_A" || return 1
+ add_rxe "$RXE_B" "$DUMMY_B" || return 1
+}
+
+cleanup_devs()
+{
+ ip link del "$DUMMY_A" 2>/dev/null
+ ip link del "$DUMMY_B" 2>/dev/null
+ wait_rdma_devs_gone
+}
+
+setup()
+{
+ OLD_MODE=$(rdma system show 2>/dev/null |
+ sed -n 's/.*netns \([^ ]*\).*/\1/p')
+ [ -n "$OLD_MODE" ] || ksft_skip "failed to read RDMA netns mode"
+
+ rdma system set netns exclusive >/dev/null 2>&1 ||
+ ksft_skip "rdma netns exclusive mode is not supported"
+ MODE_CHANGED=1
+
+ ip netns add "$NS1" || return 1
+ ip netns add "$NS2" || return 1
+}
+
+# ip netns del returns before rdma_dev_exit_net() removes the net from
+# rdma_nets. rdma_compatdev_set() returns -EBUSY until that completes, so
+# retry the mode restore instead of leaving the system in exclusive mode.
+restore_netns_mode()
+{
+ local i
+
+ [ "$MODE_CHANGED" -eq 1 ] || return 0
+
+ for i in $(seq 1 50); do
+ if rdma system set netns "$OLD_MODE" >/dev/null 2>&1; then
+ MODE_CHANGED=0
+ return 0
+ fi
+ sleep 0.1
+ done
+
+ echo "warning: failed to restore RDMA netns mode to $OLD_MODE" >&2
+ return 1
+}
+
+cleanup()
+{
+ cleanup_devs
+
+ ip netns del "$NS1" 2>/dev/null
+ ip netns del "$NS2" 2>/dev/null
+
+ restore_netns_mode
+
+ for m in "${MODS[@]}"; do
+ modprobe -r "$m" 2>/dev/null
+ done
+}
+
+rdma_supports_combined_move_rename()
+{
+ rdma dev help 2>&1 | grep -Eq 'netns .*name|name .*netns'
+}
+
+[ "$(id -u)" -eq 0 ] || ksft_skip "must be run as root"
+need_cmd ip
+need_cmd rdma
+need_cmd modprobe
+
+trap cleanup EXIT
+
+for m in "${MODS[@]}"; do
+ modinfo "$m" >/dev/null 2>&1 || ksft_skip "module $m not found"
+ modprobe "$m" || fail "failed to load $m"
+done
+
+setup || fail "failed to create net namespaces"
+
+ktap_print_header
+ktap_set_plan 6
+
+if setup_devs &&
+ rdma dev set "$RXE_A" netns "$NS1" &&
+ rdma_ns "$NS1" dev set "$RXE_A" name "$RXE_SAME" &&
+ rdma dev set "$RXE_B" netns "$NS2" &&
+ rdma_ns "$NS2" dev set "$RXE_B" name "$RXE_SAME" &&
+ rdma_dev_exists "$NS1" "$RXE_SAME" &&
+ rdma_dev_exists "$NS2" "$RXE_SAME"; then
+ ktap_test_pass "$TEST_SAME_NAMES"
+else
+ ktap_test_fail "$TEST_SAME_NAMES"
+fi
+cleanup_devs
+
+if ! setup_devs ||
+ ! rdma dev set "$RXE_A" netns "$NS1" ||
+ ! rdma_ns "$NS1" dev set "$RXE_A" name "$RXE_SAME" ||
+ ! rdma dev set "$RXE_B" netns "$NS2" ||
+ ! rdma_ns "$NS2" dev set "$RXE_B" name "$RXE_SAME"; then
+ ktap_test_fail "$TEST_MOVE_CONFLICT"
+elif rdma_ns "$NS1" dev set "$RXE_SAME" netns "$NS2" >/dev/null 2>&1; then
+ ktap_test_fail "$TEST_MOVE_CONFLICT"
+elif rdma_dev_exists "$NS1" "$RXE_SAME" &&
+ rdma_dev_exists "$NS2" "$RXE_SAME"; then
+ ktap_test_pass "$TEST_MOVE_CONFLICT"
+else
+ ktap_test_fail "$TEST_MOVE_CONFLICT"
+fi
+cleanup_devs
+
+if ! setup_devs; then
+ ktap_test_fail "$TEST_MOVE_RENAME"
+elif rdma dev set "$RXE_A" netns "$NS2" &&
+ rdma_ns "$NS2" dev set "$RXE_A" name "$RXE_NEW"; then
+ if rdma_dev_exists "$NS2" "$RXE_NEW" &&
+ ! rdma_dev_exists "" "$RXE_A"; then
+ ktap_test_pass "$TEST_MOVE_RENAME"
+ else
+ ktap_test_fail "$TEST_MOVE_RENAME"
+ fi
+else
+ ktap_test_fail "$TEST_MOVE_RENAME"
+fi
+cleanup_devs
+
+if ! rdma_supports_combined_move_rename; then
+ ktap_test_skip "$TEST_COMBINED_MOVE_RENAME"
+elif ! setup_devs; then
+ ktap_test_fail "$TEST_COMBINED_MOVE_RENAME"
+elif rdma dev set "$RXE_A" netns "$NS2" name "$RXE_NEW"; then
+ if rdma_dev_exists "$NS2" "$RXE_NEW" &&
+ ! rdma_dev_exists "" "$RXE_A"; then
+ ktap_test_pass "$TEST_COMBINED_MOVE_RENAME"
+ else
+ ktap_test_fail "$TEST_COMBINED_MOVE_RENAME"
+ fi
+else
+ ktap_test_fail "$TEST_COMBINED_MOVE_RENAME"
+fi
+cleanup_devs
+
+if ! setup_devs; then
+ ktap_test_fail "$TEST_SAME_NETNS_DUP_RENAME"
+elif rdma dev set "$RXE_A" name "$RXE_SAME" &&
+ rdma dev set "$RXE_B" name "$RXE_NEW"; then
+ if rdma dev set "$RXE_SAME" name "$RXE_NEW" >/dev/null 2>&1; then
+ ktap_test_fail "$TEST_SAME_NETNS_DUP_RENAME"
+ elif rdma_dev_exists "" "$RXE_SAME" &&
+ rdma_dev_exists "" "$RXE_NEW"; then
+ ktap_test_pass "$TEST_SAME_NETNS_DUP_RENAME"
+ else
+ ktap_test_fail "$TEST_SAME_NETNS_DUP_RENAME"
+ fi
+else
+ ktap_test_fail "$TEST_SAME_NETNS_DUP_RENAME"
+fi
+cleanup_devs
+
+if ! setup_devs; then
+ ktap_test_fail "$TEST_TEARDOWN_RETURN"
+elif ! rdma dev set "$RXE_A" name "$RXE_SAME" ||
+ ! rdma dev set "$RXE_B" netns "$NS2" ||
+ ! rdma_ns "$NS2" dev set "$RXE_B" name "$RXE_SAME" ||
+ ! rdma_dev_exists "$NS2" "$RXE_SAME"; then
+ ktap_test_fail "$TEST_TEARDOWN_RETURN"
+else
+ ip netns del "$NS2"
+ returned=$(wait_rdma_dev_on_netdev "$DUMMY_B")
+ ktap_print_msg "device returned to init_net as '${returned:-<missing>}'"
+ if rdma_dev_exists "" "$RXE_SAME" &&
+ [ -n "$returned" ] &&
+ [ "$returned" != "$RXE_SAME" ] &&
+ [ "${returned#ibdev}" != "$returned" ]; then
+ ktap_test_pass "$TEST_TEARDOWN_RETURN"
+ else
+ ktap_test_fail "$TEST_TEARDOWN_RETURN"
+ fi
+fi
+cleanup_devs
+
+ktap_finished
diff --git a/tools/testing/selftests/riscv/cfi/cfi_rv_test.h b/tools/testing/selftests/riscv/cfi/cfi_rv_test.h
index 1c8043f2b778..184df6903d01 100644
--- a/tools/testing/selftests/riscv/cfi/cfi_rv_test.h
+++ b/tools/testing/selftests/riscv/cfi/cfi_rv_test.h
@@ -56,7 +56,7 @@
#define CSR_SSP 0x011
-#ifdef __ASSEMBLY__
+#ifdef __ASSEMBLER__
#define __ASM_STR(x) x
#else
#define __ASM_STR(x) #x
diff --git a/tools/testing/selftests/riscv/hwprobe/hwprobe.c b/tools/testing/selftests/riscv/hwprobe/hwprobe.c
index 54c435af9923..eca4441ee77f 100644
--- a/tools/testing/selftests/riscv/hwprobe/hwprobe.c
+++ b/tools/testing/selftests/riscv/hwprobe/hwprobe.c
@@ -9,7 +9,7 @@ int main(int argc, char **argv)
long out;
ksft_print_header();
- ksft_set_plan(5);
+ ksft_set_plan(6);
/* Fake the CPU_SET ops. */
cpus = -1;
@@ -62,5 +62,23 @@ int main(int argc, char **argv)
pairs[1].key == 1 && pairs[1].value != 0xAAAA,
"Unknown key overwritten with -1 and doesn't block other elements\n");
+ pairs[0].key = RISCV_HWPROBE_KEY_IMA_EXT_0;
+ out = riscv_hwprobe(pairs, 1, 0, 0, 0);
+ if (out != 0)
+ ksft_exit_fail_msg("hwprobe(IMA_EXT_0) failed with %ld\n", out);
+
+ /*
+ * The RISC-V ISA manual specifies that Zfh implies Zfhmin and Zvfh
+ * implies Zvfhmin, so hwprobe must report the implied subset
+ * extensions whenever the supersets are present.
+ */
+ if ((pairs[0].value & RISCV_HWPROBE_EXT_ZFH) &&
+ !(pairs[0].value & RISCV_HWPROBE_EXT_ZFHMIN))
+ ksft_exit_fail_msg("Zfh reported without implied Zfhmin\n");
+ if ((pairs[0].value & RISCV_HWPROBE_EXT_ZVFH) &&
+ !(pairs[0].value & RISCV_HWPROBE_EXT_ZVFHMIN))
+ ksft_exit_fail_msg("Zvfh reported without implied Zvfhmin\n");
+ ksft_test_result_pass("Zfh/Zvfh imply Zfhmin/Zvfhmin\n");
+
ksft_finished();
}
diff --git a/tools/testing/selftests/rseq/.gitignore b/tools/testing/selftests/rseq/.gitignore
index ec01d164c1f0..9b6eb6c3922f 100644
--- a/tools/testing/selftests/rseq/.gitignore
+++ b/tools/testing/selftests/rseq/.gitignore
@@ -11,3 +11,5 @@ param_test_mm_cid_benchmark
param_test_mm_cid_compare_twice
syscall_errors_test
slice_test
+legacy_check
+check_optimized
diff --git a/tools/testing/selftests/rseq/rseq-x86-thread-pointer.h b/tools/testing/selftests/rseq/rseq-x86-thread-pointer.h
index d3133587d996..5a29d6bec51f 100644
--- a/tools/testing/selftests/rseq/rseq-x86-thread-pointer.h
+++ b/tools/testing/selftests/rseq/rseq-x86-thread-pointer.h
@@ -8,13 +8,11 @@
#ifndef _RSEQ_X86_THREAD_POINTER
#define _RSEQ_X86_THREAD_POINTER
-#include <features.h>
-
#ifdef __cplusplus
extern "C" {
#endif
-#if __GNUC_PREREQ (11, 1)
+#if __GNUC__ > 11 || (__GNUC__ == 11 && __GNUC_MINOR__ >= 1)
static inline void *rseq_thread_pointer(void)
{
return __builtin_thread_pointer();
diff --git a/tools/testing/selftests/rseq/rseq.c b/tools/testing/selftests/rseq/rseq.c
index be0d0a97031e..1cd633923b1d 100644
--- a/tools/testing/selftests/rseq/rseq.c
+++ b/tools/testing/selftests/rseq/rseq.c
@@ -73,7 +73,7 @@ static int rseq_ownership;
#define ORIG_RSEQ_ALLOC_SIZE 32
/*
- * Use a union to ensure we allocate a TLS area of 1024 bytes to accomodate an
+ * Use a union to ensure we allocate a TLS area of 1024 bytes to accommodate an
* rseq registration that is larger than the current rseq ABI.
*/
union rseq_tls {
diff --git a/tools/testing/selftests/syscall_user_dispatch/config b/tools/testing/selftests/syscall_user_dispatch/config
index 039e303e59d7..22c4dfe167ca 100644
--- a/tools/testing/selftests/syscall_user_dispatch/config
+++ b/tools/testing/selftests/syscall_user_dispatch/config
@@ -1 +1 @@
-CONFIG_GENERIC_ENTRY=y
+CONFIG_SYSCALL_USER_DISPATCH=y
diff --git a/tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json b/tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json
new file mode 100644
index 000000000000..ef7ca4a6775b
--- /dev/null
+++ b/tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json
@@ -0,0 +1,115 @@
+[
+ {
+ "id": "d710",
+ "name": "Release tail references after first action deletion fails",
+ "category": [
+ "actions",
+ "gact"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$TC actions flush action gact",
+ 0,
+ 1,
+ 255
+ ],
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC actions add action pass index 1",
+ "$TC actions add action pass index 2",
+ "$TC actions add action pass index 3",
+ "$TC filter add dev $DEV1 protocol ip ingress u32 match u32 0 0 action gact index 1"
+ ],
+ "cmdUnderTest": "$TC actions del action gact index 1 action gact index 2 action gact index 3",
+ "expExitCode": "255",
+ "verifyCmd": "$TC actions ls action gact",
+ "matchPattern": "total acts 3\\b.*index 1 ref 2 bind 1\\b.*index 2 ref 1 bind 0\\b.*index 3 ref 1 bind 0\\b",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress",
+ [
+ "$TC actions flush action gact",
+ 0,
+ 1,
+ 255
+ ]
+ ]
+ },
+ {
+ "id": "d711",
+ "name": "Release tail references after middle action deletion fails",
+ "category": [
+ "actions",
+ "gact"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$TC actions flush action gact",
+ 0,
+ 1,
+ 255
+ ],
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC actions add action pass index 1",
+ "$TC actions add action pass index 2",
+ "$TC actions add action pass index 3",
+ "$TC filter add dev $DEV1 protocol ip ingress u32 match u32 0 0 action gact index 2"
+ ],
+ "cmdUnderTest": "$TC actions del action gact index 1 action gact index 2 action gact index 3",
+ "expExitCode": "255",
+ "verifyCmd": "$TC actions ls action gact",
+ "matchPattern": "total acts 2\\b.*index 2 ref 2 bind 1\\b.*index 3 ref 1 bind 0\\b",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress",
+ [
+ "$TC actions flush action gact",
+ 0,
+ 1,
+ 255
+ ]
+ ]
+ },
+ {
+ "id": "d713",
+ "name": "Delete a tail action once after a failed batch",
+ "category": [
+ "actions",
+ "gact"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$TC actions flush action gact",
+ 0,
+ 1,
+ 255
+ ],
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC actions add action pass index 1",
+ "$TC actions add action pass index 2",
+ "$TC filter add dev $DEV1 protocol ip ingress u32 match u32 0 0 action gact index 1"
+ ],
+ "cmdUnderTest": "$TC actions del action gact index 1 action gact index 2",
+ "expExitCode": "255",
+ "verifyCmd": "sh -c '$TC actions del action gact index 2 && $TC actions ls action gact'",
+ "matchPattern": "total acts 1\\b.*index 1 ref 2 bind 1\\b",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress",
+ [
+ "$TC actions flush action gact",
+ 0,
+ 1,
+ 255
+ ]
+ ]
+ }
+]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/actions/ct.json b/tools/testing/selftests/tc-testing/tc-tests/actions/ct.json
index da65f838bd52..8ab48def89b6 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/actions/ct.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/actions/ct.json
@@ -702,5 +702,45 @@
"$TC qdisc del dev $DUMMY clsact",
"$TC qdisc del dev $DUMMY root handle 1:"
]
+ },
+ {
+ "id": "c7a3",
+ "name": "Verify act_ct drops a packet whose header checks fail",
+ "category": [
+ "actions",
+ "ct",
+ "scapy"
+ ],
+ "plugins": {
+ "requires": [
+ "nsPlugin",
+ "scapyPlugin"
+ ]
+ },
+ "setup": [
+ [
+ "$TC qdisc del dev $DEV1 clsact",
+ 0,
+ 1,
+ 2,
+ 255
+ ],
+ "$TC qdisc add dev $DEV1 clsact"
+ ],
+ "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol all prio 1 matchall action ct",
+ "scapy": [
+ {
+ "iface": "$DEV0",
+ "count": 10,
+ "packet": "Ether(type=0x86dd)/IPv6(nh=0, plen=0, src='::1', dst='::2')"
+ }
+ ],
+ "expExitCode": "0",
+ "verifyCmd": "$TC -s qdisc show dev $DEV1 clsact",
+ "matchPattern": "dropped 10",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 clsact"
+ ]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/actions/gact-rollback.json b/tools/testing/selftests/tc-testing/tc-tests/actions/gact-rollback.json
new file mode 100644
index 000000000000..e92a4180db68
--- /dev/null
+++ b/tools/testing/selftests/tc-testing/tc-tests/actions/gact-rollback.json
@@ -0,0 +1,78 @@
+[
+ {
+ "id": "e3b1",
+ "name": "Failed action batch releases a bound action reference",
+ "category": [
+ "actions",
+ "gact"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC actions add action pass index 1",
+ "$TC filter add dev $DEV1 protocol all ingress prio 1 matchall action gact index 1"
+ ],
+ "cmdUnderTest": "$TC actions replace action pass index 1 action goto chain 42 index 3",
+ "expExitCode": "255",
+ "verifyCmd": "$TC actions ls action gact",
+ "matchPattern": "total acts 1.*index 1 ref 2 bind 1",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ },
+ {
+ "id": "e3b2",
+ "name": "Failed action batch releases entries after a bound action",
+ "category": [
+ "actions",
+ "gact"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC actions add action pass index 1",
+ "$TC filter add dev $DEV1 protocol all ingress prio 1 matchall action gact index 1",
+ [
+ "$TC actions replace action pass index 1 action pass index 2 action goto chain 42 index 3",
+ 255
+ ]
+ ],
+ "cmdUnderTest": "$TC actions add action pass index 2",
+ "expExitCode": "0",
+ "verifyCmd": "$TC actions ls action gact",
+ "matchPattern": "total acts 2.*index 1 ref 2 bind 1.*index 2 ref 1 bind 0",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ },
+ {
+ "id": "e3b3",
+ "name": "Failed action batch releases repeated references to a bound action",
+ "category": [
+ "actions",
+ "gact"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC actions add action pass index 1",
+ "$TC filter add dev $DEV1 protocol all ingress prio 1 matchall action gact index 1"
+ ],
+ "cmdUnderTest": "$TC actions replace action pass index 1 action pass index 1 action goto chain 42 index 3",
+ "expExitCode": "255",
+ "verifyCmd": "$TC actions ls action gact",
+ "matchPattern": "total acts 1.*index 1 ref 2 bind 1",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ }
+]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/filters/route.json b/tools/testing/selftests/tc-testing/tc-tests/filters/route.json
index 05cedca67cca..2d5843aebd72 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/filters/route.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/filters/route.json
@@ -202,5 +202,215 @@
"teardown": [
"$TC qdisc del dev $DEV1 parent root drr"
]
+ },
+ {
+ "id": "a7d2",
+ "name": "Delete a route filter that was moved to another bucket",
+ "category": [
+ "filter",
+ "route"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 route from 1 to 1 classid 1:1",
+ "$TC filter change dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10001 route from 1 to 2 classid 1:1",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 200 route from 5 to 5 classid 1:5"
+ ],
+ "cmdUnderTest": "$TC filter del dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10002 route from 1 to 2",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -j filter show dev $DEV1 parent ffff:",
+ "matchJSON": [
+ {
+ "protocol": "ip",
+ "pref": 200,
+ "kind": "route",
+ "chain": 0
+ },
+ {
+ "protocol": "ip",
+ "pref": 200,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x50005",
+ "flowid": "1:5"
+ }
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ },
+ {
+ "id": "c05a",
+ "name": "Try to change a route filter onto an already used handle",
+ "category": [
+ "filter",
+ "route"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 route from 1 to 1 classid 1:1 action ok",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 route from 2 to 2 classid 1:2 action drop"
+ ],
+ "cmdUnderTest": "$TC filter change dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10001 route from 2 to 2 classid 1:1 action ok",
+ "expExitCode": "2",
+ "verifyCmd": "$TC -j filter show dev $DEV1 parent ffff:",
+ "matchJSON": [
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0
+ },
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x10001",
+ "flowid": "1:1",
+ "actions": [
+ {
+ "order": 1,
+ "kind": "gact",
+ "control_action": {
+ "type": "pass"
+ }
+ }
+ ]
+ }
+ },
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x20002",
+ "flowid": "1:2",
+ "actions": [
+ {
+ "order": 1,
+ "kind": "gact",
+ "control_action": {
+ "type": "drop"
+ }
+ }
+ ]
+ }
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ },
+ {
+ "id": "3f21",
+ "name": "Replace a route filter that shares its key with another filter",
+ "category": [
+ "filter",
+ "route"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10101 route from 1 to 1 classid 1:1",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10201 route from 1 to 1 classid 1:2"
+ ],
+ "cmdUnderTest": "$TC filter replace dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10101 route from 1 to 1 classid 1:9",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -j filter show dev $DEV1 parent ffff:",
+ "matchJSON": [
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0
+ },
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x10101",
+ "flowid": "1:9"
+ }
+ },
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x10201",
+ "flowid": "1:2"
+ }
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ },
+ {
+ "id": "9d0e",
+ "name": "Replace both route filters sharing a key",
+ "category": [
+ "filter",
+ "route"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10101 route from 1 to 1 classid 1:1",
+ "$TC filter add dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10201 route from 1 to 1 classid 1:2",
+ "$TC filter replace dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10101 route from 1 to 1 classid 1:9"
+ ],
+ "cmdUnderTest": "$TC filter replace dev $DEV1 parent ffff: protocol ip prio 100 handle 0x10201 route from 1 to 1 classid 1:8",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -j filter show dev $DEV1 parent ffff:",
+ "matchJSON": [
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0
+ },
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x10101",
+ "flowid": "1:9"
+ }
+ },
+ {
+ "protocol": "ip",
+ "pref": 100,
+ "kind": "route",
+ "chain": 0,
+ "options": {
+ "fh": "0x10201",
+ "flowid": "1:8"
+ }
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json b/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json
index b2ca9d4e991b..edc5148a8d97 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json
@@ -353,5 +353,76 @@
"teardown": [
"$TC qdisc del dev $DEV1 parent root drr"
]
+ },
+ {
+ "id": "70fd",
+ "name": "Add u32 filter when node ID pool is exhausted (4096th filter rejected)",
+ "category": [
+ "filter",
+ "u32"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DUMMY clsact",
+ "bash -c 'for i in {1..4095}; do echo filter add dev $DUMMY ingress prio 1 protocol ip u32 match u8 0 0 at 0; done | $TC -b -'"
+ ],
+ "cmdUnderTest": "$TC filter add dev $DUMMY ingress prio 1 protocol ip u32 match u8 0 0 at 0",
+ "expExitCode": "2",
+ "verifyCmd": "$TC -d filter show dev $DUMMY ingress",
+ "matchPattern": "fh 800::",
+ "matchCount": "4095",
+ "teardown": [
+ "$TC qdisc del dev $DUMMY clsact"
+ ]
+ },
+ {
+ "id": "35fc",
+ "name": "u32 manual table then auto table: auto allocation must not alias a live manual handle",
+ "category": [
+ "filter",
+ "u32"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC filter add dev $DEV1 ingress protocol ip pref 1 handle 801: u32 divisor 16"
+ ],
+ "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol ip pref 2 u32 divisor 16",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -d filter show dev $DEV1 ingress",
+ "matchPattern": "fh 801:",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
+ },
+ {
+ "id": "a6e8",
+ "name": "u32 manual table add/del does not leak its idr entry (re-adding the same handle succeeds)",
+ "category": [
+ "filter",
+ "u32"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DEV1 ingress",
+ "$TC filter add dev $DEV1 ingress protocol ip pref 1 u32 divisor 16",
+ "$TC filter add dev $DEV1 ingress protocol ip pref 5 handle 901: u32 divisor 1",
+ "$TC filter del dev $DEV1 ingress protocol ip pref 5 handle 901: u32"
+ ],
+ "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol ip pref 6 handle 901: u32 divisor 1",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -d filter show dev $DEV1 ingress",
+ "matchPattern": "fh 901: ht divisor 1",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DEV1 ingress"
+ ]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/infra/qdiscs.json b/tools/testing/selftests/tc-testing/tc-tests/infra/qdiscs.json
index a1f97a4b606e..0cf12c50fb74 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/infra/qdiscs.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/infra/qdiscs.json
@@ -1540,5 +1540,169 @@
"$TC qdisc del dev $DUMMY root",
"$IP addr del 10.10.10.10/24 dev $DUMMY || true"
]
+ },
+ {
+ "id": "fb6c",
+ "name": "Force multiq to dequeue from its child's gso_skb with qfq leaf",
+ "category": [
+ "qdisc",
+ "tbf",
+ "multiq",
+ "qfq"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "echo \"1 1 4\" > /sys/bus/netdevsim/new_device",
+ "$IP link set dev $ETH up || true",
+ "$IP l set addr 01:02:03:04:05:06 dev $ETH || true",
+ "$IP n add dev $ETH 10.10.11.1 lladdr 01:02:03:04:05:06 dev $ETH || true",
+ "$IP addr add 10.10.11.10/24 dev $ETH || true",
+ "$TC qdisc add dev $ETH root handle 1: tbf rate 88bit burst 1661b peakrate 2257333 minburst 1024 limit 7b",
+ "$TC qdisc add dev $ETH parent 1: handle 2: multiq",
+ "$TC qdisc add dev $ETH parent 2:1 handle 3: qfq",
+ "$TC class add dev $ETH classid 3:1 parent 3: qfq maxpkt 512 weight 1",
+ "$TC filter add dev $ETH parent 2: protocol all prio 1 matchall action skbedit queue_mapping 0",
+ "$TC filter add dev $ETH parent 3: protocol all prio 1 matchall classid 3:1 action ok"
+ ],
+ "cmdUnderTest": "ping -c 1 10.10.11.1 -W0.01 -I$ETH || true",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -s -j qdisc ls dev $ETH parent 1:",
+ "matchJSON": [
+ {
+ "kind": "multiq",
+ "handle": "2:",
+ "bytes": 98,
+ "packets": 1,
+ "backlog": 0,
+ "qlen": 0
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $ETH handle 1: root",
+ "echo \"1\" > /sys/bus/netdevsim/del_device"
+ ]
+ },
+ {
+ "id": "1922",
+ "name": "Force multiq to dequeue from its child's gso_skb with dualpi2 leaf",
+ "category": [
+ "qdisc",
+ "tbf",
+ "multiq",
+ "dualpi2"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "echo \"1 1 4\" > /sys/bus/netdevsim/new_device",
+ "$IP link set dev $ETH up || true",
+ "$IP l set addr 01:02:03:04:05:06 dev $ETH || true",
+ "$IP n add dev $ETH 10.10.11.1 lladdr 01:02:03:04:05:06 dev $ETH || true",
+ "$IP addr add 10.10.11.10/24 dev $ETH || true",
+ "$TC qdisc add dev $ETH root handle 1: tbf rate 88bit burst 1661b peakrate 2257333 minburst 1024 limit 7b",
+ "$TC qdisc add dev $ETH parent 1: handle 2: multiq",
+ "$TC qdisc add dev $ETH parent 2:1 handle 3: dualpi2",
+ "$TC filter add dev $ETH parent 2: protocol ip prio 1 u32 match ip dst 10.10.11.1 action skbedit queue_mapping 0",
+ "$TC filter add dev $ETH parent 3: protocol ip prio 1 u32 match ip dst 10.10.11.1 classid 3:1 action ok"
+ ],
+ "cmdUnderTest": "ping -c 1 10.10.11.1 -W0.01 -I$ETH || true",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -j -s qdisc ls dev $ETH handle 3:",
+ "matchJSON": [
+ {
+ "kind": "dualpi2",
+ "handle": "3:",
+ "bytes": 98,
+ "packets": 1,
+ "backlog": 0,
+ "qlen": 0
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $ETH handle 1: root",
+ "echo \"1\" > /sys/bus/netdevsim/del_device"
+ ]
+ },
+ {
+ "id": "476f",
+ "name": "Force taprio to dequeue from its child's gso_skb with qfq leaf",
+ "category": [
+ "qdisc",
+ "tbf",
+ "multiq",
+ "qfq"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "echo \"1 1 4\" > /sys/bus/netdevsim/new_device",
+ "$IP link set dev $ETH up || true",
+ "$IP l set addr 01:02:03:04:05:06 dev $ETH || true",
+ "$IP n add dev $ETH 10.10.11.1 lladdr 01:02:03:04:05:06 dev $ETH || true",
+ "$TC qdisc add dev $ETH root handle 1: taprio num_tc 2 map 0 1 0 0 0 0 0 0 0 0 0 0 0 0 0 0 queues 1@0 1@1 base-time 9000000000000000000 sched-entry S 03 200000 flags 0x0 clockid CLOCK_TAI",
+ "$TC qdisc add dev $ETH parent 1:1 handle 3: qfq",
+ "$TC class add dev $ETH classid 3:1 parent 3: qfq maxpkt 512 weight 1",
+ "$TC filter add dev $ETH parent 3: protocol all prio 1 matchall classid 3:1 action ok"
+ ],
+ "cmdUnderTest": "ping -c 1 10.10.11.1 -W0.01 -I$ETH || true",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -s -j qdisc ls dev $ETH",
+ "matchJSON": [
+ {
+ "kind": "taprio",
+ "handle": "1:",
+ "bytes": 98,
+ "packets": 1,
+ "backlog": 0,
+ "qlen": 0
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $ETH handle 1: root",
+ "echo \"1\" > /sys/bus/netdevsim/del_device"
+ ]
+ },
+ {
+ "id": "0235",
+ "name": "Force taprio to dequeue from its child's gso_skb with dualpi2 leaf",
+ "category": [
+ "qdisc",
+ "tbf",
+ "taprio",
+ "dualpi2"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "echo \"1 1 4\" > /sys/bus/netdevsim/new_device",
+ "$IP link set dev $ETH up || true",
+ "$IP l set addr 01:02:03:04:05:06 dev $ETH || true",
+ "$IP n add dev $ETH 10.10.11.1 lladdr 01:02:03:04:05:06 dev $ETH || true",
+ "$TC qdisc add dev $ETH root handle 1: taprio num_tc 2 map 0 1 0 0 0 0 0 0 0 0 0 0 0 0 0 0 queues 1@0 1@1 base-time 9000000000000000000 sched-entry S 03 200000 flags 0x0 clockid CLOCK_TAI",
+ "$TC qdisc replace dev $ETH parent 1:1 handle 3: dualpi2",
+ "$TC filter add dev $ETH parent 3: protocol ip prio 1 u32 match ip dst 10.10.11.1 classid 3:1 action ok"
+ ],
+ "cmdUnderTest": "ping -c 1 10.10.11.1 -W0.01 -I$ETH || true",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -j -s qdisc ls dev $ETH handle 3:",
+ "matchJSON": [
+ {
+ "kind": "dualpi2",
+ "handle": "3:",
+ "bytes": 98,
+ "packets": 1,
+ "backlog": 0,
+ "qlen": 0
+ }
+ ],
+ "teardown": [
+ "$TC qdisc del dev $ETH handle 1: root",
+ "echo \"1\" > /sys/bus/netdevsim/del_device"
+ ]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json
index 6d515d0e5ed6..a894e6f0e267 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json
@@ -213,5 +213,77 @@
"matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1p target 5ms interval 100ms",
"matchCount": "1",
"teardown": ["$TC qdisc del dev $DEV1 handle 1: root"]
+ },
+ {
+ "id": "6e44",
+ "name": "Create CODEL with 1us interval, accepted (sub-tick, uAPI locked)",
+ "category": [
+ "qdisc",
+ "codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel interval 1us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 5ms interval 0us",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
+ },
+ {
+ "id": "a8c3",
+ "name": "Create CODEL with 3us interval, accepted (two ticks)",
+ "category": [
+ "qdisc",
+ "codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel interval 3us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 5ms interval 2us",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
+ },
+ {
+ "id": "a695",
+ "name": "Create CODEL with 1024us interval boundary accepted",
+ "category": [
+ "qdisc",
+ "codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel interval 1024us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 5ms interval 1.02ms",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
+ },
+ {
+ "id": "9793",
+ "name": "Create CODEL with 1us target, accepted (target not in control law)",
+ "category": [
+ "qdisc",
+ "codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel target 1us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 0us interval 100ms",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/ets.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/ets.json
index ee09e6d6fdf3..d2eab61c099a 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/ets.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/ets.json
@@ -987,7 +987,7 @@
},
{
"id": "41f5",
- "name": "ETS offload where the sum of quanta wraps u32",
+ "name": "ETS offload with out-of-range quanta clamped",
"category": [
"qdisc",
"ets"
@@ -1002,7 +1002,7 @@
"cmdUnderTest": "$TC qdisc add dev $ETH root ets quanta 4294967294 1 1",
"expExitCode": "0",
"verifyCmd": "$TC qdisc show dev $ETH",
- "matchPattern": "qdisc ets .*bands 3 quanta 4294967294 1 1",
+ "matchPattern": "qdisc ets .*bands 3 quanta 1048576 256 256",
"matchCount": "1",
"teardown": [
"echo \"1\" > /sys/bus/netdevsim/del_device"
diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json
index 4ce62b857fd7..de6a1b8d954a 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json
@@ -316,5 +316,77 @@
"matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 1p flows 1024 quantum.*target 5ms interval 100ms memory_limit 32Mb ecn drop_batch 64",
"matchCount": "1",
"teardown": ["$TC qdisc del dev $DEV1 handle 1: root"]
+ },
+ {
+ "id": "1b4d",
+ "name": "Create FQ_CODEL with 1us interval, accepted (sub-tick, uAPI locked)",
+ "category": [
+ "qdisc",
+ "fq_codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel interval 1us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 5ms interval 0us memory_limit 32Mb ecn drop_batch 64",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
+ },
+ {
+ "id": "3540",
+ "name": "Create FQ_CODEL with 3us interval, accepted (two ticks)",
+ "category": [
+ "qdisc",
+ "fq_codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel interval 3us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 5ms interval 2us memory_limit 32Mb ecn drop_batch 64",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
+ },
+ {
+ "id": "49c5",
+ "name": "Create FQ_CODEL with 1024us interval boundary accepted",
+ "category": [
+ "qdisc",
+ "fq_codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel interval 1024us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 5ms interval 1.02ms memory_limit 32Mb ecn drop_batch 64",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
+ },
+ {
+ "id": "3e0f",
+ "name": "Create FQ_CODEL with 1us target, accepted (target not in control law)",
+ "category": [
+ "qdisc",
+ "fq_codel"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel target 1us",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 0us interval 100ms memory_limit 32Mb ecn drop_batch 64",
+ "matchCount": "1",
+ "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json
index c98c339424d4..4f6bbb8b57f9 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json
@@ -169,5 +169,39 @@
"teardown": [
"$TC qdisc del dev $DUMMY handle 1: root"
]
+ },
+ {
+ "id": "8c39",
+ "name": "HFSC classify walk still reaches leaf after lateral drift",
+ "category": [
+ "qdisc",
+ "hfsc"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "ip link set lo up",
+ "$TC qdisc add dev lo handle 1: root hfsc default 30",
+ "$TC class add dev lo parent 1: classid 1:1 hfsc rt m2 100kbit",
+ "$TC class add dev lo parent 1:1 classid 1:10 hfsc rt m2 50kbit",
+ "$TC class add dev lo parent 1: classid 1:2 hfsc rt m2 100kbit",
+ "$TC filter add dev lo parent 1: protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:1",
+ "$TC filter add dev lo parent 1:1 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:2",
+ "$TC class add dev lo parent 1:2 classid 1:20 hfsc rt m2 10kbit",
+ "$TC class add dev lo parent 1: classid 1:3 hfsc rt m2 100kbit",
+ "$TC filter add dev lo parent 1:2 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:3",
+ "$TC class add dev lo parent 1:3 classid 1:30 hfsc rt m2 10kbit",
+ "$TC class add dev lo parent 1:3 classid 1:31 hfsc rt m2 100kbit",
+ "$TC filter add dev lo parent 1:3 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:31"
+ ],
+ "cmdUnderTest": "ping -n -c 10 -W 1 127.0.0.1",
+ "expExitCode": "0",
+ "verifyCmd": "$TC -s class show dev lo",
+ "matchPattern": "class hfsc 1:31 parent 1:3 rt[^\\n]*\\n Sent [0-9]+ bytes [1-9][0-9]* pkt",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev lo handle 1: root"
+ ]
}
]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json
new file mode 100644
index 000000000000..44538b9266b6
--- /dev/null
+++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json
@@ -0,0 +1,128 @@
+[
+ {
+ "id": "e3cc",
+ "name": "HHF hh_limit rejects value above 2*HH_FLOWS_CNT cap (4294967295)",
+ "category": [
+ "qdisc",
+ "hhf"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DUMMY handle 1: root hhf"
+ ],
+ "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 4294967295",
+ "expExitCode": "2",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DUMMY handle 1: root"
+ ]
+ },
+ {
+ "id": "f681",
+ "name": "HHF hh_limit rejects 65536 (above 2*HH_FLOWS_CNT cap)",
+ "category": [
+ "qdisc",
+ "hhf"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DUMMY handle 1: root hhf"
+ ],
+ "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 65536",
+ "expExitCode": "2",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DUMMY handle 1: root"
+ ]
+ },
+ {
+ "id": "223d",
+ "name": "HHF hh_limit accepts boundary value 2048 (2*HH_FLOWS_CNT)",
+ "category": [
+ "qdisc",
+ "hhf"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DUMMY handle 1: root hhf hh_limit 100"
+ ],
+ "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 2048",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DUMMY handle 1: root"
+ ]
+ },
+ {
+ "id": "147f",
+ "name": "HHF hh_limit rejects first value above cap (2049)",
+ "category": [
+ "qdisc",
+ "hhf"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ "$TC qdisc add dev $DUMMY handle 1: root hhf"
+ ],
+ "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 2049",
+ "expExitCode": "2",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DUMMY handle 1: root"
+ ]
+ },
+ {
+ "id": "4d4f",
+ "name": "HHF add-time hh_limit 500 is preserved (init does not clobber user value)",
+ "category": [
+ "qdisc",
+ "hhf"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root hhf hh_limit 500",
+ "expExitCode": "0",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 500",
+ "matchCount": "1",
+ "teardown": [
+ "$TC qdisc del dev $DUMMY handle 1: root"
+ ]
+ },
+ {
+ "id": "ca99",
+ "name": "HHF add-time hh_limit 4294967295 is rejected (no qdisc installed)",
+ "category": [
+ "qdisc",
+ "hhf"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root hhf hh_limit 4294967295",
+ "expExitCode": "2",
+ "verifyCmd": "$TC qdisc show dev $DUMMY",
+ "matchPattern": "qdisc hhf 1: root",
+ "matchCount": "0",
+ "teardown": []
+ }
+]
diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/pfifo_fast.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/pfifo_fast.json
index 30da27fe8806..a6e25e76ecb1 100644
--- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/pfifo_fast.json
+++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/pfifo_fast.json
@@ -105,5 +105,209 @@
"teardown": [
"$TC qdisc del dev $DUMMY handle 1: root"
]
+ },
+ {
+ "id": "dbe3",
+ "name": "Set tx_queue_len to S16_MAX boundary (32767 accepted)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$IP link set dev $DUMMY txqueuelen 32767",
+ "expExitCode": "0",
+ "verifyCmd": "$IP link show dev $DUMMY",
+ "matchPattern": "qlen 32767$",
+ "matchCount": "1",
+ "teardown": []
+ },
+ {
+ "id": "b50e",
+ "name": "Reject tx_queue_len above S16_MAX at set time (32768)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "$IP link set dev $DUMMY txqueuelen 32768",
+ "expExitCode": "2",
+ "verifyCmd": "$IP link show dev $DUMMY",
+ "matchPattern": "qlen 1000$",
+ "matchCount": "1",
+ "teardown": []
+ },
+ {
+ "id": "40f8",
+ "name": "Reject tx_queue_len above S16_MAX via sysfs (32768)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "sh -c 'echo 32768 > /sys/class/net/$DUMMY/tx_queue_len'",
+ "expExitCode": "1",
+ "verifyCmd": "$IP link show dev $DUMMY",
+ "matchPattern": "qlen 1000$",
+ "matchCount": "1",
+ "teardown": []
+ },
+ {
+ "id": "4b6e",
+ "name": "Set tx_queue_len to S16_MAX via sysfs (32767 accepted)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [],
+ "cmdUnderTest": "sh -c 'echo 32767 > /sys/class/net/$DUMMY/tx_queue_len'",
+ "expExitCode": "0",
+ "verifyCmd": "$IP link show dev $DUMMY",
+ "matchPattern": "qlen 32767$",
+ "matchCount": "1",
+ "teardown": []
+ },
+ {
+ "id": "b90d",
+ "name": "Create device with tx_queue_len at S16_MAX boundary (32767 accepted)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$IP link del dev $DUMMY",
+ 0,
+ 1
+ ]
+ ],
+ "cmdUnderTest": "$IP link add dev $DUMMY txqueuelen 32767 type dummy",
+ "expExitCode": "0",
+ "verifyCmd": "$IP link show dev $DUMMY",
+ "matchPattern": "qlen 32767$",
+ "matchCount": "1",
+ "teardown": [
+ [
+ "$IP link del dev $DUMMY",
+ 0,
+ 1
+ ]
+ ]
+ },
+ {
+ "id": "57ab",
+ "name": "Reject creating device with tx_queue_len above S16_MAX (32768)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$IP link del dev $DUMMY",
+ 0,
+ 1
+ ]
+ ],
+ "cmdUnderTest": "$IP link add dev $DUMMY txqueuelen 32768 type dummy",
+ "expExitCode": "2",
+ "verifyCmd": "$IP -o link show",
+ "matchPattern": "^[0-9]+: $DUMMY",
+ "matchCount": "0",
+ "teardown": []
+ },
+ {
+ "id": "e777",
+ "name": "Reject creating device with oversized tx_queue_len (500000)",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$IP link del dev $DUMMY",
+ 0,
+ 1
+ ]
+ ],
+ "cmdUnderTest": "$IP link add dev $DUMMY txqueuelen 500000 type dummy",
+ "expExitCode": "2",
+ "verifyCmd": "$IP -o link show",
+ "matchPattern": "^[0-9]+: $DUMMY",
+ "matchCount": "0",
+ "teardown": []
+ },
+ {
+ "id": "31ac",
+ "name": "Reject veth peer nest tx_queue_len above S16_MAX at create",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$IP link del dev $DEV1",
+ 0,
+ 1
+ ]
+ ],
+ "cmdUnderTest": "$IP link add dev $DEV1 type veth peer name $DEV0 txqueuelen 500000",
+ "expExitCode": "2",
+ "verifyCmd": "$IP -o link show",
+ "matchPattern": "^[0-9]+: $DEV1",
+ "matchCount": "0",
+ "teardown": []
+ },
+ {
+ "id": "b567",
+ "name": "Accept veth peer nest tx_queue_len within S16_MAX",
+ "category": [
+ "qdisc",
+ "pfifo_fast"
+ ],
+ "plugins": {
+ "requires": "nsPlugin"
+ },
+ "setup": [
+ [
+ "$IP link del dev $DEV1",
+ 0,
+ 1
+ ]
+ ],
+ "cmdUnderTest": "$IP link add dev $DEV1 txqueuelen 100 type veth peer name $DEV0 txqueuelen 200",
+ "expExitCode": "0",
+ "verifyCmd": "$IP link show",
+ "matchPattern": "qlen (100|200)$",
+ "matchCount": "2",
+ "teardown": [
+ [
+ "$IP link del dev $DEV0",
+ 0,
+ 1
+ ]
+ ]
}
]
diff --git a/tools/testing/selftests/timers/Makefile b/tools/testing/selftests/timers/Makefile
index 32203593c62e..0e73a16874c4 100644
--- a/tools/testing/selftests/timers/Makefile
+++ b/tools/testing/selftests/timers/Makefile
@@ -1,5 +1,5 @@
# SPDX-License-Identifier: GPL-2.0
-CFLAGS += -O3 -Wl,-no-as-needed -Wall -I $(top_srcdir)
+CFLAGS += -O3 -Wl,-no-as-needed -Wall
LDLIBS += -lrt -lpthread -lm
# these are all "safe" tests that don't modify
diff --git a/tools/testing/selftests/timers/adjtick.c b/tools/testing/selftests/timers/adjtick.c
index 5b3ef708d6e9..68009a6d6de8 100644
--- a/tools/testing/selftests/timers/adjtick.c
+++ b/tools/testing/selftests/timers/adjtick.c
@@ -22,8 +22,8 @@
#include <sys/time.h>
#include <sys/timex.h>
#include <time.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define MILLION 1000000
diff --git a/tools/testing/selftests/timers/alarmtimer-suspend.c b/tools/testing/selftests/timers/alarmtimer-suspend.c
index aa66c805f6a4..120b3ce8b39e 100644
--- a/tools/testing/selftests/timers/alarmtimer-suspend.c
+++ b/tools/testing/selftests/timers/alarmtimer-suspend.c
@@ -28,8 +28,8 @@
#include <signal.h>
#include <stdlib.h>
#include <pthread.h>
-#include <include/vdso/time64.h>
#include <errno.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define UNREASONABLE_LAT (NSEC_PER_SEC * 5) /* hopefully we resume in 5 secs */
@@ -39,37 +39,6 @@ int alarmcount;
int alarm_clock_id;
struct timespec start_time;
-
-char *clockstring(int clockid)
-{
- switch (clockid) {
- case CLOCK_REALTIME:
- return "CLOCK_REALTIME";
- case CLOCK_MONOTONIC:
- return "CLOCK_MONOTONIC";
- case CLOCK_PROCESS_CPUTIME_ID:
- return "CLOCK_PROCESS_CPUTIME_ID";
- case CLOCK_THREAD_CPUTIME_ID:
- return "CLOCK_THREAD_CPUTIME_ID";
- case CLOCK_MONOTONIC_RAW:
- return "CLOCK_MONOTONIC_RAW";
- case CLOCK_REALTIME_COARSE:
- return "CLOCK_REALTIME_COARSE";
- case CLOCK_MONOTONIC_COARSE:
- return "CLOCK_MONOTONIC_COARSE";
- case CLOCK_BOOTTIME:
- return "CLOCK_BOOTTIME";
- case CLOCK_REALTIME_ALARM:
- return "CLOCK_REALTIME_ALARM";
- case CLOCK_BOOTTIME_ALARM:
- return "CLOCK_BOOTTIME_ALARM";
- case CLOCK_TAI:
- return "CLOCK_TAI";
- }
- return "UNKNOWN_CLOCKID";
-}
-
-
long long timespec_sub(struct timespec a, struct timespec b)
{
long long ret = NSEC_PER_SEC * b.tv_sec + b.tv_nsec;
@@ -129,12 +98,12 @@ int main(void)
alarmcount = 0;
if (timer_create(alarm_clock_id, &se, &tm1) == -1) {
printf("timer_create failed, %s unsupported?: %s\n",
- clockstring(alarm_clock_id), strerror(errno));
+ clock_name(alarm_clock_id), strerror(errno));
break;
}
clock_gettime(alarm_clock_id, &start_time);
- printf("Start time (%s): %ld:%ld\n", clockstring(alarm_clock_id),
+ printf("Start time (%s): %ld:%ld\n", clock_name(alarm_clock_id),
start_time.tv_sec, start_time.tv_nsec);
printf("Setting alarm for every %i seconds\n", SUSPEND_SECS);
its1.it_value = start_time;
diff --git a/tools/testing/selftests/timers/inconsistency-check.c b/tools/testing/selftests/timers/inconsistency-check.c
index e53e63e18683..d7982ac4bd18 100644
--- a/tools/testing/selftests/timers/inconsistency-check.c
+++ b/tools/testing/selftests/timers/inconsistency-check.c
@@ -28,7 +28,7 @@
#include <sys/timex.h>
#include <string.h>
#include <signal.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
/* CLOCK_HWSPECIFIC == CLOCK_SGI_CYCLE (Deprecated) */
@@ -36,35 +36,6 @@
#define CALLS_PER_LOOP 64
-char *clockstring(int clockid)
-{
- switch (clockid) {
- case CLOCK_REALTIME:
- return "CLOCK_REALTIME";
- case CLOCK_MONOTONIC:
- return "CLOCK_MONOTONIC";
- case CLOCK_PROCESS_CPUTIME_ID:
- return "CLOCK_PROCESS_CPUTIME_ID";
- case CLOCK_THREAD_CPUTIME_ID:
- return "CLOCK_THREAD_CPUTIME_ID";
- case CLOCK_MONOTONIC_RAW:
- return "CLOCK_MONOTONIC_RAW";
- case CLOCK_REALTIME_COARSE:
- return "CLOCK_REALTIME_COARSE";
- case CLOCK_MONOTONIC_COARSE:
- return "CLOCK_MONOTONIC_COARSE";
- case CLOCK_BOOTTIME:
- return "CLOCK_BOOTTIME";
- case CLOCK_REALTIME_ALARM:
- return "CLOCK_REALTIME_ALARM";
- case CLOCK_BOOTTIME_ALARM:
- return "CLOCK_BOOTTIME_ALARM";
- case CLOCK_TAI:
- return "CLOCK_TAI";
- }
- return "UNKNOWN_CLOCKID";
-}
-
/* returns 1 if a <= b, 0 otherwise */
static inline int in_order(struct timespec a, struct timespec b)
{
@@ -171,15 +142,15 @@ int main(int argc, char *argv[])
for (clockid = userclock; clockid < maxclocks; clockid++) {
if (clockid == CLOCK_HWSPECIFIC || clock_gettime(clockid, &ts)) {
- ksft_test_result_skip("%-31s\n", clockstring(clockid));
+ ksft_test_result_skip("%-31s\n", clock_name(clockid));
continue;
}
if (consistency_test(clockid, runtime)) {
- ksft_test_result_fail("%-31s\n", clockstring(clockid));
+ ksft_test_result_fail("%-31s\n", clock_name(clockid));
ksft_exit_fail();
} else {
- ksft_test_result_pass("%-31s\n", clockstring(clockid));
+ ksft_test_result_pass("%-31s\n", clock_name(clockid));
}
}
ksft_exit_pass();
diff --git a/tools/testing/selftests/timers/leap-a-day.c b/tools/testing/selftests/timers/leap-a-day.c
index 3568cfb3e815..b93cb5714b37 100644
--- a/tools/testing/selftests/timers/leap-a-day.c
+++ b/tools/testing/selftests/timers/leap-a-day.c
@@ -9,16 +9,19 @@
* kernel's leap-second behavior, as well as how well applications
* handle the leap-second discontinuity.
*
- * Usage: leap-a-day [-s] [-i <num>]
+ * Usage: leap-a-day [-w] [-i <num>] [-t]
*
* Options:
- * -s: Each iteration, set the date to 10 seconds before midnight GMT.
- * This speeds up the number of leapsecond transitions tested,
- * but because it calls settimeofday frequently, advancing the
- * time by 24 hours every ~16 seconds, it may cause application
- * disruption.
+ * -w: Only set the leap-second flag and wait for the leap second
+ * each iteration, instead of advancing the time. By default the
+ * date is set to 10 seconds before midnight GMT, which speeds up
+ * the number of leapsecond transitions tested, but because it
+ * calls settimeofday frequently, advancing the time by 24 hours
+ * every ~16 seconds, it may cause application disruption.
*
- * -i: Number of iterations to run (default: infinite)
+ * -i: Number of iterations to run (-1 = infinite, default: 10)
+ *
+ * -t: Print TAI time.
*
* Other notes: Disabling NTP prior to running this is advised, as the two
* may conflict in their commands to the kernel.
@@ -48,7 +51,7 @@
#include <string.h>
#include <signal.h>
#include <unistd.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define CLOCK_TAI 11
@@ -186,7 +189,7 @@ int main(int argc, char **argv)
int opt;
/* Process arguments */
- while ((opt = getopt(argc, argv, "sti:")) != -1) {
+ while ((opt = getopt(argc, argv, "wti:")) != -1) {
switch (opt) {
case 'w':
printf("Only setting leap-flag, not changing time. It could take up to a day for leap to trigger.\n");
diff --git a/tools/testing/selftests/timers/mqueue-lat.c b/tools/testing/selftests/timers/mqueue-lat.c
index c0d9368e4fca..fa4c3e3f58fe 100644
--- a/tools/testing/selftests/timers/mqueue-lat.c
+++ b/tools/testing/selftests/timers/mqueue-lat.c
@@ -29,7 +29,7 @@
#include <signal.h>
#include <errno.h>
#include <mqueue.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
diff --git a/tools/testing/selftests/timers/nanosleep.c b/tools/testing/selftests/timers/nanosleep.c
index a054680b3372..b45e4c855259 100644
--- a/tools/testing/selftests/timers/nanosleep.c
+++ b/tools/testing/selftests/timers/nanosleep.c
@@ -27,43 +27,9 @@
#include <sys/timex.h>
#include <string.h>
#include <signal.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
-/* CLOCK_HWSPECIFIC == CLOCK_SGI_CYCLE (Deprecated) */
-#define CLOCK_HWSPECIFIC 10
-
-#define UNSUPPORTED 0xf00f
-
-char *clockstring(int clockid)
-{
- switch (clockid) {
- case CLOCK_REALTIME:
- return "CLOCK_REALTIME";
- case CLOCK_MONOTONIC:
- return "CLOCK_MONOTONIC";
- case CLOCK_PROCESS_CPUTIME_ID:
- return "CLOCK_PROCESS_CPUTIME_ID";
- case CLOCK_THREAD_CPUTIME_ID:
- return "CLOCK_THREAD_CPUTIME_ID";
- case CLOCK_MONOTONIC_RAW:
- return "CLOCK_MONOTONIC_RAW";
- case CLOCK_REALTIME_COARSE:
- return "CLOCK_REALTIME_COARSE";
- case CLOCK_MONOTONIC_COARSE:
- return "CLOCK_MONOTONIC_COARSE";
- case CLOCK_BOOTTIME:
- return "CLOCK_BOOTTIME";
- case CLOCK_REALTIME_ALARM:
- return "CLOCK_REALTIME_ALARM";
- case CLOCK_BOOTTIME_ALARM:
- return "CLOCK_BOOTTIME_ALARM";
- case CLOCK_TAI:
- return "CLOCK_TAI";
- };
- return "UNKNOWN_CLOCKID";
-}
-
/* returns 1 if a <= b, 0 otherwise */
static inline int in_order(struct timespec a, struct timespec b)
{
@@ -92,15 +58,15 @@ int nanosleep_test(int clockid, long long ns)
/* First check abs time */
if (clock_gettime(clockid, &now))
- return UNSUPPORTED;
+ return KSFT_SKIP;
target = timespec_add(now, ns);
if (clock_nanosleep(clockid, TIMER_ABSTIME, &target, NULL))
- return UNSUPPORTED;
+ return KSFT_SKIP;
clock_gettime(clockid, &now);
if (!in_order(target, now))
- return -1;
+ return KSFT_FAIL;
/* Second check reltime */
clock_gettime(clockid, &now);
@@ -112,8 +78,8 @@ int nanosleep_test(int clockid, long long ns)
clock_gettime(clockid, &now);
if (!in_order(target, now))
- return -1;
- return 0;
+ return KSFT_FAIL;
+ return KSFT_PASS;
}
static void dummy_event_handler(int val)
@@ -132,82 +98,86 @@ static int nanosleep_test_remaining(int clockid)
sa.sa_handler = dummy_event_handler;
ret = sigaction(SIGALRM, &sa, NULL);
if (ret)
- return -1;
+ return KSFT_FAIL;
ret = timer_create(clockid, NULL, &timer);
if (ret)
- return -1;
+ return KSFT_FAIL;
itimer.it_value.tv_nsec = NSEC_PER_SEC / 4;
ret = timer_settime(timer, 0, &itimer, NULL);
if (ret)
- return -1;
+ return KSFT_FAIL;
rqtp.tv_nsec = NSEC_PER_SEC / 2;
ret = clock_nanosleep(clockid, 0, &rqtp, &rmtp);
- if (ret != EINTR)
- return -1;
- ret = timer_delete(timer);
- if (ret)
- return -1;
+ if (timer_delete(timer)) {
+ ksft_exit_fail_msg("Unable to delete the timeout timer for %s. "
+ "This might interfere with following testcases.\n",
+ clock_name(clockid));
+ }
+
+ if (ret != EINTR)
+ return KSFT_FAIL;
sa.sa_handler = SIG_DFL;
ret = sigaction(SIGALRM, &sa, NULL);
if (ret)
- return -1;
+ return KSFT_FAIL;
if (!in_order((struct timespec) {}, rmtp))
- return -1;
+ return KSFT_FAIL;
if (!in_order(rmtp, rqtp))
- return -1;
+ return KSFT_FAIL;
- return 0;
+ return KSFT_PASS;
+}
+
+static void nanosleep_test_clock(clockid_t clockid)
+{
+ long long length = 10;
+ int ret;
+
+ while (length <= (NSEC_PER_SEC * 10)) {
+ ret = nanosleep_test(clockid, length);
+ if (ret != KSFT_PASS) {
+ ksft_test_result_report(ret, "%s\n", clock_name(clockid));
+ ksft_test_result_skip("%s (remaining)\n", clock_name(clockid));
+ return;
+ }
+
+ length *= 100;
+ }
+ ksft_test_result_pass("%s\n", clock_name(clockid));
+
+ ret = nanosleep_test_remaining(clockid);
+ ksft_test_result_report(ret, "%s (remaining)\n", clock_name(clockid));
}
int main(int argc, char **argv)
{
- long long length;
- int clockid, ret;
- int max_clocks = CLOCK_TAI + 1;
+ int clockid;
+
+ static const clockid_t tested_clocks[] = {
+ CLOCK_REALTIME,
+ CLOCK_MONOTONIC,
+ CLOCK_BOOTTIME,
+ CLOCK_BOOTTIME_ALARM,
+ CLOCK_REALTIME_ALARM,
+ CLOCK_TAI,
+ };
ksft_print_header();
- ksft_set_plan(max_clocks);
-
- for (clockid = CLOCK_REALTIME; clockid < max_clocks; clockid++) {
+ ksft_set_plan(ARRAY_SIZE(tested_clocks) * 2);
- /* Skip cputime clockids since nanosleep won't increment cputime */
- if (clockid == CLOCK_PROCESS_CPUTIME_ID ||
- clockid == CLOCK_THREAD_CPUTIME_ID ||
- clockid == CLOCK_HWSPECIFIC) {
- ksft_test_result_skip("%-31s\n", clockstring(clockid));
- continue;
- }
+ for (size_t clock_index = 0; clock_index < ARRAY_SIZE(tested_clocks); clock_index++) {
+ clockid = tested_clocks[clock_index];
fflush(stdout);
- length = 10;
- while (length <= (NSEC_PER_SEC * 10)) {
- ret = nanosleep_test(clockid, length);
- if (ret == UNSUPPORTED) {
- ksft_test_result_skip("%-31s\n", clockstring(clockid));
- goto next;
- }
- if (ret < 0) {
- ksft_test_result_fail("%-31s\n", clockstring(clockid));
- ksft_exit_fail();
- }
- length *= 100;
- }
- ret = nanosleep_test_remaining(clockid);
- if (ret < 0) {
- ksft_test_result_fail("%-31s\n", clockstring(clockid));
- ksft_exit_fail();
- }
- ksft_test_result_pass("%-31s\n", clockstring(clockid));
-next:
- ret = 0;
+ nanosleep_test_clock(clockid);
}
- ksft_exit_pass();
+ ksft_finished();
}
diff --git a/tools/testing/selftests/timers/nsleep-lat.c b/tools/testing/selftests/timers/nsleep-lat.c
index a7ba1eb1e21b..5de0051ac8e3 100644
--- a/tools/testing/selftests/timers/nsleep-lat.c
+++ b/tools/testing/selftests/timers/nsleep-lat.c
@@ -24,44 +24,10 @@
#include <sys/timex.h>
#include <string.h>
#include <signal.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
-#define UNRESONABLE_LATENCY 40000000 /* 40ms in nanosecs */
-
-/* CLOCK_HWSPECIFIC == CLOCK_SGI_CYCLE (Deprecated) */
-#define CLOCK_HWSPECIFIC 10
-
-#define UNSUPPORTED 0xf00f
-
-char *clockstring(int clockid)
-{
- switch (clockid) {
- case CLOCK_REALTIME:
- return "CLOCK_REALTIME";
- case CLOCK_MONOTONIC:
- return "CLOCK_MONOTONIC";
- case CLOCK_PROCESS_CPUTIME_ID:
- return "CLOCK_PROCESS_CPUTIME_ID";
- case CLOCK_THREAD_CPUTIME_ID:
- return "CLOCK_THREAD_CPUTIME_ID";
- case CLOCK_MONOTONIC_RAW:
- return "CLOCK_MONOTONIC_RAW";
- case CLOCK_REALTIME_COARSE:
- return "CLOCK_REALTIME_COARSE";
- case CLOCK_MONOTONIC_COARSE:
- return "CLOCK_MONOTONIC_COARSE";
- case CLOCK_BOOTTIME:
- return "CLOCK_BOOTTIME";
- case CLOCK_REALTIME_ALARM:
- return "CLOCK_REALTIME_ALARM";
- case CLOCK_BOOTTIME_ALARM:
- return "CLOCK_BOOTTIME_ALARM";
- case CLOCK_TAI:
- return "CLOCK_TAI";
- };
- return "UNKNOWN_CLOCKID";
-}
+#define UNRESONABLE_LATENCY (40 * NSEC_PER_MSEC)
struct timespec timespec_add(struct timespec ts, unsigned long long ns)
{
@@ -92,58 +58,68 @@ int nanosleep_lat_test(int clockid, long long ns)
target.tv_nsec = ns%NSEC_PER_SEC;
if (clock_gettime(clockid, &start))
- return UNSUPPORTED;
+ return KSFT_SKIP;
if (clock_nanosleep(clockid, 0, &target, NULL))
- return UNSUPPORTED;
+ return KSFT_SKIP;
count = 10;
/* First check relative latency */
- clock_gettime(clockid, &start);
- for (i = 0; i < count; i++)
- clock_nanosleep(clockid, 0, &target, NULL);
- clock_gettime(clockid, &end);
+ if (clock_gettime(clockid, &start))
+ return KSFT_FAIL;
+
+ for (i = 0; i < count; i++) {
+ if (clock_nanosleep(clockid, 0, &target, NULL))
+ return KSFT_FAIL;
+ }
+
+ if (clock_gettime(clockid, &end))
+ return KSFT_FAIL;
if (((timespec_sub(start, end)/count)-ns) > UNRESONABLE_LATENCY) {
ksft_print_msg("Large rel latency: %lld ns :", (timespec_sub(start, end)/count)-ns);
- return -1;
+ return KSFT_FAIL;
}
/* Next check absolute latency */
for (i = 0; i < count; i++) {
- clock_gettime(clockid, &start);
+ if (clock_gettime(clockid, &start))
+ return KSFT_FAIL;
target = timespec_add(start, ns);
- clock_nanosleep(clockid, TIMER_ABSTIME, &target, NULL);
- clock_gettime(clockid, &end);
+ if (clock_nanosleep(clockid, TIMER_ABSTIME, &target, NULL))
+ return KSFT_FAIL;
+ if (clock_gettime(clockid, &end))
+ return KSFT_FAIL;
latency += timespec_sub(target, end);
}
if (latency/count > UNRESONABLE_LATENCY) {
ksft_print_msg("Large abs latency: %lld ns :", latency/count);
- return -1;
+ return KSFT_FAIL;
}
- return 0;
+ return KSFT_PASS;
}
-#define SKIPPED_CLOCK_COUNT 3
-
int main(int argc, char **argv)
{
long long length;
int clockid, ret;
- int max_clocks = CLOCK_TAI + 1;
- ksft_print_header();
- ksft_set_plan(max_clocks - CLOCK_REALTIME - SKIPPED_CLOCK_COUNT);
+ static const clockid_t tested_clocks[] = {
+ CLOCK_REALTIME,
+ CLOCK_MONOTONIC,
+ CLOCK_BOOTTIME,
+ CLOCK_BOOTTIME_ALARM,
+ CLOCK_REALTIME_ALARM,
+ CLOCK_TAI,
+ };
- for (clockid = CLOCK_REALTIME; clockid < max_clocks; clockid++) {
+ ksft_print_header();
+ ksft_set_plan(ARRAY_SIZE(tested_clocks));
- /* Skip cputime clockids since nanosleep won't increment cputime */
- if (clockid == CLOCK_PROCESS_CPUTIME_ID ||
- clockid == CLOCK_THREAD_CPUTIME_ID ||
- clockid == CLOCK_HWSPECIFIC)
- continue;
+ for (size_t clock_index = 0; clock_index < ARRAY_SIZE(tested_clocks); clock_index++) {
+ clockid = tested_clocks[clock_index];
length = 10;
while (length <= (NSEC_PER_SEC * 10)) {
@@ -154,12 +130,7 @@ int main(int argc, char **argv)
}
- if (ret == UNSUPPORTED) {
- ksft_test_result_skip("%s\n", clockstring(clockid));
- } else {
- ksft_test_result(ret >= 0, "%s\n",
- clockstring(clockid));
- }
+ ksft_test_result_report(ret, "%s\n", clock_name(clockid));
}
ksft_finished();
diff --git a/tools/testing/selftests/timers/posix_timers.c b/tools/testing/selftests/timers/posix_timers.c
index 2f3bac9fc6e8..a92d4b957747 100644
--- a/tools/testing/selftests/timers/posix_timers.c
+++ b/tools/testing/selftests/timers/posix_timers.c
@@ -16,10 +16,10 @@
#include <string.h>
#include <unistd.h>
#include <time.h>
-#include <include/vdso/time64.h>
#include <pthread.h>
#include <stdbool.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define DELAY 2
@@ -141,8 +141,9 @@ static void check_itimer(int which, const char *name)
ksft_test_result(check_diff(start, end) == 0, "%s\n", name);
}
-static void check_timer_create(int which, const char *name)
+static void check_timer_create(int which)
{
+ const char *name = clock_name(which);
struct timespec start, end;
struct itimerspec val = {
.it_value.tv_sec = DELAY,
@@ -455,8 +456,9 @@ static void check_delete(void)
ksft_test_result(!tsig.signals, "check_delete\n");
}
-static void check_sigev_none(int which, const char *name)
+static void check_sigev_none(int which)
{
+ const char *name = clock_name(which);
struct timespec start, now;
struct itimerspec its;
struct sigevent sev;
@@ -493,8 +495,9 @@ static void check_sigev_none(int which, const char *name)
"check_sigev_none %s\n", name);
}
-static void check_gettime(int which, const char *name)
+static void check_gettime(int which)
{
+ const char *name = clock_name(which);
struct itimerspec its, prev;
struct timespec start, now;
struct sigevent sev;
@@ -546,8 +549,9 @@ static void check_gettime(int which, const char *name)
ksft_test_result(wraps > 1, "check_gettime %s\n", name);
}
-static void check_overrun(int which, const char *name)
+static void check_overrun(int which)
{
+ const char *name = clock_name(which);
struct timespec start, now;
struct tmrsig tsig = { };
struct itimerspec its;
@@ -689,7 +693,7 @@ int main(int argc, char **argv)
check_itimer(ITIMER_VIRTUAL, "ITIMER_VIRTUAL");
check_itimer(ITIMER_PROF, "ITIMER_PROF");
check_itimer(ITIMER_REAL, "ITIMER_REAL");
- check_timer_create(CLOCK_THREAD_CPUTIME_ID, "CLOCK_THREAD_CPUTIME_ID");
+ check_timer_create(CLOCK_THREAD_CPUTIME_ID);
/*
* It's unfortunately hard to reliably test a timer expiration
@@ -700,7 +704,7 @@ int main(int argc, char **argv)
* to ensure true parallelism. So test only one thread until we
* find a better solution.
*/
- check_timer_create(CLOCK_PROCESS_CPUTIME_ID, "CLOCK_PROCESS_CPUTIME_ID");
+ check_timer_create(CLOCK_PROCESS_CPUTIME_ID);
check_timer_distribution();
if (run_sig_ign_tests) {
@@ -708,18 +712,18 @@ int main(int argc, char **argv)
check_sig_ign(1);
check_rearm();
check_delete();
- check_sigev_none(CLOCK_MONOTONIC, "CLOCK_MONOTONIC");
- check_sigev_none(CLOCK_PROCESS_CPUTIME_ID, "CLOCK_PROCESS_CPUTIME_ID");
- check_gettime(CLOCK_MONOTONIC, "CLOCK_MONOTONIC");
- check_gettime(CLOCK_PROCESS_CPUTIME_ID, "CLOCK_PROCESS_CPUTIME_ID");
- check_gettime(CLOCK_THREAD_CPUTIME_ID, "CLOCK_THREAD_CPUTIME_ID");
+ check_sigev_none(CLOCK_MONOTONIC);
+ check_sigev_none(CLOCK_PROCESS_CPUTIME_ID);
+ check_gettime(CLOCK_MONOTONIC);
+ check_gettime(CLOCK_PROCESS_CPUTIME_ID);
+ check_gettime(CLOCK_THREAD_CPUTIME_ID);
} else {
ksft_print_msg("Skipping SIG_IGN tests on kernel < 6.13\n");
}
- check_overrun(CLOCK_MONOTONIC, "CLOCK_MONOTONIC");
- check_overrun(CLOCK_PROCESS_CPUTIME_ID, "CLOCK_PROCESS_CPUTIME_ID");
- check_overrun(CLOCK_THREAD_CPUTIME_ID, "CLOCK_THREAD_CPUTIME_ID");
+ check_overrun(CLOCK_MONOTONIC);
+ check_overrun(CLOCK_PROCESS_CPUTIME_ID);
+ check_overrun(CLOCK_THREAD_CPUTIME_ID);
ksft_finished();
}
diff --git a/tools/testing/selftests/timers/raw_skew.c b/tools/testing/selftests/timers/raw_skew.c
index a7bae7d80916..0c87a8fb0d7f 100644
--- a/tools/testing/selftests/timers/raw_skew.c
+++ b/tools/testing/selftests/timers/raw_skew.c
@@ -25,7 +25,7 @@
#include <sys/time.h>
#include <sys/timex.h>
#include <time.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define shift_right(x, s) ({ \
diff --git a/tools/testing/selftests/timers/set-2038.c b/tools/testing/selftests/timers/set-2038.c
index ecc171de4728..f522a3035ec6 100644
--- a/tools/testing/selftests/timers/set-2038.c
+++ b/tools/testing/selftests/timers/set-2038.c
@@ -27,7 +27,7 @@
#include <unistd.h>
#include <time.h>
#include <sys/time.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define KTIME_MAX ((long long)~((unsigned long long)1 << 63))
diff --git a/tools/testing/selftests/timers/set-timer-lat.c b/tools/testing/selftests/timers/set-timer-lat.c
index 44d2e3614fa5..79ddba25d314 100644
--- a/tools/testing/selftests/timers/set-timer-lat.c
+++ b/tools/testing/selftests/timers/set-timer-lat.c
@@ -28,7 +28,7 @@
#include <signal.h>
#include <stdlib.h>
#include <pthread.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
/* CLOCK_HWSPECIFIC == CLOCK_SGI_CYCLE (Deprecated) */
@@ -43,36 +43,6 @@ struct timespec start_time;
long long max_latency_ns;
int timer_fired_early;
-char *clockstring(int clockid)
-{
- switch (clockid) {
- case CLOCK_REALTIME:
- return "CLOCK_REALTIME";
- case CLOCK_MONOTONIC:
- return "CLOCK_MONOTONIC";
- case CLOCK_PROCESS_CPUTIME_ID:
- return "CLOCK_PROCESS_CPUTIME_ID";
- case CLOCK_THREAD_CPUTIME_ID:
- return "CLOCK_THREAD_CPUTIME_ID";
- case CLOCK_MONOTONIC_RAW:
- return "CLOCK_MONOTONIC_RAW";
- case CLOCK_REALTIME_COARSE:
- return "CLOCK_REALTIME_COARSE";
- case CLOCK_MONOTONIC_COARSE:
- return "CLOCK_MONOTONIC_COARSE";
- case CLOCK_BOOTTIME:
- return "CLOCK_BOOTTIME";
- case CLOCK_REALTIME_ALARM:
- return "CLOCK_REALTIME_ALARM";
- case CLOCK_BOOTTIME_ALARM:
- return "CLOCK_BOOTTIME_ALARM";
- case CLOCK_TAI:
- return "CLOCK_TAI";
- }
- return "UNKNOWN_CLOCKID";
-}
-
-
long long timespec_sub(struct timespec a, struct timespec b)
{
long long ret = NSEC_PER_SEC * b.tv_sec + b.tv_nsec;
@@ -103,7 +73,7 @@ void sigalarm(int signo)
void describe_timer(int flags, int interval)
{
printf("%-22s %s %s ",
- clockstring(clock_id),
+ clock_name(clock_id),
flags ? "ABSTIME":"RELTIME",
interval ? "PERIODIC":"ONE-SHOT");
}
@@ -129,12 +99,12 @@ int setup_timer(int clock_id, int flags, int interval, timer_t *tm1)
if ((clock_id == CLOCK_REALTIME_ALARM) ||
(clock_id == CLOCK_BOOTTIME_ALARM)) {
printf("%-22s %s missing CAP_WAKE_ALARM? : [UNSUPPORTED]\n",
- clockstring(clock_id),
+ clock_name(clock_id),
flags ? "ABSTIME":"RELTIME");
/* Indicate timer isn't set, so caller doesn't wait */
return 1;
}
- printf("%s - timer_create() failed\n", clockstring(clock_id));
+ printf("%s - timer_create() failed\n", clock_name(clock_id));
return -1;
}
@@ -151,7 +121,7 @@ int setup_timer(int clock_id, int flags, int interval, timer_t *tm1)
err = timer_settime(*tm1, flags, &its1, &its2);
if (err) {
- printf("%s - timer_settime() failed\n", clockstring(clock_id));
+ printf("%s - timer_settime() failed\n", clock_name(clock_id));
return -1;
}
diff --git a/tools/testing/selftests/timers/valid-adjtimex.c b/tools/testing/selftests/timers/valid-adjtimex.c
index e1e56d3097d6..f641d5fb0902 100644
--- a/tools/testing/selftests/timers/valid-adjtimex.c
+++ b/tools/testing/selftests/timers/valid-adjtimex.c
@@ -29,7 +29,7 @@
#include <string.h>
#include <signal.h>
#include <unistd.h>
-#include <include/vdso/time64.h>
+#include "clock-helpers.h"
#include "kselftest.h"
#define ADJ_SETOFFSET 0x0100
diff --git a/tools/testing/selftests/ublk/Makefile b/tools/testing/selftests/ublk/Makefile
index 6e4fe8d1fed1..37883e9d50ec 100644
--- a/tools/testing/selftests/ublk/Makefile
+++ b/tools/testing/selftests/ublk/Makefile
@@ -23,6 +23,7 @@ TEST_PROGS += test_generic_17.sh
TEST_PROGS += test_batch_01.sh
TEST_PROGS += test_batch_02.sh
TEST_PROGS += test_batch_03.sh
+TEST_PROGS += test_batch_04.sh
TEST_PROGS += test_null_01.sh
TEST_PROGS += test_null_02.sh
@@ -34,6 +35,7 @@ TEST_PROGS += test_loop_04.sh
TEST_PROGS += test_loop_05.sh
TEST_PROGS += test_loop_06.sh
TEST_PROGS += test_loop_07.sh
+TEST_PROGS += test_loop_08.sh
TEST_PROGS += test_integrity_01.sh
TEST_PROGS += test_integrity_02.sh
@@ -53,6 +55,8 @@ TEST_PROGS += test_stripe_06.sh
TEST_PROGS += test_part_01.sh
TEST_PROGS += test_part_02.sh
+TEST_PROGS += test_params_01.sh
+
TEST_PROGS += test_shmemzc_01.sh
TEST_PROGS += test_shmemzc_02.sh
TEST_PROGS += test_shmemzc_03.sh
@@ -69,6 +73,8 @@ TEST_PROGS += test_stress_08.sh
TEST_PROGS += test_stress_09.sh
TEST_FILES := settings
+TEST_FILES += test_common.sh
+TEST_FILES += trace
TEST_GEN_PROGS_EXTENDED = kublk metadata_size
STANDALONE_UTILS := metadata_size.c
diff --git a/tools/testing/selftests/ublk/batch.c b/tools/testing/selftests/ublk/batch.c
index a54025b00917..d8d9ebed5979 100644
--- a/tools/testing/selftests/ublk/batch.c
+++ b/tools/testing/selftests/ublk/batch.c
@@ -535,7 +535,7 @@ void ublk_batch_complete_io(struct ublk_thread *t, struct ublk_queue *q,
elem = (struct ublk_batch_elem *)(cb->elem + cb->done * t->commit_buf_elem_size);
elem->tag = tag;
- elem->buf_index = ublk_batch_io_buf_idx(t, q, tag);
+ elem->buf_index = ublk_batch_io_buf_idx_next(t, q, tag);
elem->result = res;
if (!ublk_queue_no_buf(q))
diff --git a/tools/testing/selftests/ublk/kublk.c b/tools/testing/selftests/ublk/kublk.c
index 0b23c09daea5..2400b4615766 100644
--- a/tools/testing/selftests/ublk/kublk.c
+++ b/tools/testing/selftests/ublk/kublk.c
@@ -8,6 +8,13 @@
#include "kublk.h"
#define MAX_NR_TGT_ARG 64
+#define KUBLK_PARAM_LOGICAL_BS_SHIFT 9
+#define KUBLK_PARAM_PHYSICAL_BS_SHIFT 12
+#define KUBLK_PARAM_ZONE_SECTORS 128
+#define KUBLK_PARAM_NR_ZONES 16
+#define KUBLK_PARAM_DEV_SECTORS \
+ (KUBLK_PARAM_ZONE_SECTORS * KUBLK_PARAM_NR_ZONES)
+#define KUBLK_PARAM_ZONE_APPEND_SECTORS 8
unsigned int ublk_dbg_mask = UBLK_LOG;
static const struct ublk_tgt_ops *tgt_ops_list[] = {
@@ -227,6 +234,55 @@ static int ublk_ctrl_get_features(struct ublk_dev *dev,
return __ublk_ctrl_cmd(dev, &data);
}
+static int parse_param_types(const char *arg, __u32 *types)
+{
+ char buf[128], *save = NULL, *tok;
+
+ if (strlen(arg) >= sizeof(buf))
+ return -EINVAL;
+
+ strcpy(buf, arg);
+ *types = 0;
+ tok = strtok_r(buf, ",", &save);
+ while (tok) {
+ if (!strcmp(tok, "none"))
+ ;
+ else if (!strcmp(tok, "basic"))
+ *types |= UBLK_PARAM_TYPE_BASIC;
+ else if (!strcmp(tok, "zoned"))
+ *types |= UBLK_PARAM_TYPE_ZONED;
+ else
+ return -EINVAL;
+ tok = strtok_r(NULL, ",", &save);
+ }
+
+ return 0;
+}
+
+static void ublk_init_params_from_ctx(const struct dev_ctx *ctx,
+ struct ublk_params *params)
+{
+ const struct params_ctx *p = &ctx->params;
+
+ *params = (struct ublk_params) {
+ .types = p->types,
+ .basic = {
+ .logical_bs_shift = p->logical_bs_shift,
+ .physical_bs_shift = p->physical_bs_shift,
+ .io_min_shift = p->io_min_shift,
+ .io_opt_shift = p->io_opt_shift,
+ .max_sectors = p->max_sectors,
+ .chunk_sectors = p->chunk_sectors,
+ .dev_sectors = p->dev_sectors,
+ },
+ .zoned = {
+ .max_open_zones = p->max_open_zones,
+ .max_active_zones = p->max_active_zones,
+ .max_zone_append_sectors = p->max_zone_append_sectors,
+ },
+ };
+}
+
static int ublk_ctrl_update_size(struct ublk_dev *dev,
__u64 nr_sects)
{
@@ -352,6 +408,8 @@ static void ublk_ctrl_dump(struct ublk_dev *dev)
ublk_log("\tmax rq size %d daemon pid %d flags 0x%llx state %s\n",
info->max_io_buf_bytes, info->ublksrv_pid, info->flags,
ublk_dev_state_desc(dev));
+ if (info->flags & UBLK_F_IO_DESC_SIZE)
+ ublk_log("\tio_desc_size %u\n", info->io_desc_size);
if (affinity) {
char buf[512];
@@ -400,22 +458,22 @@ static struct ublk_dev *ublk_ctrl_init(void)
return dev;
}
-static int __ublk_queue_cmd_buf_sz(unsigned depth)
+static size_t __ublk_queue_cmd_buf_sz(const struct ublk_queue *q, __u16 depth)
{
- int size = depth * sizeof(struct ublksrv_io_desc);
- unsigned int page_sz = getpagesize();
+ size_t size = depth * (size_t)q->io_desc_size;
+ size_t page_sz = getpagesize();
return round_up(size, page_sz);
}
-static int ublk_queue_max_cmd_buf_sz(void)
+static size_t ublk_queue_max_cmd_buf_sz(const struct ublk_queue *q)
{
- return __ublk_queue_cmd_buf_sz(UBLK_MAX_QUEUE_DEPTH);
+ return __ublk_queue_cmd_buf_sz(q, UBLK_MAX_QUEUE_DEPTH);
}
-static int ublk_queue_cmd_buf_sz(struct ublk_queue *q)
+static size_t ublk_queue_cmd_buf_sz(const struct ublk_queue *q)
{
- return __ublk_queue_cmd_buf_sz(q->q_depth);
+ return __ublk_queue_cmd_buf_sz(q, q->q_depth);
}
static void ublk_queue_deinit(struct ublk_queue *q)
@@ -453,7 +511,7 @@ static int ublk_queue_init(struct ublk_queue *q, unsigned long long extra_flags,
struct ublk_dev *dev = q->dev;
int depth = dev->dev_info.queue_depth;
int i;
- int cmd_buf_size, io_buf_size, integrity_size;
+ size_t cmd_buf_size, io_buf_size, integrity_size;
unsigned long off;
pthread_spin_init(&q->lock, PTHREAD_PROCESS_PRIVATE);
@@ -463,12 +521,13 @@ static int ublk_queue_init(struct ublk_queue *q, unsigned long long extra_flags,
q->flags = dev->dev_info.flags;
q->flags |= extra_flags;
q->metadata_size = metadata_size;
+ q->io_desc_size = dev->dev_info.io_desc_size;
/* Cache fd in queue for fast path access */
q->ublk_fd = dev->fds[0];
cmd_buf_size = ublk_queue_cmd_buf_sz(q);
- off = UBLKSRV_CMD_BUF_OFFSET + q->q_id * ublk_queue_max_cmd_buf_sz();
+ off = UBLKSRV_CMD_BUF_OFFSET + q->q_id * ublk_queue_max_cmd_buf_sz(q);
q->io_cmd_buf = mmap(0, cmd_buf_size, PROT_READ,
MAP_SHARED | MAP_POPULATE, dev->fds[0], off);
if (q->io_cmd_buf == MAP_FAILED) {
@@ -540,9 +599,14 @@ static int ublk_thread_init(struct ublk_thread *t, unsigned long long extra_flag
unsigned max_nr_ios_per_thread = nr_ios / dev->nthreads;
max_nr_ios_per_thread += !!(nr_ios % dev->nthreads);
+ t->auto_buf_stride = max_nr_ios_per_thread;
t->nr_bufs = max_nr_ios_per_thread;
+ if ((extra_flags & UBLKS_Q_ROTATE_AUTO_BUF) &&
+ (dev->dev_info.flags & UBLK_F_AUTO_BUF_REG))
+ t->nr_bufs *= 2;
} else {
t->nr_bufs = 0;
+ t->auto_buf_stride = 0;
}
if (ublk_dev_batch_io(dev))
@@ -1436,6 +1500,8 @@ static int ublk_start_daemon(const struct dev_ctx *ctx, struct ublk_dev *dev)
extra_flags = UBLKS_Q_AUTO_BUF_REG_FALLBACK;
if (ctx->no_ublk_fixed_fd)
extra_flags |= UBLKS_Q_NO_UBLK_FIXED_FD;
+ if (ctx->rotate_auto_buf)
+ extra_flags |= UBLKS_Q_ROTATE_AUTO_BUF;
for (i = 0; i < dinfo->nr_hw_queues; i++) {
dev->q[i].dev = dev;
@@ -1708,6 +1774,7 @@ static int __cmd_dev_add(const struct dev_ctx *ctx)
info->dev_id = ctx->dev_id;
info->nr_hw_queues = nr_queues;
info->queue_depth = depth;
+ info->io_desc_size = ctx->io_desc_size;
info->flags = ctx->flags;
if ((features & UBLK_F_QUIESCE) &&
(info->flags & UBLK_F_USER_RECOVERY))
@@ -1761,6 +1828,51 @@ fail:
static int __cmd_dev_list(struct dev_ctx *ctx);
+static int cmd_dev_set_params(struct dev_ctx *ctx)
+{
+ struct ublksrv_ctrl_dev_info *info;
+ struct ublk_params params;
+ struct ublk_dev *dev;
+ __u64 features;
+ int ret, del_ret;
+
+ dev = ublk_ctrl_init();
+ if (!dev)
+ return -ENODEV;
+
+ ret = ublk_ctrl_get_features(dev, &features);
+ if (ret < 0)
+ goto out;
+
+ if (!(features & UBLK_F_CMD_IOCTL_ENCODE)) {
+ ret = -ENOTSUP;
+ goto out;
+ }
+
+ info = &dev->dev_info;
+ info->dev_id = ctx->dev_id;
+ info->nr_hw_queues = ctx->nr_hw_queues;
+ info->queue_depth = ctx->queue_depth;
+ info->io_desc_size = ctx->io_desc_size;
+ info->flags = ctx->flags;
+
+ ret = ublk_ctrl_add_dev(dev);
+ if (ret < 0)
+ goto out;
+
+ ublk_init_params_from_ctx(ctx, &params);
+
+ ret = ublk_ctrl_set_params(dev, &params);
+ printf("SET_PARAMS returned %d\n", ret);
+
+ del_ret = ublk_ctrl_del_dev(dev);
+ if (del_ret < 0 && ret == 0)
+ ret = del_ret;
+out:
+ ublk_ctrl_deinit(dev);
+ return ret < 0 ? ret : 0;
+}
+
static int cmd_dev_add(struct dev_ctx *ctx)
{
int res;
@@ -1970,6 +2082,7 @@ static int cmd_dev_get_features(void)
FEAT_NAME(UBLK_F_BATCH_IO),
FEAT_NAME(UBLK_F_NO_AUTO_PART_SCAN),
FEAT_NAME(UBLK_F_SHMEM_ZC),
+ FEAT_NAME(UBLK_F_IO_DESC_SIZE),
};
struct ublk_dev *dev;
__u64 features = 0;
@@ -2067,7 +2180,8 @@ static void __cmd_create_help(char *exe, bool recovery)
printf("\t[--nthreads threads] [--per_io_tasks]\n");
printf("\t[--integrity_capable] [--integrity_reftag] [--metadata_size SIZE] "
"[--pi_offset OFFSET] [--csum_type ip|t10dif|nvme] [--tag_size SIZE]\n");
- printf("\t[--batch|-b] [--no_auto_part_scan]\n");
+ printf("\t[--batch|-b] [--rotate_auto_buf] [--no_auto_part_scan]\n");
+ printf("\t[--io_desc_size SIZE]\n");
printf("\t[target options] [backfile1] [backfile2] ...\n");
printf("\tdefault: nr_queues=2(max 32), depth=128(max 1024), dev_id=-1(auto allocation)\n");
printf("\tdefault: nthreads=nr_queues");
@@ -2104,6 +2218,9 @@ static int cmd_dev_help(char *exe)
printf("\t --safe only stop if device has no active openers\n\n");
printf("%s list [-n dev_id] -a \n", exe);
printf("\t -a list all devices, -n list specified device, default -a \n\n");
+ printf("%s set_params [-n dev_id] [-q nr_queues] [-d depth] [-u] [--zoned]\n", exe);
+ printf("\t[--param_types basic[,zoned]|none]\n");
+ printf("\t issue ADD_DEV, SET_PARAMS and DEL_DEV without START_DEV\n\n");
printf("%s features\n", exe);
printf("%s update_size -n dev_id -s|--size size_in_bytes \n", exe);
printf("%s quiesce -n dev_id\n", exe);
@@ -2141,10 +2258,24 @@ int main(int argc, char *argv[])
{ "tag_size", 1, NULL, 0 },
{ "safe", 0, NULL, 0 },
{ "batch", 0, NULL, 'b'},
+ { "rotate_auto_buf", 0, NULL, 0 },
{ "no_auto_part_scan", 0, NULL, 0 },
{ "shmem_zc", 0, NULL, 0 },
{ "htlb", 1, NULL, 0 },
{ "rdonly_shmem_buf", 0, NULL, 0 },
+ { "io_desc_size", 1, NULL, 0 },
+ { "zoned", 0, NULL, 0 },
+ { "param_types", 1, NULL, 0 },
+ { "logical_bs_shift", 1, NULL, 0 },
+ { "physical_bs_shift", 1, NULL, 0 },
+ { "io_min_shift", 1, NULL, 0 },
+ { "io_opt_shift", 1, NULL, 0 },
+ { "max_sectors", 1, NULL, 0 },
+ { "chunk_sectors", 1, NULL, 0 },
+ { "dev_sectors", 1, NULL, 0 },
+ { "max_zone_append_sectors", 1, NULL, 0 },
+ { "max_open_zones", 1, NULL, 0 },
+ { "max_active_zones", 1, NULL, 0 },
{ 0, 0, 0, 0 }
};
const struct ublk_tgt_ops *ops = NULL;
@@ -2157,6 +2288,20 @@ int main(int argc, char *argv[])
.dev_id = -1,
.tgt_type = "unknown",
.csum_type = LBMD_PI_CSUM_NONE,
+ .io_desc_size = sizeof(struct ublksrv_io_desc),
+ .params = {
+ .types = UBLK_PARAM_TYPE_BASIC,
+ .logical_bs_shift = KUBLK_PARAM_LOGICAL_BS_SHIFT,
+ .physical_bs_shift = KUBLK_PARAM_PHYSICAL_BS_SHIFT,
+ .io_min_shift = KUBLK_PARAM_LOGICAL_BS_SHIFT,
+ .io_opt_shift = KUBLK_PARAM_PHYSICAL_BS_SHIFT,
+ .max_sectors =
+ UBLK_IO_MAX_BYTES >> KUBLK_PARAM_LOGICAL_BS_SHIFT,
+ .chunk_sectors = KUBLK_PARAM_ZONE_SECTORS,
+ .dev_sectors = KUBLK_PARAM_DEV_SECTORS,
+ .max_zone_append_sectors =
+ KUBLK_PARAM_ZONE_APPEND_SECTORS,
+ },
};
int ret = -EINVAL, i;
int tgt_argc = 1;
@@ -2228,6 +2373,8 @@ int main(int argc, char *argv[])
ctx.flags |= UBLK_F_AUTO_BUF_REG;
if (!strcmp(longopts[option_idx].name, "auto_zc_fallback"))
ctx.auto_zc_fallback = 1;
+ if (!strcmp(longopts[option_idx].name, "rotate_auto_buf"))
+ ctx.rotate_auto_buf = 1;
if (!strcmp(longopts[option_idx].name, "nthreads"))
ctx.nthreads = strtol(optarg, NULL, 10);
if (!strcmp(longopts[option_idx].name, "per_io_tasks"))
@@ -2266,6 +2413,38 @@ int main(int argc, char *argv[])
ctx.htlb_path = strdup(optarg);
if (!strcmp(longopts[option_idx].name, "rdonly_shmem_buf"))
ctx.rdonly_shmem_buf = 1;
+ if (!strcmp(longopts[option_idx].name, "io_desc_size")) {
+ ctx.flags |= UBLK_F_IO_DESC_SIZE;
+ ctx.io_desc_size = strtoul(optarg, NULL, 0);
+ }
+ if (!strcmp(longopts[option_idx].name, "zoned"))
+ ctx.flags |= UBLK_F_ZONED;
+ if (!strcmp(longopts[option_idx].name, "param_types")) {
+ ret = parse_param_types(optarg, &ctx.params.types);
+ if (ret)
+ return ret;
+ }
+ if (!strcmp(longopts[option_idx].name, "logical_bs_shift"))
+ ctx.params.logical_bs_shift = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "physical_bs_shift"))
+ ctx.params.physical_bs_shift = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "io_min_shift"))
+ ctx.params.io_min_shift = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "io_opt_shift"))
+ ctx.params.io_opt_shift = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "max_sectors"))
+ ctx.params.max_sectors = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "chunk_sectors"))
+ ctx.params.chunk_sectors = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "dev_sectors"))
+ ctx.params.dev_sectors = strtoull(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "max_zone_append_sectors"))
+ ctx.params.max_zone_append_sectors =
+ strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "max_open_zones"))
+ ctx.params.max_open_zones = strtoul(optarg, NULL, 0);
+ if (!strcmp(longopts[option_idx].name, "max_active_zones"))
+ ctx.params.max_active_zones = strtoul(optarg, NULL, 0);
break;
case '?':
/*
@@ -2335,6 +2514,13 @@ int main(int argc, char *argv[])
return -EINVAL;
}
+ if (ctx.rotate_auto_buf &&
+ !((ctx.flags & UBLK_F_AUTO_BUF_REG) &&
+ (ctx.flags & UBLK_F_BATCH_IO))) {
+ ublk_err("rotate_auto_buf requires --auto_zc and --batch\n");
+ return -EINVAL;
+ }
+
i = optind;
while (i < argc && ctx.nr_files < MAX_BACK_FILES) {
ctx.files[ctx.nr_files++] = argv[i++];
@@ -2348,7 +2534,9 @@ int main(int argc, char *argv[])
ops->parse_cmd_line(&ctx, tgt_argc, tgt_argv);
}
- if (!strcmp(cmd, "add"))
+ if (!strcmp(cmd, "set_params"))
+ ret = cmd_dev_set_params(&ctx);
+ else if (!strcmp(cmd, "add"))
ret = cmd_dev_add(&ctx);
else if (!strcmp(cmd, "recover")) {
if (ctx.dev_id < 0) {
diff --git a/tools/testing/selftests/ublk/kublk.h b/tools/testing/selftests/ublk/kublk.h
index 742c41d77df1..d98f3d612d88 100644
--- a/tools/testing/selftests/ublk/kublk.h
+++ b/tools/testing/selftests/ublk/kublk.h
@@ -63,6 +63,22 @@ struct fault_inject_ctx {
bool die_during_fetch;
};
+struct params_ctx {
+ __u32 types;
+
+ __u32 logical_bs_shift;
+ __u32 physical_bs_shift;
+ __u32 io_min_shift;
+ __u32 io_opt_shift;
+ __u32 max_sectors;
+ __u32 chunk_sectors;
+ __u64 dev_sectors;
+
+ __u32 max_open_zones;
+ __u32 max_active_zones;
+ __u32 max_zone_append_sectors;
+};
+
struct dev_ctx {
char tgt_type[16];
unsigned long flags;
@@ -82,11 +98,13 @@ struct dev_ctx {
unsigned int safe_stop:1;
unsigned int no_auto_part_scan:1;
unsigned int rdonly_shmem_buf:1;
+ unsigned int rotate_auto_buf:1;
__u32 integrity_flags;
__u8 metadata_size;
__u8 pi_offset;
__u8 csum_type;
__u8 tag_size;
+ __u16 io_desc_size;
int _evtfd;
int _shmid;
@@ -97,6 +115,8 @@ struct dev_ctx {
/* for 'update_size' command */
unsigned long long size;
+ struct params_ctx params;
+
char *htlb_path;
union {
@@ -134,6 +154,7 @@ struct ublk_io {
unsigned short buf_index;
unsigned short tgt_ios;
+ unsigned char auto_buf_phase;
void *private_data;
};
@@ -184,9 +205,11 @@ struct ublk_queue {
#define UBLKS_Q_AUTO_BUF_REG_FALLBACK (1ULL << 63)
#define UBLKS_Q_NO_UBLK_FIXED_FD (1ULL << 62)
#define UBLKS_Q_PREPARED (1ULL << 61)
+#define UBLKS_Q_ROTATE_AUTO_BUF (1ULL << 60)
__u64 flags;
int ublk_fd; /* cached ublk char device fd */
__u8 metadata_size;
+ __u16 io_desc_size;
struct ublk_io ios[UBLK_QUEUE_DEPTH];
/* used for prep io commands */
@@ -232,6 +255,7 @@ struct ublk_thread {
unsigned int io_inflight;
unsigned short nr_bufs;
+ unsigned short auto_buf_stride;
/* followings are for BATCH_IO */
unsigned short commit_buf_start;
@@ -461,9 +485,9 @@ static inline void ublk_mark_io_done(struct ublk_io *io, int res)
io->result = res;
}
-static inline const struct ublksrv_io_desc *ublk_get_iod(const struct ublk_queue *q, int tag)
+static inline const struct ublksrv_io_desc *ublk_get_iod(const struct ublk_queue *q, __u16 tag)
{
- return &q->io_cmd_buf[tag];
+ return (void *)q->io_cmd_buf + tag * (size_t)q->io_desc_size;
}
static inline void ublk_set_sqe_cmd_op(struct io_uring_sqe *sqe, __u32 cmd_op)
@@ -550,7 +574,20 @@ static inline unsigned short ublk_batch_io_buf_idx(
const struct ublk_thread *t, const struct ublk_queue *q,
unsigned tag)
{
- return ublk_queue_idx_in_thread(t, q) * q->q_depth + tag;
+ unsigned short base = ublk_queue_idx_in_thread(t, q) * q->q_depth + tag;
+
+ if (q->flags & UBLKS_Q_ROTATE_AUTO_BUF)
+ return base + q->ios[tag].auto_buf_phase * t->auto_buf_stride;
+ return base;
+}
+
+static inline unsigned short ublk_batch_io_buf_idx_next(
+ const struct ublk_thread *t, struct ublk_queue *q,
+ unsigned tag)
+{
+ if (q->flags & UBLKS_Q_ROTATE_AUTO_BUF)
+ q->ios[tag].auto_buf_phase ^= 1;
+ return ublk_batch_io_buf_idx(t, q, tag);
}
/* Queue UBLK_U_IO_PREP_IO_CMDS for a specific queue with batch elements */
diff --git a/tools/testing/selftests/ublk/test_batch_04.sh b/tools/testing/selftests/ublk/test_batch_04.sh
new file mode 100755
index 000000000000..cd5e1ff9d630
--- /dev/null
+++ b/tools/testing/selftests/ublk/test_batch_04.sh
@@ -0,0 +1,44 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# --rotate_auto_buf: COMMIT must unregister old auto_buf index before store.
+
+. "$(cd "$(dirname "$0")" && pwd)"/test_common.sh
+
+ERR_CODE=0
+
+if ! _have_feature "BATCH_IO" || ! _have_feature "AUTO_BUF_REG"; then
+ exit "$UBLK_SKIP_CODE"
+fi
+if ! _have_program fio || ! _have_program timeout; then
+ exit "$UBLK_SKIP_CODE"
+fi
+
+_prep_test "generic" "batch auto_buf unregister with rotating index"
+
+_create_backfile 0 64M
+
+dev_id=$(_add_ublk_dev_no_settle -t loop -q 1 --nthreads 1 -b --auto_zc \
+ --rotate_auto_buf "${UBLK_BACKFILES[0]}")
+_check_add_dev $TID $?
+
+for ((i = 0; i < 50; i++)); do
+ [ -b /dev/ublkb"${dev_id}" ] && break
+ sleep 0.1
+done
+[ -b /dev/ublkb"${dev_id}" ] || { _cleanup_test; _show_result $TID 1; }
+
+timeout -k 2 5 fio --name=job1 --filename=/dev/ublkb"${dev_id}" \
+ --ioengine=libaio --rw=write --direct=1 --bs=4k --iodepth=1 --size=64k \
+ > /dev/null 2>&1
+ERR_CODE=$?
+
+if [ "$ERR_CODE" -ne 0 ]; then
+ kill -9 "$(_get_ublk_daemon_pid "$dev_id" 2>/dev/null)" 2>/dev/null || true
+ sleep 0.5
+ pkill -9 fio 2>/dev/null || true
+ ERR_CODE=1
+fi
+
+_cleanup_test
+_show_result $TID $ERR_CODE
diff --git a/tools/testing/selftests/ublk/test_loop_08.sh b/tools/testing/selftests/ublk/test_loop_08.sh
new file mode 100755
index 000000000000..f7af2587482d
--- /dev/null
+++ b/tools/testing/selftests/ublk/test_loop_08.sh
@@ -0,0 +1,25 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+
+. "$(cd "$(dirname "$0")" && pwd)"/test_common.sh
+
+ERR_CODE=0
+
+if ! _have_program fio; then
+ exit "$UBLK_SKIP_CODE"
+fi
+
+_prep_test "loop" "write and verify with io_desc_size"
+
+_create_backfile 0 256M
+
+dev_id=$(_add_ublk_dev -t loop --io_desc_size 64 "${UBLK_BACKFILES[0]}")
+_check_add_dev $TID $?
+
+# run fio over the ublk disk
+_run_fio_verify_io --filename=/dev/ublkb"${dev_id}" --size=256M
+ERR_CODE=$?
+
+_cleanup_test
+
+_show_result $TID $ERR_CODE
diff --git a/tools/testing/selftests/ublk/test_params_01.sh b/tools/testing/selftests/ublk/test_params_01.sh
new file mode 100755
index 000000000000..928e72b1035d
--- /dev/null
+++ b/tools/testing/selftests/ublk/test_params_01.sh
@@ -0,0 +1,114 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+
+. "$(cd "$(dirname "$0")" && pwd)"/test_common.sh
+
+ERR_CODE=0
+
+run_set_params_success()
+{
+ local name=$1
+
+ shift
+
+ echo "$name"
+ if ! "$UBLK_PROG" set_params -q 1 -d 2 "$@"; then
+ echo "$name: SET_PARAMS check failed"
+ return 1
+ fi
+}
+
+run_set_params_failure()
+{
+ local name=$1
+
+ shift
+
+ echo "$name"
+ if "$UBLK_PROG" set_params -q 1 -d 2 "$@"; then
+ echo "$name: SET_PARAMS succeeded unexpectedly"
+ return 1
+ fi
+}
+
+run_zoned_set_params_success()
+{
+ local name=$1
+
+ shift
+
+ echo "$name"
+ if ! "$UBLK_PROG" set_params -q 1 -d 2 -u --zoned "$@"; then
+ echo "$name: SET_PARAMS check failed"
+ return 1
+ fi
+}
+
+run_zoned_set_params_failure()
+{
+ local name=$1
+
+ shift
+
+ echo "$name"
+ if "$UBLK_PROG" set_params -q 1 -d 2 -u --zoned "$@"; then
+ echo "$name: SET_PARAMS succeeded unexpectedly"
+ return 1
+ fi
+}
+
+_prep_test "params" "SET_PARAMS validation"
+
+if [ ! -c /dev/ublk-control ]; then
+ _cleanup_test
+ _show_result $TID $UBLK_SKIP_CODE
+fi
+
+run_set_params_success "valid basic params" ||
+ ERR_CODE=1
+
+run_set_params_failure "missing basic params" \
+ --param_types none ||
+ ERR_CODE=1
+
+run_set_params_failure "logical block larger than physical block" \
+ --logical_bs_shift 12 --physical_bs_shift 9 ||
+ ERR_CODE=1
+
+run_set_params_failure "too large max sectors" \
+ --max_sectors 2049 ||
+ ERR_CODE=1
+
+if _have_feature "ZONED" && _have_feature "USER_COPY"; then
+ run_zoned_set_params_success "valid zoned params" \
+ --param_types basic,zoned ||
+ ERR_CODE=1
+
+ run_zoned_set_params_failure "missing zoned params" ||
+ ERR_CODE=1
+
+ run_zoned_set_params_failure "non-power-of-2 zone size" \
+ --param_types basic,zoned \
+ --chunk_sectors 96 --dev_sectors $((96 * 16)) ||
+ ERR_CODE=1
+
+ run_zoned_set_params_failure "zero max zone append" \
+ --param_types basic,zoned \
+ --max_zone_append_sectors 0 ||
+ ERR_CODE=1
+
+ run_zoned_set_params_failure "too many open zones" \
+ --param_types basic,zoned \
+ --dev_sectors $((128 * 4)) --max_open_zones 5 ||
+ ERR_CODE=1
+
+ run_zoned_set_params_failure "too many active zones" \
+ --param_types basic,zoned \
+ --dev_sectors $((128 * 4)) --max_active_zones 5 ||
+ ERR_CODE=1
+else
+ echo "zoned ublk feature unavailable, skip zoned SET_PARAMS cases"
+fi
+
+_cleanup_test
+_show_result $TID $ERR_CODE
diff --git a/tools/testing/selftests/ublk/test_recover_03.sh b/tools/testing/selftests/ublk/test_recover_03.sh
index 2554805e5b02..92f4012178f0 100755
--- a/tools/testing/selftests/ublk/test_recover_03.sh
+++ b/tools/testing/selftests/ublk/test_recover_03.sh
@@ -29,6 +29,11 @@ _create_backfile 0 256M
_create_backfile 1 128M
_create_backfile 2 128M
+ublk_run_quiesce_recover -t null -q 2 -r 1 -b &
+ublk_run_quiesce_recover -t loop -q 2 -r 1 -b "${UBLK_BACKFILES[0]}" &
+ublk_run_quiesce_recover -t stripe -q 2 -r 1 -b "${UBLK_BACKFILES[1]}" "${UBLK_BACKFILES[2]}" &
+wait
+
ublk_run_quiesce_recover -t null -q 2 -r 1 &
ublk_run_quiesce_recover -t loop -q 2 -r 1 "${UBLK_BACKFILES[0]}" &
ublk_run_quiesce_recover -t stripe -q 2 -r 1 "${UBLK_BACKFILES[1]}" "${UBLK_BACKFILES[2]}" &
diff --git a/tools/testing/selftests/verification/test.d/rv_deadline.tc b/tools/testing/selftests/verification/test.d/rv_deadline.tc
new file mode 100644
index 000000000000..fc95267dbb82
--- /dev/null
+++ b/tools/testing/selftests/verification/test.d/rv_deadline.tc
@@ -0,0 +1,23 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0-or-later
+# description: Test deadline monitors trigger no reaction
+# requires: available_reactors deadline:monitor printk:reactor stress-ng:program
+
+load() { # returns true if there was a reaction
+ local lines_before
+ lines_before=$(dmesg | wc -l)
+ stress-ng --cpu 2 --sched deadline --sched-period 100000000 \
+ --sched-deadline 100000000 --sched-runtime 20000000 -t 5 &
+ stress-ng --cpu 2 --sched rr --sched-prio 50 --cyclic 1 \
+ --cyclic-policy rr --cyclic-prio 50 -t 5 &
+ wait
+ dmesg | tail -n +$((lines_before + 1)) | grep -q "rv: monitor [a-z]\+ does not allow event"
+}
+
+echo 1 > monitors/deadline/enable
+echo printk > monitors/deadline/reactors
+
+! load || false
+
+echo nop > monitors/deadline/reactors
+echo 0 > monitors/deadline/enable
diff --git a/tools/testing/selftests/verification/test.d/rv_monitor_enable_disable.tc b/tools/testing/selftests/verification/test.d/rv_monitor_enable_disable.tc
index f29236defb5a..61e2c8b54d9a 100644
--- a/tools/testing/selftests/verification/test.d/rv_monitor_enable_disable.tc
+++ b/tools/testing/selftests/verification/test.d/rv_monitor_enable_disable.tc
@@ -10,7 +10,7 @@ test_simple_monitor() {
grep -q "$monitor$" enabled_monitors
echo 0 > "monitors/$prefix$monitor/enable"
- ! grep -q "$monitor$" enabled_monitors
+ ! grep -q "$monitor$" enabled_monitors || false
echo "$monitor" >> enabled_monitors
grep -q 1 "monitors/$prefix$monitor/enable"
@@ -34,12 +34,12 @@ test_container_monitor() {
test -n "$nested"
echo 0 > "monitors/$monitor/enable"
- ! grep -q "^$monitor$" enabled_monitors
+ ! grep -q "^$monitor$" enabled_monitors || false
for nested_dir in "monitors/$monitor"/*; do
[ -d "$nested_dir" ] || continue
nested=$(basename "$nested_dir")
- ! grep -q "^$monitor:$nested$" enabled_monitors
+ ! grep -q "^$monitor:$nested$" enabled_monitors || false
done
echo "$monitor" >> enabled_monitors
@@ -71,5 +71,5 @@ for monitor_dir in monitors/*; do
fi
done
-! echo non_existent_monitor > enabled_monitors
-! grep -q "^non_existent_monitor$" enabled_monitors
+! echo non_existent_monitor > enabled_monitors || false
+! grep -q "^non_existent_monitor$" enabled_monitors || false
diff --git a/tools/testing/selftests/verification/test.d/rv_monitor_reactor.tc b/tools/testing/selftests/verification/test.d/rv_monitor_reactor.tc
index 2958bf849338..516a20971390 100644
--- a/tools/testing/selftests/verification/test.d/rv_monitor_reactor.tc
+++ b/tools/testing/selftests/verification/test.d/rv_monitor_reactor.tc
@@ -64,5 +64,5 @@ done
monitor=$(ls /sys/kernel/tracing/rv/monitors -1 | head -n 1)
test -f "monitors/$monitor/reactors"
-! echo non_existent_reactor > "monitors/$monitor/reactors"
-! grep -q "\\[non_existent_reactor\\]" "monitors/$monitor/reactors"
+! echo non_existent_reactor > "monitors/$monitor/reactors" || false
+! grep -q "\\[non_existent_reactor\\]" "monitors/$monitor/reactors" || false
diff --git a/tools/testing/selftests/verification/test.d/rv_stall.tc b/tools/testing/selftests/verification/test.d/rv_stall.tc
new file mode 100644
index 000000000000..515a10263ca1
--- /dev/null
+++ b/tools/testing/selftests/verification/test.d/rv_stall.tc
@@ -0,0 +1,33 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0-or-later
+# description: Test stall monitor
+# requires: available_reactors stall:monitor printk:reactor stress-ng:program
+
+THRESHOLD=/sys/module/stall/parameters/threshold_jiffies
+ORIG_THRESHOLD=$(cat $THRESHOLD)
+trap 'echo $ORIG_THRESHOLD > $THRESHOLD' EXIT
+
+load() { # returns true if there was a reaction
+ local lines_before cpu
+ cpu=$(($(nproc) - 1))
+ lines_before=$(dmesg | wc -l)
+ stress-ng --cpu 1 --taskset "$cpu" --sched rr --sched-prio 1 -t 3 &
+ stress-ng --cpu 5 --taskset "$cpu" -t 3 &
+ wait
+ dmesg | tail -n +$((lines_before + 1)) | grep -q "rv: monitor stall does not allow event"
+}
+
+echo 5000 > $THRESHOLD
+echo 1 > monitors/stall/enable
+echo printk > monitors/stall/reactors
+
+! load || false
+
+echo 0 > monitors/stall/enable
+echo 70 > $THRESHOLD
+echo 1 > monitors/stall/enable
+
+load
+
+echo nop > monitors/stall/reactors
+echo 0 > monitors/stall/enable
diff --git a/tools/testing/selftests/verification/test.d/rv_wwnr_printk.tc b/tools/testing/selftests/verification/test.d/rv_wwnr_printk.tc
index 5a59432b1d93..17e1edfb3902 100644
--- a/tools/testing/selftests/verification/test.d/rv_wwnr_printk.tc
+++ b/tools/testing/selftests/verification/test.d/rv_wwnr_printk.tc
@@ -4,11 +4,29 @@
# requires: available_reactors wwnr:monitor printk:reactor stress-ng:program
load() { # returns true if there was a reaction
- local lines_before num
+ local lines_before num load_pid ret
num=$((($(nproc) + 1) / 2))
lines_before=$(dmesg | wc -l)
- stress-ng --cpu-sched "$num" --timer "$num" -t 5 -q
- dmesg | tail -n $((lines_before + 1)) | grep -q "rv: monitor wwnr does not allow event"
+ stress-ng --cpu-sched "$num" --timer "$num" -t 5 -q &
+ load_pid=$!
+ timeout 5 dmesg -w | tail -n +$((lines_before + 1)) | \
+ grep -m 1 -q "rv: monitor wwnr does not allow event"
+ ret=$?
+ kill "$load_pid" || true
+ wait "$load_pid" || true
+ return $ret
+}
+
+# loads may flood the ringbuffer, wait for all pending printks (timeout at 2 minutes)
+wait_dmesg_flush() {
+ local last_before last_after
+ for _ in $(seq 400); do
+ last_before=$last_after
+ last_after=$(dmesg | grep "rv:" | tail -n 1 || true)
+ [ "$last_before" = "$last_after" ] && return 0
+ sleep .3
+ done
+ return 1
}
echo 1 > monitors/wwnr/enable
@@ -17,13 +35,17 @@ echo printk > monitors/wwnr/reactors
load
echo 0 > monitoring_on
-! load
+wait_dmesg_flush
+
+! load || false
echo 1 > monitoring_on
load
echo 0 > reacting_on
-! load
+wait_dmesg_flush
+
+! load || false
echo 1 > reacting_on
echo nop > monitors/wwnr/reactors
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h
new file mode 120000
index 000000000000..b84affdec559
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_82575.h \ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h
new file mode 120000
index 000000000000..9f97f4330086
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_defines.h \ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h
new file mode 120000
index 000000000000..c733634171bb
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_regs.h \ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/igb.c b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
new file mode 100644
index 000000000000..fd9e05d77ea4
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
@@ -0,0 +1,585 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include <unistd.h>
+#include <errno.h>
+#include <stdint.h>
+#include <linux/io.h>
+#include <linux/pci_regs.h>
+#include <linux/pci_ids.h>
+#include <linux/kernel.h>
+#include <linux/compiler.h>
+#include <asm/barrier.h>
+#include <linux/mii.h>
+#include <libvfio/vfio_pci_device.h>
+
+#include "e1000_regs.h"
+#include "e1000_defines.h"
+#include "e1000_82575.h"
+
+#define PCI_DEVICE_ID_INTEL_82576 0x10C9
+#define IGB_MAX_CHUNK_SIZE 1024
+#define MSIX_VECTOR 0
+#define MSIX_VECTOR_MASK (1 << MSIX_VECTOR)
+#define RING_SIZE 4096 /* Number of descriptors in ring */
+
+struct igb_tx_desc {
+ union {
+ struct {
+ u64 buffer_addr; /* Address of descriptor's data buffer */
+ u32 cmd_type_len; /* Command/Type/Length */
+ u32 olinfo_status; /* Context/Buffer info */
+ } read;
+
+ struct {
+ u64 rsvd; /* Reserved */
+ u32 nxtseq_seed; /* Next sequence seed */
+ u32 status; /* Descriptor status */
+ } wb;
+ };
+};
+
+struct igb_rx_desc {
+ union {
+ struct {
+ u64 pkt_addr; /* Packet buffer address */
+ u64 hdr_addr; /* Header buffer address */
+ } read;
+ struct {
+ u16 pkt_info; /* RSS type, Packet type */
+ u16 hdr_info; /* Split Head, buf len */
+ u32 rss; /* RSS Hash */
+ u32 status_error; /* ext status/error */
+ u16 length; /* Packet length */
+ u16 vlan; /* VLAN tag */
+ } wb; /* writeback */
+ };
+};
+
+struct igb {
+ void *bar0;
+ u32 tx_tail;
+ u32 rx_tail;
+ struct igb_tx_desc tx_ring[RING_SIZE] __attribute__((aligned(128)));
+ struct igb_rx_desc rx_ring[RING_SIZE] __attribute__((aligned(128)));
+};
+
+static inline struct igb *to_igb_state(struct vfio_pci_device *device)
+{
+ return (struct igb *)device->driver.region.vaddr;
+}
+
+static inline void igb_write32(struct igb *igb, u32 reg, u32 val)
+{
+ writel(val, igb->bar0 + reg);
+}
+
+static inline u32 igb_read32(struct igb *igb, u32 reg)
+{
+ return readl(igb->bar0 + reg);
+}
+
+static int igb_write_phy(struct igb *igb, u32 offset, u16 data)
+{
+ u32 mdic;
+ int i;
+
+ /*
+ * Write a PHY register over MDIO.
+ *
+ * A production driver would hold the SW/FW semaphore (SWSM.SWESMBI + the
+ * SW_FW_SYNC PHY bit) across the MDIO transaction to serialize against the
+ * device's management firmware. The selftest owns the assigned function
+ * exclusively on a dedicated test device with no active manageability
+ * contending for the PHY, so the sync is omitted; it should be added here
+ * if this ever needs to run on a manageability-enabled NIC.
+ */
+ mdic = (((u32)data) |
+ (offset << E1000_MDIC_REG_SHIFT) |
+ (1 << E1000_MDIC_PHY_SHIFT) |
+ E1000_MDIC_OP_WRITE);
+
+ igb_write32(igb, E1000_MDIC, mdic);
+
+ for (i = 0; i < 1000; i++) {
+ usleep(50);
+ mdic = igb_read32(igb, E1000_MDIC);
+ if (mdic & E1000_MDIC_READY)
+ break;
+ }
+
+ if (!(mdic & E1000_MDIC_READY))
+ return -1;
+
+ if (mdic & E1000_MDIC_ERROR)
+ return -1;
+
+ return 0;
+}
+
+/*
+ * Configure the device for PHY internal loopback per 82576 datasheet
+ * section 3.5.6.3.1. Force the PHY to 1Gb/s full duplex with loopback
+ * enabled, then force the MAC link state to match. Internal loopback
+ * wraps data at the end of the PHY datapath (section 3.5.6.3), so the
+ * physical link state is irrelevant.
+ *
+ * Section 3.5.6.1 directs to "Use PHY Loopback instead of MAC Loopback
+ * on the 82576", and section 3.5.6.2 states "MAC Loopback is not used
+ * on this device." RCTL.LBM_MAC is still set elsewhere as a QEMU-only
+ * accommodation; see the RCTL programming in the caller for the
+ * rationale.
+ */
+static void igb_setup_loopback(struct igb *igb)
+{
+ u32 ctrl;
+ int ret;
+
+ /*
+ * Kick the autoneg machinery solely to bring STATUS.LU up under
+ * QEMU's igb emulation: QEMU only updates STATUS.LU via its
+ * autoneg-done timer, and without LU set its receive path
+ * (e1000x_hw_rx_enabled) drops every loopback frame. On real
+ * hardware autoneg cannot complete before the next PHY write
+ * below clears the autoneg-enable bit, so this is effectively a
+ * no-op there.
+ */
+ (void)igb_write_phy(igb, MII_BMCR,
+ BMCR_ANENABLE | BMCR_ANRESTART);
+
+ /* PHY control: loopback + 1Gb/s full duplex, autoneg disabled. */
+ ret = igb_write_phy(igb, MII_BMCR,
+ BMCR_LOOPBACK |
+ BMCR_SPEED1000 |
+ BMCR_FULLDPLX);
+ VFIO_ASSERT_EQ(ret, 0, "Failed to write PHY control register");
+
+ /*
+ * Brief delay before forcing the MAC, mirroring the kernel ethtool
+ * selftest in igb_integrated_phy_loopback(). Not specified by the
+ * datasheet, but empirically required by the kernel driver.
+ */
+ usleep(50000);
+
+ /*
+ * Force the MAC to 1Gb/s full duplex with link up. Without forcing
+ * the link state the descriptor engine does not run, since the chip
+ * normally waits for a real negotiated link.
+ */
+ ctrl = igb_read32(igb, E1000_CTRL);
+ ctrl &= ~E1000_CTRL_SPD_SEL;
+ ctrl |= E1000_CTRL_FRCSPD |
+ E1000_CTRL_FRCDPX |
+ E1000_CTRL_SPD_1000 |
+ E1000_CTRL_FD |
+ E1000_CTRL_SLU;
+ igb_write32(igb, E1000_CTRL, ctrl);
+
+ /*
+ * Settling delay matching the kernel ethtool selftest's msleep(500)
+ * at the tail of igb_integrated_phy_loopback(). Not specified by
+ * the datasheet; empirical, and inherited from the kernel driver.
+ */
+ usleep(500000);
+}
+
+static int igb_probe(struct vfio_pci_device *device)
+{
+ if (!vfio_pci_device_match(device, PCI_VENDOR_ID_INTEL, PCI_DEVICE_ID_INTEL_82576))
+ return -EINVAL;
+
+ return 0;
+}
+
+static void igb_reset(struct igb *igb)
+{
+ int retries = 20;
+
+ igb_write32(igb, E1000_CTRL, igb_read32(igb, E1000_CTRL) | E1000_CTRL_RST);
+ /*
+ * Must wait at least 1 millisecond after setting the reset bit before
+ * checking if this device is ready to be used (82576 datasheet section
+ * 4.2.1.6.1). The delay also ensures the reset has taken effect and
+ * cleared EECD.AUTO_RD before it is polled below.
+ */
+ usleep(1000);
+
+ /*
+ * Poll NVM Auto Read Done rather than CTRL.RST, matching
+ * igb_get_auto_rd_done() in the igb driver: AUTO_RD implies both that
+ * the reset completed and that the device finished re-reading its
+ * configuration from NVM, which is what actually makes it usable.
+ */
+ while (retries-- > 0 && !(igb_read32(igb, E1000_EECD) & E1000_EECD_AUTO_RD))
+ usleep(1000);
+
+ /*
+ * QEMU's igb emulation does not set E1000_EECD_AUTO_RD. If we timed out,
+ * check if CTRL.RST is cleared, which is what QEMU uses to signal reset
+ * completion.
+ */
+ if (retries < 0) {
+ VFIO_ASSERT_EQ(igb_read32(igb, E1000_CTRL) & E1000_CTRL_RST, 0,
+ "Device reset did not complete (CTRL.RST not cleared)");
+ }
+
+ igb_write32(igb, E1000_IMC, 0xFFFFFFFF);
+}
+
+/*
+ * Program the device into a usable state. Split out of igb_init() so it
+ * can be reused after a device reset to re-program the registers that
+ * CTRL.RST clears. Expects bar0 to be mapped and MSI-X already enabled
+ * via VFIO.
+ */
+static void igb_hw_init(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+ u64 iova_tx, iova_rx;
+ u32 ctrl, rctl;
+ u16 cmd_reg;
+ int retries;
+
+ iova_tx = to_iova(device, igb->tx_ring);
+ iova_rx = to_iova(device, igb->rx_ring);
+
+
+
+ /* Signal that the driver is loaded */
+ ctrl = igb_read32(igb, E1000_CTRL_EXT);
+ ctrl |= E1000_CTRL_EXT_DRV_LOAD;
+ ctrl &= ~E1000_CTRL_EXT_LINK_MODE_MASK;
+ igb_write32(igb, E1000_CTRL_EXT, ctrl);
+
+ /* Enable PCI Bus Master. */
+ cmd_reg = vfio_pci_config_readw(device, PCI_COMMAND);
+ if ((cmd_reg & (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) !=
+ (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) {
+ cmd_reg |= (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY);
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd_reg);
+ }
+
+ /* Configure PHY internal loopback for testing. */
+ igb_setup_loopback(igb);
+
+ /*
+ * Disable DMA re-send on PCIe completion timeout (82576 datasheet
+ * section 8.6.1, GCR.Completion_Timeout_Resend, bit 16). The
+ * mix_and_match test intentionally submits descriptors targeting
+ * unmapped IOVAs; with the default (set) value, the device keeps
+ * retrying the failed read indefinitely, which keeps PCIe AER and
+ * IOMMU error handling busy and interferes with reset recovery.
+ */
+ ctrl = igb_read32(igb, E1000_GCR);
+ ctrl &= ~E1000_GCR_CMPL_TMOUT_RESEND;
+ igb_write32(igb, E1000_GCR, ctrl);
+
+ /* Configure TX and RX descriptor rings */
+ igb_write32(igb, E1000_TDBAL(0), (u32)iova_tx);
+ igb_write32(igb, E1000_TDBAH(0), (u32)(iova_tx >> 32));
+ igb_write32(igb, E1000_TDLEN(0), RING_SIZE * sizeof(struct igb_tx_desc));
+ igb_write32(igb, E1000_TDH(0), 0);
+ igb_write32(igb, E1000_TDT(0), 0);
+ igb_write32(igb, E1000_TXDCTL(0), E1000_TXDCTL_QUEUE_ENABLE);
+
+ igb_write32(igb, E1000_RDBAL(0), (u32)iova_rx);
+ igb_write32(igb, E1000_RDBAH(0), (u32)(iova_rx >> 32));
+ igb_write32(igb, E1000_RDLEN(0), RING_SIZE * sizeof(struct igb_rx_desc));
+ igb_write32(igb, E1000_RDH(0), 0);
+ igb_write32(igb, E1000_RDT(0), 0);
+
+ /*
+ * Select the advanced one-buffer descriptor format. Per 82576
+ * datasheet section 7.1.5.2: "SRRCTL[n].DESCTYPE must be set to a
+ * value other than 000b for the 82576 to write back the special
+ * descriptors." struct igb_rx_desc matches the advanced one-buffer
+ * writeback layout (section 7.1.5.2), so polling rx.wb.status_error
+ * requires this format. Section 8.10.2 specifies DESCTYPE[27:25].
+ *
+ * The direct write also zeroes SRRCTL.BSIZEPACKET, which is
+ * intentional: per section 7.1.3.1 a zero BSIZEPACKET falls back to
+ * the RCTL.BSIZE buffer size, whose reset default (00b) is 2048
+ * bytes -- ample for the loopback frames here.
+ */
+ igb_write32(igb, E1000_SRRCTL(0), E1000_SRRCTL_DESCTYPE_ADV_ONEBUF);
+
+ igb_write32(igb, E1000_RXDCTL(0), E1000_RXDCTL_QUEUE_ENABLE);
+
+ /*
+ * Enable Receiver and Transmitter. RCTL.LBM_MAC is set in addition
+ * to PHY loopback as a QEMU-only accommodation: QEMU's emulated igb
+ * does not honor PHY register 0 bit 14 (PHY internal loopback) and
+ * relies on RCTL.LBM_MAC to wrap TX descriptors back to the RX
+ * queue. Datasheet 8.10.1 (RCTL register) advises "When using the
+ * internal PHY, LBM should remain set to 00b", so setting LBM_MAC
+ * here deviates from datasheet guidance; empirically the bit has
+ * no observable effect on real 82576 hardware because MAC loopback
+ * is not implemented (datasheet 3.5.6.2). Setting both lets the
+ * selftest work on both real hardware and QEMU without conditional
+ * code paths.
+ */
+ rctl = E1000_RCTL_EN | /* Receiver Enable */
+ E1000_RCTL_UPE | /* Unicast Promiscuous (for dummy MAC) */
+ E1000_RCTL_MPE | /* Multicast Promiscuous */
+ E1000_RCTL_BAM | /* Broadcast Accept Mode */
+ E1000_RCTL_LBM_MAC | /* MAC Loopback - for QEMU emulation only */
+ E1000_RCTL_SECRC; /* Strip CRC (needed for memcmp) */
+ igb_write32(igb, E1000_RCTL, rctl);
+ igb_write32(igb, E1000_TCTL, E1000_TCTL_EN | E1000_TCTL_PSP);
+
+ /*
+ * Wait for TX and RX queues to be enabled. Per the RXDCTL/TXDCTL
+ * register definitions (8.10.10/8.12.13), the per-queue enable bit
+ * "remains zero" until the global RCTL.RXEN/TCTL.TXEN are set, so
+ * E1000_RCTL_EN and E1000_TCTL_EN must already be written above.
+ */
+ retries = 2000;
+ while (retries-- > 0) {
+ if ((igb_read32(igb, E1000_TXDCTL(0)) & E1000_TXDCTL_QUEUE_ENABLE) &&
+ (igb_read32(igb, E1000_RXDCTL(0)) & E1000_RXDCTL_QUEUE_ENABLE))
+ break;
+ usleep(10);
+ }
+ VFIO_ASSERT_GE(retries, 0);
+
+ /*
+ * Program MSI-X interrupt routing per 82576 datasheet:
+ *
+ * GPIE (section 7.3.2.11, Table 7-47): set Multiple_MSIX (bit 4) to
+ * route interrupt causes through IVAR mapping, and EIAME (bit 30)
+ * to apply EIAM on MSI-X assertion (without EIAME, EIAM only
+ * applies on EICR read/write).
+ *
+ * EIAC (section 8.8.5): enable auto-clear of EICR for vector 0.
+ * Without auto-clear the cause stays set after delivery and the
+ * test can see spurious interrupts on the next memcpy batch.
+ *
+ * EIAM (section 8.8.6): enable auto-mask of EIMS for vector 0 on
+ * MSI-X assertion (effective because EIAME is set).
+ *
+ * IVAR (section 7.3.1.2, register definition in 8.8.13): map RX
+ * cause 0 to MSI-X vector 0 and mark the entry valid.
+ */
+ igb_write32(igb, E1000_GPIE, E1000_GPIE_MSIX_MODE | E1000_GPIE_EIAME);
+ igb_write32(igb, E1000_EIAC, MSIX_VECTOR_MASK);
+ igb_write32(igb, E1000_EIAM, MSIX_VECTOR_MASK);
+
+ /* Map vector 0 to interrupt cause 0 and mark it valid */
+ igb_write32(igb, E1000_IVAR0, E1000_IVAR_VALID);
+
+ /* Enable interrupts on vector 0 */
+ igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK);
+
+ /* Initialize driver state and capability limits */
+ igb->tx_tail = 0;
+ igb->rx_tail = 0;
+
+ device->driver.max_memcpy_size = IGB_MAX_CHUNK_SIZE;
+ device->driver.max_memcpy_count = RING_SIZE - 1;
+ device->driver.msi = MSIX_VECTOR;
+}
+
+static void igb_init(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ VFIO_ASSERT_GE(device->driver.region.size, sizeof(struct igb));
+
+ igb->bar0 = device->bars[0].vaddr;
+
+ igb_reset(igb);
+
+ /*
+ * Enable MSI-X via VFIO before device-side register programming.
+ * vfio_pci_msix_enable() only touches the VFIO IRQ machinery and the
+ * PCI MSI-X capability via config space; it has no ordering
+ * dependency on the device-side writes performed by igb_hw_init().
+ * Placing it here keeps igb_hw_init() reusable from the reset
+ * recovery path (which calls vfio_pci_irq_reenable() instead).
+ */
+ vfio_pci_msix_enable(device, MSIX_VECTOR, 1);
+
+ igb_hw_init(device);
+}
+
+static void igb_remove(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ igb_write32(igb, E1000_RCTL, 0);
+ igb_write32(igb, E1000_TCTL, 0);
+ igb_reset(igb);
+
+ vfio_pci_msix_disable(device);
+}
+
+static void igb_irq_disable(struct igb *igb)
+{
+ igb_write32(igb, E1000_EIMC, MSIX_VECTOR_MASK);
+}
+
+static void igb_irq_enable(struct igb *igb)
+{
+ igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK);
+}
+
+static void igb_irq_clear(struct igb *igb)
+{
+ /*
+ * Use write-to-clear (datasheet 7.3.4.2). In MSI-X mode with EIAC
+ * programmed, section 8.8.5 explicitly states "If any bits are set
+ * in EIAC, the EICR register should not be read", which rules out
+ * the read-to-clear path in 7.3.4.3. Bits not in EIAC are still
+ * cleared by writing 1.
+ */
+ igb_write32(igb, E1000_EICR, 0xFFFFFFFF);
+}
+
+static void igb_memcpy_start(struct vfio_pci_device *device, iova_t src,
+ iova_t dst, u64 size, u64 count)
+{
+ struct igb *igb = to_igb_state(device);
+ struct igb_rx_desc *rx;
+ struct igb_tx_desc *tx;
+ u32 i;
+
+ VFIO_ASSERT_GE(size, 60,
+ "IGB driver requires memcpy size to be at least 60 bytes (Ethernet minimum payload size)");
+
+ igb_irq_disable(igb);
+
+ for (i = 0; i < count; i++) {
+ tx = &igb->tx_ring[igb->tx_tail];
+ rx = &igb->rx_ring[igb->rx_tail];
+
+ memset(tx, 0, sizeof(struct igb_tx_desc));
+ memset(rx, 0, sizeof(struct igb_rx_desc));
+
+ rx->read.pkt_addr = cpu_to_le64(dst);
+ rx->read.hdr_addr = cpu_to_le64(0);
+
+ tx->read.buffer_addr = cpu_to_le64(src);
+ /*
+ * Build an advanced data descriptor per 82576 datasheet
+ * section 7.2.2.3. DEXT marks the descriptor as advanced
+ * (required by hardware); DTYP=data selects the data
+ * descriptor; IFCS asks the MAC to append the Ethernet
+ * FCS (without it the frame is dropped as malformed);
+ * EOP marks end of packet. DTALEN is the buffer length
+ * in bits 15:0 of cmd_type_len.
+ */
+ tx->read.cmd_type_len = cpu_to_le32((uint32_t)size |
+ E1000_ADVTXD_DTYP_DATA |
+ E1000_ADVTXD_DCMD_DEXT |
+ E1000_ADVTXD_DCMD_IFCS |
+ E1000_ADVTXD_DCMD_EOP);
+ /*
+ * PAYLEN (section 7.2.2.3.11) is the total payload size
+ * in olinfo_status[31:14].
+ */
+ tx->read.olinfo_status =
+ cpu_to_le32((uint32_t)size << E1000_ADVTXD_PAYLEN_SHIFT);
+
+ igb->tx_tail = (igb->tx_tail + 1) % RING_SIZE;
+ igb->rx_tail = (igb->rx_tail + 1) % RING_SIZE;
+ }
+
+ igb_write32(igb, E1000_RDT(0), igb->rx_tail);
+ igb_write32(igb, E1000_TDT(0), igb->tx_tail);
+}
+
+/*
+ * Reset the device via VFIO_DEVICE_RESET (PCIe FLR on the 82576) and
+ * re-program it. VFIO_DEVICE_RESET tears down the kernel-side MSI-X
+ * trigger but leaves user-side eventfds intact, so re-arm the trigger
+ * via vfio_pci_irq_reenable() before reprogramming so any caller-cached
+ * eventfd remains valid.
+ *
+ * FLR clears device-side state to power-on reset values (datasheet
+ * 4.2.1.5.1: a PF FLR is "equivalent to a D0->D3->D0 transition"), so
+ * EIMS and EICR come back as 0 from their register-defined initial
+ * values, and igb_hw_init() resets tx_tail/rx_tail to 0. The next
+ * igb_memcpy_start() will memset each descriptor it touches before
+ * submission, so no explicit IMC/EICR writes or ring memsets are
+ * needed here.
+ */
+static void igb_error_reset_and_reinit(struct vfio_pci_device *device)
+{
+ vfio_pci_device_reset(device);
+ vfio_pci_msix_reenable(device, MSIX_VECTOR, 1);
+ igb_hw_init(device);
+}
+
+static int igb_memcpy_wait(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+ struct igb_rx_desc *rx;
+ u32 status = 0;
+ u32 prev_tail;
+ int retries;
+
+ prev_tail = (igb->rx_tail + RING_SIZE - 1) % RING_SIZE;
+ rx = &igb->rx_ring[prev_tail];
+
+ /*
+ * Real 82576 hardware processes the descriptor ring at line rate.
+ * max_memcpy_size = (RING_SIZE - 1) * IGB_MAX_CHUNK_SIZE ~= 4 MB,
+ * split into 4095 1 KB frames. At 1 Gb/s (~125 MB/s) the worst
+ * valid memcpy takes ~32 ms on the wire, plus per-frame preamble,
+ * SFD, IFG and FCS overhead (~3%) and descriptor fetch/writeback
+ * latency. Wait up to ~200 ms before declaring the device hung;
+ * ~6x the line-rate floor leaves comfortable headroom for host
+ * scheduling jitter while keeping the intentional invalid-DMA
+ * tests bounded.
+ */
+ retries = 200;
+ while (retries-- > 0) {
+ status = le32_to_cpu(READ_ONCE(rx->wb.status_error));
+ if (status & 1)
+ break;
+ usleep(1000);
+ }
+
+ if (status & 1)
+ /*
+ * Ensure the test code doesn't speculatively read the DMA
+ * destination buffer before we have verified that the
+ * descriptor writeback is complete.
+ */
+ rmb();
+
+ igb_irq_clear(igb);
+
+ igb_irq_enable(igb);
+
+ if (status & 1)
+ return 0;
+
+ /*
+ * The descriptor never completed. On real 82576 hardware this
+ * typically follows a DMA-read fault from one of the intentional
+ * unmapped-IOVA tests; the fault leaves the descriptor engine
+ * unable to service subsequent valid descriptors. CTRL.RST alone
+ * reinitializes the queue registers but leaves the engine wedged
+ * for the current process, so a broader VFIO_DEVICE_RESET (FLR)
+ * is required.
+ */
+ igb_error_reset_and_reinit(device);
+
+ return -ETIMEDOUT;
+}
+
+static void igb_send_msi(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ igb_write32(igb, E1000_EICS, MSIX_VECTOR_MASK);
+}
+
+const struct vfio_pci_driver_ops igb_ops = {
+ .name = "igb",
+ .probe = igb_probe,
+ .init = igb_init,
+ .remove = igb_remove,
+ .memcpy_start = igb_memcpy_start,
+ .memcpy_wait = igb_memcpy_wait,
+ .send_msi = igb_send_msi,
+};
diff --git a/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h
new file mode 100644
index 000000000000..edce130fd008
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h
@@ -0,0 +1,352 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+ */
+#ifndef _NV_FALCON_HW_H_
+#define _NV_FALCON_HW_H_
+
+#include <linux/types.h>
+
+/* PMC (Power Management Controller) Registers */
+#define NV_PMC_BOOT_0 0x00000000
+#define NV_PMC_ENABLE 0x00000200
+#define NV_PMC_ENABLE_PWR 0x00002000
+#define NV_PMC_ENABLE_HUB 0x20000000
+
+/* Falcon Base Pages for Different Engines */
+#define NV_PPWR_FALCON_BASE 0x10a000
+#define NV_PGSP_FALCON_BASE 0x110000
+
+/* Falcon Common Register Offsets (relative to base_page) */
+#define NV_FALCON_DMACTL_OFFSET 0x010c
+#define NV_FALCON_ENGINE_RESET_OFFSET 0x03c0
+
+/* DMEM Control Register Flags */
+#define NV_PPWR_FALCON_DMEMC_AINCR_TRUE 0x01000000
+#define NV_PPWR_FALCON_DMEMC_AINCW_TRUE 0x02000000
+
+/* Falcon DMEM port offsets (for port 0) */
+#define NV_FALCON_DMEMC_OFFSET 0x1c0
+#define NV_FALCON_DMEMD_OFFSET 0x1c4
+
+/* DMA Register Offsets (relative to base_page) */
+#define NV_FALCON_DMA_ADDR_LOW_OFFSET 0x110
+#define NV_FALCON_DMA_MEM_OFFSET 0x114
+#define NV_FALCON_DMA_CMD_OFFSET 0x118
+#define NV_FALCON_DMA_BLOCK_OFFSET 0x11c
+#define NV_FALCON_DMA_ADDR_HIGH_OFFSET 0x128
+
+/* DMA Global Address Top Bits Register */
+#define NV_GPU_DMA_ADDR_TOP_BITS_REG 0x100f04
+
+/* DMA Command Register Bit Definitions */
+#define NV_FALCON_DMA_CMD_WRITE_BIT 0x20
+#define NV_FALCON_DMA_CMD_SIZE_SHIFT 8
+#define NV_FALCON_DMA_CMD_DONE_BIT 0x2
+
+/*
+ * Falcon DMA is synchronous, so a transfer size and count larger than
+ * its per-operation maximum adds no value.
+ */
+
+/* DMA block size and alignment */
+#define NV_FALCON_DMA_MIN_TRANSFER_SIZE 4
+#define NV_FALCON_DMA_MAX_TRANSFER_SIZE 256
+#define NV_FALCON_DMA_BLOCK_SIZE 256
+#define NV_FALCON_DMA_MAX_TRANSFER_COUNT 1
+
+/* DMACTL register bits */
+#define NV_FALCON_DMACTL_DMEM_SCRUBBING 0x1
+#define NV_FALCON_DMACTL_READY_MASK 0x6
+
+/* Falcon Core Selection Register */
+#define NV_FALCON_CORE_SELECT_OFFSET 0x1668
+#define NV_FALCON_CORE_SELECT_MASK 0x30
+
+/* Falcon mailbox register (for Ada+ reset check) */
+#define NV_FALCON_MAILBOX_TEST_OFFSET 0x40c
+#define NV_FALCON_MAILBOX_RESET_MAGIC 0xbadf5620
+
+/* Falcon Message Queue Register Offsets (relative to base_page) */
+#define NV_FALCON_QUEUE_HEAD_BASE_OFFSET 0x2c00
+#define NV_FALCON_QUEUE_TAIL_BASE_OFFSET 0x2c04
+#define NV_FALCON_QUEUE_STRIDE 0x8
+#define NV_FALCON_MSG_QUEUE_HEAD_BASE_OFFSET 0x2c80
+#define NV_FALCON_MSG_QUEUE_TAIL_BASE_OFFSET 0x2c84
+
+/* FSP Falcon Base Pages */
+#define NV_FSP_FALCON_BASE 0x8f0100
+/* base_page = cpuctl & ~0xfff */
+#define NV_FSP_FALCON_BASE_PAGE 0x8f0000
+#define NV_FSP_EMEM_BASE 0x8f2000
+
+/* FSP EMEM Port Offsets (relative to FSP EMEM base) */
+#define NV_FSP_EMEMC_OFFSET 0xac0
+#define NV_FSP_EMEMD_OFFSET 0xac4
+#define NV_FSP_EMEM_PORT_STRIDE 0x8
+
+/* EMEM Control Register Flags (same as DMEM) */
+#define NV_FALCON_EMEMC_AINCR 0x01000000
+#define NV_FALCON_EMEMC_AINCW 0x02000000
+
+/* FSP RPC channel configuration */
+#define NV_FSP_RPC_CHANNEL_SIZE 1024
+#define NV_FSP_RPC_MAX_PACKET_SIZE 1024
+#define NV_FSP_RPC_CHANNEL_HOPPER 2
+#define NV_FSP_RPC_EMEM_BASE \
+ (NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_RPC_CHANNEL_SIZE)
+
+/* FSP EMEM port 2 registers (pre-computed for Hopper channel 2) */
+#define NV_FSP_EMEM_PORT2_CTRL (NV_FSP_EMEM_BASE + NV_FSP_EMEMC_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_EMEM_PORT_STRIDE)
+#define NV_FSP_EMEM_PORT2_DATA (NV_FSP_EMEM_BASE + NV_FSP_EMEMD_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_EMEM_PORT_STRIDE)
+
+/* FSP queue register offsets (pre-computed for Hopper channel 2) */
+#define NV_FSP_QUEUE_HEAD \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_QUEUE_HEAD_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+#define NV_FSP_QUEUE_TAIL \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_QUEUE_TAIL_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+#define NV_FSP_MSG_QUEUE_HEAD \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_MSG_QUEUE_HEAD_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+#define NV_FSP_MSG_QUEUE_TAIL \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_MSG_QUEUE_TAIL_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+
+/* MCTP Header */
+#define NV_MCTP_HDR_SEID_SHIFT 16
+#define NV_MCTP_HDR_SEID_MASK 0xff
+#define NV_MCTP_HDR_SEQ_SHIFT 28
+#define NV_MCTP_HDR_SEQ_MASK 0x3
+#define NV_MCTP_HDR_EOM_BIT 0x40000000
+#define NV_MCTP_HDR_SOM_BIT 0x80000000
+
+/* MCTP Message Header */
+#define NV_MCTP_MSG_TYPE_SHIFT 0
+#define NV_MCTP_MSG_TYPE_MASK 0x7f
+#define NV_MCTP_MSG_TYPE_VENDOR_DEFINED 0x7e
+#define NV_MCTP_MSG_VENDOR_ID_SHIFT 8
+#define NV_MCTP_MSG_VENDOR_ID_MASK 0xffff
+#define NV_MCTP_MSG_VENDOR_ID_NVIDIA 0x10de
+#define NV_MCTP_MSG_NVDM_TYPE_SHIFT 24
+#define NV_MCTP_MSG_NVDM_TYPE_MASK 0xff
+
+/* NVDM response type */
+#define NV_NVDM_TYPE_RESPONSE 0x15
+
+/* Minimum response size: mctp_hdr + msg_hdr + status_hdr + type + status */
+#define NV_FSP_RPC_MIN_RESPONSE_WORDS 5
+
+/* FBIF (Frame Buffer Interface) Registers */
+/* Legacy PMU FBIF offsets (Kepler, Maxwell Gen1) */
+#define NV_PMU_LEGACY_FBIF_CTL_OFFSET 0x624
+#define NV_PMU_LEGACY_FBIF_TRANSCFG_OFFSET 0x600
+
+/* PMU FBIF offsets */
+#define NV_PMU_FBIF_CTL_OFFSET 0xe24
+#define NV_PMU_FBIF_TRANSCFG_OFFSET 0xe00
+
+/* GSP FBIF offsets */
+#define NV_GSP_FBIF_CTL_OFFSET 0x624
+#define NV_GSP_FBIF_TRANSCFG_OFFSET 0x600
+
+/* OFA Falcon Base Page and FBIF offsets (used for Hopper+ DMA) */
+#define NV_OFA_FALCON_BASE 0x844000
+#define NV_OFA_FBIF_CTL_OFFSET 0x424
+#define NV_OFA_FBIF_TRANSCFG_OFFSET 0x400
+
+/* OFA DMA support check register (Hopper+) */
+#define NV_OFA_DMA_SUPPORT_CHECK_REG 0x8443c0
+
+/* FSP NVDM command types */
+#define NV_NVDM_TYPE_FBDMA 0x22
+#define NV_FBDMA_SUBCMD_ENABLE 0x1
+
+/* FBIF CTL2 offset (relative to fbif_ctl) */
+#define NV_FBIF_CTL2_OFFSET 0x60
+
+/* FBIF TRANSCFG register bits */
+#define NV_FBIF_TRANSCFG_TARGET_MASK 0x3
+#define NV_FBIF_TRANSCFG_SYSMEM_DEFAULT 0x5
+
+/* FBIF CTL register bits */
+#define NV_FBIF_CTL_ALLOW_PHYS_MODE 0x10
+#define NV_FBIF_CTL_ALLOW_FULL_PHYS_MODE 0x80
+
+/* Memory clear register offsets */
+#define NV_MEM_CLEAR_OFFSET 0x100b20
+#define NV_BOOT_COMPLETE_OFFSET 0x118234
+#define NV_BOOT_COMPLETE_SUCCESS 0x3ff
+
+/* FSP boot complete register (Hopper+) */
+#define NV_FSP_BOOT_COMPLETE_OFFSET 0x200bc
+#define NV_FSP_BOOT_COMPLETE_SUCCESS 0xff
+
+enum gpu_arch {
+ GPU_ARCH_UNKNOWN = -1,
+ GPU_ARCH_KEPLER = 0,
+ GPU_ARCH_MAXWELL_GEN1,
+ GPU_ARCH_MAXWELL_GEN2,
+ GPU_ARCH_PASCAL,
+ GPU_ARCH_PASCAL_10X,
+ GPU_ARCH_VOLTA,
+ GPU_ARCH_TURING,
+ GPU_ARCH_AMPERE,
+ GPU_ARCH_ADA,
+ GPU_ARCH_HOPPER,
+};
+
+enum falcon_type {
+ FALCON_TYPE_PMU_LEGACY = 0,
+ FALCON_TYPE_PMU,
+ FALCON_TYPE_GSP,
+ FALCON_TYPE_OFA,
+};
+
+struct falcon {
+ u32 base_page;
+ u32 dmactl;
+ u32 engine_reset;
+ u32 fbif_ctl;
+ u32 fbif_ctl2;
+ u32 fbif_transcfg;
+ u32 dmem_control_reg;
+ u32 dmem_data_reg;
+ bool no_outside_reset;
+};
+
+struct gpu_properties {
+ u32 pmc_enable_mask;
+ bool memory_clear_supported;
+ enum falcon_type falcon_type;
+};
+
+static const u32 verified_gpu_map[] = {
+ 0x0e40a0a2, /* K520 */
+ 0x0e6000a1, /* GTX660 */
+ 0x0e63a0a1, /* K4000 */
+ 0x0f22d0a1, /* K80 */
+ 0x108000a1, /* GT635 */
+ 0x117010a2, /* GTX750 */
+ 0x117020a2, /* GTX745 */
+ 0x124320a1, /* M60 */
+ 0x130000a1, /* P100 */
+ 0x134000a1, /* P4 */
+ 0x132000a1, /* P40 */
+ 0x140000a1, /* V100 */
+ 0x164000a1, /* T4 */
+ 0xb77000a1, /* A16 */
+ 0x170000a1, /* A100 */
+ 0xb72000a1, /* A10 */
+ 0x180000a1, /* H100 */
+ 0x194000a1, /* L4 */
+ 0x192000a1, /* L40S */
+};
+
+#define VERIFIED_GPU_MAP_SIZE ARRAY_SIZE(verified_gpu_map)
+
+static const struct gpu_properties gpu_properties_map[] = {
+ [GPU_ARCH_KEPLER] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR | NV_PMC_ENABLE_HUB,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU_LEGACY,
+ },
+ [GPU_ARCH_MAXWELL_GEN1] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR | NV_PMC_ENABLE_HUB,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU_LEGACY,
+ },
+ [GPU_ARCH_MAXWELL_GEN2] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_PASCAL] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_PASCAL_10X] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_VOLTA] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_GSP,
+ },
+ [GPU_ARCH_TURING] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_GSP,
+ },
+ [GPU_ARCH_AMPERE] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_GSP,
+ },
+ [GPU_ARCH_ADA] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_HOPPER] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_OFA,
+ },
+};
+
+static const struct falcon falcon_map[] = {
+ [FALCON_TYPE_PMU_LEGACY] = {
+ .base_page = NV_PPWR_FALCON_BASE,
+ .dmactl = NV_PPWR_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_PPWR_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_PPWR_FALCON_BASE + NV_PMU_LEGACY_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_PPWR_FALCON_BASE +
+ NV_PMU_LEGACY_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_PPWR_FALCON_BASE + NV_PMU_LEGACY_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = false,
+ },
+ [FALCON_TYPE_PMU] = {
+ .base_page = NV_PPWR_FALCON_BASE,
+ .dmactl = NV_PPWR_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_PPWR_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = false,
+ },
+ [FALCON_TYPE_GSP] = {
+ .base_page = NV_PGSP_FALCON_BASE,
+ .dmactl = NV_PGSP_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_PGSP_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_PGSP_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_PGSP_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = false,
+ },
+ [FALCON_TYPE_OFA] = {
+ .base_page = NV_OFA_FALCON_BASE,
+ .dmactl = NV_OFA_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_OFA_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_OFA_FALCON_BASE + NV_OFA_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_OFA_FALCON_BASE + NV_OFA_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_OFA_FALCON_BASE + NV_OFA_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_OFA_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_OFA_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = true,
+ },
+};
+
+#endif /* _NV_FALCON_HW_H_ */
diff --git a/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c
new file mode 100644
index 000000000000..c08aa81c44f4
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c
@@ -0,0 +1,783 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+ */
+#include <stdint.h>
+#include <strings.h>
+#include <unistd.h>
+#include <stdbool.h>
+#include <string.h>
+#include <time.h>
+
+#include <linux/errno.h>
+#include <linux/io.h>
+#include <linux/pci_ids.h>
+
+#include <libvfio.h>
+
+#include "hw.h"
+
+struct gpu_device {
+ enum gpu_arch arch;
+ void *bar0;
+ bool is_memory_clear_supported;
+ const struct falcon *falcon;
+ u32 pmc_enable_mask;
+ bool fsp_dma_enabled;
+
+ /* Pending memcpy parameters, set by memcpy_start() */
+ u64 memcpy_src;
+ u64 memcpy_dst;
+ u64 memcpy_size;
+};
+
+static inline struct gpu_device *to_gpu_device(struct vfio_pci_device *device)
+{
+ return device->driver.region.vaddr;
+}
+
+static enum gpu_arch nv_gpu_arch_lookup(u32 pmc_boot_0)
+{
+ u32 arch = (pmc_boot_0 >> 24) & 0x1f;
+
+ switch (arch) {
+ case 0x0e:
+ case 0x0f:
+ case 0x10:
+ return GPU_ARCH_KEPLER;
+ case 0x11:
+ return GPU_ARCH_MAXWELL_GEN1;
+ case 0x12:
+ return GPU_ARCH_MAXWELL_GEN2;
+ case 0x13:
+ /* P100 (impl 0) uses PMC reset; P4/P40 use engine reset */
+ if (((pmc_boot_0 >> 20) & 0xf) == 0)
+ return GPU_ARCH_PASCAL;
+ return GPU_ARCH_PASCAL_10X;
+ case 0x14:
+ return GPU_ARCH_VOLTA;
+ case 0x16:
+ return GPU_ARCH_TURING;
+ case 0x17:
+ return GPU_ARCH_AMPERE;
+ case 0x18:
+ return GPU_ARCH_HOPPER;
+ case 0x19:
+ return GPU_ARCH_ADA;
+ default:
+ return GPU_ARCH_UNKNOWN;
+ }
+}
+
+static inline u32 gpu_read32(struct gpu_device *gpu, u32 offset)
+{
+ return readl(gpu->bar0 + offset);
+}
+
+static inline void gpu_write32(struct gpu_device *gpu, u32 offset, u32 value)
+{
+ writel(value, gpu->bar0 + offset);
+}
+
+static u64 get_elapsed_ms(struct timespec *start)
+{
+ struct timespec now;
+
+ clock_gettime(CLOCK_MONOTONIC, &now);
+
+ return (now.tv_sec - start->tv_sec) * 1000
+ + (now.tv_nsec - start->tv_nsec) / 1000000;
+}
+
+static int gpu_poll_register(struct vfio_pci_device *device,
+ const char *name, u32 offset,
+ u32 expected, u32 mask, u32 timeout_ms)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ struct timespec start;
+ u64 elapsed_ms;
+ u32 value;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+
+ for (;;) {
+ value = gpu_read32(gpu, offset);
+ if ((value & mask) == expected)
+ return 0;
+
+ elapsed_ms = get_elapsed_ms(&start);
+
+ if (elapsed_ms >= timeout_ms)
+ break;
+
+ usleep(1000);
+ }
+
+ dev_err(device,
+ "Timeout polling %s (0x%x): value=0x%x expected=0x%x mask=0x%x after %lu ms\n",
+ name, offset, value, expected, mask, elapsed_ms);
+ return -ETIMEDOUT;
+}
+
+static int fsp_poll_queue(struct vfio_pci_device *device, const char *name,
+ u32 head_reg, u32 tail_reg, bool wait_empty,
+ u32 timeout_ms)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ struct timespec start;
+ u64 elapsed_ms;
+ u32 head, tail;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+
+ for (;;) {
+ head = gpu_read32(gpu, head_reg);
+ tail = gpu_read32(gpu, tail_reg);
+ if (wait_empty ? (head == tail) : (head != tail))
+ return 0;
+
+ elapsed_ms = get_elapsed_ms(&start);
+
+ if (elapsed_ms >= timeout_ms)
+ break;
+
+ usleep(1000);
+ }
+
+ dev_err(device,
+ "Timeout polling %s: head=0x%x tail=0x%x wait_empty=%d after %lu ms\n",
+ name, head, tail, wait_empty, elapsed_ms);
+ return -ETIMEDOUT;
+}
+
+static void fsp_emem_write(struct vfio_pci_device *device, u32 offset,
+ const u32 *data, u32 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 i;
+
+ /* Configure port with auto-increment for read and write */
+ gpu_write32(gpu, NV_FSP_EMEM_PORT2_CTRL,
+ offset | NV_FALCON_EMEMC_AINCR | NV_FALCON_EMEMC_AINCW);
+
+ for (i = 0; i < count; i++)
+ gpu_write32(gpu, NV_FSP_EMEM_PORT2_DATA, data[i]);
+}
+
+static void fsp_emem_read(struct vfio_pci_device *device, u32 offset,
+ u32 *data, u32 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 i;
+
+ /* Configure port with auto-increment for read and write */
+ gpu_write32(gpu, NV_FSP_EMEM_PORT2_CTRL,
+ offset | NV_FALCON_EMEMC_AINCR | NV_FALCON_EMEMC_AINCW);
+
+ for (i = 0; i < count; i++)
+ data[i] = gpu_read32(gpu, NV_FSP_EMEM_PORT2_DATA);
+}
+
+static int fsp_rpc_send_data(struct vfio_pci_device *device, const u32 *data,
+ u32 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ int ret;
+
+ ret = fsp_poll_queue(device, "fsp_cmd_queue_empty",
+ NV_FSP_QUEUE_HEAD, NV_FSP_QUEUE_TAIL, true, 1000);
+ if (ret)
+ return ret;
+
+ fsp_emem_write(device, NV_FSP_RPC_EMEM_BASE, data, count);
+
+ /* Update queue head/tail to signal data is ready */
+ gpu_write32(gpu, NV_FSP_QUEUE_TAIL,
+ NV_FSP_RPC_EMEM_BASE + (count - 1) * 4);
+ gpu_write32(gpu, NV_FSP_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE);
+
+ return ret;
+}
+
+static int fsp_rpc_receive_data(struct vfio_pci_device *device, u32 *data,
+ u32 max_count, u32 timeout_ms)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 head, tail;
+ u32 msg_size_words;
+ int ret;
+
+ ret = fsp_poll_queue(device, "fsp_msg_queue_ready",
+ NV_FSP_MSG_QUEUE_HEAD, NV_FSP_MSG_QUEUE_TAIL,
+ false, timeout_ms);
+ if (ret)
+ return ret;
+
+ head = gpu_read32(gpu, NV_FSP_MSG_QUEUE_HEAD);
+ tail = gpu_read32(gpu, NV_FSP_MSG_QUEUE_TAIL);
+
+ msg_size_words = (tail - head + 4) / 4;
+ if (msg_size_words > max_count)
+ msg_size_words = max_count;
+
+ fsp_emem_read(device, NV_FSP_RPC_EMEM_BASE, data, msg_size_words);
+
+ /* Reset message queue tail to acknowledge receipt */
+ gpu_write32(gpu, NV_FSP_MSG_QUEUE_TAIL, head);
+
+ return msg_size_words;
+}
+
+static void fsp_reset_rpc_state(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 head, tail;
+
+ head = gpu_read32(gpu, NV_FSP_QUEUE_HEAD);
+ tail = gpu_read32(gpu, NV_FSP_QUEUE_TAIL);
+
+ if (head == tail) {
+ head = gpu_read32(gpu, NV_FSP_MSG_QUEUE_HEAD);
+ tail = gpu_read32(gpu, NV_FSP_MSG_QUEUE_TAIL);
+ if (head == tail)
+ return;
+ }
+
+ /* Best-effort drain; timeout is expected if no pending message. */
+ fsp_poll_queue(device, "fsp_msg_queue_drain",
+ NV_FSP_MSG_QUEUE_HEAD, NV_FSP_MSG_QUEUE_TAIL,
+ false, 5000);
+
+ gpu_write32(gpu, NV_FSP_QUEUE_TAIL, NV_FSP_RPC_EMEM_BASE);
+ gpu_write32(gpu, NV_FSP_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE);
+ gpu_write32(gpu, NV_FSP_MSG_QUEUE_TAIL, NV_FSP_RPC_EMEM_BASE);
+ gpu_write32(gpu, NV_FSP_MSG_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE);
+}
+
+static inline u32 mctp_header_build(u8 seid, u8 seq, bool som, bool eom)
+{
+ u32 hdr = 0;
+
+ hdr |= (seid & NV_MCTP_HDR_SEID_MASK) << NV_MCTP_HDR_SEID_SHIFT;
+ hdr |= (seq & NV_MCTP_HDR_SEQ_MASK) << NV_MCTP_HDR_SEQ_SHIFT;
+ if (som)
+ hdr |= NV_MCTP_HDR_SOM_BIT;
+ if (eom)
+ hdr |= NV_MCTP_HDR_EOM_BIT;
+
+ return hdr;
+}
+
+static inline u32 mctp_msg_header_build(u8 nvdm_type)
+{
+ u32 hdr = 0;
+
+ hdr |= (NV_MCTP_MSG_TYPE_VENDOR_DEFINED & NV_MCTP_MSG_TYPE_MASK)
+ << NV_MCTP_MSG_TYPE_SHIFT;
+ hdr |= (NV_MCTP_MSG_VENDOR_ID_NVIDIA & NV_MCTP_MSG_VENDOR_ID_MASK)
+ << NV_MCTP_MSG_VENDOR_ID_SHIFT;
+ hdr |= (nvdm_type & NV_MCTP_MSG_NVDM_TYPE_MASK)
+ << NV_MCTP_MSG_NVDM_TYPE_SHIFT;
+
+ return hdr;
+}
+
+static inline u8 mctp_msg_header_get_nvdm_type(u32 hdr)
+{
+ return (hdr >> NV_MCTP_MSG_NVDM_TYPE_SHIFT) &
+ NV_MCTP_MSG_NVDM_TYPE_MASK;
+}
+
+static int fsp_rpc_send_cmd(struct vfio_pci_device *device, u8 nvdm_type,
+ const u32 *data, u32 data_count, u32 timeout_ms)
+{
+ u32 max_packet_words = NV_FSP_RPC_MAX_PACKET_SIZE / 4;
+ u32 packet[256];
+ u32 resp_buf[256];
+ u32 total_words;
+ int resp_words;
+ u8 resp_nvdm_type;
+ int ret;
+
+ total_words = 2 + data_count;
+ if (total_words > max_packet_words)
+ return -EINVAL;
+
+ packet[0] = mctp_header_build(0, 0, true, true);
+ packet[1] = mctp_msg_header_build(nvdm_type);
+
+ if (data_count > 0)
+ memcpy(&packet[2], data, data_count * sizeof(u32));
+
+ ret = fsp_rpc_send_data(device, packet, total_words);
+ if (ret)
+ return ret;
+
+ resp_words = fsp_rpc_receive_data(device, resp_buf, 256, timeout_ms);
+ if (resp_words < 0)
+ return resp_words;
+
+ if (resp_words < NV_FSP_RPC_MIN_RESPONSE_WORDS)
+ return -EPROTO;
+
+ resp_nvdm_type = mctp_msg_header_get_nvdm_type(resp_buf[1]);
+ if (resp_nvdm_type != NV_NVDM_TYPE_RESPONSE)
+ return -EPROTO;
+
+ if (resp_buf[3] != nvdm_type)
+ return -EPROTO;
+
+ if (resp_buf[4] != 0)
+ return -resp_buf[4];
+
+ return 0;
+}
+
+static int fsp_init(struct vfio_pci_device *device)
+{
+ int ret;
+
+ ret = gpu_poll_register(device, "fsp_boot_complete",
+ NV_FSP_BOOT_COMPLETE_OFFSET,
+ NV_FSP_BOOT_COMPLETE_SUCCESS, 0xffffffff, 5000);
+ if (ret)
+ return ret;
+
+ fsp_reset_rpc_state(device);
+ return ret;
+}
+
+static int fsp_fbdma_enable(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 cmd_data = NV_FBDMA_SUBCMD_ENABLE;
+ int ret = 0;
+
+ if (gpu->fsp_dma_enabled)
+ return ret;
+
+ ret = fsp_rpc_send_cmd(device, NV_NVDM_TYPE_FBDMA, &cmd_data, 1, 5000);
+ if (ret)
+ return ret;
+
+ gpu->fsp_dma_enabled = true;
+ return ret;
+}
+
+static bool fsp_check_ofa_dma_support(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 val = gpu_read32(gpu, NV_OFA_DMA_SUPPORT_CHECK_REG);
+
+ return (val >> 16) != 0xbadf;
+}
+
+static u32 size_to_dma_encoding(u64 size)
+{
+ VFIO_ASSERT_LE(size, NV_FALCON_DMA_MAX_TRANSFER_SIZE);
+ VFIO_ASSERT_GE(size, NV_FALCON_DMA_MIN_TRANSFER_SIZE);
+ VFIO_ASSERT_EQ(size & (size - 1), 0, "size must be power-of-2\n");
+
+ return ffs(size) - 3;
+}
+
+static void falcon_dmem_port_configure(struct vfio_pci_device *device,
+ u32 offset, bool auto_inc_read,
+ bool auto_inc_write)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 memc_value = offset;
+
+ /* Set auto-increment flags */
+ if (auto_inc_read)
+ memc_value |= NV_PPWR_FALCON_DMEMC_AINCR_TRUE;
+ if (auto_inc_write)
+ memc_value |= NV_PPWR_FALCON_DMEMC_AINCW_TRUE;
+
+ gpu_write32(gpu, falcon->dmem_control_reg, memc_value);
+}
+
+static void falcon_select_core_falcon(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 core_select_reg = falcon->base_page + NV_FALCON_CORE_SELECT_OFFSET;
+ u32 core_select;
+
+ core_select = gpu_read32(gpu, core_select_reg);
+
+ /* Clear bits 4:5 to select falcon core (not RISCV) */
+ core_select &= ~NV_FALCON_CORE_SELECT_MASK;
+
+ gpu_write32(gpu, core_select_reg, core_select);
+}
+
+static int falcon_enable(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 mailbox_test_reg;
+ u32 mailbox_val;
+
+ if (falcon->no_outside_reset)
+ return 0;
+
+ /* Ada-specific: Check if falcon needs reset before enable */
+ if (gpu->arch == GPU_ARCH_ADA) {
+ mailbox_test_reg = falcon->base_page +
+ NV_FALCON_MAILBOX_TEST_OFFSET;
+ mailbox_val = gpu_read32(gpu, mailbox_test_reg);
+ if (mailbox_val == NV_FALCON_MAILBOX_RESET_MAGIC)
+ gpu_write32(gpu, falcon->engine_reset, 1);
+ }
+
+ /* Enable the falcon based on control method */
+ if (gpu->pmc_enable_mask != 0) {
+ u32 pmc_enable;
+
+ /* Enable via PMC_ENABLE register */
+ pmc_enable = gpu_read32(gpu, NV_PMC_ENABLE);
+ gpu_write32(gpu, NV_PMC_ENABLE,
+ pmc_enable | gpu->pmc_enable_mask);
+ } else {
+ /* Enable by deasserting engine reset */
+ gpu_write32(gpu, falcon->engine_reset, 0);
+ }
+
+ if (gpu->arch < GPU_ARCH_HOPPER) {
+ falcon_select_core_falcon(device);
+
+ /* Wait for DMACTL to be ready (bits 1:2 should be 0) */
+ return gpu_poll_register(device, "falcon_dmactl",
+ falcon->dmactl, 0,
+ NV_FALCON_DMACTL_READY_MASK, 1000);
+ }
+
+ return 0;
+}
+
+static void falcon_disable(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 pmc_enable;
+
+ if (falcon->no_outside_reset)
+ return;
+
+ if (gpu->pmc_enable_mask != 0) {
+ /* Disable via PMC_ENABLE */
+ pmc_enable = gpu_read32(gpu, NV_PMC_ENABLE);
+ gpu_write32(gpu, NV_PMC_ENABLE,
+ pmc_enable & ~gpu->pmc_enable_mask);
+ } else {
+ /* Disable by asserting engine reset */
+ gpu_write32(gpu, falcon->engine_reset, 1);
+ }
+}
+
+static int falcon_reset(struct vfio_pci_device *device)
+{
+ falcon_disable(device);
+
+ return falcon_enable(device);
+}
+
+static int nv_falcon_dma_init(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon;
+ u32 transcfg;
+ u32 dmactl;
+ u32 ctl;
+ int ret = 0;
+
+ falcon = gpu->falcon;
+
+ vfio_pci_cmd_set(device, PCI_COMMAND_MASTER);
+
+ if (gpu->arch >= GPU_ARCH_HOPPER) {
+ ret = fsp_init(device);
+ if (ret) {
+ dev_err(device, "Failed to init FSP: %d\n", ret);
+ return ret;
+ }
+
+ ret = fsp_fbdma_enable(device);
+ if (ret) {
+ dev_err(device,
+ "Failed to enable FSP FBDMA: %d\n", ret);
+ return ret;
+ }
+
+ if (!fsp_check_ofa_dma_support(device)) {
+ dev_err(device,
+ "OFA DMA not supported with current firmware\n");
+ return -EOPNOTSUPP;
+ }
+ }
+
+ if (gpu->is_memory_clear_supported) {
+ /* For Turing+, wait for boot to complete first */
+ if (gpu->arch >= GPU_ARCH_TURING) {
+ /* Wait for boot complete - Hopper+ uses FSP register */
+ if (gpu->arch >= GPU_ARCH_HOPPER) {
+ ret = gpu_poll_register(device,
+ "fsp_boot_complete",
+ NV_FSP_BOOT_COMPLETE_OFFSET,
+ NV_FSP_BOOT_COMPLETE_SUCCESS,
+ 0xffffffff, 5000);
+ } else {
+ ret = gpu_poll_register(device,
+ "boot_complete",
+ NV_BOOT_COMPLETE_OFFSET,
+ NV_BOOT_COMPLETE_SUCCESS,
+ 0xffffffff, 5000);
+ }
+ if (ret)
+ return ret;
+
+ ret = gpu_poll_register(device,
+ "memory_clear_finished",
+ NV_MEM_CLEAR_OFFSET, 0x1, 0xffffffff, 5000);
+ if (ret)
+ return ret;
+ }
+ }
+
+ ret = falcon_reset(device);
+ if (ret)
+ return ret;
+
+ falcon_dmem_port_configure(device, 0, false, false);
+
+ transcfg = gpu_read32(gpu, falcon->fbif_transcfg);
+ transcfg &= ~NV_FBIF_TRANSCFG_TARGET_MASK;
+ transcfg |= NV_FBIF_TRANSCFG_SYSMEM_DEFAULT;
+ gpu_write32(gpu, falcon->fbif_transcfg, transcfg);
+
+ gpu_write32(gpu, falcon->fbif_ctl2, 0x1);
+
+ ctl = gpu_read32(gpu, falcon->fbif_ctl);
+ ctl |= NV_FBIF_CTL_ALLOW_PHYS_MODE | NV_FBIF_CTL_ALLOW_FULL_PHYS_MODE;
+ gpu_write32(gpu, falcon->fbif_ctl, ctl);
+
+ dmactl = gpu_read32(gpu, falcon->dmactl);
+ dmactl &= ~NV_FALCON_DMACTL_DMEM_SCRUBBING;
+ gpu_write32(gpu, falcon->dmactl, dmactl);
+
+ return ret;
+}
+
+static int nv_falcon_dma(struct vfio_pci_device *device,
+ u64 address, u64 size,
+ bool write)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 dma_cmd;
+ int ret;
+
+ gpu_write32(gpu, NV_GPU_DMA_ADDR_TOP_BITS_REG,
+ (address >> 47) & 0x1ffff);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_ADDR_HIGH_OFFSET,
+ (address >> 40) & 0x7f);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_ADDR_LOW_OFFSET,
+ (address >> 8) & 0xffffffff);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_BLOCK_OFFSET,
+ address & 0xff);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_MEM_OFFSET, 0);
+
+ dma_cmd = size_to_dma_encoding(size) << NV_FALCON_DMA_CMD_SIZE_SHIFT;
+
+ /* Set direction: write (DMEM->mem) or read (mem->DMEM) */
+ if (write)
+ dma_cmd |= NV_FALCON_DMA_CMD_WRITE_BIT;
+
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_CMD_OFFSET, dma_cmd);
+
+ ret = gpu_poll_register(device, "dma_done",
+ falcon->base_page + NV_FALCON_DMA_CMD_OFFSET,
+ NV_FALCON_DMA_CMD_DONE_BIT,
+ NV_FALCON_DMA_CMD_DONE_BIT, 1000);
+ if (ret)
+ dev_err(device, "Failed DMA %s (addr=0x%lx, size=%lu)\n",
+ write ? "write" : "read", address, size);
+
+ return ret;
+}
+
+static int nv_falcon_memcpy_chunk(struct vfio_pci_device *device,
+ iova_t src, iova_t dst, u64 size)
+{
+ int ret;
+
+ ret = nv_falcon_dma(device, src, size, false);
+ if (ret)
+ return ret;
+
+ return nv_falcon_dma(device, dst, size, true);
+}
+
+static int nv_falcon_probe(struct vfio_pci_device *device)
+{
+ enum gpu_arch gpu_arch;
+ u32 pmc_boot_0;
+ void *bar0;
+ int i;
+
+ if (vfio_pci_config_readw(device, PCI_VENDOR_ID) !=
+ PCI_VENDOR_ID_NVIDIA)
+ return -ENODEV;
+
+ if (vfio_pci_config_readw(device, PCI_CLASS_DEVICE) >> 8 !=
+ PCI_BASE_CLASS_DISPLAY)
+ return -ENODEV;
+
+ /* Get BAR0 pointer for reading GPU registers */
+ bar0 = device->bars[0].vaddr;
+ if (!bar0)
+ return -ENODEV;
+
+ /* Read PMC_BOOT_0 register from BAR0 to identify GPU */
+ pmc_boot_0 = readl(bar0 + NV_PMC_BOOT_0);
+
+ /* Look up GPU architecture to verify this is a supported GPU */
+ gpu_arch = nv_gpu_arch_lookup(pmc_boot_0);
+ if (gpu_arch == GPU_ARCH_UNKNOWN) {
+ dev_err(device,
+ "Unsupported GPU architecture for PMC_BOOT_0: 0x%x\n",
+ pmc_boot_0);
+ return -ENODEV;
+ }
+
+ /* Check verified GPU map */
+ for (i = 0; i < VERIFIED_GPU_MAP_SIZE; i++) {
+ if (verified_gpu_map[i] == pmc_boot_0)
+ return 0;
+ }
+
+ dev_info(device,
+ "Unvalidated GPU: PMC_BOOT_0: 0x%x, possibly not supported\n",
+ pmc_boot_0);
+
+ return 0;
+}
+
+static void nv_falcon_init(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct gpu_properties *props;
+ u32 pmc_boot_0;
+ int ret;
+
+ VFIO_ASSERT_GE(device->driver.region.size, sizeof(*gpu));
+
+ /* Read PMC_BOOT_0 register from BAR0 to identify GPU */
+ pmc_boot_0 = readl(device->bars[0].vaddr + NV_PMC_BOOT_0);
+
+ /* Look up GPU architecture */
+ gpu->arch = nv_gpu_arch_lookup(pmc_boot_0);
+
+ props = &gpu_properties_map[gpu->arch];
+
+ /* Populate GPU structure */
+ gpu->bar0 = device->bars[0].vaddr;
+ gpu->is_memory_clear_supported = props->memory_clear_supported;
+ gpu->falcon = &falcon_map[props->falcon_type];
+ gpu->pmc_enable_mask = props->pmc_enable_mask;
+
+ /* Initialize falcon for DMA */
+ ret = nv_falcon_dma_init(device);
+ VFIO_ASSERT_EQ(ret, 0, "Failed to initialize falcon DMA: %d\n", ret);
+
+ device->driver.max_memcpy_size = NV_FALCON_DMA_MAX_TRANSFER_SIZE;
+ device->driver.max_memcpy_count = NV_FALCON_DMA_MAX_TRANSFER_COUNT;
+}
+
+static void nv_falcon_remove(struct vfio_pci_device *device)
+{
+ falcon_disable(device);
+ vfio_pci_cmd_clear(device, PCI_COMMAND_MASTER);
+}
+
+/*
+ * Falcon DMA can only process one transfer at a time,
+ * so the actual work is deferred to memcpy_wait() to conform to the
+ * memcpy_start()/memcpy_wait() contract.
+ */
+static void nv_falcon_memcpy_start(struct vfio_pci_device *device,
+ iova_t src, iova_t dst, u64 size, u64 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+
+ VFIO_ASSERT_EQ(count, 1);
+ VFIO_ASSERT_EQ(size & (NV_FALCON_DMA_MIN_TRANSFER_SIZE - 1), 0,
+ "size 0x%lx must be %u-byte aligned\n",
+ (unsigned long)size, NV_FALCON_DMA_MIN_TRANSFER_SIZE);
+
+ gpu->memcpy_src = src;
+ gpu->memcpy_dst = dst;
+ gpu->memcpy_size = size;
+}
+
+/*
+ * Return the largest power-of-2 bytes we can transfer from @addr
+ * without crossing a DMA block boundary.
+ */
+static u64 dma_block_remain(u64 addr)
+{
+ u64 offset = addr & (NV_FALCON_DMA_BLOCK_SIZE - 1);
+
+ if (!offset)
+ return NV_FALCON_DMA_BLOCK_SIZE;
+
+ /* Lowest set bit of the offset is the largest aligned chunk */
+ return 1ULL << (ffs(offset) - 1);
+}
+
+static u64 rounddown_pow_of_two(u64 x)
+{
+ return 1ULL << (63 - __builtin_clzll(x));
+}
+
+static int nv_falcon_memcpy_wait(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ iova_t src = gpu->memcpy_src;
+ iova_t dst = gpu->memcpy_dst;
+ u64 remaining = gpu->memcpy_size;
+ int ret = 0;
+
+ /*
+ * Falcon DMA supports power-of-2 transfer sizes in [4, 256] and
+ * cannot cross 256-byte block boundaries. Decompose the request
+ * into the largest valid chunk at each step.
+ */
+ while (remaining) {
+ u64 chunk = rounddown_pow_of_two(remaining);
+
+ chunk = min(chunk, dma_block_remain(src));
+ chunk = min(chunk, dma_block_remain(dst));
+
+ ret = nv_falcon_memcpy_chunk(device, src, dst, chunk);
+ if (ret)
+ break;
+
+ src += chunk;
+ dst += chunk;
+ remaining -= chunk;
+ }
+
+ return ret;
+}
+
+const struct vfio_pci_driver_ops nv_falcon_ops = {
+ .name = "nv_falcon",
+ .probe = nv_falcon_probe,
+ .init = nv_falcon_init,
+ .remove = nv_falcon_remove,
+ .memcpy_start = nv_falcon_memcpy_start,
+ .memcpy_wait = nv_falcon_memcpy_wait,
+};
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/assert.h b/tools/testing/selftests/vfio/lib/include/libvfio/assert.h
index 77b68c7129a6..9fff88f6e4e1 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/assert.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/assert.h
@@ -3,6 +3,7 @@
#define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_ASSERT_H
#include <stdio.h>
+#include <stdlib.h>
#include <string.h>
#include <sys/ioctl.h>
@@ -45,6 +46,23 @@
VFIO_LOG_AND_EXIT(_fmt, ##__VA_ARGS__); \
} while (0)
+#define malloc_assert(_size) ({ \
+ size_t __size = (_size); \
+ void *__ptr = malloc(__size); \
+ VFIO_ASSERT_NOT_NULL(__ptr, "malloc(%zu) failed", \
+ __size); \
+ __ptr; \
+})
+
+#define calloc_assert(_nmemb, _size) ({ \
+ size_t __nmemb = (_nmemb); \
+ size_t __size = (_size); \
+ void *__ptr = calloc(__nmemb, __size); \
+ VFIO_ASSERT_NOT_NULL(__ptr, "calloc(%zu, %zu) failed", \
+ __nmemb, __size); \
+ __ptr; \
+})
+
#define ioctl_assert(_fd, _op, _arg) do { \
void *__arg = (_arg); \
int __ret = ioctl((_fd), (_op), (__arg)); \
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
index 3eabead717bb..e19bd94b8dd2 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
@@ -43,6 +43,7 @@ void vfio_pci_device_free(struct vfio_pci_device *device);
struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu);
void vfio_pci_device_cleanup(struct vfio_pci_device *device);
+int __vfio_pci_device_reset(struct vfio_pci_device *device);
void vfio_pci_device_reset(struct vfio_pci_device *device);
void vfio_pci_config_access(struct vfio_pci_device *device, bool write,
@@ -67,9 +68,25 @@ void vfio_pci_config_access(struct vfio_pci_device *device, bool write,
#define vfio_pci_config_writew(_d, _o, _v) vfio_pci_config_write(_d, _o, _v, u16)
#define vfio_pci_config_writel(_d, _o, _v) vfio_pci_config_write(_d, _o, _v, u32)
+static inline void vfio_pci_cmd_set(struct vfio_pci_device *device, u16 bits)
+{
+ u16 cmd = vfio_pci_config_readw(device, PCI_COMMAND);
+
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd | bits);
+}
+
+static inline void vfio_pci_cmd_clear(struct vfio_pci_device *device, u16 bits)
+{
+ u16 cmd = vfio_pci_config_readw(device, PCI_COMMAND);
+
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd & ~bits);
+}
+
void vfio_pci_irq_enable(struct vfio_pci_device *device, u32 index,
u32 vector, int count);
void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index);
+void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index,
+ u32 vector, int count);
void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector);
static inline void fcntl_set_nonblock(int fd)
@@ -94,6 +111,12 @@ static inline void vfio_pci_msi_disable(struct vfio_pci_device *device)
vfio_pci_irq_disable(device, VFIO_PCI_MSI_IRQ_INDEX);
}
+static inline void vfio_pci_msi_reenable(struct vfio_pci_device *device,
+ u32 vector, int count)
+{
+ vfio_pci_irq_reenable(device, VFIO_PCI_MSI_IRQ_INDEX, vector, count);
+}
+
static inline void vfio_pci_msix_enable(struct vfio_pci_device *device,
u32 vector, int count)
{
@@ -105,6 +128,12 @@ static inline void vfio_pci_msix_disable(struct vfio_pci_device *device)
vfio_pci_irq_disable(device, VFIO_PCI_MSIX_IRQ_INDEX);
}
+static inline void vfio_pci_msix_reenable(struct vfio_pci_device *device,
+ u32 vector, int count)
+{
+ vfio_pci_irq_reenable(device, VFIO_PCI_MSIX_IRQ_INDEX, vector, count);
+}
+
static inline int __to_iova(struct vfio_pci_device *device, void *vaddr, iova_t *iova)
{
return __iommu_hva2iova(device->iommu, vaddr, iova);
diff --git a/tools/testing/selftests/vfio/lib/iommu.c b/tools/testing/selftests/vfio/lib/iommu.c
index 035dac069d60..b6f3c5c84e01 100644
--- a/tools/testing/selftests/vfio/lib/iommu.c
+++ b/tools/testing/selftests/vfio/lib/iommu.c
@@ -286,8 +286,7 @@ static struct vfio_iommu_type1_info *vfio_iommu_get_info(int container_fd)
{
struct vfio_iommu_type1_info *info;
- info = malloc(sizeof(*info));
- VFIO_ASSERT_NOT_NULL(info);
+ info = malloc_assert(sizeof(*info));
*info = (struct vfio_iommu_type1_info) {
.argsz = sizeof(*info),
@@ -324,8 +323,7 @@ static struct iommu_iova_range *vfio_iommu_iova_ranges(struct iommu *iommu,
cap_range = container_of(hdr, struct vfio_iommu_type1_info_cap_iova_range, header);
VFIO_ASSERT_GT(cap_range->nr_iovas, 0);
- ranges = calloc(cap_range->nr_iovas, sizeof(*ranges));
- VFIO_ASSERT_NOT_NULL(ranges);
+ ranges = calloc_assert(cap_range->nr_iovas, sizeof(*ranges));
for (u32 i = 0; i < cap_range->nr_iovas; i++) {
ranges[i] = (struct iommu_iova_range){
@@ -357,8 +355,7 @@ static struct iommu_iova_range *iommufd_iova_ranges(struct iommu *iommu,
VFIO_ASSERT_EQ(errno, EMSGSIZE);
VFIO_ASSERT_GT(query.num_iovas, 0);
- ranges = calloc(query.num_iovas, sizeof(*ranges));
- VFIO_ASSERT_NOT_NULL(ranges);
+ ranges = calloc_assert(query.num_iovas, sizeof(*ranges));
query.allowed_iovas = (uintptr_t)ranges;
@@ -424,8 +421,7 @@ struct iommu *iommu_init(const char *iommu_mode)
struct iommu *iommu;
int version;
- iommu = calloc(1, sizeof(*iommu));
- VFIO_ASSERT_NOT_NULL(iommu);
+ iommu = calloc_assert(1, sizeof(*iommu));
INIT_LIST_HEAD(&iommu->dma_regions);
diff --git a/tools/testing/selftests/vfio/lib/iova_allocator.c b/tools/testing/selftests/vfio/lib/iova_allocator.c
index 8c1cc86b70cd..4a660f636f49 100644
--- a/tools/testing/selftests/vfio/lib/iova_allocator.c
+++ b/tools/testing/selftests/vfio/lib/iova_allocator.c
@@ -29,8 +29,7 @@ struct iova_allocator *iova_allocator_init(struct iommu *iommu)
ranges = iommu_iova_ranges(iommu, &nranges);
VFIO_ASSERT_NOT_NULL(ranges);
- allocator = malloc(sizeof(*allocator));
- VFIO_ASSERT_NOT_NULL(allocator);
+ allocator = malloc_assert(sizeof(*allocator));
*allocator = (struct iova_allocator){
.ranges = ranges,
@@ -90,4 +89,3 @@ next_range:
allocator->range_offset = 0;
}
}
-
diff --git a/tools/testing/selftests/vfio/lib/libvfio.mk b/tools/testing/selftests/vfio/lib/libvfio.mk
index 67942b085068..bcfa74ae040e 100644
--- a/tools/testing/selftests/vfio/lib/libvfio.mk
+++ b/tools/testing/selftests/vfio/lib/libvfio.mk
@@ -15,6 +15,9 @@ LIBVFIO_C += drivers/ioat/ioat.c
LIBVFIO_C += drivers/dsa/dsa.c
endif
+LIBVFIO_C += drivers/nv_falcon/nv_falcon.c
+LIBVFIO_C += drivers/igb/igb.c
+
LIBVFIO_OUTPUT := $(OUTPUT)/libvfio
LIBVFIO_O := $(patsubst %.c, $(LIBVFIO_OUTPUT)/%.o, $(LIBVFIO_C))
diff --git a/tools/testing/selftests/vfio/lib/sysfs.c b/tools/testing/selftests/vfio/lib/sysfs.c
index 11415448b2e2..98a46a2543cd 100644
--- a/tools/testing/selftests/vfio/lib/sysfs.c
+++ b/tools/testing/selftests/vfio/lib/sysfs.c
@@ -107,8 +107,7 @@ char *sysfs_sriov_vf_bdf_get(const char *pf_bdf, int i)
char *out_vf_bdf;
/* Fit "0000:00:00.0" */
- out_vf_bdf = calloc(16, sizeof(char));
- VFIO_ASSERT_NOT_NULL(out_vf_bdf);
+ out_vf_bdf = calloc_assert(16, sizeof(char));
snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/virtfn%d", pf_bdf, i);
readlink_base(path, "%s", out_vf_bdf);
diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_device.c b/tools/testing/selftests/vfio/lib/vfio_pci_device.c
index 94dc5fcecbeb..4063a0e2b3df 100644
--- a/tools/testing/selftests/vfio/lib/vfio_pci_device.c
+++ b/tools/testing/selftests/vfio/lib/vfio_pci_device.c
@@ -1,5 +1,6 @@
// SPDX-License-Identifier: GPL-2.0-only
#include <dirent.h>
+#include <errno.h>
#include <fcntl.h>
#include <libgen.h>
#include <stdint.h>
@@ -30,13 +31,11 @@
static void vfio_pci_irq_set(struct vfio_pci_device *device,
u32 index, u32 vector, u32 count, int *fds)
{
- u8 buf[sizeof(struct vfio_irq_set) + sizeof(int) * count];
- struct vfio_irq_set *irq = (void *)&buf;
- int *irq_fds = (void *)&irq->data;
+ size_t argsz = sizeof(struct vfio_irq_set) + sizeof(int) * count;
+ struct vfio_irq_set *irq;
- memset(buf, 0, sizeof(buf));
-
- irq->argsz = sizeof(buf);
+ irq = calloc_assert(1, argsz);
+ irq->argsz = argsz;
irq->flags = VFIO_IRQ_SET_ACTION_TRIGGER;
irq->index = index;
irq->start = vector;
@@ -44,12 +43,13 @@ static void vfio_pci_irq_set(struct vfio_pci_device *device,
if (count) {
irq->flags |= VFIO_IRQ_SET_DATA_EVENTFD;
- memcpy(irq_fds, fds, sizeof(int) * count);
+ memcpy(irq->data, fds, sizeof(int) * count);
} else {
irq->flags |= VFIO_IRQ_SET_DATA_NONE;
}
ioctl_assert(device->fd, VFIO_DEVICE_SET_IRQS, irq);
+ free(irq);
}
void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector)
@@ -106,6 +106,28 @@ void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index)
vfio_pci_irq_set(device, index, 0, 0, NULL);
}
+/*
+ * Re-issue VFIO_DEVICE_SET_IRQS for an already-enabled vector range using
+ * the existing eventfds. Intended for drivers that need to re-arm device
+ * interrupts after a VFIO_DEVICE_RESET, which tears down the kernel-side
+ * IRQ trigger but leaves user-side eventfds intact. Recreating the
+ * eventfds would invalidate any test-fixture cache of the fd, so this
+ * helper deliberately preserves them.
+ */
+void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index,
+ u32 vector, int count)
+{
+ int i;
+
+ check_supported_irq_index(index);
+
+ for (i = vector; i < vector + count; i++)
+ VFIO_ASSERT_GE(device->msi_eventfds[i], 0,
+ "vector %d eventfd not allocated\n", i);
+
+ vfio_pci_irq_set(device, index, vector, count, device->msi_eventfds + vector);
+}
+
static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index,
struct vfio_irq_info *irq_info)
{
@@ -118,15 +140,20 @@ static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index,
static int vfio_device_feature_ioctl(int fd, u32 flags, void *data,
size_t data_size)
{
- u8 buffer[sizeof(struct vfio_device_feature) + data_size] = {};
- struct vfio_device_feature *feature = (void *)buffer;
+ size_t argsz = sizeof(struct vfio_device_feature) + data_size;
+ struct vfio_device_feature *feature;
+ int ret;
+ feature = calloc_assert(1, argsz);
memcpy(feature->data, data, data_size);
- feature->argsz = sizeof(buffer);
+ feature->argsz = argsz;
feature->flags = flags;
- return ioctl(fd, VFIO_DEVICE_FEATURE, feature);
+ ret = ioctl(fd, VFIO_DEVICE_FEATURE, feature);
+ free(feature);
+
+ return ret;
}
static void vfio_device_feature_set(int fd, u16 feature, void *data, size_t data_size)
@@ -233,9 +260,26 @@ void vfio_pci_config_access(struct vfio_pci_device *device, bool write,
write ? "write to" : "read from", config);
}
+int __vfio_pci_device_reset(struct vfio_pci_device *device)
+{
+ if (ioctl(device->fd, VFIO_DEVICE_RESET, NULL))
+ return -errno;
+
+ return 0;
+}
+
void vfio_pci_device_reset(struct vfio_pci_device *device)
{
- ioctl_assert(device->fd, VFIO_DEVICE_RESET, NULL);
+ int retries = 20;
+ int r;
+
+ do {
+ r = __vfio_pci_device_reset(device);
+ if (r == -EAGAIN)
+ usleep(10000);
+ } while (r == -EAGAIN && retries-- > 0);
+
+ VFIO_ASSERT_EQ(r, 0, "ioctl(device->fd, VFIO_DEVICE_RESET) failed\n");
}
void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf)
@@ -343,8 +387,7 @@ const char *vfio_pci_get_cdev_path(const char *bdf)
char *cdev_path;
DIR *dir;
- cdev_path = calloc(PATH_MAX, 1);
- VFIO_ASSERT_NOT_NULL(cdev_path);
+ cdev_path = calloc_assert(PATH_MAX, 1);
snprintf_assert(dir_path, sizeof(dir_path), "/sys/bus/pci/devices/%s/vfio-dev/", bdf);
@@ -425,8 +468,7 @@ struct vfio_pci_device *vfio_pci_device_alloc(const char *bdf, struct iommu *iom
{
struct vfio_pci_device *device;
- device = calloc(1, sizeof(*device));
- VFIO_ASSERT_NOT_NULL(device);
+ device = calloc_assert(1, sizeof(*device));
VFIO_ASSERT_NOT_NULL(iommu);
device->iommu = iommu;
diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
index 6827f4a6febe..5e65434d2318 100644
--- a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
+++ b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
@@ -6,12 +6,16 @@
extern struct vfio_pci_driver_ops dsa_ops;
extern struct vfio_pci_driver_ops ioat_ops;
#endif
+extern struct vfio_pci_driver_ops nv_falcon_ops;
+extern struct vfio_pci_driver_ops igb_ops;
static struct vfio_pci_driver_ops *driver_ops[] = {
#ifdef __x86_64__
&dsa_ops,
&ioat_ops,
#endif
+ &nv_falcon_ops,
+ &igb_ops,
};
void vfio_pci_driver_probe(struct vfio_pci_device *device)
@@ -106,7 +110,21 @@ int vfio_pci_driver_memcpy_wait(struct vfio_pci_device *device)
int vfio_pci_driver_memcpy(struct vfio_pci_device *device,
iova_t src, iova_t dst, u64 size)
{
- vfio_pci_driver_memcpy_start(device, src, dst, size, 1);
+ struct vfio_pci_driver *driver = &device->driver;
+ u64 offset = 0;
+
+ while (offset < size) {
+ u64 chunk = min(size - offset, driver->max_memcpy_size);
+ int ret;
+
+ vfio_pci_driver_memcpy_start(device, src + offset,
+ dst + offset, chunk, 1);
+ ret = vfio_pci_driver_memcpy_wait(device);
+ if (ret)
+ return ret;
+
+ offset += chunk;
+ }
- return vfio_pci_driver_memcpy_wait(device);
+ return 0;
}
diff --git a/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c b/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c
index 33b0c31fe2ed..e1a54e153cd3 100644
--- a/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c
+++ b/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c
@@ -45,8 +45,8 @@ FIXTURE_SETUP(vfio_pci_device_init_perf_test)
int i;
self->iommu = iommu_init(variant->iommu_mode);
- self->threads = calloc(nr_devices, sizeof(self->threads[0]));
- self->thread_args = calloc(nr_devices, sizeof(self->thread_args[0]));
+ self->threads = calloc_assert(nr_devices, sizeof(self->threads[0]));
+ self->thread_args = calloc_assert(nr_devices, sizeof(self->thread_args[0]));
pthread_barrier_init(&self->barrier, NULL, nr_devices);
diff --git a/tools/testing/selftests/vfio/vfio_pci_driver_test.c b/tools/testing/selftests/vfio/vfio_pci_driver_test.c
index afa0480ddd9b..761bf117d624 100644
--- a/tools/testing/selftests/vfio/vfio_pci_driver_test.c
+++ b/tools/testing/selftests/vfio/vfio_pci_driver_test.c
@@ -11,11 +11,18 @@
static const char *device_bdf;
-#define ASSERT_NO_MSI(_eventfd) do { \
- u64 __value; \
- \
- ASSERT_EQ(-1, read(_eventfd, &__value, 8)); \
- ASSERT_EQ(EAGAIN, errno); \
+#define fcntl_set_msi_nonblock(_self) do { \
+ if (_self->device->driver.ops->send_msi) \
+ fcntl_set_nonblock(_self->msi_fd); \
+} while (0)
+
+#define ASSERT_NO_MSI(_self) do { \
+ u64 __value; \
+ \
+ if (!_self->device->driver.ops->send_msi) \
+ break; \
+ ASSERT_EQ(-1, read(_self->msi_fd, &__value, 8)); \
+ ASSERT_EQ(EAGAIN, errno); \
} while (0)
static void region_setup(struct iommu *iommu,
@@ -89,12 +96,12 @@ FIXTURE_SETUP(vfio_pci_driver_test)
self->msi_fd = self->device->msi_eventfds[driver->msi];
/*
- * Use the maximum size supported by the device for memcpy operations,
- * slimmed down to fit into the memcpy region (divided by 2 so src and
- * dst regions do not overlap).
+ * Use 4x the driver's max_memcpy_size to exercise the chunking
+ * logic in vfio_pci_driver_memcpy(). Cap to half the memcpy
+ * region so src and dst do not overlap.
*/
- self->size = self->device->driver.max_memcpy_size;
- self->size = min(self->size, self->memcpy_region.size / 2);
+ self->size = min_t(u64, driver->max_memcpy_size * 4,
+ self->memcpy_region.size / 2);
self->src = self->memcpy_region.vaddr;
self->dst = self->src + self->size;
@@ -129,7 +136,7 @@ TEST_F(vfio_pci_driver_test, init_remove)
TEST_F(vfio_pci_driver_test, memcpy_success)
{
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
memset(self->src, 'x', self->size);
memset(self->dst, 'y', self->size);
@@ -140,12 +147,12 @@ TEST_F(vfio_pci_driver_test, memcpy_success)
self->size));
ASSERT_EQ(0, memcmp(self->src, self->dst, self->size));
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
TEST_F(vfio_pci_driver_test, memcpy_from_unmapped_iova)
{
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
/*
* Ignore the return value since not all devices will detect and report
@@ -154,12 +161,12 @@ TEST_F(vfio_pci_driver_test, memcpy_from_unmapped_iova)
vfio_pci_driver_memcpy(self->device, self->unmapped_iova,
self->dst_iova, self->size);
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
TEST_F(vfio_pci_driver_test, memcpy_to_unmapped_iova)
{
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
/*
* Ignore the return value since not all devices will detect and report
@@ -168,13 +175,16 @@ TEST_F(vfio_pci_driver_test, memcpy_to_unmapped_iova)
vfio_pci_driver_memcpy(self->device, self->src_iova,
self->unmapped_iova, self->size);
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
TEST_F(vfio_pci_driver_test, send_msi)
{
u64 value;
+ if (!self->device->driver.ops->send_msi)
+ SKIP(return, "Driver does not support send_msi()\n");
+
vfio_pci_driver_send_msi(self->device);
ASSERT_EQ(8, read(self->msi_fd, &value, 8));
ASSERT_EQ(1, value);
@@ -201,6 +211,9 @@ TEST_F(vfio_pci_driver_test, mix_and_match)
self->dst_iova,
self->size);
+ if (!self->device->driver.ops->send_msi)
+ continue;
+
vfio_pci_driver_send_msi(self->device);
ASSERT_EQ(8, read(self->msi_fd, &value, 8));
ASSERT_EQ(1, value);
@@ -211,9 +224,10 @@ TEST_F_TIMEOUT(vfio_pci_driver_test, memcpy_storm, 60)
{
struct vfio_pci_driver *driver = &self->device->driver;
u64 total_size;
+ u64 size;
u64 count;
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
/*
* Perform up to 250GiB worth of DMA reads and writes across several
@@ -221,16 +235,17 @@ TEST_F_TIMEOUT(vfio_pci_driver_test, memcpy_storm, 60)
* will take too long.
*/
total_size = 250UL * SZ_1G;
- count = min(total_size / self->size, driver->max_memcpy_count);
+ size = min(driver->max_memcpy_size, self->memcpy_region.size / 2);
+ count = min(total_size / size, driver->max_memcpy_count);
- printf("Kicking off %lu memcpys of size 0x%lx\n", count, self->size);
+ printf("Kicking off %lu memcpys of size 0x%lx\n", count, size);
vfio_pci_driver_memcpy_start(self->device,
self->src_iova,
self->dst_iova,
- self->size, count);
+ size, count);
ASSERT_EQ(0, vfio_pci_driver_memcpy_wait(self->device));
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
static bool device_has_selftests_driver(const char *bdf)
diff --git a/tools/testing/selftests/wireguard/qemu/kernel.config b/tools/testing/selftests/wireguard/qemu/kernel.config
index bb89d2dfaa2a..368fd9d600c8 100644
--- a/tools/testing/selftests/wireguard/qemu/kernel.config
+++ b/tools/testing/selftests/wireguard/qemu/kernel.config
@@ -63,7 +63,7 @@ CONFIG_FILE_LOCKING=y
CONFIG_POSIX_TIMERS=y
CONFIG_DEVTMPFS=y
CONFIG_PROC_FS=y
-CONFIG_PROC_SYSCTL=y
+CONFIG_SYSCTL=y
CONFIG_SYSFS=y
CONFIG_TMPFS=y
CONFIG_CONSOLE_LOGLEVEL_DEFAULT=15
diff --git a/tools/testing/selftests/x86/Makefile b/tools/testing/selftests/x86/Makefile
index 434065215d12..d478b13cc8d5 100644
--- a/tools/testing/selftests/x86/Makefile
+++ b/tools/testing/selftests/x86/Makefile
@@ -13,7 +13,7 @@ CAN_BUILD_WITH_NOPIE := $(shell ./check_cc.sh "$(CC)" trivial_program.c -no-pie)
TARGETS_C_BOTHBITS := single_step_syscall sysret_ss_attrs syscall_nt test_mremap_vdso \
check_initial_reg_state sigreturn iopl ioperm \
test_vsyscall mov_ss_trap sigtrap_loop \
- syscall_arg_fault fsgsbase_restore sigaltstack
+ syscall_arg_fault fsgsbase_restore sigaltstack int_signal
TARGETS_C_BOTHBITS += nx_stack
TARGETS_C_32BIT_ONLY := entry_from_vm86 test_syscall_vdso unwind_vdso \
test_FCMOV test_FCOMI test_FISTTP \
diff --git a/tools/testing/selftests/x86/int_signal.c b/tools/testing/selftests/x86/int_signal.c
new file mode 100644
index 000000000000..22676dac72b5
--- /dev/null
+++ b/tools/testing/selftests/x86/int_signal.c
@@ -0,0 +1,311 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/* Check the signal context for INT instructions with IDT and FRED entry. */
+#define _GNU_SOURCE
+
+#include <cpuid.h>
+#include <errno.h>
+#include <stdbool.h>
+#include <stddef.h>
+#include <stdint.h>
+#include <sys/ptrace.h>
+#include <sys/user.h>
+#include <sys/wait.h>
+#include <unistd.h>
+#include <ucontext.h>
+
+#include "helpers.h"
+
+#ifdef __x86_64__
+#define REG_IP REG_RIP
+#define USER_IP rip
+#define STACK_PTR "%rsp"
+#else
+#define REG_IP REG_EIP
+#define USER_IP eip
+#define STACK_PTR "%esp"
+#endif
+
+/*
+ * Each instruction has normal and single-step entry points. Resume at the
+ * NOP after handling its signal, then expect a trace trap after that NOP
+ * when TF is set. Explicit labels avoid assuming the kernel's saved IP.
+ */
+#define PROBE(name, insn) \
+ extern void name(void); \
+ extern void name##_tf(void); \
+ extern const char name##_end[], name##_step[]; \
+ asm(".pushsection .text\n" \
+ ".globl " #name "_tf\n" \
+ ".type " #name "_tf, @function\n" \
+ #name "_tf:\n" \
+ "pushf\n" \
+ "orl $0x100, (" STACK_PTR ")\n" \
+ "popf\n" \
+ ".globl " #name "\n" \
+ ".type " #name ", @function\n" \
+ #name ":\n" insn "\n" \
+ ".globl " #name "_end\n" \
+ #name "_end:\nnop\n" \
+ ".globl " #name "_step\n" \
+ #name "_step:\nret\n" \
+ ".size " #name ", .-" #name "\n" \
+ ".size " #name "_tf, .-" #name "_tf\n" \
+ ".popsection\n")
+
+PROBE(int1, ".byte 0xcd, 0x01");
+PROBE(int29, ".byte 0xcd, 0x29");
+PROBE(int2c, ".byte 0xcd, 0x2c");
+PROBE(int2d, ".byte 0xcd, 0x2d");
+PROBE(prefixed_int2d, ".byte 0x66, 0xcd, 0x2d");
+PROBE(long_int2d, ".fill 13, 1, 0x2e\n.byte 0xcd, 0x2d");
+PROBE(int81, ".byte 0xcd, 0x81");
+PROBE(intff, ".byte 0xcd, 0xff");
+PROBE(short_int3, ".byte 0xcc");
+PROBE(long_int3, ".byte 0xcd, 0x03");
+PROBE(int4, ".byte 0xcd, 0x04");
+PROBE(ud2, ".byte 0x0f, 0x0b");
+PROBE(hlt, ".byte 0xf4");
+
+struct test {
+ const char *name;
+ void (*run)(void);
+ void (*run_tf)(void);
+ const char *end, *step;
+ int signo, trap, error, ip_offset, flags, code;
+};
+
+#define TEST(name, sig, trap, error, offset, flags, code) \
+ { #name, name, name##_tf, name##_end, name##_step, \
+ sig, trap, error, offset, flags, code }
+
+#define GP(name, error) \
+ TEST(name, SIGSEGV, 13, error, 0, X86_EFLAGS_RF, SI_KERNEL)
+
+static const struct test tests[] = {
+ GP(int1, 0x00a),
+ GP(int29, 0x14a),
+ GP(int2c, 0x162),
+ GP(int2d, 0x16a),
+ GP(prefixed_int2d, 0x16a),
+ GP(long_int2d, 0x16a),
+ GP(int81, 0x40a),
+ GP(intff, 0x7fa),
+ GP(hlt, 0),
+ TEST(short_int3, SIGTRAP, 3, 0, 1, 0, SI_KERNEL),
+ TEST(long_int3, SIGTRAP, 3, 0, 2, 0, SI_KERNEL),
+ TEST(int4, SIGSEGV, 4, 0, 2, 0, SI_KERNEL),
+ TEST(ud2, SIGILL, 6, 0, 0, X86_EFLAGS_RF, ILL_ILLOPN),
+};
+
+static const struct test *active;
+static volatile sig_atomic_t seen, signo, trap, error, ip_offset, flags;
+static volatile sig_atomic_t code, addr_ok, single_step, stepped, step_ok;
+
+static void handler(int sig, siginfo_t *info, void *context)
+{
+ ucontext_t *uc = context;
+ uintptr_t ip = uc->uc_mcontext.gregs[REG_IP];
+ uintptr_t start = (uintptr_t)active->run;
+ uintptr_t end = (uintptr_t)active->end;
+
+ if (seen && single_step && sig == SIGTRAP) {
+ if (stepped++) {
+ ksft_print_msg("%s: second trace trap at %#lx\n",
+ active->name, (unsigned long)ip);
+ _exit(KSFT_FAIL);
+ }
+ step_ok = ip == (uintptr_t)active->step &&
+ uc->uc_mcontext.gregs[REG_TRAPNO] == 1 &&
+ info->si_code == TRAP_TRACE;
+ uc->uc_mcontext.gregs[REG_EFL] &= ~X86_EFLAGS_TF;
+ return;
+ }
+
+ if (seen || ip < start || ip > end) {
+ ksft_print_msg("%s: unexpected signal %d at %#lx\n",
+ active->name, sig, (unsigned long)ip);
+ _exit(KSFT_FAIL);
+ }
+
+ signo = sig;
+ trap = uc->uc_mcontext.gregs[REG_TRAPNO];
+ error = uc->uc_mcontext.gregs[REG_ERR];
+ ip_offset = ip - start;
+ flags = uc->uc_mcontext.gregs[REG_EFL] & (X86_EFLAGS_RF | X86_EFLAGS_TF);
+ code = info->si_code;
+ /* force_sig() reports no address, force_sig_fault() reports the IP. */
+ addr_ok = info->si_addr == (code == SI_KERNEL ? NULL : (void *)ip);
+ seen = 1;
+ uc->uc_mcontext.gregs[REG_IP] = end;
+}
+
+static void wait_for_child(pid_t child, int *status)
+{
+ pid_t ret;
+
+ do {
+ ret = waitpid(child, status, 0);
+ } while (ret < 0 && errno == EINTR);
+ if (ret != child)
+ ksft_exit_fail_perror("waitpid");
+}
+
+/* Resume the tracee and check where the next stop lands. */
+static bool resume_to(pid_t child, int *status, int request, int sig,
+ const void *ip, const char *what)
+{
+ struct user_regs_struct regs;
+
+ if (ptrace(request, child, 0, 0))
+ return false;
+ wait_for_child(child, status);
+ if (!WIFSTOPPED(*status)) {
+ ksft_print_msg("%s: tracee did not stop\n", what);
+ return false;
+ }
+ if (WSTOPSIG(*status) != sig) {
+ ksft_print_msg("%s: stopped with signal %d, expected %d\n",
+ what, WSTOPSIG(*status), sig);
+ return false;
+ }
+ if (ptrace(PTRACE_GETREGS, child, 0, &regs))
+ return false;
+ if ((unsigned long)regs.USER_IP != (unsigned long)ip) {
+ ksft_print_msg("%s: stopped at %#lx, expected %#lx\n", what,
+ (unsigned long)regs.USER_IP, (unsigned long)ip);
+ return false;
+ }
+ return true;
+}
+
+static bool set_ip(pid_t child, const void *ip, bool tf)
+{
+ struct user_regs_struct regs;
+
+ if (ptrace(PTRACE_GETREGS, child, 0, &regs))
+ return false;
+ regs.USER_IP = (unsigned long)ip;
+ if (tf)
+ regs.eflags |= X86_EFLAGS_TF;
+ return !ptrace(PTRACE_SETREGS, child, 0, &regs);
+}
+
+/*
+ * Exercise the tracer paths that resume through the fault frame rather than
+ * sigreturn. A stale FRED software event flag on that frame traps before the
+ * NOP executes instead of after it.
+ */
+static void test_ptrace(void)
+{
+ bool into = false, step = false, cont = false;
+ pid_t child;
+ int status;
+
+ child = fork();
+ if (child < 0)
+ ksft_exit_fail_perror("fork");
+ if (!child) {
+ if (ptrace(PTRACE_TRACEME, 0, 0, 0))
+ _exit(KSFT_FAIL);
+ /* Start from a breakpoint frame, not the syscall frame of raise(). */
+ asm volatile("int3");
+ _exit(KSFT_FAIL);
+ }
+
+ wait_for_child(child, &status);
+ if (!WIFSTOPPED(status) || WSTOPSIG(status) != SIGTRAP)
+ goto out;
+ if (ptrace(PTRACE_SETOPTIONS, child, 0, PTRACE_O_EXITKILL))
+ goto out;
+
+ /* Single-step into the INT. The fault must report the INT's address. */
+ if (!set_ip(child, int2d, false))
+ goto out;
+ into = resume_to(child, &status, PTRACE_SINGLESTEP, SIGSEGV, int2d,
+ "single-step into INT");
+ if (!into)
+ goto out;
+
+ /* Suppress SIGSEGV and single-step the NOP. */
+ if (!set_ip(child, int2d_end, false))
+ goto out;
+ step = resume_to(child, &status, PTRACE_SINGLESTEP, SIGTRAP, int2d_step,
+ "single-step after INT");
+ if (!step)
+ goto out;
+
+ /* Fault again, then suppress SIGSEGV and continue with TF set. */
+ if (!set_ip(child, int2d, false))
+ goto out;
+ if (!resume_to(child, &status, PTRACE_CONT, SIGSEGV, int2d,
+ "continue to INT"))
+ goto out;
+ if (!set_ip(child, int2d_end, true))
+ goto out;
+ cont = resume_to(child, &status, PTRACE_CONT, SIGTRAP, int2d_step,
+ "continue with TF after INT");
+out:
+ if (WIFSTOPPED(status)) {
+ kill(child, SIGKILL);
+ wait_for_child(child, &status);
+ }
+ ksft_test_result(into, "ptrace single-step into INT faults at the INT\n");
+ ksft_test_result(step, "ptrace single-step after suppressing SIGSEGV\n");
+ ksft_test_result(cont, "ptrace continue with TF after suppressing SIGSEGV\n");
+}
+
+static bool cpu_has_fred(void)
+{
+ unsigned int eax, ebx, ecx, edx;
+
+ if (__get_cpuid_max(0, NULL) < 7)
+ return false;
+ __cpuid_count(7, 1, eax, ebx, ecx, edx);
+ return eax & (1 << 17);
+}
+
+int main(void)
+{
+ unsigned int i, tf;
+ int expected_flags, ok;
+
+ ksft_print_header();
+ ksft_set_plan(2 * ARRAY_SIZE(tests) + 3);
+ ksft_print_msg("CPU %s FRED\n", cpu_has_fred() ? "supports" : "lacks");
+ sethandler(SIGSEGV, handler, 0);
+ sethandler(SIGTRAP, handler, 0);
+ sethandler(SIGILL, handler, 0);
+
+ for (tf = 0; tf < 2; tf++) {
+ for (i = 0; i < ARRAY_SIZE(tests); i++) {
+ active = &tests[i];
+ single_step = tf;
+ seen = signo = trap = error = ip_offset = flags = 0;
+ code = addr_ok = stepped = step_ok = 0;
+ expected_flags = active->flags | (tf ? X86_EFLAGS_TF : 0);
+ if (tf)
+ active->run_tf();
+ else
+ active->run();
+
+ ok = seen && signo == active->signo && trap == active->trap &&
+ error == active->error && ip_offset == active->ip_offset &&
+ flags == expected_flags && code == active->code && addr_ok &&
+ (!tf || (stepped && step_ok));
+ ksft_test_result(ok, "%s%s\n", active->name, tf ? " with TF" : "");
+ if (!ok) {
+ ksft_print_msg("got signal=%d trap=%d error=%#x ip=%d\n",
+ signo, trap, error, ip_offset);
+ ksft_print_msg("got flags=%#x code=%d addr_ok=%d step_ok=%d\n",
+ flags, code, addr_ok, step_ok);
+ ksft_print_msg("expected signal=%d trap=%d error=%#x ip=%d\n",
+ active->signo, active->trap, active->error,
+ active->ip_offset);
+ ksft_print_msg("expected flags=%#x code=%d\n",
+ expected_flags, active->code);
+ }
+ }
+ }
+ test_ptrace();
+ ksft_finished();
+}
diff --git a/tools/testing/selftests/zram/zram_lib.sh b/tools/testing/selftests/zram/zram_lib.sh
index 21ec1966de76..0d44d83888f9 100755
--- a/tools/testing/selftests/zram/zram_lib.sh
+++ b/tools/testing/selftests/zram/zram_lib.sh
@@ -37,7 +37,7 @@ kernel_gte()
if [ $kernel_major -gt $major ]; then
return 0
- elif [[ $kernel_major -eq $major && $kernel_minor -ge $minor ]]; then
+ elif [ $kernel_major -eq $major ] && [ $kernel_minor -ge $minor ]; then
return 0
fi