diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c index d40cb5dd446c..b8df2bc9a9a0 100644 --- a/kernel/bpf/hashtab.c +++ b/kernel/bpf/hashtab.c @@ -493,10 +493,54 @@ static void htab_pcpu_mem_dtor(void *obj, void *ctx) bpf_obj_free_fields(hrec->record, per_cpu_ptr(pptr, cpu)); } +/* + * bpf_ma_set_dtor() duplicates the map's btf_record. For kptr fields whose + * btf is the program BTF (MEM_ALLOC kptrs, e.g. objects allocated with + * bpf_obj_new()/bpf_percpu_obj_new()) btf_record_dup() only borrows the + * reference, like btf_parse_fields() did for the map's own record. The + * duplicated record is released later from the deferred bpf_mem_alloc + * destructor workqueue, by which time the program BTF may already have been + * freed (the map dropped its own reference in bpf_map_free()), so reading + * field->kptr.btf there would be a use-after-free. + * + * Hold a reference on non-kernel (program) BTF for the lifetime of the + * duplicated record and release it before the record is freed. After the + * last btf_put() the object is only destroyed after an RCU grace period, so + * btf_record_free() can still safely read the field descriptors. + */ +static void htab_record_prog_btf_ref(struct btf_record *rec, bool get) +{ + int i; + + if (IS_ERR_OR_NULL(rec)) + return; + + for (i = 0; i < rec->cnt; i++) { + const struct btf_field *field = &rec->fields[i]; + + switch (field->type) { + case BPF_KPTR_UNREF: + case BPF_KPTR_REF: + case BPF_KPTR_PERCPU: + case BPF_UPTR: + if (field->kptr.btf && !btf_is_kernel(field->kptr.btf)) { + if (get) + btf_get(field->kptr.btf); + else + btf_put(field->kptr.btf); + } + break; + default: + break; + } + } +} + static void htab_dtor_ctx_free(void *ctx) { struct htab_btf_record *hrec = ctx; + htab_record_prog_btf_ref(hrec->record, false); btf_record_free(hrec->record); kfree(ctx); } @@ -521,6 +565,7 @@ static int bpf_ma_set_dtor(struct bpf_map *map, struct bpf_mem_alloc *ma, kfree(hrec); return err; } + htab_record_prog_btf_ref(hrec->record, true); bpf_mem_alloc_set_dtor(ma, dtor, htab_dtor_ctx_free, hrec); return 0; } @@ -2864,14 +2909,56 @@ static int rhtab_map_alloc_check(union bpf_attr *attr) return htab_map_alloc_check(attr); } -static void rhtab_check_and_free_fields(struct bpf_rhtab *rhtab, - struct rhtab_elem *elem) +static void rhtab_cancel_fields(struct bpf_rhtab *rhtab, + struct rhtab_elem *elem) { if (IS_ERR_OR_NULL(rhtab->map.record)) return; - bpf_obj_free_fields(rhtab->map.record, - rhtab_elem_value(elem, rhtab->map.key_size)); + /* + * Only cancel NMI-safe fields (timer, workqueue, task_work) here. + * RHASH values can also carry referenced kptrs (and per-cpu kptrs), + * whose destructors must not run from arbitrary BPF execution + * contexts (e.g. NMI); leave them attached to the recycled element + * and let rhtab_mem_dtor() destroy them once the element is + * eventually freed. This matches the hash map semantics introduced + * by a3a81d247651 ("bpf: Cancel special fields on map value + * recycle"). + */ + bpf_map_free_internal_structs(&rhtab->map, + rhtab_elem_value(elem, rhtab->map.key_size)); +} + +/* + * Initialize special fields of a freshly allocated rhtab element, but keep + * kptr fields untouched. A recycled element may carry a referenced kptr from + * its previous life: the delete path only cancels NMI-safe fields (matching + * the hash map semantics), so the kptr reference stays owned by the element + * until rhtab_mem_dtor() destroys it. Zeroing it here (as + * check_and_init_map_value() would) would drop the reference without + * releasing it. + */ +static void rhtab_init_map_value(struct bpf_map *map, void *value) +{ + struct btf_record *rec = map->record; + int i; + + if (IS_ERR_OR_NULL(rec)) + return; + + for (i = 0; i < rec->cnt; i++) { + struct btf_field *field = &rec->fields[i]; + void *field_ptr = value + field->offset; + + switch (field->type) { + case BPF_KPTR_UNREF: + case BPF_KPTR_REF: + case BPF_KPTR_PERCPU: + continue; + default: + bpf_obj_init_field(field, field_ptr); + } + } } static void rhtab_mem_dtor(void *obj, void *ctx) @@ -2963,8 +3050,8 @@ static int rhtab_delete_elem(struct bpf_rhtab *rhtab, struct rhtab_elem *elem, v rhtab_read_elem_value(&rhtab->map, copy, elem, flags); check_and_init_map_value(&rhtab->map, copy); } - /* Release internal structs: kptr, bpf_timer, task_work, wq */ - rhtab_check_and_free_fields(rhtab, elem); + /* Cancel NMI-safe fields; full destruction happens in rhtab_mem_dtor */ + rhtab_cancel_fields(rhtab, elem); bpf_mem_cache_free_rcu(&rhtab->ma, elem); return 0; } @@ -3022,10 +3109,11 @@ static long rhtab_map_update_existing(struct bpf_map *map, struct rhtab_elem *el * BPF_F_LOCK, matching arraymap semantics. * * copy_map_value() skips special-field offsets, so old timers/ - * kptrs/etc. still sit in the slot. Cancel them after the copy - * to match arraymap's update semantics. + * kptrs/etc. still sit in the slot. Cancel the NMI-safe ones after + * the copy to match arraymap's update semantics; referenced kptrs + * stay attached and are destroyed by rhtab_mem_dtor(). */ - rhtab_check_and_free_fields(rhtab, elem); + rhtab_cancel_fields(rhtab, elem); return 0; } @@ -3066,7 +3154,14 @@ static long rhtab_map_update_elem(struct bpf_map *map, void *key, void *value, u memcpy(elem->data, key, map->key_size); copy_map_value(map, rhtab_elem_value(elem, map->key_size), value); - check_and_init_map_value(map, rhtab_elem_value(elem, map->key_size)); + /* + * Initialize special fields of the (possibly recycled) element, but + * leave kptr slots alone: a recycled element may still own a + * referenced kptr that rhtab_mem_dtor() will release, so zeroing it + * here would leak the reference. Fresh memory from the bpf mem + * allocator is zeroed, so skipping the kptr init is safe there too. + */ + rhtab_init_map_value(map, rhtab_elem_value(elem, map->key_size)); /* Prevent deadlock for NMI programs attempting to take bucket lock */ bpf_disable_instrumentation(); diff --git a/tools/testing/selftests/bpf/prog_tests/rhtab_fields.c b/tools/testing/selftests/bpf/prog_tests/rhtab_fields.c new file mode 100644 index 000000000000..29de05bcbd4b --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/rhtab_fields.c @@ -0,0 +1,213 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 KylinSoft Co., Ltd. */ + +#include +#include +#include "rhtab_fields.skel.h" + +#define RECYCLE_LOOPS 2000 +#define LK_MAGIC 0x52484142 + +/* Userspace view of the BPF value types (layouts must match the progs). */ +struct lock_kptr_val_user { + __u32 lock; + __u32 pad; + __u64 tsk; + __u32 magic; + __u32 pad2; +}; + +static __u64 read_counter(struct rhtab_fields *skel, u32 idx) +{ + __u64 vals[libbpf_num_possible_cpus()]; + __u64 sum = 0; + int i, err; + + err = bpf_map_lookup_elem(bpf_map__fd(skel->maps.counters), &idx, vals); + if (!ASSERT_OK(err, "lookup_counter")) + return 0; + for (i = 0; i < libbpf_num_possible_cpus(); i++) + sum += vals[i]; + return sum; +} + +/* Returns the program retval; asserts the test_run itself succeeded. */ +static int run_prog(struct rhtab_fields *skel, const char *name) +{ + LIBBPF_OPTS(bpf_test_run_opts, topts); + struct bpf_program *prog; + int err; + + prog = bpf_object__find_program_by_name(skel->obj, name); + if (!ASSERT_OK_PTR(prog, name)) + return -1; + err = bpf_prog_test_run_opts(bpf_program__fd(prog), &topts); + if (!ASSERT_OK(err, name)) + return -1; + return topts.retval; +} + +static void recycle_loop(struct rhtab_fields *skel, int map_fd, + const char *init, const char *del, + const char *upd, const char *probe) +{ + u64 zero = 0; + u32 key = 0; + int i; + + for (i = 0; i < RECYCLE_LOOPS; i++) { + if (run_prog(skel, init) != 0) { + /* Element may be gone; recreate and retry once. */ + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &zero, BPF_ANY), + "recreate_elem")) + return; + if (!ASSERT_OK(run_prog(skel, init), init)) + return; + } + if (!ASSERT_OK(run_prog(skel, del), del)) + return; + if (!ASSERT_OK(run_prog(skel, upd), upd)) + return; + if (!ASSERT_OK(run_prog(skel, probe), probe)) + return; + } +} + +static void subtest_lock_kptr(struct rhtab_fields *skel) +{ + struct lock_kptr_val_user val = {}; + struct lock_kptr_val_user out = {}; + u64 nonnull_before; + u32 key = 0; + int map_fd; + + map_fd = bpf_map__fd(skel->maps.lkmap); + + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &val, BPF_ANY), + "create_elem")) + return; + + /* Spin lock must be usable from the syscall path (BPF_F_LOCK). */ + val.magic = LK_MAGIC; + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &val, BPF_F_LOCK), + "locked_update")) + return; + if (!ASSERT_OK(bpf_map_lookup_elem_flags(map_fd, &key, &out, BPF_F_LOCK), + "locked_lookup")) + return; + ASSERT_EQ(out.magic, LK_MAGIC, "locked_lookup_magic"); + + /* + * Delete/re-insert recycle cycles: the referenced kptr must be + * inherited on recycled elements (zeroing it would leak the + * reference) and the plain magic bytes must round-trip every time. + */ + nonnull_before = read_counter(skel, 1); + recycle_loop(skel, map_fd, "lk_init", "lk_del", "lk_upd", "lk_probe"); + ASSERT_GT(read_counter(skel, 1), nonnull_before, "recycle_xchg_non_null"); + ASSERT_EQ(read_counter(skel, 3), RECYCLE_LOOPS, "recycle_magic_roundtrip"); + + /* The spin lock must still work after many recycles. */ + val.magic = LK_MAGIC + 1; + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &val, BPF_F_LOCK), + "post_recycle_locked_update")) + return; + memset(&out, 0, sizeof(out)); + if (!ASSERT_OK(bpf_map_lookup_elem_flags(map_fd, &key, &out, BPF_F_LOCK), + "post_recycle_locked_lookup")) + return; + ASSERT_EQ(out.magic, LK_MAGIC + 1, "post_recycle_locked_magic"); +} + +static void subtest_timer(struct rhtab_fields *skel) +{ + u64 zero = 0; + u32 key = 0; + int fired, map_fd; + + map_fd = bpf_map__fd(skel->maps.tmap); + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &zero, BPF_ANY), + "create_elem")) + return; + + if (!ASSERT_OK(run_prog(skel, "arm_timer"), "arm_timer_first")) + return; + usleep(300000); + if (!ASSERT_GT(skel->bss->timer_fired, 0, "timer_fired_first")) + return; + + /* Deleting the element must cancel the timer. */ + fired = skel->bss->timer_fired; + if (!ASSERT_OK(bpf_map_delete_elem(map_fd, &key), "delete_elem")) + return; + usleep(300000); + ASSERT_EQ(skel->bss->timer_fired, fired, "timer_cancelled_after_delete"); + + /* + * Re-insert (may recycle the freed element): the timer field must be + * re-initialized so a fresh timer can be armed again. + */ + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &zero, BPF_ANY), + "recreate_elem")) + return; + if (!ASSERT_OK(run_prog(skel, "arm_timer"), "arm_timer_second")) + return; + usleep(300000); + ASSERT_GT(skel->bss->timer_fired, fired, "timer_fired_second"); +} + +static void subtest_kptr_untrusted(struct rhtab_fields *skel) +{ + u64 nonnull_before; + u64 zero = 0; + u32 key = 0; + int map_fd; + + map_fd = bpf_map__fd(skel->maps.umap); + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &zero, BPF_ANY), + "create_elem")) + return; + + /* The untrusted kptr must survive the recycle like a referenced one. */ + nonnull_before = read_counter(skel, 5); + recycle_loop(skel, map_fd, "u_init", "u_del", "u_upd", "u_probe"); + ASSERT_GT(read_counter(skel, 5), nonnull_before, "recycle_unref_non_null"); +} + +static void subtest_kptr_percpu(struct rhtab_fields *skel) +{ + u64 nonnull_before; + u64 zero = 0; + u32 key = 0; + int map_fd; + + map_fd = bpf_map__fd(skel->maps.pcmap); + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, &zero, BPF_ANY), + "create_elem")) + return; + + /* The per-cpu kptr reference must survive the recycle (no leak). */ + nonnull_before = read_counter(skel, 7); + recycle_loop(skel, map_fd, "pc_init", "pc_del", "pc_upd", "pc_probe"); + ASSERT_GT(read_counter(skel, 7), nonnull_before, "recycle_pcpu_non_null"); +} + +void test_rhtab_fields(void) +{ + struct rhtab_fields *skel; + + skel = rhtab_fields__open_and_load(); + if (!ASSERT_OK_PTR(skel, "open_and_load")) + return; + + if (test__start_subtest("lock_kptr")) + subtest_lock_kptr(skel); + if (test__start_subtest("timer")) + subtest_timer(skel); + if (test__start_subtest("kptr_untrusted")) + subtest_kptr_untrusted(skel); + if (test__start_subtest("kptr_percpu")) + subtest_kptr_percpu(skel); + + rhtab_fields__destroy(skel); +} diff --git a/tools/testing/selftests/bpf/prog_tests/rhtab_kptr.c b/tools/testing/selftests/bpf/prog_tests/rhtab_kptr.c new file mode 100644 index 000000000000..13158d74cbc1 --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/rhtab_kptr.c @@ -0,0 +1,146 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 KylinSoft Co., Ltd. */ + +#include +#include +#include +#include +#include "rhtab_kptr.skel.h" + +static __u64 read_counter(struct rhtab_kptr *skel, u32 idx) +{ + __u64 vals[libbpf_num_possible_cpus()]; + __u64 sum = 0; + int i, err; + + err = bpf_map_lookup_elem(bpf_map__fd(skel->maps.counters), &idx, vals); + if (!ASSERT_OK(err, "lookup_counter")) + return 0; + for (i = 0; i < libbpf_num_possible_cpus(); i++) + sum += vals[i]; + return sum; +} + +void test_rhtab_kptr(void) +{ + struct perf_event_attr attr = { + .type = PERF_TYPE_HARDWARE, + .config = PERF_COUNT_HW_CPU_CYCLES, + .freq = 1, + .sample_freq = read_perf_max_sample_freq(), + .size = sizeof(struct perf_event_attr), + }; + LIBBPF_OPTS(bpf_test_run_opts, topts); + struct rhtab_kptr *skel; + __u32 key = 0; + __u64 zero = 0; + __u64 nonnull_before; + int pmu_fd, i, err; + + skel = rhtab_kptr__open_and_load(); + if (!ASSERT_OK_PTR(skel, "open_and_load")) + return; + + /* Create the element and stash a referenced task kptr in it. */ + if (!ASSERT_OK(bpf_map_update_elem(bpf_map__fd(skel->maps.rhtab), + &key, &zero, BPF_ANY), "create_elem")) + goto out; + if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(skel->progs.init_elem), + &topts), "test_run_init") || + !ASSERT_EQ(topts.retval, 0, "init_ret")) + goto out; + + pmu_fd = syscall(__NR_perf_event_open, &attr, -1, 0, -1, 0); + if (pmu_fd >= 0) { + skel->links.nmi_update = bpf_program__attach_perf_event(skel->progs.nmi_update, + pmu_fd); + if (!ASSERT_OK_PTR(skel->links.nmi_update, "attach_perf_event")) { + close(pmu_fd); + goto out; + } + + /* Let the NMI handler overwrite the element, and make sure it + * actually ran before probing (otherwise the probe would pass + * vacuously even on an unfixed kernel). + */ + for (i = 0; i < 20 && read_counter(skel, 1) == 0; i++) + usleep(100000); + ASSERT_GT(read_counter(skel, 1), 0, "nmi_update_ran"); + + bpf_link__destroy(skel->links.nmi_update); + skel->links.nmi_update = NULL; + close(pmu_fd); + + /* + * The old kptr must still be attached to the element: the + * NMI update path only cancels NMI-safe fields, mirroring + * hash map semantics. Before the fix the kptr was released + * from the NMI context and the probe below would see NULL. + */ + topts.retval = 0; + if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(skel->progs.probe_elem), + &topts), "test_run_probe") || + !ASSERT_EQ(topts.retval, 0, "probe_ret")) + goto out; + + ASSERT_EQ(read_counter(skel, 2), 1, "xchg_non_null"); + ASSERT_EQ(read_counter(skel, 3), 0, "xchg_null"); + } else { + test__skip(); + } + + /* + * Now exercise the delete/re-insert recycle path. The delete only + * cancels NMI-safe fields, so the freed element still owns the kptr. + * If the re-insertion recycles that element, the kptr must be + * inherited; zeroing it (as check_and_init_map_value() did before + * the fix) leaks the reference and probe_elem() observes NULL. + * Fresh memory handed out by the allocator is zeroed, so NULL probes + * are expected too; only require that the inherited kptr survives at + * least one recycle. + */ + nonnull_before = read_counter(skel, 2); + for (i = 0; i < 2000; i++) { + topts.retval = 0; + err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.init_elem), + &topts); + if (err || topts.retval) { + /* Element may be gone; recreate and retry once. */ + if (!ASSERT_OK(bpf_map_update_elem(bpf_map__fd(skel->maps.rhtab), + &key, &zero, BPF_ANY), + "recreate_elem")) + goto out; + topts.retval = 0; + err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.init_elem), + &topts); + } + if (!ASSERT_OK(err, "test_run_init_loop") || + !ASSERT_EQ(topts.retval, 0, "init_loop_ret")) + goto out; + + topts.retval = 0; + if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(skel->progs.del_elem), + &topts), "test_run_del")) + goto out; + topts.retval = 0; + if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(skel->progs.upd_elem), + &topts), "test_run_upd")) + goto out; + topts.retval = 0; + if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(skel->progs.probe_elem), + &topts), "test_run_probe")) + goto out; + } + + /* + * Plain (non-special) value bytes must survive the recycle path: + * every probe must observe the magic value written by upd_elem() in + * the same iteration, regardless of whether the element memory was + * recycled or freshly allocated. + */ + ASSERT_EQ(read_counter(skel, 4), 2000, "recycle_magic_roundtrip"); + + ASSERT_GT(read_counter(skel, 2), nonnull_before, "recycle_xchg_non_null"); +out: + rhtab_kptr__destroy(skel); +} diff --git a/tools/testing/selftests/bpf/progs/rhtab_fields.c b/tools/testing/selftests/bpf/progs/rhtab_fields.c new file mode 100644 index 000000000000..85335f19f172 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/rhtab_fields.c @@ -0,0 +1,305 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 KylinSoft Co., Ltd. */ + +/* + * Combined special-field tests for BPF_MAP_TYPE_RHASH. Each map carries a + * different field combination and is exercised through delete/re-insert + * cycles so the bpf memory allocator recycles element memory: + * + * 1. lkmap: bpf_spin_lock + referenced kptr + plain data in one value. + * After every recycle the spin lock must still be usable (initialized by + * the alloc path), the referenced kptr must be inherited instead of + * zeroed (zeroing would leak the reference), and the plain bytes must + * round-trip. + * 2. tmap: bpf_timer. The delete path must cancel the timer, and a recycled + * element must be able to arm a fresh timer again. + * 3. umap: untrusted (unreferenced) kptr. The inherited pointer must be + * preserved on recycle, matching hash map behavior. + * 4. pcmap: per-cpu kptr. Like the referenced kptr, the per-cpu reference + * must not be dropped on recycle. + */ + +#include +#include +#include "bpf_experimental.h" + +char LICENSE[] SEC("license") = "GPL"; + +struct lock_kptr_val { + struct bpf_spin_lock lock; + struct task_struct __kptr * tsk; + __u32 magic; +}; + +struct { + __uint(type, BPF_MAP_TYPE_RHASH); + __uint(max_entries, 16); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, __u32); + __type(value, struct lock_kptr_val); +} lkmap SEC(".maps"); + +struct timer_val { + struct bpf_timer timer; + __u64 data; +}; + +struct { + __uint(type, BPF_MAP_TYPE_RHASH); + __uint(max_entries, 16); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, __u32); + __type(value, struct timer_val); +} tmap SEC(".maps"); + +struct unref_val { + struct task_struct __kptr_untrusted * tsk; +}; + +struct { + __uint(type, BPF_MAP_TYPE_RHASH); + __uint(max_entries, 16); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, __u32); + __type(value, struct unref_val); +} umap SEC(".maps"); + +struct pcval { + __u64 v; +}; + +struct pcpu_val { + struct pcval __percpu_kptr * pc; +}; + +struct { + __uint(type, BPF_MAP_TYPE_RHASH); + __uint(max_entries, 16); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, __u32); + __type(value, struct pcpu_val); +} pcmap SEC(".maps"); + +struct { + __uint(type, BPF_MAP_TYPE_PERCPU_ARRAY); + __uint(max_entries, 9); + __type(key, __u32); + __type(value, __u64); +} counters SEC(".maps"); + +/* 0: lk init ok, 1: lk probe xchg non-NULL, 2: lk probe xchg NULL, + * 3: lk probe magic ok, 4: u init ok, 5: u probe ptr non-NULL, + * 6: pc init ok, 7: pc probe xchg non-NULL, 8: pc probe xchg NULL + */ +static __always_inline void bump(u32 idx) +{ + u64 *v = bpf_map_lookup_elem(&counters, &idx); + + if (v) + (*v)++; +} + +extern struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym; +extern void bpf_task_release(struct task_struct *p) __ksym; + +int timer_fired; + +/* Map 1: spin lock + referenced kptr + plain data. */ + +SEC("syscall") +int lk_init(void *ctx) +{ + struct lock_kptr_val *val; + struct task_struct *task, *old; + u32 key = 0; + + val = bpf_map_lookup_elem(&lkmap, &key); + if (!val) + return 1; + task = bpf_task_acquire(bpf_get_current_task_btf()); + if (!task) + return 2; + old = bpf_kptr_xchg(&val->tsk, task); + if (old) + bpf_task_release(old); + bump(0); + return 0; +} + +SEC("syscall") +int lk_del(void *ctx) +{ + u32 key = 0; + + bpf_map_delete_elem(&lkmap, &key); + return 0; +} + +SEC("syscall") +int lk_upd(void *ctx) +{ + struct lock_kptr_val val = { .magic = 0x52484142 }; + u32 key = 0; + + bpf_map_update_elem(&lkmap, &key, &val, BPF_ANY); + return 0; +} + +SEC("syscall") +int lk_probe(void *ctx) +{ + struct lock_kptr_val *val; + struct task_struct *old; + u32 key = 0; + + val = bpf_map_lookup_elem(&lkmap, &key); + if (!val) + return 1; + old = bpf_kptr_xchg(&val->tsk, NULL); + if (old) { + bpf_task_release(old); + bump(1); + } else { + bump(2); + } + if (val->magic == 0x52484142) + bump(3); + return 0; +} + +/* Map 2: bpf_timer. */ + +static int timer_cb(void *map, void *key, struct timer_val *value) +{ + timer_fired++; + return 0; +} + +SEC("syscall") +int arm_timer(void *ctx) +{ + struct timer_val *val; + u32 key = 0; + + val = bpf_map_lookup_elem(&tmap, &key); + if (!val) + return 1; + /* 1 == CLOCK_MONOTONIC */ + if (bpf_timer_init(&val->timer, &tmap, 1)) + return 2; + bpf_timer_set_callback(&val->timer, timer_cb); + if (bpf_timer_start(&val->timer, 50000, 0)) + return 3; + return 0; +} + +/* Map 3: untrusted kptr. */ + +SEC("syscall") +int u_init(void *ctx) +{ + struct unref_val *val; + u32 key = 0; + + val = bpf_map_lookup_elem(&umap, &key); + if (!val) + return 1; + val->tsk = bpf_get_current_task_btf(); + bump(4); + return 0; +} + +SEC("syscall") +int u_del(void *ctx) +{ + u32 key = 0; + + bpf_map_delete_elem(&umap, &key); + return 0; +} + +SEC("syscall") +int u_upd(void *ctx) +{ + struct unref_val val = {}; + u32 key = 0; + + bpf_map_update_elem(&umap, &key, &val, BPF_ANY); + return 0; +} + +SEC("syscall") +int u_probe(void *ctx) +{ + struct unref_val *val; + u32 key = 0; + + val = bpf_map_lookup_elem(&umap, &key); + if (!val) + return 1; + if (val->tsk) + bump(5); + val->tsk = NULL; + return 0; +} + +/* Map 4: per-cpu kptr. */ + +SEC("syscall") +int pc_init(void *ctx) +{ + struct pcpu_val *val; + struct pcval *p, *old; + u32 key = 0; + + val = bpf_map_lookup_elem(&pcmap, &key); + if (!val) + return 1; + p = bpf_percpu_obj_new(struct pcval); + if (!p) + return 2; + old = bpf_kptr_xchg(&val->pc, p); + if (old) + bpf_percpu_obj_drop(old); + bump(6); + return 0; +} + +SEC("syscall") +int pc_del(void *ctx) +{ + u32 key = 0; + + bpf_map_delete_elem(&pcmap, &key); + return 0; +} + +SEC("syscall") +int pc_upd(void *ctx) +{ + struct pcpu_val val = {}; + u32 key = 0; + + bpf_map_update_elem(&pcmap, &key, &val, BPF_ANY); + return 0; +} + +SEC("syscall") +int pc_probe(void *ctx) +{ + struct pcpu_val *val; + struct pcval *old; + u32 key = 0; + + val = bpf_map_lookup_elem(&pcmap, &key); + if (!val) + return 1; + old = bpf_kptr_xchg(&val->pc, NULL); + if (old) { + bpf_percpu_obj_drop(old); + bump(7); + } else { + bump(8); + } + return 0; +} diff --git a/tools/testing/selftests/bpf/progs/rhtab_kptr.c b/tools/testing/selftests/bpf/progs/rhtab_kptr.c new file mode 100644 index 000000000000..fd6bd63cb405 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/rhtab_kptr.c @@ -0,0 +1,132 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 KylinSoft Co., Ltd. */ + +/* + * Verify that the rhtab update/delete recycle paths do not eagerly destroy + * referenced kptrs. rhtab must match the hash map semantics introduced by + * commit a3a81d247651 ("bpf: Cancel special fields on map value recycle"): + * only NMI-safe fields (timer, workqueue, task_work) are cancelled on + * update/delete, while kptrs stay attached to the recycled element until it + * is eventually freed. + * + * Two paths are exercised: + * 1. a perf_event (NMI) program overwrites an existing element; without the + * fix the NMI update releases the old kptr and probe_elem() observes + * NULL; + * 2. the element is deleted and re-inserted; the re-insertion may recycle + * the freed element, and zeroing the inherited kptr slot (as + * check_and_init_map_value() did before the fix) would drop the + * reference without releasing it. probe_elem() must observe the + * inherited non-NULL pointer, and plain (non-special) value bytes must + * still round-trip through the recycled element. + */ +#include +#include + +char LICENSE[] SEC("license") = "GPL"; + +struct val_t { + struct task_struct __kptr * tsk; + __u32 magic; +}; + +struct { + __uint(type, BPF_MAP_TYPE_RHASH); + __uint(max_entries, 16); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, __u32); + __type(value, struct val_t); +} rhtab SEC(".maps"); + +struct { + __uint(type, BPF_MAP_TYPE_PERCPU_ARRAY); + __uint(max_entries, 5); + __type(key, __u32); + __type(value, __u64); +} counters SEC(".maps"); + +/* 0: init ok, 1: nmi update ok, 2: probe xchg non-NULL, 3: probe xchg NULL, + * 4: probe saw expected magic value + */ +static __always_inline void bump(u32 idx) +{ + u64 *v = bpf_map_lookup_elem(&counters, &idx); + + if (v) + (*v)++; +} + +extern struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym; +extern void bpf_task_release(struct task_struct *p) __ksym; + +SEC("perf_event") +int nmi_update(struct bpf_perf_event_data *ctx) +{ + struct val_t val = {}; + u32 key = 0; + + if (bpf_map_update_elem(&rhtab, &key, &val, BPF_ANY) == 0) + bump(1); + return 0; +} + +SEC("syscall") +int init_elem(void *ctx) +{ + struct val_t *val; + struct task_struct *task, *old; + u32 key = 0; + + val = bpf_map_lookup_elem(&rhtab, &key); + if (!val) + return 1; + task = bpf_task_acquire(bpf_get_current_task_btf()); + if (!task) + return 2; + old = bpf_kptr_xchg(&val->tsk, task); + if (old) + bpf_task_release(old); + bump(0); + return 0; +} + +SEC("syscall") +int del_elem(void *ctx) +{ + u32 key = 0; + + bpf_map_delete_elem(&rhtab, &key); + return 0; +} + +SEC("syscall") +int upd_elem(void *ctx) +{ + struct val_t val = { .magic = 0x52484153 }; /* "RHAS" */ + u32 key = 0; + + bpf_map_update_elem(&rhtab, &key, &val, BPF_ANY); + return 0; +} + +SEC("syscall") +int probe_elem(void *ctx) +{ + struct val_t *val; + struct task_struct *old; + u32 key = 0; + + val = bpf_map_lookup_elem(&rhtab, &key); + if (!val) + return 1; + old = bpf_kptr_xchg(&val->tsk, NULL); + if (old) { + bpf_task_release(old); + bump(2); + } else { + bump(3); + } + if (val->magic == 0x52484153) + bump(4); + return 0; +}