From 0c5a9b3b41f1906ba9c326a34eae7087dfcafbb2 Mon Sep 17 00:00:00 2001 From: Varun R Mallya Date: Mon, 22 Jun 2026 03:36:52 +0530 Subject: [PATCH 01/25] Add kernel selftest equivalent roadmap tests --- pythonbpf/maps/__init__.py | 11 ++++- pythonbpf/maps/maps.py | 20 ++++++++ pythonbpf/maps/maps_pass.py | 8 ++++ tests/README.md | 11 ++++- tests/conftest.py | 14 ++++-- tests/framework/collector.py | 2 +- .../maps/array_map_lookup_update.py | 35 ++++++++++++++ .../ringbuf/reserve_submit_discard.py | 48 +++++++++++++++++++ tests/test_config.toml | 4 ++ 9 files changed, 146 insertions(+), 7 deletions(-) create mode 100644 tests/kernel_selftest_equivalent/maps/array_map_lookup_update.py create mode 100644 tests/kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py diff --git a/pythonbpf/maps/__init__.py b/pythonbpf/maps/__init__.py index eb2007da..fb64f8de 100644 --- a/pythonbpf/maps/__init__.py +++ b/pythonbpf/maps/__init__.py @@ -1,5 +1,12 @@ -from .maps import HashMap, PerfEventArray, RingBuffer +from .maps import ArrayMap, HashMap, PerfEventArray, RingBuffer from .maps_pass import maps_proc from .map_types import BPFMapType -__all__ = ["HashMap", "PerfEventArray", "maps_proc", "RingBuffer", "BPFMapType"] +__all__ = [ + "ArrayMap", + "HashMap", + "PerfEventArray", + "maps_proc", + "RingBuffer", + "BPFMapType", +] diff --git a/pythonbpf/maps/maps.py b/pythonbpf/maps/maps.py index 583e9570..12cd3a48 100644 --- a/pythonbpf/maps/maps.py +++ b/pythonbpf/maps/maps.py @@ -26,6 +26,26 @@ def update(self, key, value, flags=None): raise KeyError(f"Key {key} not found in map") +class ArrayMap: + def __init__(self, key, value, max_entries): + self.key = key + self.value = value + self.max_entries = max_entries + self.entries = {} + + def lookup(self, key): + return self.entries.get(key) + + def update(self, key, value, flags=None): + self.entries[key] = value + + def delete(self, key): + if key in self.entries: + del self.entries[key] + else: + raise KeyError(f"Key {key} not found in map") + + class PerfEventArray: def __init__(self, key_size, value_size): self.key_type = key_size diff --git a/pythonbpf/maps/maps_pass.py b/pythonbpf/maps/maps_pass.py index ca078454..91aa35c3 100644 --- a/pythonbpf/maps/maps_pass.py +++ b/pythonbpf/maps/maps_pass.py @@ -138,6 +138,14 @@ def process_hash_map(map_name, rval, compilation_context): return map_global +@MapProcessorRegistry.register("ArrayMap") +def process_array_map(map_name, rval, compilation_context): + """Document the planned BPF_ARRAY map support with an explicit failure.""" + raise NotImplementedError( + "ArrayMap is not implemented yet; add BPF_MAP_TYPE_ARRAY metadata support" + ) + + @MapProcessorRegistry.register("PerfEventArray") def process_perf_event_map(map_name, rval, compilation_context): """Process a BPF_PERF_EVENT_ARRAY map declaration""" diff --git a/tests/README.md b/tests/README.md index 6b63fd45..c7d18bae 100644 --- a/tests/README.md +++ b/tests/README.md @@ -88,6 +88,14 @@ All xfails use `strict = True`: if a test starts **passing** it shows up as **XP 2. Run `make test` — the file is discovered and tested automatically at all levels. 3. If the test is expected to fail, add it to `tests/test_config.toml` instead of `passing_tests/`. +## Kernel selftest equivalents + +`tests/kernel_selftest_equivalent/` contains PythonBPF versions of important +kernel BPF selftests from `bpf-next/tools/testing/selftests/bpf`. These tests +describe features PythonBPF should grow next. They are collected by default and +must be listed as strict expected failures in `tests/test_config.toml` until the +corresponding feature lands. + ## Directory structure ``` @@ -104,5 +112,6 @@ tests/ │ ├── compiler.py ← wrappers around compile_to_ir() + _run_llc() │ └── verifier.py ← bpftool subprocess wrapper ├── passing_tests/ ← programs that should compile and verify cleanly -└── failing_tests/ ← programs with known issues (declared in test_config.toml) +├── failing_tests/ ← programs with known issues (declared in test_config.toml) +└── kernel_selftest_equivalent/ ← kernel-selftest-inspired feature roadmap tests ``` diff --git a/tests/conftest.py b/tests/conftest.py index 42ab30ed..bcea4d33 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -16,6 +16,7 @@ """ import logging +import warnings import pytest @@ -25,11 +26,15 @@ # ── vmlinux availability ──────────────────────────────────────────────────── try: - import vmlinux # noqa: F401 + with warnings.catch_warnings(): + warnings.simplefilter("ignore", DeprecationWarning) + import vmlinux # noqa: F401 VMLINUX_AVAILABLE = True -except ImportError: + VMLINUX_SKIP_REASON = "" +except Exception as exc: VMLINUX_AVAILABLE = False + VMLINUX_SKIP_REASON = f"vmlinux.py not usable for current kernel: {exc}" # ── pytest_generate_tests: parametrize on bpf_test_file ─────────────────── @@ -65,7 +70,10 @@ def pytest_collection_modifyitems(items): # vmlinux skip if case.needs_vmlinux and not VMLINUX_AVAILABLE: item.add_marker( - pytest.mark.skip(reason="vmlinux.py not available for current kernel") + pytest.mark.skip( + reason=VMLINUX_SKIP_REASON + or "vmlinux.py not available for current kernel" + ) ) continue diff --git a/tests/framework/collector.py b/tests/framework/collector.py index bdafc149..856e4b04 100644 --- a/tests/framework/collector.py +++ b/tests/framework/collector.py @@ -33,7 +33,7 @@ def collect_all_test_files() -> list[BpfTestCase]: xfail_map: dict = config.get("xfail", {}) cases = [] - for subdir in ("passing_tests", "failing_tests"): + for subdir in ("passing_tests", "failing_tests", "kernel_selftest_equivalent"): for py_file in sorted((TESTS_DIR / subdir).rglob("*.py")): if py_file.name == "vmlinux.py": # Not a test case: the per-directory symlink to the master diff --git a/tests/kernel_selftest_equivalent/maps/array_map_lookup_update.py b/tests/kernel_selftest_equivalent/maps/array_map_lookup_update.py new file mode 100644 index 00000000..f84c2d3b --- /dev/null +++ b/tests/kernel_selftest_equivalent/maps/array_map_lookup_update.py @@ -0,0 +1,35 @@ +# Adapted from bpf-next/tools/testing/selftests/bpf/progs/test_map_ops.c +# and bpf-next/tools/testing/selftests/bpf/progs/bpf_iter_bpf_array_map.c. + +from ctypes import c_int32, c_uint64, c_void_p + +from pythonbpf import bpf, bpfglobal, compile, map, section +from pythonbpf.maps import ArrayMap + + +@bpf +@map +def counters() -> ArrayMap: + return ArrayMap(key=c_int32, value=c_uint64, max_entries=8) + + +@bpf +@section("tracepoint/syscalls/sys_enter_getpid") +def array_map_lookup_update(ctx: c_void_p) -> c_int32: + counters.update(0, 1) + + current = counters.lookup(0) + if current: + next_value = current + 1 + counters.update(0, next_value) + + return c_int32(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py b/tests/kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py new file mode 100644 index 00000000..0c3b1314 --- /dev/null +++ b/tests/kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py @@ -0,0 +1,48 @@ +# Adapted from bpf-next/tools/testing/selftests/bpf/progs/test_ringbuf.c. + +from ctypes import c_int32, c_uint64, c_void_p + +from pythonbpf import bpf, bpfglobal, compile, map, section, struct +from pythonbpf.helper import pid +from pythonbpf.maps import RingBuffer + + +@bpf +@struct +class sample_t: + pid: c_uint64 + seq: c_uint64 + value: c_uint64 + + +@bpf +@map +def events() -> RingBuffer: + return RingBuffer(max_entries=4096) + + +@bpf +@section("tracepoint/syscalls/sys_enter_getpid") +def ringbuf_reserve_submit_discard(ctx: c_void_p) -> c_int32: + first = events.reserve(24) + if first: + sample = sample_t(first) + sample.pid = pid() + sample.seq = 0 + sample.value = 7 + events.submit(first, 0) + + second = events.reserve(24) + if second: + events.discard(second, 0) + + return c_int32(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/test_config.toml b/tests/test_config.toml index 8a6255a3..894c8162 100644 --- a/tests/test_config.toml +++ b/tests/test_config.toml @@ -32,3 +32,7 @@ "failing_tests/vmlinux/assignment_handling.py" = {reason = "Assigning vmlinux enum value (XDP_PASS) to a local variable not yet supported", level = "ir"} "failing_tests/xdp_pass.py" = {reason = "XDP program using vmlinux structs (struct_xdp_md) and complex map/struct interaction not yet supported", level = "ir"} + +"kernel_selftest_equivalent/maps/array_map_lookup_update.py" = {reason = "ArrayMap / BPF_MAP_TYPE_ARRAY support is planned but not implemented yet", level = "ir"} + +"kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py" = {reason = "RingBuffer reserve/typed record/discard workflow is planned but not implemented yet", level = "ir"} From a64f01d6b3efb79a1c47b436d39bcc02fe129b3e Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 14 Aug 2026 22:34:07 +0530 Subject: [PATCH 02/25] Tests: Register a vmlinux subdirectory under kernel_selftest_equivalent Ports that import from vmlinux need the same treatment as tests/passing_tests/vmlinux/ -- skipped, not failed, when no vmlinux.py has been generated for the running kernel. Without this they fail outright wherever one is absent, CI included. Co-Authored-By: Claude Opus 5 (1M context) --- tests/framework/collector.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/framework/collector.py b/tests/framework/collector.py index 856e4b04..59b96084 100644 --- a/tests/framework/collector.py +++ b/tests/framework/collector.py @@ -7,7 +7,10 @@ TESTS_DIR = Path(__file__).parent.parent CONFIG_FILE = TESTS_DIR / "test_config.toml" -VMLINUX_TEST_DIRS_PASSING = {"passing_tests/vmlinux"} +VMLINUX_TEST_DIRS_PASSING = { + "passing_tests/vmlinux", + "kernel_selftest_equivalent/vmlinux", +} VMLINUX_TEST_DIRS_FAILING = { "failing_tests/vmlinux", "failing_tests/xdp", From 7911436996d1e48b835fa43cc8256350b23b3e2d Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 14 Aug 2026 22:34:08 +0530 Subject: [PATCH 03/25] Tests: Port three kernel BPF selftests Ports of programs from tools/testing/selftests/bpf/progs/ in the Linux tree, each naming its upstream original in a header comment. tracing/tracepoint_sched_switch.py test_tracepoint.c tracing/get_cgroup_id.py get_cgroup_id_kern.c tracing/autoattach.py test_autoattach.c Deliberately a small spike rather than bulk coverage. The three sample different global-variable shapes, since substituting for globals is what porting the rest of the corpus will mostly consist of: none at all (the control case), a scalar read plus a scalar write, and two flags written from two programs on two different attach points. They also widen the range of program types under test. tracepoint/sched/sched_switch and raw_tp/sys_enter were previously unexercised; @section writes its string straight into the ELF with no allowlist, so that pass-through had only ever been tested against a handful of types. Only the BPF half of each selftest is ported -- upstream pairs every program with a userspace driver in prog_tests/ that loads and asserts, whereas this framework compiles and verifies but never runs. tests/README.md now says so explicitly, along with the WORKAROUND(globals) convention marking each map substituted for a global. Co-Authored-By: Claude Opus 5 (1M context) --- tests/README.md | 48 +++++++++++++++++-- .../tracing/autoattach.py | 42 ++++++++++++++++ .../tracing/get_cgroup_id.py | 47 ++++++++++++++++++ .../tracing/tracepoint_sched_switch.py | 27 +++++++++++ 4 files changed, 159 insertions(+), 5 deletions(-) create mode 100644 tests/kernel_selftest_equivalent/tracing/autoattach.py create mode 100644 tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py create mode 100644 tests/kernel_selftest_equivalent/tracing/tracepoint_sched_switch.py diff --git a/tests/README.md b/tests/README.md index c7d18bae..50f7dd48 100644 --- a/tests/README.md +++ b/tests/README.md @@ -91,10 +91,48 @@ All xfails use `strict = True`: if a test starts **passing** it shows up as **XP ## Kernel selftest equivalents `tests/kernel_selftest_equivalent/` contains PythonBPF versions of important -kernel BPF selftests from `bpf-next/tools/testing/selftests/bpf`. These tests -describe features PythonBPF should grow next. They are collected by default and -must be listed as strict expected failures in `tests/test_config.toml` until the -corresponding feature lands. +kernel BPF selftests from `bpf-next/tools/testing/selftests/bpf`. Each file names +its upstream original in a header comment. + +The directory holds two kinds of test, and both are useful: + +- **Ports that pass.** A program PythonBPF can already express. These widen the + range of program types under test — `raw_tp`, `perf_event`, + `tracepoint/sched/*` and others that nothing else exercises. +- **Roadmap tests that fail.** A program describing a feature PythonBPF should + grow next. These must be listed as **strict** expected failures in + `tests/test_config.toml` until the feature lands, at which point they turn up + as XPASS and should be promoted. + +### What a passing port proves — and does not + +A kernel selftest is two halves: the BPF program under `progs/`, and a userspace +driver under `prog_tests/` that loads it through a skeleton, triggers it and +asserts on the result. **Only the BPF half is ported**, because this framework +compiles and verifies programs but never runs them. + +So a passing test here says PythonBPF emits a loadable, verifiable object for +that program type and feature mix. It does not say the program behaves the way +the kernel's version does. Treat it as a compiler assertion, not a semantic one. + +### `WORKAROUND(globals)` + +The selftest corpus overwhelmingly reports results through global variables: the +program writes a global and the driver reads it back. PythonBPF has no global +variable support, so each becomes a one-entry `HashMap` keyed by index, tagged in +a comment naming the variable it replaces: + +```bash +grep -rn "WORKAROUND(globals)" tests/kernel_selftest_equivalent/ +``` + +This is deliberate scaffolding, not the intended shape — the tag exists so the +sweep is mechanical once real globals land. It is not a cosmetic substitution +either: it changes what a future userspace driver would read. + +Anything importing from `vmlinux` belongs in `vmlinux/`, which is registered in +`VMLINUX_TEST_DIRS_PASSING` so it is skipped rather than failed where no +`vmlinux.py` has been generated. ## Directory structure @@ -113,5 +151,5 @@ tests/ │ └── verifier.py ← bpftool subprocess wrapper ├── passing_tests/ ← programs that should compile and verify cleanly ├── failing_tests/ ← programs with known issues (declared in test_config.toml) -└── kernel_selftest_equivalent/ ← kernel-selftest-inspired feature roadmap tests +└── kernel_selftest_equivalent/ ← ports of kernel selftests + feature roadmap tests ``` diff --git a/tests/kernel_selftest_equivalent/tracing/autoattach.py b/tests/kernel_selftest_equivalent/tracing/autoattach.py new file mode 100644 index 00000000..7a898699 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/autoattach.py @@ -0,0 +1,42 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_autoattach.c +# +# Two programs on different raw tracepoints, each recording that it ran. The +# upstream test asserts both fired after bpf_object__attach_skeleton(). +# +# WORKAROUND(globals): upstream uses `bool prog1_called` / `bool prog2_called`. +# PythonBPF has no global variable support yet, so both live in one HashMap +# keyed by program number. Replace with real globals once they land. + +from pythonbpf import bpf, map, section, bpfglobal, compile +from pythonbpf.maps import HashMap +from ctypes import c_void_p, c_int64, c_int32, c_uint64 + + +# WORKAROUND(globals): key 1 -> prog1_called, key 2 -> prog2_called +@bpf +@map +def called() -> HashMap: + return HashMap(key=c_int32, value=c_uint64, max_entries=2) + + +@bpf +@section("raw_tp/sys_enter") +def prog1(ctx: c_void_p) -> c_int64: + called.update(1, 1) + return c_int64(0) + + +@bpf +@section("raw_tp/sys_exit") +def prog2(ctx: c_void_p) -> c_int64: + called.update(2, 1) + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py b/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py new file mode 100644 index 00000000..b4462484 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py @@ -0,0 +1,47 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/get_cgroup_id_kern.c +# +# Upstream records the cgroup id of a process whose pid matches one the +# userspace half of the test set beforehand. +# +# WORKAROUND(globals): upstream uses the file-scope variables `cg_id` and +# `expected_pid` to pass values in and out. PythonBPF has no global variable +# support yet, so each becomes a one-entry HashMap keyed by 0. Replace these +# with real globals once they land; grep for WORKAROUND(globals). + +from pythonbpf import bpf, map, section, bpfglobal, compile +from pythonbpf.maps import HashMap +from pythonbpf.helper import pid, get_current_cgroup_id +from ctypes import c_void_p, c_int64, c_int32, c_uint64 + + +# WORKAROUND(globals): stands in for `__u64 expected_pid;` +@bpf +@map +def expected_pid() -> HashMap: + return HashMap(key=c_int32, value=c_uint64, max_entries=1) + + +# WORKAROUND(globals): stands in for `__u64 cg_id;` +@bpf +@map +def cg_id() -> HashMap: + return HashMap(key=c_int32, value=c_uint64, max_entries=1) + + +@bpf +@section("tracepoint/syscalls/sys_enter_nanosleep") +def trace(ctx: c_void_p) -> c_int64: + process_id = pid() + want = expected_pid.lookup(0) + if want == process_id: + cg_id.update(0, get_current_cgroup_id()) + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/tracepoint_sched_switch.py b/tests/kernel_selftest_equivalent/tracing/tracepoint_sched_switch.py new file mode 100644 index 00000000..9252e0fd --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/tracepoint_sched_switch.py @@ -0,0 +1,27 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_tracepoint.c +# +# Upstream is a bare handler on sched/sched_switch, used to prove the program +# attaches to a non-syscall tracepoint. Kept faithful: the point is the +# attachment surface, not the body. +# +# Upstream declares the tracepoint argument layout as a struct taken from +# /sys/kernel/tracing/events/sched/sched_switch/format. PythonBPF does not read +# tracepoint formats, so the context stays opaque. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("tracepoint/sched/sched_switch") +def oncpu(ctx: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() From 0459b3a3c2b32b2e004f97bb1ef7ada66f0cfe16 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 14 Aug 2026 22:34:09 +0530 Subject: [PATCH 04/25] Tests: Add perf_skip as a nested-struct-access roadmap test Ported faithfully from test_perf_skip.c, which reads the sampled instruction pointer out of a perf_event context: ctx.regs.ip. That is two levels of struct field access and PythonBPF supports one, so this lands as a strict expected failure describing the gap. _allocate_for_attribute in allocation_pass.py declines to allocate unless the attribute's base is a plain Name, so the nested form is skipped. One level on the same context works today -- ctx.sample_period compiles and llc's fine. Worth noting for whoever picks this up: the failure surfaces as 'SyntaxError: Undefined variable actual', naming the assignment target rather than the nested access responsible. The allocation pass declines quietly and the expression pass then trips over the missing symbol. Co-Authored-By: Claude Opus 5 (1M context) --- .../vmlinux/perf_skip.py | 60 +++++++++++++++++++ tests/test_config.toml | 2 + 2 files changed, 62 insertions(+) create mode 100644 tests/kernel_selftest_equivalent/vmlinux/perf_skip.py diff --git a/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py b/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py new file mode 100644 index 00000000..a691774b --- /dev/null +++ b/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py @@ -0,0 +1,60 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_perf_skip.c +# +# A perf_event program that reports whether the sampled instruction pointer is +# the one userspace asked about. Upstream: +# +# uintptr_t ip; +# +# SEC("perf_event") +# int handler(struct bpf_perf_event_data *data) +# { +# /* Skip events that have the correct ip. */ +# return ip != PT_REGS_IP(&data->regs); +# } +# +# ROADMAP: this is a strict expected failure. `ctx.regs.ip` is two levels of +# struct field access, and PythonBPF supports only one -- +# `_allocate_for_attribute` in allocation_pass.py bails out unless the +# attribute's base is a plain Name. One level works today: `ctx.sample_period` +# on this same context compiles fine. +# +# Note the failure surfaces as `SyntaxError: Undefined variable actual`, naming +# the assignment target rather than the nested access that caused it -- the +# allocation pass declines to allocate and logs at debug level, then the +# expression pass fails later on the missing symbol. Worth improving alongside +# nested access support. +# +# WORKAROUND(globals): upstream uses `uintptr_t ip` to receive the address to +# compare against. PythonBPF has no global variable support yet, so it becomes a +# one-entry HashMap keyed by 0. Replace with a real global once they land. + +from pythonbpf import bpf, map, section, bpfglobal, compile +from pythonbpf.maps import HashMap +from vmlinux import struct_bpf_perf_event_data +from ctypes import c_int64, c_int32, c_uint64 + + +# WORKAROUND(globals): stands in for `uintptr_t ip;` +@bpf +@map +def expected_ip() -> HashMap: + return HashMap(key=c_int32, value=c_uint64, max_entries=1) + + +@bpf +@section("perf_event") +def handler(ctx: struct_bpf_perf_event_data) -> c_int64: + want = expected_ip.lookup(0) + actual = ctx.regs.ip + if want == actual: + return c_int64(0) + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/test_config.toml b/tests/test_config.toml index 894c8162..b71e48c6 100644 --- a/tests/test_config.toml +++ b/tests/test_config.toml @@ -36,3 +36,5 @@ "kernel_selftest_equivalent/maps/array_map_lookup_update.py" = {reason = "ArrayMap / BPF_MAP_TYPE_ARRAY support is planned but not implemented yet", level = "ir"} "kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py" = {reason = "RingBuffer reserve/typed record/discard workflow is planned but not implemented yet", level = "ir"} + +"kernel_selftest_equivalent/vmlinux/perf_skip.py" = {reason = "Nested struct field access (ctx.regs.ip) not supported; one level such as ctx.sample_period works", level = "ir"} From 507f07510dba80456b98daa8f10d41243c0be168 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 14 Aug 2026 22:35:01 +0530 Subject: [PATCH 05/25] Tests: Record what the first selftest porting spike found The four ports were the experiment; this is the result. Covers whether LLM-assisted porting is worth continuing (yes, in feature-sized increments), what a passing port actually proves (a compiler assertion, not a behavioural one), and the four distinct global-variable shapes the spike turned up. Includes the finding that libbpf implements globals as single-element BPF_MAP_TYPE_ARRAY maps, so a one-element ArrayMap -- not a HashMap -- is the structurally faithful stand-in, and that the ELF half of global support already works today via the machinery behind @bpfglobal. Co-Authored-By: Claude Opus 5 (1M context) --- .../PORTING-NOTES.md | 111 ++++++++++++++++++ 1 file changed, 111 insertions(+) create mode 100644 tests/kernel_selftest_equivalent/PORTING-NOTES.md diff --git a/tests/kernel_selftest_equivalent/PORTING-NOTES.md b/tests/kernel_selftest_equivalent/PORTING-NOTES.md new file mode 100644 index 00000000..645d3d22 --- /dev/null +++ b/tests/kernel_selftest_equivalent/PORTING-NOTES.md @@ -0,0 +1,111 @@ +# Porting kernel selftests: what the first spike found + +Four programs from `tools/testing/selftests/bpf/progs/` were ported as an experiment, +to answer two questions before anyone commits to doing this at scale: + +1. Is LLM-assisted porting of kernel selftests viable? +2. What must real global-variable support actually handle? + +Short answers: **viable, with a caveat about what a passing port proves**; and +**four distinct global shapes showed up in four programs**, which is the more +actionable finding. + +## The spike + +| Port | Upstream | Section | Outcome | +|---|---|---|---| +| `tracing/tracepoint_sched_switch.py` | `test_tracepoint.c` | `tracepoint/sched/sched_switch` | passes | +| `tracing/get_cgroup_id.py` | `get_cgroup_id_kern.c` | `tracepoint/syscalls/sys_enter_nanosleep` | passes | +| `tracing/autoattach.py` | `test_autoattach.c` | `raw_tp/sys_enter`, `raw_tp/sys_exit` | passes | +| `vmlinux/perf_skip.py` | `test_perf_skip.c` | `perf_event` | strict xfail — nested ctx access | + +## 1. Is it viable? + +**Yes, for programs inside the envelope — three of four compiled and passed `llc` on the +first attempt.** The mechanical part of a port (decorators, ctypes annotations, map +declarations, helper names) is regular enough to be reliable. + +The failure was not a translation error. `perf_skip` needs `ctx.regs.ip`, which PythonBPF +genuinely cannot express, and no amount of care in the port changes that. That is the +useful kind of failure: it converts into a roadmap test that documents the gap. + +Two caveats that matter more than the pass rate: + +**A passing port proves less than the test it came from.** A kernel selftest is two +halves — the BPF program, and a `prog_tests/` driver that loads it through a skeleton, +triggers it, and asserts on the result. Only the BPF half is portable here, because this +framework compiles and verifies but never runs. Everything ported becomes a compiler +assertion: *PythonBPF emits a loadable, verifiable object for this program type and +feature mix*. That is worth having — it is how the `raw_tp` and `perf_event` program types +came under test at all — but it is not what "we ported the kernel's selftests" sounds +like. Closing that gap needs a runtime test tier, which is a much larger piece of work. + +**Selection is the expensive step, not translation.** Of 820 real programs, 28 are +portable today. Picking those out required scoring the whole corpus against the compiler's +actual envelope; guessing from filenames does not work. The classifier that did it is +worth keeping around and re-running after each feature lands. + +**Recommendation: viable and worth continuing, in small increments tied to features.** +Port a handful, let them reveal the next gap, fix the gap, port more. Bulk porting ahead of +the features would just produce a large pile of xfails. + +## 2. What real globals must support + +Every port that touches a global currently substitutes a one-entry `HashMap`, tagged +`WORKAROUND(globals)`. Four programs produced four distinct shapes: + +| Shape | Example | What globals must support | +|---|---|---| +| none | `tracepoint_sched_switch` | — (control case) | +| scalar in + scalar out | `get_cgroup_id` | read a global, write a different one | +| flags across programs | `autoattach` | two programs in one object sharing global state | +| scalar in, compared against ctx | `perf_skip` | read-only input set by userspace before attach | +| array + cursor *(next increment)* | `cgroup_preorder` | indexed writes and read-modify-write on a global | + +The last row is not in this spike but is the recommended next port precisely because it is +the most demanding shape: `result[idx++] = N` needs an array global *and* a read-modify-write +cursor, which together constrain the design more than anything here does. + +### A design note worth acting on + +**libbpf implements global variables as single-element `BPF_MAP_TYPE_ARRAY` maps.** +`.bss`, `.data` and `.rodata` become internal array maps at load time. Two consequences: + +- A one-element **`ArrayMap`** is the structurally faithful stand-in for a global, not a + `HashMap`. `HashMap` is used here only because `ArrayMap` is still a placeholder that + raises `NotImplementedError`. Landing `ArrayMap` first would make the eventual migration + to real globals close to mechanical. +- **Most of the ELF work is already done.** `@bpfglobal` is vestigial — a metadata carrier + for `LICENSE` — but the machinery behind it already emits globals that LLVM places into + `.bss` and `.data` correctly, and that libbpf already recognises: + + ``` + libbpf: map 'g.bss' (global data): at sec_idx 5, offset 0, flags 0. + libbpf: map 'g.data' (global data): at sec_idx 6, offset 0, flags 0. + ``` + + What is missing is narrower than "implement global variables": name resolution in + `expr_pass.get_operand_value` (which resolves against `local_sym_tab`, then vmlinux + enums, then gives up), a Python-level surface for declaring one, and userspace access + through `pylibbpf`. + +## 3. Incidental findings + +- **Nested struct field access fails with a misleading error.** `ctx.regs.ip` reports + `SyntaxError: Undefined variable actual` — naming the assignment target rather than the + nested access that caused it. `_allocate_for_attribute` declines to allocate when the + attribute's base is not a plain `Name`, logging at debug level, and the expression pass + then trips over the missing symbol. The diagnostic should name the real cause. +- **One level of nested-context access already works.** `ctx.sample_period` on + `struct_bpf_perf_event_data` compiles and `llc`s cleanly, so `perf_event` contexts are + usable today for anything that does not need `regs`. +- **`@section` really does accept anything.** `tc`, `socket`, `fentry/…`, `lsm/…`, + `cgroup_skb/egress`, `netfilter` and `tp_btf/…` all compile and land in the ELF verbatim. + Program type is not a constraint; the context type is. + +## Re-running the corpus scoring + +The audit behind this spike scored all 976 programs against the compiler's envelope. It is +worth re-running after each feature lands, to see what the change unlocked. The blocker +histogram at the time of writing, over 820 real programs: globals 52%, verifier-test +annotations 27%, typed program macros 26%, kfuncs 20%, inline asm 16%, loops 12%. From 3ffba47a80481fdc0e9ba198d76638643999bdce Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 15:15:58 +0000 Subject: [PATCH 06/25] Tests: Replace the WORKAROUND(globals) HashMaps with real @bpfglobal scalars The selftest spike substituted a one-entry HashMap for every upstream global because PythonBPF had no global variable support. Integer-scalar @bpfglobal support has since landed, so the three affected ports now declare the upstream globals directly: get_cgroup_id reads expected_pid and writes cg_id, autoattach shares prog1_called/prog2_called between two programs, and perf_skip reads ip. perf_skip stays a strict xfail on the nested ctx.regs.ip access. Update the tests README and porting notes accordingly; the WORKAROUND tag no longer exists in the tree. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01DLMW3X7opqAPH6bjmTVkr7 --- tests/README.md | 17 ++----- .../PORTING-NOTES.md | 22 +++++---- .../tracing/autoattach.py | 49 ++++++++++++++----- .../tracing/get_cgroup_id.py | 48 ++++++++++-------- .../vmlinux/perf_skip.py | 22 ++++----- 5 files changed, 90 insertions(+), 68 deletions(-) diff --git a/tests/README.md b/tests/README.md index 50f7dd48..a2314bec 100644 --- a/tests/README.md +++ b/tests/README.md @@ -115,20 +115,13 @@ So a passing test here says PythonBPF emits a loadable, verifiable object for that program type and feature mix. It does not say the program behaves the way the kernel's version does. Treat it as a compiler assertion, not a semantic one. -### `WORKAROUND(globals)` +### Globals The selftest corpus overwhelmingly reports results through global variables: the -program writes a global and the driver reads it back. PythonBPF has no global -variable support, so each becomes a one-entry `HashMap` keyed by index, tagged in -a comment naming the variable it replaces: - -```bash -grep -rn "WORKAROUND(globals)" tests/kernel_selftest_equivalent/ -``` - -This is deliberate scaffolding, not the intended shape — the tag exists so the -sweep is mechanical once real globals land. It is not a cosmetic substitution -either: it changes what a future userspace driver would read. +program writes a global and the driver reads it back. Ports keep that shape with +`@bpfglobal` scalars, so a future userspace driver reads the same `.bss`/`.data` +values the kernel's driver does. Anything a global cannot yet hold (arrays, +structs, strings) is a roadmap test, not a workaround. Anything importing from `vmlinux` belongs in `vmlinux/`, which is registered in `VMLINUX_TEST_DIRS_PASSING` so it is skipped rather than failed where no diff --git a/tests/kernel_selftest_equivalent/PORTING-NOTES.md b/tests/kernel_selftest_equivalent/PORTING-NOTES.md index 645d3d22..2e9b4022 100644 --- a/tests/kernel_selftest_equivalent/PORTING-NOTES.md +++ b/tests/kernel_selftest_equivalent/PORTING-NOTES.md @@ -51,16 +51,18 @@ the features would just produce a large pile of xfails. ## 2. What real globals must support -Every port that touches a global currently substitutes a one-entry `HashMap`, tagged -`WORKAROUND(globals)`. Four programs produced four distinct shapes: - -| Shape | Example | What globals must support | -|---|---|---| -| none | `tracepoint_sched_switch` | — (control case) | -| scalar in + scalar out | `get_cgroup_id` | read a global, write a different one | -| flags across programs | `autoattach` | two programs in one object sharing global state | -| scalar in, compared against ctx | `perf_skip` | read-only input set by userspace before attach | -| array + cursor *(next increment)* | `cgroup_preorder` | indexed writes and read-modify-write on a global | +At the time of the spike every port that touched a global substituted a one-entry +`HashMap`, tagged `WORKAROUND(globals)`. Integer-scalar `@bpfglobal` support has since +landed and the sweep is done: the three ports below now declare the upstream globals +directly. Four programs produced four distinct shapes: + +| Shape | Example | What globals must support | Status | +|---|---|---|---| +| none | `tracepoint_sched_switch` | — (control case) | passes | +| scalar in + scalar out | `get_cgroup_id` | read a global, write a different one | passes with `@bpfglobal` | +| flags across programs | `autoattach` | two programs in one object sharing global state | passes with `@bpfglobal` | +| scalar in, compared against ctx | `perf_skip` | read-only input set by userspace before attach | global fine; still xfail on `ctx.regs.ip` | +| array + cursor *(next increment)* | `cgroup_preorder` | indexed writes and read-modify-write on a global | needs array globals | The last row is not in this spike but is the recommended next port precisely because it is the most demanding shape: `result[idx++] = N` needs an array global *and* a read-modify-write diff --git a/tests/kernel_selftest_equivalent/tracing/autoattach.py b/tests/kernel_selftest_equivalent/tracing/autoattach.py index 7a898699..b6746f5b 100644 --- a/tests/kernel_selftest_equivalent/tracing/autoattach.py +++ b/tests/kernel_selftest_equivalent/tracing/autoattach.py @@ -1,35 +1,58 @@ # Ported from Linux tools/testing/selftests/bpf/progs/test_autoattach.c # # Two programs on different raw tracepoints, each recording that it ran. The -# upstream test asserts both fired after bpf_object__attach_skeleton(). +# upstream test asserts both fired after bpf_object__attach_skeleton(): # -# WORKAROUND(globals): upstream uses `bool prog1_called` / `bool prog2_called`. -# PythonBPF has no global variable support yet, so both live in one HashMap -# keyed by program number. Replace with real globals once they land. +# bool prog1_called = false; +# bool prog2_called = false; +# +# SEC("raw_tp/sys_enter") +# int prog1(const void *ctx) +# { +# prog1_called = true; +# return 0; +# } +# +# SEC("raw_tp/sys_exit") +# int prog2(const void *ctx) +# { +# prog2_called = true; +# return 0; +# } +# +# Both flags are @bpfglobal scalars shared by the two programs in one object. +# They are c_uint64 rather than bool because integer scalars are the only +# global type today; the driver-side check is the same either way. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_uint64 -from pythonbpf import bpf, map, section, bpfglobal, compile -from pythonbpf.maps import HashMap -from ctypes import c_void_p, c_int64, c_int32, c_uint64 + +@bpf +@bpfglobal +def prog1_called() -> c_uint64: + return c_uint64(0) -# WORKAROUND(globals): key 1 -> prog1_called, key 2 -> prog2_called @bpf -@map -def called() -> HashMap: - return HashMap(key=c_int32, value=c_uint64, max_entries=2) +@bpfglobal +def prog2_called() -> c_uint64: + return c_uint64(0) @bpf @section("raw_tp/sys_enter") def prog1(ctx: c_void_p) -> c_int64: - called.update(1, 1) + global prog1_called + prog1_called = 1 return c_int64(0) @bpf @section("raw_tp/sys_exit") def prog2(ctx: c_void_p) -> c_int64: - called.update(2, 1) + global prog2_called + prog2_called = 1 return c_int64(0) diff --git a/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py b/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py index b4462484..506978dd 100644 --- a/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py +++ b/tests/kernel_selftest_equivalent/tracing/get_cgroup_id.py @@ -1,40 +1,48 @@ # Ported from Linux tools/testing/selftests/bpf/progs/get_cgroup_id_kern.c # # Upstream records the cgroup id of a process whose pid matches one the -# userspace half of the test set beforehand. +# userspace half of the test set beforehand: # -# WORKAROUND(globals): upstream uses the file-scope variables `cg_id` and -# `expected_pid` to pass values in and out. PythonBPF has no global variable -# support yet, so each becomes a one-entry HashMap keyed by 0. Replace these -# with real globals once they land; grep for WORKAROUND(globals). +# __u64 cg_id; +# __u64 expected_pid; +# +# SEC("tracepoint/syscalls/sys_enter_nanosleep") +# int trace(void *ctx) +# { +# __u32 pid = bpf_get_current_pid_tgid(); +# +# if (expected_pid == pid) +# cg_id = bpf_get_current_cgroup_id(); +# +# return 0; +# } +# +# Both file-scope variables are @bpfglobal scalars, so the userspace half reads +# `cg_id` back out of the object's .bss exactly as the kernel's driver does. -from pythonbpf import bpf, map, section, bpfglobal, compile -from pythonbpf.maps import HashMap +from pythonbpf import bpf, section, bpfglobal, compile from pythonbpf.helper import pid, get_current_cgroup_id -from ctypes import c_void_p, c_int64, c_int32, c_uint64 +from ctypes import c_void_p, c_int64, c_uint64 -# WORKAROUND(globals): stands in for `__u64 expected_pid;` @bpf -@map -def expected_pid() -> HashMap: - return HashMap(key=c_int32, value=c_uint64, max_entries=1) +@bpfglobal +def cg_id() -> c_uint64: + return c_uint64(0) -# WORKAROUND(globals): stands in for `__u64 cg_id;` @bpf -@map -def cg_id() -> HashMap: - return HashMap(key=c_int32, value=c_uint64, max_entries=1) +@bpfglobal +def expected_pid() -> c_uint64: + return c_uint64(0) @bpf @section("tracepoint/syscalls/sys_enter_nanosleep") def trace(ctx: c_void_p) -> c_int64: - process_id = pid() - want = expected_pid.lookup(0) - if want == process_id: - cg_id.update(0, get_current_cgroup_id()) + global cg_id + if expected_pid == pid(): + cg_id = get_current_cgroup_id() return c_int64(0) diff --git a/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py b/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py index a691774b..fc146710 100644 --- a/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py +++ b/tests/kernel_selftest_equivalent/vmlinux/perf_skip.py @@ -12,6 +12,9 @@ # return ip != PT_REGS_IP(&data->regs); # } # +# `ip` is a @bpfglobal the driver sets before attaching; the program only +# reads it, so no `global` statement is needed. +# # ROADMAP: this is a strict expected failure. `ctx.regs.ip` is two levels of # struct field access, and PythonBPF supports only one -- # `_allocate_for_attribute` in allocation_pass.py bails out unless the @@ -23,30 +26,23 @@ # allocation pass declines to allocate and logs at debug level, then the # expression pass fails later on the missing symbol. Worth improving alongside # nested access support. -# -# WORKAROUND(globals): upstream uses `uintptr_t ip` to receive the address to -# compare against. PythonBPF has no global variable support yet, so it becomes a -# one-entry HashMap keyed by 0. Replace with a real global once they land. -from pythonbpf import bpf, map, section, bpfglobal, compile -from pythonbpf.maps import HashMap +from pythonbpf import bpf, section, bpfglobal, compile from vmlinux import struct_bpf_perf_event_data -from ctypes import c_int64, c_int32, c_uint64 +from ctypes import c_int64, c_uint64 -# WORKAROUND(globals): stands in for `uintptr_t ip;` @bpf -@map -def expected_ip() -> HashMap: - return HashMap(key=c_int32, value=c_uint64, max_entries=1) +@bpfglobal +def ip() -> c_uint64: + return c_uint64(0) @bpf @section("perf_event") def handler(ctx: struct_bpf_perf_event_data) -> c_int64: - want = expected_ip.lookup(0) actual = ctx.regs.ip - if want == actual: + if ip == actual: return c_int64(0) return c_int64(1) From 0bdea85020b7ee31c1b9aed53797254ef176fab6 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 15:59:35 +0000 Subject: [PATCH 07/25] Core: Store a context field into a narrower slot through convert() A sub-register context field is loaded zero-extended to i64, and the assignment path accepted it only into a 64-bit slot or one whose type was exactly the field's declared type. The second case never worked in practice: `data_end = skb.data_end` into a c_uint32 global died in llvmlite with "cannot store i64 to i32*" because the value was still i64. Route the store through convert(), sizing from the physical value and signing from the field, like every other integer store since the signedness work: a no-op into an i64 slot, a trunc into anything narrower. Found by porting cgroup_skb_direct_packet_access.c, which is that exact statement. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01DLMW3X7opqAPH6bjmTVkr7 --- pythonbpf/assign_pass.py | 22 ++++--------- .../vmlinux/ctx_field_narrow_store.py | 32 +++++++++++++++++++ 2 files changed, 39 insertions(+), 15 deletions(-) create mode 100644 tests/passing_tests/vmlinux/ctx_field_narrow_store.py diff --git a/pythonbpf/assign_pass.py b/pythonbpf/assign_pass.py index f243d8ae..f6526bd2 100644 --- a/pythonbpf/assign_pass.py +++ b/pythonbpf/assign_pass.py @@ -197,22 +197,14 @@ def handle_variable_assignment( logger.info("Handling assignment to struct field") field_ir_type = ctypes_to_ir(val_type.type.__name__) # Sub-register-width context fields are zero-extended to i64 by - # load_ctx_field, so val is already i64 even though the field type - # says otherwise (c_uint for xdp_md, c_ushort for pt_regs.cs/ss). - if ( - isinstance(field_ir_type, ir.IntType) - and field_ir_type.width < 64 - and isinstance(var_type, ir.IntType) - and var_type.width == 64 + # load_ctx_field, so val may be wider than the field type says + # (c_uint for xdp_md, c_ushort for pt_regs.cs/ss). convert() + # sizes from the physical value and signs from the field, so it + # is a no-op into an i64 slot and a trunc into a narrower one. + if isinstance(field_ir_type, ir.IntType) and isinstance( + var_type, ir.IntType ): - builder.store(val, var_ptr) - logger.info( - f"Assigned zero-extended i{field_ir_type.width} context field " - f"to {var_name} (i64)" - ) - return True - # TODO: handling only ctype struct fields for now. Handle other stuff too later. - elif var_type == field_ir_type: + val = convert(builder, val, field_ir_type, var_type) builder.store(val, var_ptr) logger.info(f"Assigned ctype struct field to {var_name}") return True diff --git a/tests/passing_tests/vmlinux/ctx_field_narrow_store.py b/tests/passing_tests/vmlinux/ctx_field_narrow_store.py new file mode 100644 index 00000000..5f0d8bff --- /dev/null +++ b/tests/passing_tests/vmlinux/ctx_field_narrow_store.py @@ -0,0 +1,32 @@ +# A context field is loaded widened to i64, but it can be stored into a slot +# of its own declared width, or narrower: the store truncates like any other +# integer conversion. The cgroup_skb_direct_packet_access.c selftest does +# exactly this with `__u32 data_end = skb->data_end`. +from ctypes import c_int64, c_uint16, c_uint32 +from pythonbpf import bpf, section, bpfglobal, compile +from vmlinux import struct_xdp_md + + +@bpf +@bpfglobal +def ifindex() -> c_uint32: + return c_uint32(0) + + +@bpf +@section("xdp") +def prog(ctx: struct_xdp_md) -> c_int64: + global ifindex + ifindex = ctx.ingress_ifindex + queue = c_uint16(0) + queue = ctx.rx_queue_index + return c_int64(queue) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() From 5e70b8f0fcf76af5352ac547d4800648d26cc225 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 15:59:35 +0000 Subject: [PATCH 08/25] Tests: Port sixteen more kernel selftests now that scalar globals exist Every program on the selftest audit's Tier 1 and Tier 2 lists that the compiler can express today: xdp_dummy, priv_prog, test_xdp_link, xdp_tx, tc_dummy, test_tc_bpf, cgroup_mprog, cgroup_skb_direct_packet_access, test_signed_loader, test_signed_loader_data, test_netfilter_link_attach, kprobe_multi_empty, uprobe_multi_bench, uprobe_multi_usdt, test_link_pinning and test_xdp_devmap_helpers. Fifteen pass at every level, bringing tc, tcx, cgroup/getsockopt, cgroup_skb, socket, netfilter, kprobe.multi, uprobe.multi, usdt and tp_btf program types under test for the first time. test_xdp_devmap_helpers is a negative fixture upstream: reading egress_ifindex is only legal with expected_attach_type = BPF_XDP_DEVMAP and the driver asserts the plain load fails. It is a strict verifier-level xfail here for the same reason, until expected_attach_type can be expressed. The porting notes record the batch and why each remaining program on the two lists is still out of reach. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01DLMW3X7opqAPH6bjmTVkr7 --- .../PORTING-NOTES.md | 48 +++++++++++++++ .../cgroup/cgroup_mprog.py | 41 +++++++++++++ .../netfilter/netfilter_link_attach.py | 23 ++++++++ .../socket/signed_loader.py | 23 ++++++++ .../socket/signed_loader_data.py | 37 ++++++++++++ .../kernel_selftest_equivalent/tc/tc_dummy.py | 22 +++++++ .../tracing/kprobe_multi_empty.py | 22 +++++++ .../tracing/link_pinning.py | 58 +++++++++++++++++++ .../tracing/uprobe_multi_bench.py | 39 +++++++++++++ .../tracing/uprobe_multi_usdt.py | 30 ++++++++++ .../cgroup_skb_direct_packet_access.py | 41 +++++++++++++ .../vmlinux/tc_bpf.py | 43 ++++++++++++++ .../vmlinux/xdp_devmap_helpers.py | 34 +++++++++++ .../vmlinux/xdp_tx.py | 26 +++++++++ .../xdp/priv_prog.py | 24 ++++++++ .../xdp/xdp_dummy.py | 31 ++++++++++ .../xdp/xdp_link.py | 29 ++++++++++ tests/test_config.toml | 7 +++ 18 files changed, 578 insertions(+) create mode 100644 tests/kernel_selftest_equivalent/cgroup/cgroup_mprog.py create mode 100644 tests/kernel_selftest_equivalent/netfilter/netfilter_link_attach.py create mode 100644 tests/kernel_selftest_equivalent/socket/signed_loader.py create mode 100644 tests/kernel_selftest_equivalent/socket/signed_loader_data.py create mode 100644 tests/kernel_selftest_equivalent/tc/tc_dummy.py create mode 100644 tests/kernel_selftest_equivalent/tracing/kprobe_multi_empty.py create mode 100644 tests/kernel_selftest_equivalent/tracing/link_pinning.py create mode 100644 tests/kernel_selftest_equivalent/tracing/uprobe_multi_bench.py create mode 100644 tests/kernel_selftest_equivalent/tracing/uprobe_multi_usdt.py create mode 100644 tests/kernel_selftest_equivalent/vmlinux/cgroup_skb_direct_packet_access.py create mode 100644 tests/kernel_selftest_equivalent/vmlinux/tc_bpf.py create mode 100644 tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py create mode 100644 tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py create mode 100644 tests/kernel_selftest_equivalent/xdp/priv_prog.py create mode 100644 tests/kernel_selftest_equivalent/xdp/xdp_dummy.py create mode 100644 tests/kernel_selftest_equivalent/xdp/xdp_link.py diff --git a/tests/kernel_selftest_equivalent/PORTING-NOTES.md b/tests/kernel_selftest_equivalent/PORTING-NOTES.md index 2e9b4022..e3addc6d 100644 --- a/tests/kernel_selftest_equivalent/PORTING-NOTES.md +++ b/tests/kernel_selftest_equivalent/PORTING-NOTES.md @@ -91,6 +91,54 @@ cursor, which together constrain the design more than anything here does. enums, then gives up), a Python-level surface for declaring one, and userspace access through `pylibbpf`. +## Second batch: everything portable after globals + +With scalar globals in, the audit's Tier 1 and Tier 2 lists were re-read against the +compiler and every program it can express was ported. Sixteen more, fifteen of which +pass at every level; the sixteenth passes at IR and llc and is rejected by the verifier +exactly as its upstream driver asserts it must be: + +| Port | Upstream | Section | Outcome | +|---|---|---|---| +| `xdp/xdp_dummy.py` | `xdp_dummy.c` | `xdp` x2 | passes | +| `xdp/priv_prog.py` | `priv_prog.c` | `xdp` | passes | +| `xdp/xdp_link.py` | `test_xdp_link.c` | `xdp`, `tc` | passes | +| `vmlinux/xdp_tx.py` | `xdp_tx.c` | `xdp` | passes | +| `tc/tc_dummy.py` | `tc_dummy.c` | `tc` | passes | +| `vmlinux/tc_bpf.py` | `test_tc_bpf.c` | `tc`, `tcx/ingress` | passes (direct packet access) | +| `cgroup/cgroup_mprog.py` | `cgroup_mprog.c` | `cgroup/getsockopt` x4 | passes | +| `vmlinux/cgroup_skb_direct_packet_access.py` | `cgroup_skb_direct_packet_access.c` | `cgroup_skb/ingress` | passes, after a compiler fix | +| `socket/signed_loader.py` | `test_signed_loader.c` | `socket` | passes | +| `socket/signed_loader_data.py` | `test_signed_loader_data.c` | `socket` | passes (.data global) | +| `netfilter/netfilter_link_attach.py` | `test_netfilter_link_attach.c` | `netfilter` | passes | +| `tracing/kprobe_multi_empty.py` | `kprobe_multi_empty.c` | `kprobe.multi/` | passes | +| `tracing/uprobe_multi_bench.py` | `uprobe_multi_bench.c` | `uprobe.multi/...` | passes (`count += 1`) | +| `tracing/uprobe_multi_usdt.py` | `uprobe_multi_usdt.c` | `usdt` | passes | +| `tracing/link_pinning.py` | `test_link_pinning.c` | `raw_tp/sys_enter`, `tp_btf/sys_enter` | passes | +| `vmlinux/xdp_devmap_helpers.py` | `test_xdp_devmap_helpers.c` | `xdp` | verifier xfail by design | + +**One compiler bug fell out.** `data_end = skb->data_end` into a `__u32` global failed +with `cannot store i64 to i32*`: context fields are loaded widened to i64, and the +assignment path only accepted them into 64-bit slots or slots of exactly the field's +type, never narrowing. It now goes through `convert()` like every other integer store. +`passing_tests/vmlinux/ctx_field_narrow_store.py` pins it. + +**What is still not portable, and why**, from the same two lists: + +| Program | Blocker | +|---|---| +| `metadata_used.c`, `metadata_unused.c` | `char[]` `.rodata` globals: only integer scalars can be globals | +| `test_log_buf.c`, `cgroup_preorder.c`, `uprobe_multi_pid_filter.c`, `test_build_id.c` | array globals | +| `token_kallsyms.c`, `test_btf_ext.c`, `test_static_linked*.c` | BPF-to-BPF calls (`__weak` / `__noinline` subprogs) | +| `test_trace_ext.c`, `freplace_get_constant.c` | `freplace` needs a target program to load against | +| `test_subskeleton*.c` | extern symbols, `__kconfig`, static linking | +| `test_pkt_md_access.c` | narrow type-punned loads of `__sk_buff` fields | +| `test_xdp_attach_fail.c` | tracepoint `__data_loc` pointer arithmetic on a custom ctx struct | +| `sockopt_multi.c` | writes to context fields and through `optval` | +| `tracing_struct_many_args.c` | `BPF_PROG2` multi-argument entry | +| `bpf_nop_bench.c` | `bpf_loop`-based benchmark macro | +| `test_tcp_estats.c` | large; inlinable helpers and struct-heavy, not attempted yet | + ## 3. Incidental findings - **Nested struct field access fails with a misleading error.** `ctx.regs.ip` reports diff --git a/tests/kernel_selftest_equivalent/cgroup/cgroup_mprog.py b/tests/kernel_selftest_equivalent/cgroup/cgroup_mprog.py new file mode 100644 index 00000000..f4717b5e --- /dev/null +++ b/tests/kernel_selftest_equivalent/cgroup/cgroup_mprog.py @@ -0,0 +1,41 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/cgroup_mprog.c +# +# Four identical cgroup/getsockopt programs. Upstream attaches them in +# various orders with BPF_F_BEFORE/BPF_F_AFTER and checks the resulting +# multi-prog chain; the bodies only need to exist and return "allow". + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("cgroup/getsockopt") +def getsockopt_1(ctx: c_void_p) -> c_int64: + return c_int64(1) + + +@bpf +@section("cgroup/getsockopt") +def getsockopt_2(ctx: c_void_p) -> c_int64: + return c_int64(1) + + +@bpf +@section("cgroup/getsockopt") +def getsockopt_3(ctx: c_void_p) -> c_int64: + return c_int64(1) + + +@bpf +@section("cgroup/getsockopt") +def getsockopt_4(ctx: c_void_p) -> c_int64: + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/netfilter/netfilter_link_attach.py b/tests/kernel_selftest_equivalent/netfilter/netfilter_link_attach.py new file mode 100644 index 00000000..57408901 --- /dev/null +++ b/tests/kernel_selftest_equivalent/netfilter/netfilter_link_attach.py @@ -0,0 +1,23 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_netfilter_link_attach.c +# +# A netfilter program that accepts everything (NF_ACCEPT is 1). Upstream +# attaches it with every combination of protocol family, hook and priority +# and checks which the kernel rejects. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("netfilter") +def nf_link_attach_test(ctx: c_void_p) -> c_int64: + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/socket/signed_loader.py b/tests/kernel_selftest_equivalent/socket/signed_loader.py new file mode 100644 index 00000000..8f4974ec --- /dev/null +++ b/tests/kernel_selftest_equivalent/socket/signed_loader.py @@ -0,0 +1,23 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_signed_loader.c +# +# A minimal, map-less socket filter. Upstream drives it through libbpf's +# light-skeleton loader to test signed-program loading; a socket filter +# needs no attach resolution and no maps keeps the loader trivial. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("socket") +def probe(ctx: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/socket/signed_loader_data.py b/tests/kernel_selftest_equivalent/socket/signed_loader_data.py new file mode 100644 index 00000000..43bc53d1 --- /dev/null +++ b/tests/kernel_selftest_equivalent/socket/signed_loader_data.py @@ -0,0 +1,37 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_signed_loader_data.c +# +# The signed-loader fixture with one initialised global, so the object has a +# .data map that the loader must seed. Upstream checks that a signed loader +# keeps the attested initial value: +# +# __u64 magic = 0x5eed1234abad1deaULL; +# +# SEC("socket") +# int probe(void *ctx) +# { +# return (int)magic; +# } + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int32, c_uint64 + + +@bpf +@bpfglobal +def magic() -> c_uint64: + return c_uint64(0x5EED1234ABAD1DEA) + + +@bpf +@section("socket") +def probe(ctx: c_void_p) -> c_int32: + return c_int32(magic) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tc/tc_dummy.py b/tests/kernel_selftest_equivalent/tc/tc_dummy.py new file mode 100644 index 00000000..8f8797d7 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tc/tc_dummy.py @@ -0,0 +1,22 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/tc_dummy.c +# +# A classifier that returns TC_ACT_OK for everything. Upstream is the fixture +# behind the tc_links and tc_opts attach-order tests. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("tc") +def entry(skb: c_void_p) -> c_int64: + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/kprobe_multi_empty.py b/tests/kernel_selftest_equivalent/tracing/kprobe_multi_empty.py new file mode 100644 index 00000000..a7d8ffca --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/kprobe_multi_empty.py @@ -0,0 +1,22 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/kprobe_multi_empty.c +# +# An empty kprobe.multi program. Upstream attaches it to every function in +# the kernel's available_filter_functions list to benchmark attach time. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("kprobe.multi/") +def test_kprobe_empty(ctx: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/link_pinning.py b/tests/kernel_selftest_equivalent/tracing/link_pinning.py new file mode 100644 index 00000000..9065e544 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/link_pinning.py @@ -0,0 +1,58 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_link_pinning.c +# +# Two programs, one raw_tp and one tp_btf on the same tracepoint, each +# copying a global set by userspace into a global read by userspace. +# Upstream pins the link, closes every fd, then checks the program still +# fires by bumping `in` and watching `out` follow: +# +# int in = 0; +# int out = 0; +# +# SEC("raw_tp/sys_enter") +# int raw_tp_prog(const void *ctx) +# { +# out = in; +# return 0; +# } +# +# `in` is renamed `in_val` because `in` is a Python keyword. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_int32 + + +@bpf +@bpfglobal +def in_val() -> c_int32: + return c_int32(0) + + +@bpf +@bpfglobal +def out() -> c_int32: + return c_int32(0) + + +@bpf +@section("raw_tp/sys_enter") +def raw_tp_prog(ctx: c_void_p) -> c_int64: + global out + out = in_val + return c_int64(0) + + +@bpf +@section("tp_btf/sys_enter") +def tp_btf_prog(ctx: c_void_p) -> c_int64: + global out + out = in_val + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/uprobe_multi_bench.py b/tests/kernel_selftest_equivalent/tracing/uprobe_multi_bench.py new file mode 100644 index 00000000..890bfc80 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/uprobe_multi_bench.py @@ -0,0 +1,39 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/uprobe_multi_bench.c +# +# Count how many times the multi-uprobe fires. Upstream attaches it to +# thousands of uprobe_multi_func_* symbols and reads `count` back: +# +# int count; +# +# SEC("uprobe.multi/./uprobe_multi:uprobe_multi_func_*") +# int uprobe_bench(struct pt_regs *ctx) +# { +# count++; +# return 0; +# } + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_int32 + + +@bpf +@bpfglobal +def count() -> c_int32: + return c_int32(0) + + +@bpf +@section("uprobe.multi/./uprobe_multi:uprobe_multi_func_*") +def uprobe_bench(ctx: c_void_p) -> c_int64: + global count + count += 1 + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/uprobe_multi_usdt.py b/tests/kernel_selftest_equivalent/tracing/uprobe_multi_usdt.py new file mode 100644 index 00000000..01ed34c0 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/uprobe_multi_usdt.py @@ -0,0 +1,30 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/uprobe_multi_usdt.c +# +# The USDT flavour of the multi-uprobe counter. Upstream attaches it to a +# USDT probe in the test binary and asserts on `count` from userspace. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_int32 + + +@bpf +@bpfglobal +def count() -> c_int32: + return c_int32(0) + + +@bpf +@section("usdt") +def usdt0(ctx: c_void_p) -> c_int64: + global count + count += 1 + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/vmlinux/cgroup_skb_direct_packet_access.py b/tests/kernel_selftest_equivalent/vmlinux/cgroup_skb_direct_packet_access.py new file mode 100644 index 00000000..c290fb3c --- /dev/null +++ b/tests/kernel_selftest_equivalent/vmlinux/cgroup_skb_direct_packet_access.py @@ -0,0 +1,41 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/cgroup_skb_direct_packet_access.c +# +# A cgroup_skb program that records skb->data_end into a global. Upstream +# asserts from userspace that the value is non-zero, proving cgroup_skb +# programs get direct packet access: +# +# __u32 data_end; +# +# SEC("cgroup_skb/ingress") +# int direct_packet_access(struct __sk_buff *skb) +# { +# data_end = skb->data_end; +# return 1; +# } + +from pythonbpf import bpf, section, bpfglobal, compile +from vmlinux import struct___sk_buff +from ctypes import c_int64, c_uint32 + + +@bpf +@bpfglobal +def data_end() -> c_uint32: + return c_uint32(0) + + +@bpf +@section("cgroup_skb/ingress") +def direct_packet_access(skb: struct___sk_buff) -> c_int64: + global data_end + data_end = skb.data_end + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/vmlinux/tc_bpf.py b/tests/kernel_selftest_equivalent/vmlinux/tc_bpf.py new file mode 100644 index 00000000..cf6710d7 --- /dev/null +++ b/tests/kernel_selftest_equivalent/vmlinux/tc_bpf.py @@ -0,0 +1,43 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_tc_bpf.c +# +# `cls` is a dummy classifier for the TC-BPF API test. `pkt_ptr` is the one +# that matters: it derives a packet pointer from skb->data, bounds-checks it +# against skb->data_end, and is loaded without CAP_SYS_ADMIN/CAP_PERFMON to +# prove direct packet access works for a plain tcx program. Upstream: +# +# struct iphdr *iph = (void *)(long)skb->data + sizeof(struct ethhdr); +# +# if ((long)(iph + 1) > (long)skb->data_end) +# return 1; +# return 0; +# +# sizeof(struct ethhdr) + sizeof(struct iphdr) is 14 + 20 = 34. + +from pythonbpf import bpf, section, bpfglobal, compile +from vmlinux import struct___sk_buff +from ctypes import c_void_p, c_int64 + + +@bpf +@section("tc") +def cls(skb: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@section("tcx/ingress") +def pkt_ptr(skb: struct___sk_buff) -> c_int64: + data = c_void_p(skb.data) + data_end = c_void_p(skb.data_end) + if data + 34 > data_end: + return c_int64(1) + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py b/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py new file mode 100644 index 00000000..1e9de0a4 --- /dev/null +++ b/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py @@ -0,0 +1,34 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_xdp_devmap_helpers.c +# +# Reads xdp_md->egress_ifindex, which only exists for programs loaded with +# expected_attach_type = BPF_XDP_DEVMAP. Upstream loads it *without* that +# type and asserts the load fails, so the program is a negative fixture: +# +# unsigned int len = data_end - data; +# bpf_trace_printk(fmt, sizeof(fmt), +# ctx->ingress_ifindex, ctx->egress_ifindex, len); +# return XDP_PASS; + +from pythonbpf import bpf, section, bpfglobal, compile +from pythonbpf.helper import XDP_PASS +from vmlinux import struct_xdp_md +from ctypes import c_int64 + + +@bpf +@section("xdp") +def xdpdm_devlog(ctx: struct_xdp_md) -> c_int64: + length = ctx.data_end - ctx.data + ingress = ctx.ingress_ifindex + egress = ctx.egress_ifindex + print(f"devmap redirect: dev {ingress} -> dev {egress} len {length}") + return XDP_PASS + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py b/tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py new file mode 100644 index 00000000..8d40ffed --- /dev/null +++ b/tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py @@ -0,0 +1,26 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/xdp_tx.c +# +# Bounce every packet back out of the interface it arrived on. Upstream is +# the transmit side of the veth XDP tests. +# +# XDP_TX comes from vmlinux because pythonbpf.helper exports only XDP_PASS +# and XDP_DROP; that is why this lives under vmlinux/. + +from pythonbpf import bpf, section, bpfglobal, compile +from vmlinux import XDP_TX +from ctypes import c_void_p, c_int64 + + +@bpf +@section("xdp") +def xdp_tx(xdp: c_void_p) -> c_int64: + return c_int64(XDP_TX) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/xdp/priv_prog.py b/tests/kernel_selftest_equivalent/xdp/priv_prog.py new file mode 100644 index 00000000..181d6840 --- /dev/null +++ b/tests/kernel_selftest_equivalent/xdp/priv_prog.py @@ -0,0 +1,24 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/priv_prog.c +# +# An XDP program that drops everything. Upstream loads it from an +# unprivileged process to check the CAP_BPF/CAP_NET_ADMIN gating; the +# program itself is the smallest privileged-type program there is. + +from pythonbpf import bpf, section, bpfglobal, compile +from pythonbpf.helper import XDP_DROP +from ctypes import c_void_p, c_int64 + + +@bpf +@section("xdp") +def xdp_prog1(xdp: c_void_p) -> c_int64: + return XDP_DROP + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/xdp/xdp_dummy.py b/tests/kernel_selftest_equivalent/xdp/xdp_dummy.py new file mode 100644 index 00000000..82686de1 --- /dev/null +++ b/tests/kernel_selftest_equivalent/xdp/xdp_dummy.py @@ -0,0 +1,31 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/xdp_dummy.c +# +# Two XDP programs that pass every packet. Upstream is the fixture that a +# dozen prog_tests attach and detach to exercise XDP link plumbing; the +# second program's odd name is deliberate, it is what the kallsyms test looks +# for. + +from pythonbpf import bpf, section, bpfglobal, compile +from pythonbpf.helper import XDP_PASS +from ctypes import c_void_p, c_int64 + + +@bpf +@section("xdp") +def xdp_dummy_prog(ctx: c_void_p) -> c_int64: + return XDP_PASS + + +@bpf +@section("xdp") +def __x64_sys_nop(ctx: c_void_p) -> c_int64: + return XDP_PASS + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/xdp/xdp_link.py b/tests/kernel_selftest_equivalent/xdp/xdp_link.py new file mode 100644 index 00000000..d94df5ab --- /dev/null +++ b/tests/kernel_selftest_equivalent/xdp/xdp_link.py @@ -0,0 +1,29 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_xdp_link.c +# +# One XDP and one TC handler in the same object. Upstream attaches the XDP +# one through bpf_link and checks that legacy netlink attach of the same +# program is refused while the link exists. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("xdp") +def xdp_handler(xdp: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@section("tc") +def tc_handler(skb: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/test_config.toml b/tests/test_config.toml index 122c2be2..fa8c9c84 100644 --- a/tests/test_config.toml +++ b/tests/test_config.toml @@ -66,3 +66,10 @@ "kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py" = {reason = "RingBuffer reserve/typed record/discard workflow is planned but not implemented yet", level = "ir"} "kernel_selftest_equivalent/vmlinux/perf_skip.py" = {reason = "Nested struct field access (ctx.regs.ip) not supported; one level such as ctx.sample_period works", level = "ir"} + +# Upstream is a negative fixture: reading xdp_md->egress_ifindex is only legal +# for programs loaded with expected_attach_type = BPF_XDP_DEVMAP, and the +# prog_tests driver asserts that a plain load *fails*. The kernel rejects it +# here with "invalid bpf_context access off=20 size=4", which is the pass +# condition upstream. PythonBPF has no way to set expected_attach_type yet. +"kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py" = {reason = "Negative fixture: egress_ifindex needs expected_attach_type=BPF_XDP_DEVMAP, which cannot be set yet; the verifier rejection is upstream's pass condition", level = "verifier"} From e0d4d1eadadd08c4f640c7c7f4bd2628c02781ab Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 16:31:15 +0000 Subject: [PATCH 09/25] Tools: Check in the selftest corpus classifier The porting assessment was produced by a one-off script that scored every program under tools/testing/selftests/bpf/progs against the compiler's envelope; the notes recommended keeping it so the numbers can be regenerated after each feature lands. This is that script, rebuilt: it strips comments, skips include-shims and header-only fixtures, and reports per program the hard blockers (language gaps) and soft flags (porting costs), with a histogram and a near-miss listing. The helper, map and construct lists at the top are the envelope and are maintained by hand. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01DLMW3X7opqAPH6bjmTVkr7 --- tools/selftest-audit.py | 327 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 327 insertions(+) create mode 100755 tools/selftest-audit.py diff --git a/tools/selftest-audit.py b/tools/selftest-audit.py new file mode 100755 index 00000000..9a4edc5f --- /dev/null +++ b/tools/selftest-audit.py @@ -0,0 +1,327 @@ +#!/usr/bin/env python3 +"""Score the kernel's BPF selftest programs against what PythonBPF can express. + +Point it at a checkout of tools/testing/selftests/bpf/progs and it prints, for +every real program there, the constructs that keep it out of PythonBPF today. +Programs with no hard blocker are the porting candidates; the soft flags on +them say what a port has to rewrite by hand. + + python3 tools/selftest-audit.py path/to/linux/tools/testing/selftests/bpf/progs + python3 tools/selftest-audit.py progs/ --json > audit.json + python3 tools/selftest-audit.py progs/ --histogram + +This is a heuristic scan of the C source, not a compiler: it has false +negatives in both directions, and the envelope it encodes (the helper, map and +construct lists below) must be kept in step with the compiler by hand. Re-run +it after a feature lands to see what the feature unlocked. The lists were last +reconciled against pythonbpf/helper and pythonbpf/maps when scalar @bpfglobal +support and integer signedness merged. +""" + +import argparse +import json +import re +import sys +from collections import Counter +from pathlib import Path + +# ── the envelope ───────────────────────────────────────────────────────────── + +# Kernel helpers pythonbpf/helper can emit (bpf_trace_printk is `print`). +SUPPORTED_HELPERS = { + "bpf_get_current_cgroup_id", + "bpf_get_current_comm", + "bpf_get_current_pid_tgid", + "bpf_get_current_uid_gid", + "bpf_get_prandom_u32", + "bpf_get_smp_processor_id", + "bpf_get_stack", + "bpf_ktime_get_ns", + "bpf_map_delete_elem", + "bpf_map_lookup_elem", + "bpf_map_update_elem", + "bpf_perf_event_output", + "bpf_printk", + "bpf_trace_printk", + "bpf_probe_read", + "bpf_probe_read_kernel", + "bpf_probe_read_kernel_str", + "bpf_ringbuf_output", + "bpf_ringbuf_reserve", + "bpf_ringbuf_submit", + "bpf_skb_store_bytes", +} + +# BPF_MAP_TYPE_* that pythonbpf/maps lowers (ArrayMap is still a stub). +SUPPORTED_MAP_TYPES = {"HASH", "PERF_EVENT_ARRAY", "RINGBUF"} + +# Things that look like helper calls but are libbpf macros, not helpers. +NOT_HELPERS = { + "bpf_htons", + "bpf_ntohs", + "bpf_htonl", + "bpf_ntohl", + "bpf_be64_to_cpu", + "bpf_cpu_to_be64", + "bpf_printk_", +} + +# (label, kind, regex). kind is "hard" (a language gap) or "soft" (a porting +# cost a careful rewrite can absorb). Order does not matter. +PATTERNS = [ + # control flow + ("loop", "hard", r"\b(for|while)\s*\(|\bbpf_for\b|\bbpf_repeat\b|\bbpf_loop\s*\("), + ("goto", "hard", r"\bgoto\s+\w+"), + ("switch", "hard", r"\bswitch\s*\("), + ("ternary", "soft", r"\?[^?:]*:"), + # entry-point shape + ( + "typed_prog_macro", + "hard", + r"\bBPF_(PROG|PROG2|KPROBE|KRETPROBE|KSYSCALL|KPROBE_SYSCALL|UPROBE|URETPROBE|USDT|" + r"TRACE_\w+|ITER\w*|LSM\w*)\s*\(", + ), + ("struct_ops", "hard", r'SEC\s*\(\s*"\.?struct_ops'), + ("freplace", "hard", r'SEC\s*\(\s*"freplace'), + ("sleepable_or_special_sec", "soft", r'SEC\s*\(\s*"\?'), + # verifier-test harness + ( + "verifier_annotation", + "hard", + r"\b__(failure|success|msg|retval|naked|log_level|flag|arch_\w+|description|" + r"jited|xlated|caps_unpriv|load_if_JITed|not_msg|failure_unpriv|success_unpriv)\b", + ), + # functions + ("subprog_call", "hard", r"\b__noinline\b|\b__weak\b|\bSEC\s*\(\s*\"\?"), + ( + "static_helper", + "soft", + r"\bstatic\s+(__always_inline\s+|inline\s+|__noinline\s+)?\w[\w\s\*]*\s+\**\w+\s*\([^;]*\)\s*\{", + ), + # kernel features + ( + "kfunc", + "hard", + r"__ksym\b|bpf_experimental\.h|bpf_kfuncs\.h|\bbpf_(obj_new|obj_drop|refcount|task_from|" + r"task_acquire|task_release|cgroup_acquire|cgroup_release|cpumask_\w+|rbtree_\w+|list_\w+|" + r"rcu_read_lock|rcu_read_unlock|arena_\w+|key_put|lookup_user_key|dynptr_\w+|iter_\w+|" + r"wq_\w+|timer_\w+|throw|percpu_obj_\w+|res_spin_\w+|preempt_\w+|local_irq_\w+|" + r"session_\w+|get_dentry_xattr|get_file_xattr|kptr_xchg|sk_assign|xdp_\w+|skb_\w+)\s*\(", + ), + ("inline_asm", "hard", r"\basm\s*(volatile)?\s*\(|__asm__"), + ("atomic", "hard", r"__sync_\w+|__atomic_\w+|\bbpf_spin_(lock|unlock)\b"), + ("tail_call", "hard", r"\bbpf_tail_call\w*\s*\("), + ( + "core_read", + "hard", + r"\bBPF_CORE_READ\w*\b|\bbpf_core_\w+|__builtin_preserve\w*|\bbpf_probe_read_user\w*", + ), + ( + "builtin", + "hard", + r"__builtin_(memcpy|memset|memcmp|bswap\w*|ctz|clz|popcount|expect)\b|\b(memcpy|memset|memcmp)\s*\(", + ), + ( + "endian_macro", + "hard", + r"\bbpf_(htons|ntohs|htonl|ntohl|be64_to_cpu|cpu_to_be64)\s*\(", + ), + ("kconfig_or_extern", "hard", r"__kconfig\b|^\s*extern\s"), + ("arena_or_iter_sec", "hard", r'SEC\s*\(\s*"(iter|arena)'), + # data + ("ctx_field_write", "soft", r"\bctx\s*->\s*\w+\s*(\+|-|\||&|\^|<<|>>)?=[^=]"), + ("local_struct", "soft", r"^\s+struct\s+\w+\s+\w+\s*(=\s*\{|;)"), + ( + "local_array", + "soft", + r"^\s+(const\s+)?(char|__u8|__u16|__u32|__u64|int|long|unsigned|u8|u16|u32|u64|__s\d+)\s+(?!_?_?license\b)\w+\s*\[[^\]]*\]", + ), + ( + "string_or_char_global", + "hard", + r"^(static\s+)?(volatile\s+)?(const\s+)?(volatile\s+)?char\s+\w+\s*\[[^\]]*\]\s*(SEC\s*\(\s*\"\.rodata\"\s*\))?\s*=", + ), +] + +# File-scope declarations that are not ordinary scalar globals. +ARRAY_GLOBAL = re.compile( + r"^(static\s+)?(volatile\s+)?(const\s+)?(volatile\s+)?" + r"(struct\s+\w+|__?[us]\d+|u\d+|s\d+|int|long|short|char|bool|unsigned\s+\w+|" + r"uintptr_t|size_t|__wsum|__be\d+|__le\d+|\w+_t)\s*\**\s*\w+\s*\[", + re.M, +) +STRUCT_GLOBAL = re.compile( + r"^(static\s+)?(volatile\s+)?(const\s+)?(volatile\s+)?struct\s+\w+\s+\w+\s*(=|;)", + re.M, +) +ANON_STRUCT_GLOBAL = re.compile(r"^struct\s*\{", re.M) +MAP_DECL = re.compile(r"__uint\s*\(\s*type\s*,\s*BPF_MAP_TYPE_(\w+)\s*\)") +SEC_RE = re.compile(r'SEC\s*\(\s*"([^"]+)"\s*\)') +HELPER_CALL = re.compile(r"\b(bpf_\w+)\s*\(") +INCLUDE_C = re.compile(r'^\s*#include\s+"[^"]+\.c"', re.M) +BTF_DUMP_FIXTURE = re.compile(r"btf_dump|btf__|__attribute__\(\(btf_decl_tag", re.I) + + +def strip_comments(src: str) -> str: + src = re.sub(r"/\*.*?\*/", "", src, flags=re.S) + return re.sub(r"//[^\n]*", "", src) + + +def classify(path: Path) -> dict | None: + raw = path.read_text(errors="replace") + src = strip_comments(raw) + secs = [s for s in SEC_RE.findall(src) if s not in ("license", ".maps", "version")] + is_prog = bool(secs) and not INCLUDE_C.search(src) + if not is_prog: + return None # wrapper shim, header-only fixture, or library file + + hard: set[str] = set() + soft: set[str] = set() + detail: dict[str, list[str]] = {} + + for label, kind, rx in PATTERNS: + if re.search(rx, src, re.M): + (hard if kind == "hard" else soft).add(label) + + maps = set(MAP_DECL.findall(src)) + bad_maps = sorted(m for m in maps if m not in SUPPORTED_MAP_TYPES) + if bad_maps: + hard.add("unsupported_map") + detail["unsupported_map"] = bad_maps + if re.search(r'SEC\s*\(\s*"\.maps"\s*\)', src) and not maps: + # a map with no __uint(type) is a legacy bpf_map_def or an extern + hard.add("legacy_map_def") + + helpers = set(HELPER_CALL.findall(src)) - NOT_HELPERS + helpers = { + h for h in helpers if not h.startswith("bpf_map_") or h in SUPPORTED_HELPERS + } + unsupported = sorted( + h + for h in helpers + if h not in SUPPORTED_HELPERS + and not h.startswith(("bpf_core_", "bpf_probe_read_user")) + ) + if unsupported: + hard.add("unsupported_helper") + detail["unsupported_helper"] = unsupported + + # file-scope globals that scalar @bpfglobal cannot hold. Map declarations + # are anonymous structs too, so take them out first. + no_maps = re.sub( + r"struct\s*\{[^}]*\}\s*\w+\s*SEC\s*\(\s*\"\.maps\"\s*\)\s*;", + "", + src, + flags=re.S, + ) + body_stripped = re.sub(r"\{[^{}]*\}", "{}", no_maps) # crude: drop innermost bodies + for _ in range(6): + body_stripped = re.sub(r"\{[^{}]*\}", "{}", body_stripped) + top = "\n".join( + line + for line in body_stripped.splitlines() + if not line.lstrip().startswith(("#", "SEC", "}", "{")) + ) + if ARRAY_GLOBAL.search(top) and not re.search( + r"^char\s+_?_?license", top, re.M | re.I + ): + hard.add("array_global") + elif ARRAY_GLOBAL.search(top): + # licence/version aside, any other array is still a blocker + others = [ + m.group(0) + for m in ARRAY_GLOBAL.finditer(top) + if "license" not in m.group(0).lower() + ] + if others: + hard.add("array_global") + if STRUCT_GLOBAL.search(top) or ANON_STRUCT_GLOBAL.search(top): + hard.add("struct_global") + + # a global written from a static helper etc. is fine; a global at all is + # informational now that scalars are supported + if re.search( + r"^(volatile\s+)?(const\s+)?(volatile\s+)?(__?[us]\d+|u\d+|s\d+|int|long|short|bool|unsigned\s+\w+|uintptr_t|size_t)\s+\w+\s*(=[^=]|;)", + top, + re.M, + ): + soft.add("scalar_global") + + return { + "file": path.name, + "bytes": len(raw), + "sections": sorted(set(secs)), + "hard": sorted(hard), + "soft": sorted(soft), + "detail": detail, + } + + +def main() -> int: + ap = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + ap.add_argument( + "progs", type=Path, help="tools/testing/selftests/bpf/progs checkout" + ) + ap.add_argument( + "--json", action="store_true", help="emit one JSON object per program" + ) + ap.add_argument( + "--histogram", action="store_true", help="print the blocker histogram" + ) + ap.add_argument( + "--all", + action="store_true", + help="list every program, not only the portable ones", + ) + ap.add_argument( + "--max-hard", + type=int, + default=0, + help="list programs with at most this many hard blockers", + ) + args = ap.parse_args() + + results = [] + skipped = 0 + for c in sorted(args.progs.glob("*.c")): + r = classify(c) + if r is None: + skipped += 1 + else: + results.append(r) + + if args.json: + for r in results: + print(json.dumps(r)) + return 0 + + real = len(results) + clean = [r for r in results if not r["hard"]] + print( + f"{real} real programs ({skipped} shims/fixtures skipped); {len(clean)} with no hard blocker\n" + ) + + if args.histogram: + hist = Counter(b for r in results for b in r["hard"]) + for label, n in hist.most_common(): + print(f" {label:28s} {n:4d} {100 * n / real:4.0f}%") + print() + + rows = ( + results if args.all else [r for r in results if len(r["hard"]) <= args.max_hard] + ) + rows.sort(key=lambda r: (len(r["hard"]), r["bytes"])) + for r in rows: + flags = " ".join(r["hard"]) or "-" + soft = " ".join(r["soft"]) or "-" + extra = "; ".join(f"{k}={','.join(v)}" for k, v in r["detail"].items()) + print( + f"{r['file']:44s} {r['bytes']:6d} hard: {flags:30s} soft: {soft} {extra}" + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From 298c140ee7949ada9aee0bec7e8054ea25648f45 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 16:31:15 +0000 Subject: [PATCH 10/25] Tests: Port five more selftests found by re-running the corpus audit veristat_foo (three socket programs) is clean. test_perf_link, test_enable_stats and test_cgroup_link count with __sync_fetch_and_add, which becomes a plain `x += 1` tagged WORKAROUND(atomics) for a mechanical sweep once atomics exist. connect4_dropper compares against bpf_htons(port), written out as shifts on the low 16 bits. All five pass at IR, llc and verifier level, and bring perf_event, raw_tracepoint/sys_enter, cgroup_skb/egress and cgroup/connect4 under test. The porting notes gain the third batch, the atomics tag, the current blocker histogram from the checked-in audit tool, and how to re-run it. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01DLMW3X7opqAPH6bjmTVkr7 --- tests/README.md | 6 ++ .../PORTING-NOTES.md | 68 +++++++++++++++++-- .../cgroup/cgroup_link.py | 58 ++++++++++++++++ .../socket/veristat_foo.py | 35 ++++++++++ .../tracing/enable_stats.py | 43 ++++++++++++ .../tracing/perf_link.py | 43 ++++++++++++ .../vmlinux/connect4_dropper.py | 53 +++++++++++++++ 7 files changed, 302 insertions(+), 4 deletions(-) create mode 100644 tests/kernel_selftest_equivalent/cgroup/cgroup_link.py create mode 100644 tests/kernel_selftest_equivalent/socket/veristat_foo.py create mode 100644 tests/kernel_selftest_equivalent/tracing/enable_stats.py create mode 100644 tests/kernel_selftest_equivalent/tracing/perf_link.py create mode 100644 tests/kernel_selftest_equivalent/vmlinux/connect4_dropper.py diff --git a/tests/README.md b/tests/README.md index a2314bec..d972939e 100644 --- a/tests/README.md +++ b/tests/README.md @@ -123,6 +123,12 @@ program writes a global and the driver reads it back. Ports keep that shape with values the kernel's driver does. Anything a global cannot yet hold (arrays, structs, strings) is a roadmap test, not a workaround. +The one substitution still in use is `WORKAROUND(atomics)`: upstream counters +incremented with `__sync_fetch_and_add` are plain `x += 1` here, tagged on the +line so the sweep is mechanical once atomics land. `PORTING-NOTES.md` in that +directory records every port, its rewrite if any, and why the rest of the +corpus is out of reach; `tools/selftest-audit.py` regenerates that scoring. + Anything importing from `vmlinux` belongs in `vmlinux/`, which is registered in `VMLINUX_TEST_DIRS_PASSING` so it is skipped rather than failed where no `vmlinux.py` has been generated. diff --git a/tests/kernel_selftest_equivalent/PORTING-NOTES.md b/tests/kernel_selftest_equivalent/PORTING-NOTES.md index e3addc6d..a5929e34 100644 --- a/tests/kernel_selftest_equivalent/PORTING-NOTES.md +++ b/tests/kernel_selftest_equivalent/PORTING-NOTES.md @@ -139,6 +139,62 @@ type, never narrowing. It now goes through `convert()` like every other integer | `bpf_nop_bench.c` | `bpf_loop`-based benchmark macro | | `test_tcp_estats.c` | large; inlinable helpers and struct-heavy, not attempted yet | +## Third batch: what the re-run audit found + +`tools/selftest-audit.py` is the corpus classifier, rebuilt and checked in. Run against +the current upstream `progs/` it reports 847 real programs, 25 with no hard blocker. All +but two of those 25 were already ported or are unportable for a reason a regex cannot see +(`bpf_nop_bench.c` hides a loop in a macro, `test_pkt_md_access.c` type-puns narrow loads, +`tracing_struct_int128.c` indexes the raw ctx array and needs bpf_testmod to load). The +programs with exactly one blocker were read by hand for anything a documented rewrite +could absorb. Five more ports, all passing at every level: + +| Port | Upstream | Section | Rewrite | +|---|---|---|---| +| `socket/veristat_foo.py` | `veristat_foo.c` | `socket` x3 | none | +| `tracing/perf_link.py` | `test_perf_link.c` | `perf_event` | `WORKAROUND(atomics)` | +| `tracing/enable_stats.py` | `test_enable_stats.c` | `raw_tracepoint/sys_enter` | `WORKAROUND(atomics)` | +| `cgroup/cgroup_link.py` | `test_cgroup_link.c` | `cgroup_skb/egress` x2 | `WORKAROUND(atomics)` | +| `vmlinux/connect4_dropper.py` | `connect4_dropper.c` | `cgroup/connect4` | `bpf_htons` written as shifts | + +### `WORKAROUND(atomics)` + +Three upstream programs count with `__sync_fetch_and_add(&x, 1)`. PythonBPF has no atomic +operations, so the ports do `x += 1`, a plain read-modify-write, and tag the line. As with +the earlier globals tag this is scaffolding for a mechanical sweep once atomics land: + +```bash +grep -rn "WORKAROUND(atomics)" tests/kernel_selftest_equivalent/ +``` + +It is not a cosmetic substitution: the upstream drivers run these programs from many +CPUs at once and the exact count matters there, which is precisely what a non-atomic +increment loses. + +### The blocker histogram now + +Over 847 real programs, hard blockers only; a program usually hits several: + +| Blocker | Programs | Share | +|---|---|---| +| unsupported helper | 496 | 59% | +| kfuncs | 284 | 34% | +| unsupported map type | 273 | 32% | +| verifier-test annotations | 238 | 28% | +| typed program macros (`BPF_PROG`, `BPF_KPROBE`) | 234 | 28% | +| BPF-to-BPF calls | 176 | 21% | +| `goto` | 143 | 17% | +| inline asm | 142 | 17% | +| struct globals | 126 | 15% | +| CO-RE reads | 116 | 14% | +| loops | 109 | 13% | +| array globals | 105 | 12% | +| atomics | 86 | 10% | + +Globals no longer appear as a blocker at all. The next unlocks by count are helpers (a +long tail, but `bpf_get_current_task`, `bpf_ktime_get_boot_ns` and the `bpf_probe_read_user*` +family recur), array maps, and typed program arguments. + ## 3. Incidental findings - **Nested struct field access fails with a misleading error.** `ctx.regs.ip` reports @@ -155,7 +211,11 @@ type, never narrowing. It now goes through `convert()` like every other integer ## Re-running the corpus scoring -The audit behind this spike scored all 976 programs against the compiler's envelope. It is -worth re-running after each feature lands, to see what the change unlocked. The blocker -histogram at the time of writing, over 820 real programs: globals 52%, verifier-test -annotations 27%, typed program macros 26%, kfuncs 20%, inline asm 16%, loops 12%. +```bash +git clone --depth 1 --filter=blob:none --sparse https://github.com/torvalds/linux +git -C linux sparse-checkout set --no-cone tools/testing/selftests/bpf/progs +python3 tools/selftest-audit.py linux/tools/testing/selftests/bpf/progs --histogram --max-hard 1 +``` + +Re-run it after each feature lands. The envelope it encodes (helper, map and construct +lists at the top of the script) is maintained by hand and must move with the compiler. diff --git a/tests/kernel_selftest_equivalent/cgroup/cgroup_link.py b/tests/kernel_selftest_equivalent/cgroup/cgroup_link.py new file mode 100644 index 00000000..a7870832 --- /dev/null +++ b/tests/kernel_selftest_equivalent/cgroup/cgroup_link.py @@ -0,0 +1,58 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_cgroup_link.c +# +# Two cgroup_skb/egress programs, each counting its own runs. Upstream +# attaches one through a cgroup bpf_link, then swaps in the other with +# bpf_link_update() and checks the right counter moves: +# +# int calls = 0; +# int alt_calls = 0; +# +# SEC("cgroup_skb/egress") +# int egress(struct __sk_buff *skb) +# { +# __sync_fetch_and_add(&calls, 1); +# return 1; +# } +# +# WORKAROUND(atomics): the upstream increments are atomic. PythonBPF has no +# atomic operations, so these are plain read-modify-writes of the globals. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_int32 + + +@bpf +@bpfglobal +def calls() -> c_int32: + return c_int32(0) + + +@bpf +@bpfglobal +def alt_calls() -> c_int32: + return c_int32(0) + + +@bpf +@section("cgroup_skb/egress") +def egress(skb: c_void_p) -> c_int64: + global calls + calls += 1 # WORKAROUND(atomics): __sync_fetch_and_add(&calls, 1) + return c_int64(1) + + +@bpf +@section("cgroup_skb/egress") +def egress_alt(skb: c_void_p) -> c_int64: + global alt_calls + alt_calls += 1 # WORKAROUND(atomics): __sync_fetch_and_add(&alt_calls, 1) + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/socket/veristat_foo.py b/tests/kernel_selftest_equivalent/socket/veristat_foo.py new file mode 100644 index 00000000..1fad158d --- /dev/null +++ b/tests/kernel_selftest_equivalent/socket/veristat_foo.py @@ -0,0 +1,35 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/veristat_foo.c +# +# Three empty socket filters. Upstream exists only to exercise veristat's +# program-name filters, so the bodies are irrelevant and the names are the +# test. Ported for the same reason: three programs, one section, one object. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64 + + +@bpf +@section("socket") +def foo(ctx: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@section("socket") +def bar(ctx: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@section("socket") +def buz(ctx: c_void_p) -> c_int64: + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/enable_stats.py b/tests/kernel_selftest_equivalent/tracing/enable_stats.py new file mode 100644 index 00000000..a1dc4781 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/enable_stats.py @@ -0,0 +1,43 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_enable_stats.c +# +# A raw tracepoint program that counts its runs. Upstream enables +# BPF_STATS_RUN_TIME, triggers the program, and checks run_time_ns and +# run_cnt in bpf_prog_info alongside `count`: +# +# __u64 count = 0; +# +# SEC("raw_tracepoint/sys_enter") +# int test_enable_stats(void *ctx) +# { +# __sync_fetch_and_add(&count, 1); +# return 0; +# } +# +# WORKAROUND(atomics): the upstream increment is atomic. PythonBPF has no +# atomic operations, so this is a plain read-modify-write of the global. + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_uint64 + + +@bpf +@bpfglobal +def count() -> c_uint64: + return c_uint64(0) + + +@bpf +@section("raw_tracepoint/sys_enter") +def test_enable_stats(ctx: c_void_p) -> c_int64: + global count + count += 1 # WORKAROUND(atomics): __sync_fetch_and_add(&count, 1) + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/tracing/perf_link.py b/tests/kernel_selftest_equivalent/tracing/perf_link.py new file mode 100644 index 00000000..7cbc8b87 --- /dev/null +++ b/tests/kernel_selftest_equivalent/tracing/perf_link.py @@ -0,0 +1,43 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/test_perf_link.c +# +# A perf_event program that counts how often it runs. Upstream attaches it +# through a perf_event bpf_link and checks `run_cnt` moved: +# +# int run_cnt = 0; +# +# SEC("perf_event") +# int handler(struct pt_regs *ctx) +# { +# __sync_fetch_and_add(&run_cnt, 1); +# return 0; +# } +# +# WORKAROUND(atomics): the upstream increment is atomic. PythonBPF has no +# atomic operations, so this is a plain read-modify-write of the global. +# Replace with the atomic form once atomics land; grep for WORKAROUND(atomics). + +from pythonbpf import bpf, section, bpfglobal, compile +from ctypes import c_void_p, c_int64, c_int32 + + +@bpf +@bpfglobal +def run_cnt() -> c_int32: + return c_int32(0) + + +@bpf +@section("perf_event") +def handler(ctx: c_void_p) -> c_int64: + global run_cnt + run_cnt += 1 # WORKAROUND(atomics): __sync_fetch_and_add(&run_cnt, 1) + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/kernel_selftest_equivalent/vmlinux/connect4_dropper.py b/tests/kernel_selftest_equivalent/vmlinux/connect4_dropper.py new file mode 100644 index 00000000..d893450c --- /dev/null +++ b/tests/kernel_selftest_equivalent/vmlinux/connect4_dropper.py @@ -0,0 +1,53 @@ +# Ported from Linux tools/testing/selftests/bpf/progs/connect4_dropper.c +# +# A cgroup/connect4 hook that rejects TCP connects to one port, which +# userspace writes into `port` before attaching: +# +# int port; +# +# SEC("cgroup/connect4") +# int connect_v4_dropper(struct bpf_sock_addr *ctx) +# { +# if (ctx->type != SOCK_STREAM) +# return VERDICT_PROCEED; +# if (ctx->user_port == bpf_htons(port)) +# return VERDICT_REJECT; +# return VERDICT_PROCEED; +# } +# +# bpf_htons() is a byte swap, written out here as shifts on the low 16 bits; +# SOCK_STREAM is 1. + +from pythonbpf import bpf, section, bpfglobal, compile +from vmlinux import struct_bpf_sock_addr +from ctypes import c_int64, c_int32 + +VERDICT_REJECT = 0 +VERDICT_PROCEED = 1 +SOCK_STREAM = 1 + + +@bpf +@bpfglobal +def port() -> c_int32: + return c_int32(0) + + +@bpf +@section("cgroup/connect4") +def connect_v4_dropper(ctx: struct_bpf_sock_addr) -> c_int64: + if ctx.type != 1: + return c_int64(1) + port_be = ((port & 0xFF) << 8) | ((port >> 8) & 0xFF) + if ctx.user_port == port_be: + return c_int64(0) + return c_int64(1) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() From 598e3b4efaef43dd6745002583cf4dba4259c835 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:15:55 +0530 Subject: [PATCH 11/25] Tests: Skip vmlinux tests only when vmlinux.py is absent except Exception turned every import-time defect in the generated module into a skip of every vmlinux test, so CI could pass with no vmlinux coverage. ImportError (no module) skips; anything else propagates. Co-Authored-By: Claude Fable 5.1 --- tests/conftest.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index bcea4d33..447a786f 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -32,9 +32,13 @@ VMLINUX_AVAILABLE = True VMLINUX_SKIP_REASON = "" -except Exception as exc: +except ImportError as exc: + # No vmlinux.py: the tests that need it are skipped. Any other exception + # propagates. A vmlinux.py that exists but does not import is a defect in + # the generator, and hiding it behind skips would pass CI with no vmlinux + # coverage at all. VMLINUX_AVAILABLE = False - VMLINUX_SKIP_REASON = f"vmlinux.py not usable for current kernel: {exc}" + VMLINUX_SKIP_REASON = f"vmlinux.py not importable: {exc}" # ── pytest_generate_tests: parametrize on bpf_test_file ─────────────────── From cf80aadd35a6d1d211b2976c3bf4008f34b7e028 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:15:58 +0530 Subject: [PATCH 12/25] Tools: Fix four classifier mistakes in the selftest audit - The subprog_call pattern also matched SEC("?..."), so every non-autoloaded program counted as a BPF-to-BPF call blocker, in contradiction with the soft rule for the same syntax. - The kfunc pattern matched any bpf_xdp_* or bpf_skb_* call, which are ordinary helpers (bpf_skb_store_bytes is in SUPPORTED_HELPERS). Only the kfunc families keep those prefixes; an unsupported helper is still a hard blocker, classified as one. - strip_comments removed // inside string literals, truncating section names like SEC("uprobe//proc/self/exe:func") and classifying the file as a shim. Strings are matched first and kept. - Every bpf_map_* helper not in SUPPORTED_HELPERS was dropped before the unsupported-helper test, so bpf_map_push_elem and friends never showed up as blockers. The membership test classifies them now. Co-Authored-By: Claude Fable 5.1 --- tools/selftest-audit.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/tools/selftest-audit.py b/tools/selftest-audit.py index 9a4edc5f..051a5192 100755 --- a/tools/selftest-audit.py +++ b/tools/selftest-audit.py @@ -92,7 +92,7 @@ r"jited|xlated|caps_unpriv|load_if_JITed|not_msg|failure_unpriv|success_unpriv)\b", ), # functions - ("subprog_call", "hard", r"\b__noinline\b|\b__weak\b|\bSEC\s*\(\s*\"\?"), + ("subprog_call", "hard", r"\b__noinline\b|\b__weak\b"), ( "static_helper", "soft", @@ -106,7 +106,8 @@ r"task_acquire|task_release|cgroup_acquire|cgroup_release|cpumask_\w+|rbtree_\w+|list_\w+|" r"rcu_read_lock|rcu_read_unlock|arena_\w+|key_put|lookup_user_key|dynptr_\w+|iter_\w+|" r"wq_\w+|timer_\w+|throw|percpu_obj_\w+|res_spin_\w+|preempt_\w+|local_irq_\w+|" - r"session_\w+|get_dentry_xattr|get_file_xattr|kptr_xchg|sk_assign|xdp_\w+|skb_\w+)\s*\(", + r"session_\w+|get_dentry_xattr|get_file_xattr|kptr_xchg|sk_assign|" + r"xdp_metadata_\w+|xdp_flow_lookup|skb_flow_lookup)\s*\(", ), ("inline_asm", "hard", r"\basm\s*(volatile)?\s*\(|__asm__"), ("atomic", "hard", r"__sync_\w+|__atomic_\w+|\bbpf_spin_(lock|unlock)\b"), @@ -162,9 +163,13 @@ BTF_DUMP_FIXTURE = re.compile(r"btf_dump|btf__|__attribute__\(\(btf_decl_tag", re.I) +_STRING_OR_COMMENT = re.compile(r'("(?:\\.|[^"\\\n])*")|/\*.*?\*/|//[^\n]*', re.S) + + def strip_comments(src: str) -> str: - src = re.sub(r"/\*.*?\*/", "", src, flags=re.S) - return re.sub(r"//[^\n]*", "", src) + """Remove C comments, leaving string literals alone: a // inside a string + such as SEC("uprobe//proc/self/exe:func") is part of the section name.""" + return _STRING_OR_COMMENT.sub(lambda m: m.group(1) or "", src) def classify(path: Path) -> dict | None: @@ -193,9 +198,6 @@ def classify(path: Path) -> dict | None: hard.add("legacy_map_def") helpers = set(HELPER_CALL.findall(src)) - NOT_HELPERS - helpers = { - h for h in helpers if not h.startswith("bpf_map_") or h in SUPPORTED_HELPERS - } unsupported = sorted( h for h in helpers From 057b68c4d72cd8fc8ee24d13f7e8783a832787e7 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:16:26 +0530 Subject: [PATCH 13/25] Core: Let every integer ctypes constructor declare a slot of its width _allocate_for_call recognised only c_int32, c_int64, c_uint32, c_uint64 and c_void_p, so queue = c_uint16(0) fell through to an i64 slot and a later store of a u32 field into it never narrowed. Any integer ctypes constructor now declares a slot of its width, which is what the ctx_field_narrow_store test claimed to pin; the IR assertion pins it. Co-Authored-By: Claude Fable 5.1 --- pythonbpf/allocation_pass.py | 6 ++++-- tests/test_signedness_ir.py | 6 ++++++ 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/pythonbpf/allocation_pass.py b/pythonbpf/allocation_pass.py index aaf25391..aff095ad 100644 --- a/pythonbpf/allocation_pass.py +++ b/pythonbpf/allocation_pass.py @@ -6,7 +6,7 @@ from pythonbpf.helper import HelperHandlerRegistry from pythonbpf.vmlinux_parser.dependency_node import Field from .expr import VmlinuxHandlerRegistry -from pythonbpf.type_deducer import ctypes_to_ir, IntTy, signedness +from pythonbpf.type_deducer import ctypes_to_ir, is_ctypes, IntTy, signedness from pythonbpf.expr.type_inference import infer_int_type from pythonbpf.maps import BPFMapType @@ -110,7 +110,9 @@ def _allocate_for_call(builder, var_name, rval, local_sym_tab, compilation_conte call_type = rval.func.id # C type constructors - if call_type in ("c_int32", "c_int64", "c_uint32", "c_uint64", "c_void_p"): + if is_ctypes(call_type) and isinstance(ctypes_to_ir(call_type), ir.IntType): + # Any integer ctypes constructor, c_uint16 included, declares a + # slot of that width; the value is converted into it at the store. ir_type = ctypes_to_ir(call_type) var = builder.alloca(ir_type, name=var_name) var.align = ir_type.width // 8 diff --git a/tests/test_signedness_ir.py b/tests/test_signedness_ir.py index 2166ed88..5a4b31d8 100644 --- a/tests/test_signedness_ir.py +++ b/tests/test_signedness_ir.py @@ -68,6 +68,12 @@ [r"\bsub i64", r"trunc i64 .* to i32", r"zext i32 .* to i64"], [], ), + # c_uint16(0) declares a 16-bit slot, so a u32 ctx field stored into it + # is truncated to 16 bits (C: __u16 queue = ctx->rx_queue_index). + "vmlinux/ctx_field_narrow_store.py": ( + [r'%"queue" = alloca i16', r"trunc i64 .* to i16"], + [r'%"queue" = alloca i64'], + ), # A bool widens with zext and an integer narrows to it by != 0, never by # trunc; sext of an i1 would return -1 for True. "signedness/bool_int.py": ( From efe58fc396bbfdc3d2ebe6265062b858828105e5 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:17:00 +0530 Subject: [PATCH 14/25] Core: Treat an integer vmlinux field as its IntTy in assignments The Field branch of handle_variable_assignment rebuilt the field's type by name and duplicated the integer convert() path, and ctypes_to_ir raises NotImplementedError for a pointer or array ctype before the logged error is reached. An integer Field is now normalised to its declared IntTy (field_int_type) before the integer path, in both variable and struct-field assignment, so ctx fields go through the same convert() call as everything else; the Field branch keeps only the non-integer error. Co-Authored-By: Claude Fable 5.1 --- pythonbpf/assign_pass.py | 39 +++++++++++++++++++-------------------- 1 file changed, 19 insertions(+), 20 deletions(-) diff --git a/pythonbpf/assign_pass.py b/pythonbpf/assign_pass.py index f6526bd2..dc876768 100644 --- a/pythonbpf/assign_pass.py +++ b/pythonbpf/assign_pass.py @@ -5,7 +5,7 @@ from llvmlite import ir from pythonbpf.expr import eval_expr, convert from pythonbpf.helper import emit_probe_read_kernel_str_call -from pythonbpf.type_deducer import ctypes_to_ir +from pythonbpf.type_deducer import field_int_type from pythonbpf.vmlinux_parser.dependency_node import Field logger = logging.getLogger(__name__) @@ -40,6 +40,10 @@ def handle_struct_field_assignment( return val, val_type = val_result + if isinstance(val_type, Field): + field_ty = field_int_type(val_type) + if field_ty is not None: + val_type = field_ty # Special case: i8* string to [N x i8] char array if _is_char_array(field_type) and _is_i8_ptr(val_type): @@ -150,6 +154,13 @@ def handle_variable_assignment( logger.info( f"Evaluated value for {var_name}: {val} of type {val_type}, expected {var_type}" ) + # An integer vmlinux field is, for conversion purposes, its declared IntTy + # (width and sign from the ctype), so it takes the same convert() path as + # every other integer below instead of a special case. + if isinstance(val_type, Field): + field_ty = field_int_type(val_type) + if field_ty is not None: + val_type = field_ty if isinstance(val_type, ir.IntType) and isinstance(var_type, ir.IntType): # The descriptor may be narrower than the constant carrying the value @@ -194,25 +205,13 @@ def handle_variable_assignment( ) return False if isinstance(val_type, Field): - logger.info("Handling assignment to struct field") - field_ir_type = ctypes_to_ir(val_type.type.__name__) - # Sub-register-width context fields are zero-extended to i64 by - # load_ctx_field, so val may be wider than the field type says - # (c_uint for xdp_md, c_ushort for pt_regs.cs/ss). convert() - # sizes from the physical value and signs from the field, so it - # is a no-op into an i64 slot and a trunc into a narrower one. - if isinstance(field_ir_type, ir.IntType) and isinstance( - var_type, ir.IntType - ): - val = convert(builder, val, field_ir_type, var_type) - builder.store(val, var_ptr) - logger.info(f"Assigned ctype struct field to {var_name}") - return True - else: - logger.error( - f"Failed to assign ctype struct field to {var_name}: {val_type} != {var_type}" - ) - return False + # Integer fields were normalised to their IntTy above; what is + # left is a pointer, array or struct field, which has no path + # into this slot. + logger.error( + f"Failed to assign ctype struct field to {var_name}: {val_type} != {var_type}" + ) + return False elif isinstance(val_type, ir.IntType) and isinstance(var_type, ir.PointerType): # NOTE: This is assignment to a PTR_TO_MAP_VALUE_OR_NULL logger.info( From 3de98327926a5d2914ea51976e62cfc0dcfffe3a Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:18:52 +0530 Subject: [PATCH 15/25] Core: Implement ArrayMap as BPF_MAP_TYPE_ARRAY process_array_map was a stub raising NotImplementedError while ArrayMap was exported from pythonbpf.maps as if usable. The lowering is the hash map's with a different type constant: the same lookup/update/delete helpers, and the kernel fixes the key at a 4-byte index. The lookup allocation path accepts ARRAY alongside HASH, the audit tool lists ARRAY as supported, and the array_map_lookup_update case is a passing test now instead of a strict xfail. Co-Authored-By: Claude Fable 5.1 --- pythonbpf/allocation_pass.py | 2 +- pythonbpf/maps/maps_pass.py | 18 +++++++++++++++--- .../PORTING-NOTES.md | 6 +++--- tests/test_config.toml | 1 - tools/selftest-audit.py | 4 ++-- 5 files changed, 21 insertions(+), 10 deletions(-) diff --git a/pythonbpf/allocation_pass.py b/pythonbpf/allocation_pass.py index aff095ad..c7dcc931 100644 --- a/pythonbpf/allocation_pass.py +++ b/pythonbpf/allocation_pass.py @@ -208,7 +208,7 @@ def _allocate_for_map_method( return map_params = map_sym_tab[map_name].params - if map_params["type"] != BPFMapType.HASH: + if map_params["type"] not in (BPFMapType.HASH, BPFMapType.ARRAY): logger.warning( "Map method lookup used on non-hash map, using fallback allocation" ) diff --git a/pythonbpf/maps/maps_pass.py b/pythonbpf/maps/maps_pass.py index 108600b8..ae0d2202 100644 --- a/pythonbpf/maps/maps_pass.py +++ b/pythonbpf/maps/maps_pass.py @@ -145,10 +145,22 @@ def process_hash_map(map_name, rval, compilation_context): @MapProcessorRegistry.register("ArrayMap") def process_array_map(map_name, rval, compilation_context): - """Document the planned BPF_ARRAY map support with an explicit failure.""" - raise NotImplementedError( - "ArrayMap is not implemented yet; add BPF_MAP_TYPE_ARRAY metadata support" + """Process a BPF_ARRAY map declaration: the same lowering as a hash map + with BPF_MAP_TYPE_ARRAY, through the same lookup/update/delete helpers; + the kernel requires a 4-byte key (an index).""" + logger.info(f"Processing ArrayMap: {map_name}") + map_params = _parse_map_params(rval, expected_args=["key", "value", "max_entries"]) + map_params["type"] = BPFMapType.ARRAY + + logger.info(f"Map parameters: {map_params}") + map_global = create_bpf_map(compilation_context, map_name, map_params) + create_map_debug_info( + compilation_context, + map_global.var, + map_name, + map_params, ) + return map_global @MapProcessorRegistry.register("PerfEventArray") diff --git a/tests/kernel_selftest_equivalent/PORTING-NOTES.md b/tests/kernel_selftest_equivalent/PORTING-NOTES.md index a5929e34..d09f299e 100644 --- a/tests/kernel_selftest_equivalent/PORTING-NOTES.md +++ b/tests/kernel_selftest_equivalent/PORTING-NOTES.md @@ -74,9 +74,9 @@ cursor, which together constrain the design more than anything here does. `.bss`, `.data` and `.rodata` become internal array maps at load time. Two consequences: - A one-element **`ArrayMap`** is the structurally faithful stand-in for a global, not a - `HashMap`. `HashMap` is used here only because `ArrayMap` is still a placeholder that - raises `NotImplementedError`. Landing `ArrayMap` first would make the eventual migration - to real globals close to mechanical. + `HashMap`. The `HashMap` stand-ins predate `ArrayMap`, which lowers now + (`BPF_MAP_TYPE_ARRAY`, the same helpers as `HashMap`); real `@bpfglobal` scalars have + landed since, so the migration is to those. - **Most of the ELF work is already done.** `@bpfglobal` is vestigial — a metadata carrier for `LICENSE` — but the machinery behind it already emits globals that LLVM places into `.bss` and `.data` correctly, and that libbpf already recognises: diff --git a/tests/test_config.toml b/tests/test_config.toml index fa8c9c84..719137f5 100644 --- a/tests/test_config.toml +++ b/tests/test_config.toml @@ -61,7 +61,6 @@ "failing_tests/loops/for_map_items.py" = {reason = "for/while loops not implemented: no sugar over bpf_for_each_map_elem()-style map iteration exists yet", level = "ir"} -"kernel_selftest_equivalent/maps/array_map_lookup_update.py" = {reason = "ArrayMap / BPF_MAP_TYPE_ARRAY support is planned but not implemented yet", level = "ir"} "kernel_selftest_equivalent/ringbuf/reserve_submit_discard.py" = {reason = "RingBuffer reserve/typed record/discard workflow is planned but not implemented yet", level = "ir"} diff --git a/tools/selftest-audit.py b/tools/selftest-audit.py index 051a5192..26f6798a 100755 --- a/tools/selftest-audit.py +++ b/tools/selftest-audit.py @@ -52,8 +52,8 @@ "bpf_skb_store_bytes", } -# BPF_MAP_TYPE_* that pythonbpf/maps lowers (ArrayMap is still a stub). -SUPPORTED_MAP_TYPES = {"HASH", "PERF_EVENT_ARRAY", "RINGBUF"} +# BPF_MAP_TYPE_* that pythonbpf/maps lowers. +SUPPORTED_MAP_TYPES = {"ARRAY", "HASH", "PERF_EVENT_ARRAY", "RINGBUF"} # Things that look like helper calls but are libbpf macros, not helpers. NOT_HELPERS = { From 49e74858a624f705db909e7472053a5b30760d3c Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:18:54 +0530 Subject: [PATCH 16/25] Tests: Let a verifier-level xfail name the rejection it expects Every xfail is strict with raises=Exception, so a verifier-level negative fixture counted any rejection as the expected one, including one caused by a codegen regression, or by bpftool failing. A verifier entry may now carry match = "..."; the harness attaches it as a verifier_match marker and test_kernel_verifier fails, through pytest.fail (not swallowed by raises=Exception), when the program is rejected for any other reason. xdp_devmap_helpers pins its documented "invalid bpf_context access". Co-Authored-By: Claude Fable 5.1 --- pyproject.toml | 1 + tests/conftest.py | 4 ++++ tests/framework/bpf_test_case.py | 1 + tests/framework/collector.py | 2 ++ .../{vmlinux => xdp}/xdp_tx.py | 0 tests/test_config.toml | 4 +++- tests/test_verifier.py | 11 ++++++++++- 7 files changed, 21 insertions(+), 2 deletions(-) rename tests/kernel_selftest_equivalent/{vmlinux => xdp}/xdp_tx.py (100%) diff --git a/pyproject.toml b/pyproject.toml index 9307f626..3c82f196 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -63,6 +63,7 @@ python_files = ["test_*.py"] markers = [ "verifier: requires sudo/root for kernel verifier tests (not run by default)", "vmlinux: requires vmlinux.py for current kernel", + "verifier_match: substring a verifier-level xfail expects in the rejection", ] log_cli = false diff --git a/tests/conftest.py b/tests/conftest.py index 447a786f..f70dee2d 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -103,6 +103,10 @@ def pytest_collection_modifyitems(items): raises=Exception, ) ) + # A verifier-level xfail may name the rejection it expects; + # any other rejection is then a real failure, not an xfail. + if item_level == "verifier" and case.xfail_match: + item.add_marker(pytest.mark.verifier_match(case.xfail_match)) # ── caplog level fixture: capture ERROR+ from pythonbpf ─────────────────── diff --git a/tests/framework/bpf_test_case.py b/tests/framework/bpf_test_case.py index a993166e..cc1609bf 100644 --- a/tests/framework/bpf_test_case.py +++ b/tests/framework/bpf_test_case.py @@ -23,6 +23,7 @@ class BpfTestCase: is_expected_fail: bool = False xfail_reason: str = "" xfail_level: str = "ir" # one of LEVELS + xfail_match: str = "" # verifier level: substring the rejection must contain needs_vmlinux: bool = False skip_reason: str = "" diff --git a/tests/framework/collector.py b/tests/framework/collector.py index 59b96084..2d5287d6 100644 --- a/tests/framework/collector.py +++ b/tests/framework/collector.py @@ -49,6 +49,7 @@ def collect_all_test_files() -> list[BpfTestCase]: is_expected_fail = xfail_entry is not None xfail_reason = xfail_entry.get("reason", "") if xfail_entry else "" xfail_level = xfail_entry.get("level", "ir") if xfail_entry else "ir" + xfail_match = xfail_entry.get("match", "") if xfail_entry else "" cases.append( BpfTestCase( @@ -56,6 +57,7 @@ def collect_all_test_files() -> list[BpfTestCase]: rel_path=rel, is_expected_fail=is_expected_fail, xfail_reason=xfail_reason, + xfail_match=xfail_match, xfail_level=xfail_level, needs_vmlinux=needs_vmlinux, ) diff --git a/tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py b/tests/kernel_selftest_equivalent/xdp/xdp_tx.py similarity index 100% rename from tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py rename to tests/kernel_selftest_equivalent/xdp/xdp_tx.py diff --git a/tests/test_config.toml b/tests/test_config.toml index 719137f5..1f010014 100644 --- a/tests/test_config.toml +++ b/tests/test_config.toml @@ -6,6 +6,8 @@ # level "ir" = fails during pythonbpf IR generation (exception or ERROR log) # level "llc" = IR generates but llc rejects it # level "verifier" = IR and llc both succeed, but the kernel verifier rejects it +# A verifier-level entry may add match = "..." : the rejection must contain +# that text, otherwise the test fails instead of counting as expected. # # A failure at one level implies failure at every later one, so the declared # level marks that level and all later ones xfail. @@ -71,4 +73,4 @@ # prog_tests driver asserts that a plain load *fails*. The kernel rejects it # here with "invalid bpf_context access off=20 size=4", which is the pass # condition upstream. PythonBPF has no way to set expected_attach_type yet. -"kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py" = {reason = "Negative fixture: egress_ifindex needs expected_attach_type=BPF_XDP_DEVMAP, which cannot be set yet; the verifier rejection is upstream's pass condition", level = "verifier"} +"kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py" = {reason = "Negative fixture: egress_ifindex needs expected_attach_type=BPF_XDP_DEVMAP, which cannot be set yet; the verifier rejection is upstream's pass condition", level = "verifier", match = "invalid bpf_context access"} diff --git a/tests/test_verifier.py b/tests/test_verifier.py index 3966e3f6..bd540583 100644 --- a/tests/test_verifier.py +++ b/tests/test_verifier.py @@ -54,7 +54,7 @@ def _get_rejection_reason(verifier_test_file: Path, output) -> str: _verifier_test_files(), ids=_verifier_test_ids(), ) -def test_kernel_verifier(verifier_test_file: Path, tmp_path, caplog): +def test_kernel_verifier(verifier_test_file: Path, tmp_path, caplog, request): """Compile the BPF test and verify it passes the kernel verifier.""" ll_path = tmp_path / "output.ll" obj_path = tmp_path / "output.o" @@ -70,4 +70,13 @@ def test_kernel_verifier(verifier_test_file: Path, tmp_path, caplog): assert obj_path.exists() and obj_path.stat().st_size > 0 ok, output = verify_object(obj_path) + expected = request.node.get_closest_marker("verifier_match") + if not ok and expected is not None and expected.args[0] not in output.stderr: + # pytest.fail raises an OutcomeException, which the xfail marker's + # raises=Exception does not swallow: a rejection for the wrong + # reason is a failure, not the expected one. + pytest.fail( + f"{verifier_test_file.name}: rejected, but not for the expected reason " + f"{expected.args[0]!r}:\n{output.stderr}" + ) assert ok, _get_rejection_reason(verifier_test_file, output) From d64932774c1553f9f9f512a345b92f798db43eb6 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:18:56 +0530 Subject: [PATCH 17/25] Docs: Refresh the audit numbers after the classifier fixes Co-Authored-By: Claude Fable 5.1 --- .../PORTING-NOTES.md | 23 +++++++++++-------- 1 file changed, 14 insertions(+), 9 deletions(-) diff --git a/tests/kernel_selftest_equivalent/PORTING-NOTES.md b/tests/kernel_selftest_equivalent/PORTING-NOTES.md index d09f299e..5aa69455 100644 --- a/tests/kernel_selftest_equivalent/PORTING-NOTES.md +++ b/tests/kernel_selftest_equivalent/PORTING-NOTES.md @@ -142,7 +142,7 @@ type, never narrowing. It now goes through `convert()` like every other integer ## Third batch: what the re-run audit found `tools/selftest-audit.py` is the corpus classifier, rebuilt and checked in. Run against -the current upstream `progs/` it reports 847 real programs, 25 with no hard blocker. All +the current upstream `progs/` it reports 854 real programs, 25 with no hard blocker. All but two of those 25 were already ported or are unportable for a reason a regex cannot see (`bpf_nop_bench.c` hides a loop in a macro, `test_pkt_md_access.c` type-puns narrow loads, `tracing_struct_int128.c` indexes the raw ctx array and needs bpf_testmod to load). The @@ -173,23 +173,28 @@ increment loses. ### The blocker histogram now -Over 847 real programs, hard blockers only; a program usually hits several: +Over 854 real programs, hard blockers only; a program usually hits several: | Blocker | Programs | Share | |---|---|---| -| unsupported helper | 496 | 59% | -| kfuncs | 284 | 34% | +| unsupported helper | 504 | 59% | | unsupported map type | 273 | 32% | +| kfuncs | 242 | 28% | | verifier-test annotations | 238 | 28% | -| typed program macros (`BPF_PROG`, `BPF_KPROBE`) | 234 | 28% | -| BPF-to-BPF calls | 176 | 21% | +| typed program macros (`BPF_PROG`, `BPF_KPROBE`) | 237 | 28% | | `goto` | 143 | 17% | | inline asm | 142 | 17% | +| BPF-to-BPF calls | 128 | 15% | | struct globals | 126 | 15% | | CO-RE reads | 116 | 14% | -| loops | 109 | 13% | -| array globals | 105 | 12% | -| atomics | 86 | 10% | +| loops | 110 | 13% | +| array globals | 109 | 13% | +| atomics | 87 | 10% | + +Earlier revisions of this table over-counted kfuncs (the pattern matched every +`bpf_skb_*`/`bpf_xdp_*` helper) and BPF-to-BPF calls (it matched `SEC("?...")`), and +under-counted real programs by seven (a `//` inside a section name was stripped as a +comment). The portable set of 25 was unaffected. Globals no longer appear as a blocker at all. The next unlocks by count are helpers (a long tail, but `bpf_get_current_task`, `bpf_ktime_get_boot_ns` and the `bpf_probe_read_user*` From 55fb6d6b9a42947a01917f762a28528ba12a7473 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 01:25:35 +0530 Subject: [PATCH 18/25] Core: Export every XDP action from pythonbpf.helper helpers.py defined all five and return_utils mapped them, but only XDP_DROP and XDP_PASS were exported, so the xdp_tx port had to import XDP_TX from vmlinux and live under vmlinux/, skipped on any host without a generated module. It imports from pythonbpf.helper and lives under xdp/ now, returning XDP_TX bare like its siblings and the C original, which is the shape the return fast path resolves without vmlinux. Co-Authored-By: Claude Fable 5.1 --- pythonbpf/helper/__init__.py | 6 ++++++ tests/kernel_selftest_equivalent/xdp/xdp_tx.py | 7 ++----- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/pythonbpf/helper/__init__.py b/pythonbpf/helper/__init__.py index bd4fe174..129de1d2 100644 --- a/pythonbpf/helper/__init__.py +++ b/pythonbpf/helper/__init__.py @@ -18,8 +18,11 @@ skb_store_bytes, get_current_cgroup_id, get_stack, + XDP_ABORTED, XDP_DROP, XDP_PASS, + XDP_TX, + XDP_REDIRECT, ) @@ -86,6 +89,9 @@ def helper_call_handler(call, compilation_context, builder, func, local_sym_tab) "uid", "skb_store_bytes", "get_stack", + "XDP_ABORTED", "XDP_DROP", "XDP_PASS", + "XDP_TX", + "XDP_REDIRECT", ] diff --git a/tests/kernel_selftest_equivalent/xdp/xdp_tx.py b/tests/kernel_selftest_equivalent/xdp/xdp_tx.py index 8d40ffed..128f4383 100644 --- a/tests/kernel_selftest_equivalent/xdp/xdp_tx.py +++ b/tests/kernel_selftest_equivalent/xdp/xdp_tx.py @@ -3,18 +3,15 @@ # Bounce every packet back out of the interface it arrived on. Upstream is # the transmit side of the veth XDP tests. # -# XDP_TX comes from vmlinux because pythonbpf.helper exports only XDP_PASS -# and XDP_DROP; that is why this lives under vmlinux/. - from pythonbpf import bpf, section, bpfglobal, compile -from vmlinux import XDP_TX +from pythonbpf.helper import XDP_TX from ctypes import c_void_p, c_int64 @bpf @section("xdp") def xdp_tx(xdp: c_void_p) -> c_int64: - return c_int64(XDP_TX) + return XDP_TX # bare, like the C: resolved by the return fast path @bpf From 3a984dd9e2c8bc9a3b3baf2b3b6551ad4d51258e Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 10:15:49 +0530 Subject: [PATCH 19/25] Revert "Core: Export every XDP action from pythonbpf.helper" This reverts commit 55fb6d6b9a42947a01917f762a28528ba12a7473. --- pythonbpf/helper/__init__.py | 6 ------ tests/kernel_selftest_equivalent/xdp/xdp_tx.py | 7 +++++-- 2 files changed, 5 insertions(+), 8 deletions(-) diff --git a/pythonbpf/helper/__init__.py b/pythonbpf/helper/__init__.py index 129de1d2..bd4fe174 100644 --- a/pythonbpf/helper/__init__.py +++ b/pythonbpf/helper/__init__.py @@ -18,11 +18,8 @@ skb_store_bytes, get_current_cgroup_id, get_stack, - XDP_ABORTED, XDP_DROP, XDP_PASS, - XDP_TX, - XDP_REDIRECT, ) @@ -89,9 +86,6 @@ def helper_call_handler(call, compilation_context, builder, func, local_sym_tab) "uid", "skb_store_bytes", "get_stack", - "XDP_ABORTED", "XDP_DROP", "XDP_PASS", - "XDP_TX", - "XDP_REDIRECT", ] diff --git a/tests/kernel_selftest_equivalent/xdp/xdp_tx.py b/tests/kernel_selftest_equivalent/xdp/xdp_tx.py index 128f4383..8d40ffed 100644 --- a/tests/kernel_selftest_equivalent/xdp/xdp_tx.py +++ b/tests/kernel_selftest_equivalent/xdp/xdp_tx.py @@ -3,15 +3,18 @@ # Bounce every packet back out of the interface it arrived on. Upstream is # the transmit side of the veth XDP tests. # +# XDP_TX comes from vmlinux because pythonbpf.helper exports only XDP_PASS +# and XDP_DROP; that is why this lives under vmlinux/. + from pythonbpf import bpf, section, bpfglobal, compile -from pythonbpf.helper import XDP_TX +from vmlinux import XDP_TX from ctypes import c_void_p, c_int64 @bpf @section("xdp") def xdp_tx(xdp: c_void_p) -> c_int64: - return XDP_TX # bare, like the C: resolved by the return fast path + return c_int64(XDP_TX) @bpf From ad6adb14c92ecad37ca849b4301142767c74e610 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 10:26:22 +0530 Subject: [PATCH 20/25] x --- docs/getting-started/quickstart.md | 2 +- docs/user-guide/decorators.md | 2 +- docs/user-guide/structs.md | 3 +- examples/xdp_pass.py | 2 +- pythonbpf/functions/functions_pass.py | 16 +--------- pythonbpf/functions/return_utils.py | 32 ------------------- pythonbpf/helper/__init__.py | 4 --- pythonbpf/helper/helpers.py | 7 ---- .../{ => vmlinux}/direct_assign.py | 2 +- .../failing_tests/{ => vmlinux}/named_arg.py | 2 +- tests/failing_tests/{ => vmlinux}/xdp_pass.py | 2 +- .../{xdp => vmlinux}/priv_prog.py | 2 +- .../vmlinux/xdp_devmap_helpers.py | 2 +- .../{xdp => vmlinux}/xdp_dummy.py | 2 +- .../{xdp => vmlinux}/xdp_tx.py | 3 +- .../passing_tests/return/xdp_name_shadowed.py | 10 +++--- .../{return/xdp.py => vmlinux/return_xdp.py} | 2 +- tests/test_config.toml | 2 +- 18 files changed, 21 insertions(+), 76 deletions(-) rename tests/failing_tests/{ => vmlinux}/direct_assign.py (96%) rename tests/failing_tests/{ => vmlinux}/named_arg.py (95%) rename tests/failing_tests/{ => vmlinux}/xdp_pass.py (97%) rename tests/kernel_selftest_equivalent/{xdp => vmlinux}/priv_prog.py (93%) rename tests/kernel_selftest_equivalent/{xdp => vmlinux}/xdp_dummy.py (94%) rename tests/kernel_selftest_equivalent/{xdp => vmlinux}/xdp_tx.py (78%) rename tests/passing_tests/{return/xdp.py => vmlinux/return_xdp.py} (88%) diff --git a/docs/getting-started/quickstart.md b/docs/getting-started/quickstart.md index 2283adf4..b463478c 100644 --- a/docs/getting-started/quickstart.md +++ b/docs/getting-started/quickstart.md @@ -188,7 +188,7 @@ def trace_open(ctx: c_void_p) -> c_int64: For network packet processing: ```python -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS @section("xdp") def xdp_pass(ctx: c_void_p) -> c_int64: diff --git a/docs/user-guide/decorators.md b/docs/user-guide/decorators.md index ff8b2422..c884fc6c 100644 --- a/docs/user-guide/decorators.md +++ b/docs/user-guide/decorators.md @@ -108,7 +108,7 @@ def trace_open_return(ctx): For network packet processing at the earliest point: ```python -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from ctypes import c_void_p, c_int64 @section("xdp") diff --git a/docs/user-guide/structs.md b/docs/user-guide/structs.md index 5c68c23c..adee503d 100644 --- a/docs/user-guide/structs.md +++ b/docs/user-guide/structs.md @@ -290,7 +290,8 @@ class MyStruct: ```python from pythonbpf import bpf, struct, map, section from pythonbpf.maps import RingBuffer -from pythonbpf.helper import ktime, XDP_PASS +from pythonbpf.helper import ktime +from vmlinux import XDP_PASS from ctypes import c_void_p, c_int64, c_uint8, c_uint16, c_uint32, c_uint64 @bpf diff --git a/examples/xdp_pass.py b/examples/xdp_pass.py index ea294fff..1f49f037 100644 --- a/examples/xdp_pass.py +++ b/examples/xdp_pass.py @@ -1,5 +1,5 @@ from pythonbpf import bpf, map, section, bpfglobal, compile, compile_to_ir -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from pythonbpf.maps import HashMap from ctypes import c_int64, c_void_p diff --git a/pythonbpf/functions/functions_pass.py b/pythonbpf/functions/functions_pass.py index 382ef38e..46fb71c3 100644 --- a/pythonbpf/functions/functions_pass.py +++ b/pythonbpf/functions/functions_pass.py @@ -30,7 +30,7 @@ LocalSymbol, ) from .function_debug_info import generate_function_debug_info -from .return_utils import handle_none_return, handle_xdp_return, is_xdp_name +from .return_utils import handle_none_return from .function_metadata import get_probe_string, is_global_function, infer_return_type @@ -341,20 +341,6 @@ def handle_return( logger.info(f"Handling return statement: {ast.dump(stmt)}") if stmt.value is None: return handle_none_return(builder) - elif ( - isinstance(stmt.value, ast.Name) - and is_xdp_name(stmt.value.id) - and stmt.value.id not in local_sym_tab - and ( - compilation_context is None - or stmt.value.id not in compilation_context.bpf_globals - ) - ): - # The XDP fast path resolves names like XDP_PASS from the helper - # constant table, but only as a fallback: a local or @bpfglobal of the - # same name shadows it, mirroring C (a local shadows an enum constant) - # and the resolution order everywhere else in the compiler. - return handle_xdp_return(stmt, builder, ret_type) else: # Fallback for now if ctx not passed, but caller should pass it if compilation_context is None: diff --git a/pythonbpf/functions/return_utils.py b/pythonbpf/functions/return_utils.py index a05c704d..5979fa61 100644 --- a/pythonbpf/functions/return_utils.py +++ b/pythonbpf/functions/return_utils.py @@ -1,44 +1,12 @@ import logging -import ast from llvmlite import ir logger: logging.Logger = logging.getLogger(__name__) -XDP_ACTIONS = { - "XDP_ABORTED": 0, - "XDP_DROP": 1, - "XDP_PASS": 2, - "XDP_TX": 3, - "XDP_REDIRECT": 4, -} - def handle_none_return(builder) -> bool: """Handle return or return None -> returns 0.""" builder.ret(ir.Constant(ir.IntType(64), 0)) logger.debug("Generated default return: 0") return True - - -def is_xdp_name(name: str) -> bool: - """Check if a name is an XDP action""" - return name in XDP_ACTIONS - - -def handle_xdp_return(stmt: ast.Return, builder, ret_type) -> bool: - """Handle XDP returns""" - if not isinstance(stmt.value, ast.Name): - return False - - action_name = stmt.value.id - - if action_name not in XDP_ACTIONS: - raise ValueError( - f"Unknown XDP action: {action_name}. Available: {XDP_ACTIONS.keys()}" - ) - - value = XDP_ACTIONS[action_name] - builder.ret(ir.Constant(ret_type, value)) - logger.debug(f"Generated XDP action return: {action_name} = {value}") - return True diff --git a/pythonbpf/helper/__init__.py b/pythonbpf/helper/__init__.py index bd4fe174..6a26dcff 100644 --- a/pythonbpf/helper/__init__.py +++ b/pythonbpf/helper/__init__.py @@ -18,8 +18,6 @@ skb_store_bytes, get_current_cgroup_id, get_stack, - XDP_DROP, - XDP_PASS, ) @@ -86,6 +84,4 @@ def helper_call_handler(call, compilation_context, builder, func, local_sym_tab) "uid", "skb_store_bytes", "get_stack", - "XDP_DROP", - "XDP_PASS", ] diff --git a/pythonbpf/helper/helpers.py b/pythonbpf/helper/helpers.py index 253c4b08..832f16fc 100644 --- a/pythonbpf/helper/helpers.py +++ b/pythonbpf/helper/helpers.py @@ -60,10 +60,3 @@ def get_stack(buf, flags=0): def get_current_cgroup_id(): """Get the current cgroup ID""" return ctypes.c_int64(0) - - -XDP_ABORTED = ctypes.c_int64(0) -XDP_DROP = ctypes.c_int64(1) -XDP_PASS = ctypes.c_int64(2) -XDP_TX = ctypes.c_int64(3) -XDP_REDIRECT = ctypes.c_int64(4) diff --git a/tests/failing_tests/direct_assign.py b/tests/failing_tests/vmlinux/direct_assign.py similarity index 96% rename from tests/failing_tests/direct_assign.py rename to tests/failing_tests/vmlinux/direct_assign.py index a7843133..f3362b40 100644 --- a/tests/failing_tests/direct_assign.py +++ b/tests/failing_tests/vmlinux/direct_assign.py @@ -1,5 +1,5 @@ from pythonbpf import bpf, map, section, bpfglobal, compile -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from pythonbpf.maps import HashMap from ctypes import c_void_p, c_int64 diff --git a/tests/failing_tests/named_arg.py b/tests/failing_tests/vmlinux/named_arg.py similarity index 95% rename from tests/failing_tests/named_arg.py rename to tests/failing_tests/vmlinux/named_arg.py index 19139df8..bea5e22f 100644 --- a/tests/failing_tests/named_arg.py +++ b/tests/failing_tests/vmlinux/named_arg.py @@ -1,5 +1,5 @@ from pythonbpf import bpf, map, section, bpfglobal, compile -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from pythonbpf.maps import HashMap from ctypes import c_void_p, c_int64 diff --git a/tests/failing_tests/xdp_pass.py b/tests/failing_tests/vmlinux/xdp_pass.py similarity index 97% rename from tests/failing_tests/xdp_pass.py rename to tests/failing_tests/vmlinux/xdp_pass.py index c8510dcd..b37a9b82 100644 --- a/tests/failing_tests/xdp_pass.py +++ b/tests/failing_tests/vmlinux/xdp_pass.py @@ -1,6 +1,6 @@ from pythonbpf import bpf, map, section, bpfglobal, compile_to_ir from pythonbpf.maps import HashMap -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from vmlinux import TASK_COMM_LEN # noqa: F401 from vmlinux import struct_qspinlock # noqa: F401 diff --git a/tests/kernel_selftest_equivalent/xdp/priv_prog.py b/tests/kernel_selftest_equivalent/vmlinux/priv_prog.py similarity index 93% rename from tests/kernel_selftest_equivalent/xdp/priv_prog.py rename to tests/kernel_selftest_equivalent/vmlinux/priv_prog.py index 181d6840..976144f6 100644 --- a/tests/kernel_selftest_equivalent/xdp/priv_prog.py +++ b/tests/kernel_selftest_equivalent/vmlinux/priv_prog.py @@ -5,7 +5,7 @@ # program itself is the smallest privileged-type program there is. from pythonbpf import bpf, section, bpfglobal, compile -from pythonbpf.helper import XDP_DROP +from vmlinux import XDP_DROP from ctypes import c_void_p, c_int64 diff --git a/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py b/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py index 1e9de0a4..771b017f 100644 --- a/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py +++ b/tests/kernel_selftest_equivalent/vmlinux/xdp_devmap_helpers.py @@ -10,7 +10,7 @@ # return XDP_PASS; from pythonbpf import bpf, section, bpfglobal, compile -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from vmlinux import struct_xdp_md from ctypes import c_int64 diff --git a/tests/kernel_selftest_equivalent/xdp/xdp_dummy.py b/tests/kernel_selftest_equivalent/vmlinux/xdp_dummy.py similarity index 94% rename from tests/kernel_selftest_equivalent/xdp/xdp_dummy.py rename to tests/kernel_selftest_equivalent/vmlinux/xdp_dummy.py index 82686de1..c715cac7 100644 --- a/tests/kernel_selftest_equivalent/xdp/xdp_dummy.py +++ b/tests/kernel_selftest_equivalent/vmlinux/xdp_dummy.py @@ -6,7 +6,7 @@ # for. from pythonbpf import bpf, section, bpfglobal, compile -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS from ctypes import c_void_p, c_int64 diff --git a/tests/kernel_selftest_equivalent/xdp/xdp_tx.py b/tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py similarity index 78% rename from tests/kernel_selftest_equivalent/xdp/xdp_tx.py rename to tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py index 8d40ffed..d432eb74 100644 --- a/tests/kernel_selftest_equivalent/xdp/xdp_tx.py +++ b/tests/kernel_selftest_equivalent/vmlinux/xdp_tx.py @@ -3,8 +3,7 @@ # Bounce every packet back out of the interface it arrived on. Upstream is # the transmit side of the veth XDP tests. # -# XDP_TX comes from vmlinux because pythonbpf.helper exports only XDP_PASS -# and XDP_DROP; that is why this lives under vmlinux/. +# XDP actions are vmlinux enum constants, like every kernel constant. from pythonbpf import bpf, section, bpfglobal, compile from vmlinux import XDP_TX diff --git a/tests/passing_tests/return/xdp_name_shadowed.py b/tests/passing_tests/return/xdp_name_shadowed.py index 0caacd82..b188dd26 100644 --- a/tests/passing_tests/return/xdp_name_shadowed.py +++ b/tests/passing_tests/return/xdp_name_shadowed.py @@ -1,7 +1,9 @@ -# A local named after an XDP action must shadow the helper constant table, -# in return position too. clang agrees: a local legally shadows an enum -# constant, and the local's value is what returns (tests/c-form reference). -# Before the fix this returned the hardcoded 2 while XDP_PASS held 55. +# A local named after an XDP action is just a local: it shadows the vmlinux +# enum constant of that name (when vmlinux is imported) exactly as a local +# shadows an enum constant in C, and its value is what returns. This once +# went through a special-cased return path that ignored the local and +# returned the hardcoded 2 while XDP_PASS held 55; that path is gone, and +# return resolves names like every other expression. from pythonbpf import bpf, section, bpfglobal, compile from ctypes import c_void_p, c_int64 diff --git a/tests/passing_tests/return/xdp.py b/tests/passing_tests/vmlinux/return_xdp.py similarity index 88% rename from tests/passing_tests/return/xdp.py rename to tests/passing_tests/vmlinux/return_xdp.py index 3c0f5d8c..3978c729 100644 --- a/tests/passing_tests/return/xdp.py +++ b/tests/passing_tests/vmlinux/return_xdp.py @@ -1,6 +1,6 @@ from pythonbpf import bpf, section, bpfglobal, compile from ctypes import c_void_p, c_int64 -from pythonbpf.helper import XDP_PASS +from vmlinux import XDP_PASS @bpf diff --git a/tests/test_config.toml b/tests/test_config.toml index 1f010014..ada07a15 100644 --- a/tests/test_config.toml +++ b/tests/test_config.toml @@ -26,7 +26,7 @@ "failing_tests/vmlinux/args_test.py" = {reason = "struct_trace_event_raw_sys_enter args field access not supported", level = "ir"} -"failing_tests/xdp_pass.py" = {reason = "XDP program using vmlinux structs (struct_xdp_md) and complex map/struct interaction not yet supported", level = "ir"} +"failing_tests/vmlinux/xdp_pass.py" = {reason = "XDP program using vmlinux structs (struct_xdp_md) and complex map/struct interaction not yet supported", level = "ir"} "failing_tests/globals_read_before_shadow.py" = {reason = "Reading a name above the assignment that makes it a local shadowing a global is UnboundLocalError in Python, and a compile error here", level = "ir"} From 49ed2cf12774b68aece181918ef0668e27a991de Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 10:28:56 +0530 Subject: [PATCH 21/25] Core: Report an unknown name in an operand as undefined get_typed_operand fell through its Name branch for a name that is not a local, a global or a vmlinux constant, and raised the generic "Unsupported operand type" meant for unknown node kinds. Since return goes through it, forgetting `from vmlinux import XDP_PASS` produced that message; it says "Undefined variable XDP_PASS" now, like every other read. Co-Authored-By: Claude Fable 5.1 --- pythonbpf/expr/expr_pass.py | 1 + 1 file changed, 1 insertion(+) diff --git a/pythonbpf/expr/expr_pass.py b/pythonbpf/expr/expr_pass.py index c454e704..b9984b83 100644 --- a/pythonbpf/expr/expr_pass.py +++ b/pythonbpf/expr/expr_pass.py @@ -254,6 +254,7 @@ def get_typed_operand(func, compilation_context, operand, builder, local_sym_tab vmlinux_result = VmlinuxHandlerRegistry.handle_name(operand.id) if vmlinux_result is not None: return vmlinux_result # (i64 constant, its C rank) + raise SyntaxError(f"Undefined variable {operand.id}") elif isinstance(operand, ast.Constant): if isinstance(operand.value, (int, bool)): v = int(operand.value) From d3fa7d7e42b9a810676ce20d38bef9edc4b37f3a Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 22:20:00 +0530 Subject: [PATCH 22/25] Core: Print integers of any width in f-strings The printk formatter accepted 64- and 32-bit integers only and raised for anything else. That went unnoticed while c_uint16(0) and friends silently made 64-bit slots; with those slots now their declared width, printing any 8- or 16-bit local failed. Every integer argument is already widened to 64 bits per its sign before the call, so the 64-bit format fits every width, and the sign now chooses %lld or %llu (an unsigned value above INT_MAX printed as negative before). Co-Authored-By: Claude Opus 5.5 (1M context) --- pythonbpf/helper/printk_formatter.py | 16 +++++----------- 1 file changed, 5 insertions(+), 11 deletions(-) diff --git a/pythonbpf/helper/printk_formatter.py b/pythonbpf/helper/printk_formatter.py index 47262c99..1b96ca8a 100644 --- a/pythonbpf/helper/printk_formatter.py +++ b/pythonbpf/helper/printk_formatter.py @@ -1,5 +1,6 @@ import ast import logging +from pythonbpf.type_deducer import signedness from llvmlite import ir from pythonbpf.expr import eval_expr, get_base_type_and_depth, deref_to_depth, convert @@ -155,17 +156,10 @@ def _process_attr_in_fval(attr_node, fmt_parts, exprs, local_sym_tab, struct_sym def _populate_fval(ftype, node, fmt_parts, exprs): """Populate format parts and expressions based on field type.""" if isinstance(ftype, ir.IntType): - # TODO: We print as signed integers only for now - if ftype.width == 64: - fmt_parts.append("%lld") - exprs.append(node) - elif ftype.width == 32: - fmt_parts.append("%d") - exprs.append(node) - else: - raise NotImplementedError( - f"Unsupported integer width in f-string: {ftype.width}" - ) + # Every integer argument is widened to 64 bits (per its sign) before + # the call, so the 64-bit format fits any width; the sign picks it. + fmt_parts.append("%lld" if signedness(ftype) else "%llu") + exprs.append(node) elif isinstance(ftype, ir.PointerType): target, depth = get_base_type_and_depth(ftype) if isinstance(target, ir.IntType): From 4e8f767e6d0c7489bb2c9ada6f6e7a73674aa9e4 Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 22:20:05 +0530 Subject: [PATCH 23/25] Core: Size and describe integers by their real width Two places took an integer's width as 32 or 64 bits. Ten sites computed byte sizes as width // 8 (alignment, struct layout, globals), which is zero for a 1-bit type; they now share byte_size(), which rounds up to whole bytes. And debug info described every map key or value, and every struct field, that was not 32 bits wide as unsigned long long: a c_uint8 or c_uint16 map value was declared 8 bytes wide in BTF, so the kernel copied 8 bytes from a smaller slot. get_int_type() now gives the real width and sign everywhere, globals included. Co-Authored-By: Claude Opus 5.5 (1M context) --- pythonbpf/allocation_pass.py | 10 +++++----- pythonbpf/debuginfo/debug_info_generator.py | 21 +++++++++++++++++++++ pythonbpf/globals_pass.py | 17 +++-------------- pythonbpf/maps/map_debug_info.py | 16 +++++++--------- pythonbpf/structs/struct_type.py | 5 +++-- pythonbpf/structs/structs_pass.py | 8 ++++---- pythonbpf/type_deducer.py | 6 ++++++ 7 files changed, 49 insertions(+), 34 deletions(-) diff --git a/pythonbpf/allocation_pass.py b/pythonbpf/allocation_pass.py index c7dcc931..10ff7b2f 100644 --- a/pythonbpf/allocation_pass.py +++ b/pythonbpf/allocation_pass.py @@ -6,7 +6,7 @@ from pythonbpf.helper import HelperHandlerRegistry from pythonbpf.vmlinux_parser.dependency_node import Field from .expr import VmlinuxHandlerRegistry -from pythonbpf.type_deducer import ctypes_to_ir, is_ctypes, IntTy, signedness +from pythonbpf.type_deducer import ctypes_to_ir, is_ctypes, IntTy, signedness, byte_size from pythonbpf.expr.type_inference import infer_int_type from pythonbpf.maps import BPFMapType @@ -115,7 +115,7 @@ def _allocate_for_call(builder, var_name, rval, local_sym_tab, compilation_conte # slot of that width; the value is converted into it at the store. ir_type = ctypes_to_ir(call_type) var = builder.alloca(ir_type, name=var_name) - var.align = ir_type.width // 8 + var.align = byte_size(ir_type) local_sym_tab[var_name] = LocalSymbol(var, ir_type) logger.info(f"Pre-allocated {var_name} as {call_type}") @@ -472,7 +472,7 @@ def _allocate_for_attribute( tmp_name = f"{struct_var}_{field_name}_tmp" tmp_ir_type = ir.IntType(field_size_bits) tmp_var = builder.alloca(tmp_ir_type, name=tmp_name) - tmp_var.align = tmp_ir_type.width // 8 + tmp_var.align = byte_size(tmp_ir_type) local_sym_tab[tmp_name] = LocalSymbol(tmp_var, tmp_ir_type) logger.info( f"Pre-allocated temp {tmp_name} (i{field_size_bits}) for vmlinux field read {vmlinux_struct_name}.{field_name}" @@ -527,8 +527,8 @@ def _allocate_with_type(builder, var_name, ir_type): def _get_alignment(ir_type): """Get appropriate alignment for IR type.""" if isinstance(ir_type, ir.IntType): - return ir_type.width // 8 + return byte_size(ir_type) elif isinstance(ir_type, ir.ArrayType) and isinstance(ir_type.element, ir.IntType): - return ir_type.element.width // 8 + return byte_size(ir_type.element) else: return 8 # Default: pointer size diff --git a/pythonbpf/debuginfo/debug_info_generator.py b/pythonbpf/debuginfo/debug_info_generator.py index 8dc31ee6..3be8f287 100644 --- a/pythonbpf/debuginfo/debug_info_generator.py +++ b/pythonbpf/debuginfo/debug_info_generator.py @@ -49,6 +49,27 @@ def get_basic_type(self, name: str, size: int, encoding: int) -> Any: ) return self._type_cache[key] + def get_int_type(self, ty) -> Any: + """Debug type for an integer IR type, by width and sign (read from the + IntTy descriptor, signed for a plain ir.IntType). A 1-bit type is C's + _Bool, which occupies one byte. BTF, and so a map's key and value + sizes, come from this, so the width must be the real one.""" + from pythonbpf.type_deducer import signedness + + width, signed = ty.width, signedness(ty) + if width == 1: + return self.get_basic_type("_Bool", 8, dc.DW_ATE_boolean) + base = {8: "char", 16: "short", 32: "int", 64: "long long"}.get(width) + if base is None: + raise ValueError(f"no debug type for a {width}-bit integer") + if width == 8: + encoding = dc.DW_ATE_signed_char if signed else dc.DW_ATE_unsigned_char + else: + encoding = dc.DW_ATE_signed if signed else dc.DW_ATE_unsigned + return self.get_basic_type( + base if signed else f"unsigned {base}", width, encoding + ) + def get_uint8_type(self) -> Any: """Get debug info for signed 8-bit integer""" return self.get_basic_type("char", 8, dc.DW_ATE_unsigned) diff --git a/pythonbpf/globals_pass.py b/pythonbpf/globals_pass.py index c5dc0ba5..ee06978e 100644 --- a/pythonbpf/globals_pass.py +++ b/pythonbpf/globals_pass.py @@ -3,16 +3,13 @@ from logging import Logger import logging -from .type_deducer import ctypes_to_ir, is_signed_ctype +from .type_deducer import ctypes_to_ir, byte_size from .symbols import BpfGlobalSymbol from .debuginfo import DebugInfoGenerator from .expr import VmlinuxHandlerRegistry -from .debuginfo import dwarf_constants as dc logger: Logger = logging.getLogger(__name__) -_C_NAME_BY_WIDTH = {8: "char", 16: "short", 32: "int", 64: "long long"} - def populate_global_symbol_table(tree, compilation_context): """ @@ -78,7 +75,7 @@ def _emit_global(module: ir.Module, node, name): # (align 4 for i32, align 8 for i64). llc derives the BTF DATASEC layout # from these symbols, so the alignment should mirror the C reference in # tests/c-form/global_vars.bpf.c. - gvar.align = ty.width // 8 if isinstance(ty, ir.IntType) else 8 + gvar.align = byte_size(ty) if isinstance(ty, ir.IntType) else 8 gvar.linkage = "dso_local" gvar.global_constant = False return gvar @@ -93,15 +90,7 @@ def _emit_global_debug_info(compilation_context, gvar, name, ctype_name): future skeleton can tell which variable lives at which offset. """ generator = DebugInfoGenerator(compilation_context.module) - width = gvar.value_type.width - signed = is_signed_ctype(ctype_name) - base = _C_NAME_BY_WIDTH[width] - if width == 8: - encoding = dc.DW_ATE_signed_char if signed else dc.DW_ATE_unsigned_char - else: - encoding = dc.DW_ATE_signed if signed else dc.DW_ATE_unsigned - cname = base if signed else f"unsigned {base}" - di_type = generator.get_basic_type(cname, width, encoding) + di_type = generator.get_int_type(ctypes_to_ir(ctype_name)) dv = generator.create_global_var_debug_info(name, di_type, is_local=False) gvar.set_metadata("dbg", dv) diff --git a/pythonbpf/maps/map_debug_info.py b/pythonbpf/maps/map_debug_info.py index e8795703..19011896 100644 --- a/pythonbpf/maps/map_debug_info.py +++ b/pythonbpf/maps/map_debug_info.py @@ -1,3 +1,4 @@ +from pythonbpf.type_deducer import ctypes_to_ir, is_ctypes import logging from llvmlite import ir from pythonbpf.debuginfo import DebugInfoGenerator @@ -122,11 +123,12 @@ def _get_key_val_dbg_type(name, generator, structs_sym_tab): # Fallback to basic types logger.info(f"No struct named {name}, falling back to basic type") - # NOTE: Only handling int and long for now - if name in ["c_int32", "c_uint32"]: - return generator.get_uint32_type() + # A ctypes integer: its real width and sign, so the map's key or value + # size in BTF matches what the program stores (a c_uint8 value is 1 byte). + if is_ctypes(name): + return generator.get_int_type(ctypes_to_ir(name)) - # Default fallback for now + logger.warning(f"No debug type for map key/value {name}, defaulting to u64") return generator.get_uint64_type() @@ -136,11 +138,7 @@ def _get_struct_debug_type(struct_obj, generator, structs_sym_tab): for fld in struct_obj.fields.keys(): fld_type = struct_obj.field_type(fld) if isinstance(fld_type, ir.IntType): - if fld_type.width == 32: - fld_dbg_type = generator.get_uint32_type() - else: - # NOTE: Assuming 64-bit for all other int types - fld_dbg_type = generator.get_uint64_type() + fld_dbg_type = generator.get_int_type(fld_type) elif isinstance(fld_type, ir.ArrayType): # NOTE: Array types have u8 elements only for now # Debug info generation should fail for other types diff --git a/pythonbpf/structs/struct_type.py b/pythonbpf/structs/struct_type.py index 90abf056..9266c2cd 100644 --- a/pythonbpf/structs/struct_type.py +++ b/pythonbpf/structs/struct_type.py @@ -1,3 +1,4 @@ +from pythonbpf.type_deducer import byte_size from llvmlite import ir @@ -24,9 +25,9 @@ def gep(self, builder, ptr, field_name): def field_size(self, field_name): fld = self.fields[field_name] if isinstance(fld, ir.ArrayType): - return fld.count * (fld.element.width // 8) + return fld.count * byte_size(fld.element) elif isinstance(fld, ir.IntType): - return fld.width // 8 + return byte_size(fld) elif isinstance(fld, ir.PointerType): return 8 diff --git a/pythonbpf/structs/structs_pass.py b/pythonbpf/structs/structs_pass.py index bfc9d88d..74a4dcf4 100644 --- a/pythonbpf/structs/structs_pass.py +++ b/pythonbpf/structs/structs_pass.py @@ -1,7 +1,7 @@ import ast import logging from llvmlite import ir -from pythonbpf.type_deducer import ctypes_to_ir +from pythonbpf.type_deducer import ctypes_to_ir, byte_size from .struct_type import StructType logger = logging.getLogger(__name__) @@ -79,11 +79,11 @@ def calc_struct_size(field_types): curr_offset = 0 for ftype in field_types: if isinstance(ftype, ir.IntType): - fsize = ftype.width // 8 + fsize = byte_size(ftype) alignment = fsize elif isinstance(ftype, ir.ArrayType): - fsize = ftype.count * (ftype.element.width // 8) - alignment = ftype.element.width // 8 + fsize = ftype.count * byte_size(ftype.element) + alignment = byte_size(ftype.element) elif isinstance(ftype, ir.PointerType): # We won't encounter this rn, but for the future fsize = 8 diff --git a/pythonbpf/type_deducer.py b/pythonbpf/type_deducer.py index f6f412c4..8d0400e9 100644 --- a/pythonbpf/type_deducer.py +++ b/pythonbpf/type_deducer.py @@ -41,6 +41,12 @@ def describe(self) -> str: return f"{'i' if self.signed else 'u'}{self.width}" +def byte_size(ty) -> int: + """Bytes an integer type occupies in memory: its width rounded up to whole + bytes, so a 1-bit bool still takes one byte, as C's _Bool does.""" + return (ty.width + 7) // 8 + + def int_literal_type(value: int) -> IntTy: """C's type for an integer constant: `int` if the value fits, else `long long`. Literals and enum constants alike; an enum constant is an From 20a1e2f3a1ac09f0da30c0dada5d35c18121049d Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 22:20:07 +0530 Subject: [PATCH 24/25] Core: Accept c_bool as a ctypes type c_bool was missing from the ctypes table, so c_bool(5) was an unknown call, its slot was never allocated, and the error read "Undefined variable". It is C's _Bool: one bit, like the True/False locals, never negative, and a value narrows to it by comparing with zero, which the signedness rules already apply to 1-bit types. It works as a local, a struct field, a map value and a global; in BTF it is a one-byte _Bool. Co-Authored-By: Claude Opus 5.5 (1M context) --- pythonbpf/type_deducer.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/pythonbpf/type_deducer.py b/pythonbpf/type_deducer.py index 8d0400e9..1854b3da 100644 --- a/pythonbpf/type_deducer.py +++ b/pythonbpf/type_deducer.py @@ -113,6 +113,9 @@ def signedness(ty) -> bool: "c_longlong": 64, # A pointer-sized integer; treated as unsigned like uintptr_t. "c_void_p": 64, + # C's _Bool: one bit to LLVM (as True/False locals already are). It is + # never negative, and a value narrows to it by comparing with zero. + "c_bool": 1, } From a5eeaafdf3415ac14e88e3956d85fb4d4cbd2fcd Mon Sep 17 00:00:00 2001 From: Pragyansh Chaturvedi Date: Fri, 25 Sep 2026 22:20:09 +0530 Subject: [PATCH 25/25] Tests: Context fields into slots of every width, and c_bool ctx_field_into_slots.py stores 8-, 32- and 64-bit sk_buff fields into locals declared at every width and into struct fields of every width, with IR assertions on the slot widths and the truncation before each store. It is the compact form of a 52-case matrix (four field widths, nine local and four struct-field destinations) compiled on master and here: every case compiles, nothing sign-extends a field on the way in, and where master compiled the IR is identical except where master put a narrow local in a 64-bit slot. c_bool.py covers the local, struct field, map value and global. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/passing_tests/signedness/c_bool.py | 49 +++++++++++++++++++ .../vmlinux/ctx_field_into_slots.py | 47 ++++++++++++++++++ tests/test_signedness_ir.py | 19 +++++++ 3 files changed, 115 insertions(+) create mode 100644 tests/passing_tests/signedness/c_bool.py create mode 100644 tests/passing_tests/vmlinux/ctx_field_into_slots.py diff --git a/tests/passing_tests/signedness/c_bool.py b/tests/passing_tests/signedness/c_bool.py new file mode 100644 index 00000000..e8089042 --- /dev/null +++ b/tests/passing_tests/signedness/c_bool.py @@ -0,0 +1,49 @@ +# c_bool is C's _Bool: a value narrows to it by comparing with zero (5 is +# true), and it widens to 0 or 1 (zero-extended, never sign-extended), as a +# local, a struct field, a map value and a global. In BTF it is a one-byte +# _Bool, so the map's value size is 1. +from ctypes import c_bool, c_int64, c_uint32, c_void_p +from pythonbpf import bpf, map, struct, section, bpfglobal, compile +from pythonbpf.maps import HashMap + + +@bpf +@struct +class flags: + on: c_bool + n: c_uint32 + + +@bpf +@map +def seen() -> HashMap: + return HashMap(key=c_uint32, value=c_bool, max_entries=4) + + +@bpf +@bpfglobal +def armed() -> c_bool: + return c_bool(1) + + +@bpf +@section("tracepoint/raw_syscalls/sys_enter") +def prog(ctx: c_void_p) -> c_int64: + global armed + b = c_bool(5) + f = flags() + f.on = 2 + k = c_uint32(1) + seen.update(k, c_bool(1)) + armed = 7 + print(f"{b} {f.on} {armed}") + return c_int64(b) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/passing_tests/vmlinux/ctx_field_into_slots.py b/tests/passing_tests/vmlinux/ctx_field_into_slots.py new file mode 100644 index 00000000..c13a12ef --- /dev/null +++ b/tests/passing_tests/vmlinux/ctx_field_into_slots.py @@ -0,0 +1,47 @@ +# A context field stored into a slot of any integer width: the field is loaded +# at its declared width, zero-extended (all of these are unsigned), then cut to +# the slot's width, as C's `__u8 x = skb->len` does. Covers locals declared +# with every width and struct fields of every width. +from ctypes import c_int8, c_uint16, c_int32, c_uint64, c_uint8, c_uint32, c_int64 +from pythonbpf import bpf, struct, section, bpfglobal, compile +from vmlinux import struct___sk_buff + + +@bpf +@struct +class rec: + f8: c_uint8 + f16: c_uint16 + f32: c_uint32 + f64: c_uint64 + + +@bpf +@section("tc") +def prog(ctx: struct___sk_buff) -> c_int64: + a8 = c_int8(0) + a8 = ctx.len # u32 -> i8 + b16 = c_uint16(0) + b16 = ctx.len # u32 -> u16 + c32 = c_int32(0) + c32 = ctx.tstamp # u64 -> i32 + d64 = c_uint64(0) + d64 = ctx.tstamp_type # u8 -> u64 + e = ctx.len # undeclared: a 64-bit slot + r = rec() + r.f8 = ctx.len + r.f16 = ctx.tstamp + r.f32 = ctx.tstamp + r.f64 = ctx.tstamp_type + print(f"{a8} {b16} {c32}") + print(f"{d64} {e}") + return c_int64(0) + + +@bpf +@bpfglobal +def LICENSE() -> str: + return "GPL" + + +compile() diff --git a/tests/test_signedness_ir.py b/tests/test_signedness_ir.py index 5a4b31d8..96b003e4 100644 --- a/tests/test_signedness_ir.py +++ b/tests/test_signedness_ir.py @@ -74,6 +74,25 @@ [r'%"queue" = alloca i16', r"trunc i64 .* to i16"], [r'%"queue" = alloca i64'], ), + # A context field into slots of every width: loaded, zero-extended, + # truncated to the slot; nothing sign-extended on the way in. + "vmlinux/ctx_field_into_slots.py": ( + [ + r'%"a8" = alloca i8', + r'%"b16" = alloca i16', + r'%"c32" = alloca i32', + r'%"d64" = alloca i64', + r'trunc i64 %[^\n]* to i8\n\s*store i8 [^\n]*%"a8"', + r'trunc i64 %[^\n]* to i16\n\s*store i16 [^\n]*%"b16"', + r'trunc i64 %[^\n]* to i32\n\s*store i32 [^\n]*%"c32"', + ], + [r'alloca i64[^\n]*\n[^\n]*%"a8"'], + ), + # c_bool: narrowing is != 0, widening is zext; never trunc to i1 or sext. + "signedness/c_bool.py": ( + [r"icmp ne i64 5, 0", r"icmp ne i64 2, 0", r"icmp ne i64 7, 0", r"zext i1"], + [r"sext i1 ", r"trunc i64 [^\n]* to i1"], + ), # A bool widens with zext and an integer narrows to it by != 0, never by # trunc; sext of an i1 would return -1 for True. "signedness/bool_int.py": (