Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
21 commits
Select commit Hold shift + click to select a range
ed57883
feat(memtrack): capture allocation stacks in eBPF
not-matthias Aug 28, 2026
8574ab5
feat(memtrack): add userspace stack-capture module
not-matthias Aug 28, 2026
2f5e7e5
feat(memtrack): enable stack capture through the tracker
not-matthias Sep 1, 2026
a8e9430
test(memtrack): cover allocation stack capture
not-matthias Aug 28, 2026
e296c7b
refactor(runner): move ELF artifact pipeline to executor/shared
not-matthias Aug 28, 2026
e6e0268
feat(runner): add MemtrackMetadata sharing ModuleArtifacts with walltime
not-matthias Aug 28, 2026
3b7be16
feat(memtrack): record mapped modules for offline stack attribution
not-matthias Aug 28, 2026
7d2bfa8
feat(runner): write memtrack module artifacts and metadata
not-matthias Aug 28, 2026
e20cd5d
docs(memtrack): describe the mapping recorder
not-matthias Aug 28, 2026
b7cba60
test(memtrack): add nested allocation fixtures
not-matthias Sep 1, 2026
af91938
test(memtrack): assert nested stack identities
not-matthias Sep 1, 2026
ba1e5d8
refactor(memtrack): capture module mappings with perf
not-matthias Sep 1, 2026
a721f61
perf(runner-shared): pre-size the frame output buffer
not-matthias Sep 2, 2026
2d4ded1
ci: benchmark runner-shared in memory mode
not-matthias Sep 2, 2026
559363a
refactor(runner-shared): inline pid-key deserializer into metadata
not-matthias Sep 3, 2026
8c69119
fix(memtrack): skip allocator symbols aliased at an attached offset
not-matthias Sep 3, 2026
2f7d478
perf(memtrack): switch to mimalloc to reduce memory usage and fragmen…
not-matthias Sep 4, 2026
774482d
perf(memtrack): resolve stack fp chains off the ring poll thread
not-matthias Sep 4, 2026
c9e0bf6
fix(memtrack): read tracking_enabled from a global instead of an arra…
not-matthias Sep 7, 2026
c66179d
fix(memtrack): detach BPF links through forked fd holders
not-matthias Sep 7, 2026
a939b33
feat(runner): add experimental flag to capture allocation stacks
not-matthias Sep 7, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -93,7 +93,7 @@ jobs:
# Each memtrack integration test binary runs its cases serially
# (eBPF tracker can't overlap with itself in one process), so we
# shard at the test-binary level to parallelize across jobs.
test: [c_tests, cpp_tests, rust_tests, spawn_tests, dlopen_tests, rss_tests]
test: [c_tests, cpp_tests, rust_tests, spawn_tests, dlopen_tests, rss_tests, stack_tests]
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
Expand Down Expand Up @@ -133,7 +133,7 @@ jobs:
strategy:
fail-fast: false
matrix:
mode: [simulation, walltime]
mode: [simulation, walltime, memory]
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
Expand Down
74 changes: 74 additions & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -111,6 +111,7 @@ ipc-channel = "0.20"
itertools = "0.14.0"
rayon = "1.12"
linux-perf-event-reader = "0.10.2" # matches the version linux-perf-data resolves to
perf-event-open-sys = "6.0"
env_logger = "0.11.10"
tempfile = "3.27.0"
object = { version = "0.39", default-features = false, features = ["read_core", "elf"] }
Expand Down
38 changes: 36 additions & 2 deletions crates/memtrack/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,11 +20,20 @@ Control plane: `src/ipc.rs` exposes an out-of-band `ipc-channel` protocol (`Enab

Allocator discovery (`src/allocators/`): `AllocatorLib::find_all()` = dynamic (glob shared libs incl. `/nix/store/*` hints) + static-linked (scan build-dir ELF symbols) + env (`CODSPEED_MEMTRACK_BINARIES`). Each `AllocatorKind` (`Libc`/`LibCpp`/`Jemalloc`/`Mimalloc`/`Tcmalloc`) maps to best-effort attach helpers; only libc must succeed.

Mapping collection (`src/perf_mappings.rs`) uses Linux's native per-CPU perf event stream, not an LSM/BPF availability gate. `PerfMappingPoller` opens a `PERF_TYPE_SOFTWARE` dummy event with `PERF_ATTR_INHERIT` and `PERF_ATTR_MMAP2` on every online CPU for the tracked process, mmaps a perf ring per CPU, and drains those rings on a poll thread. It keeps executable mappings with absolute paths from `PERF_RECORD_MMAP2` and emits the single artifact representation, `MemtrackEventKind::Mapping` inside `MemtrackArtifact.events`, carrying the mapping's pid/tid/timestamp/address/path/device/inode/file offset/length. Opening or enabling any perf event requires the host's perf permissions (for example an allowed `perf_event_paranoid` policy or `CAP_PERFMON`); a permission error is returned from `Tracker::spawn` rather than silently disabling mapping collection. Kernel `PERF_RECORD_LOST` records, ring overruns, and malformed records increment the shared mapping-loss counter. `Tracker::dropped_events_count()` includes that counter with BPF ring-buffer drops, and `codspeed-memtrack track` aborts when the total is non-zero because the artifact is incomplete.`

### Event stream compatibility

Session relies on Rust's declaration-order field drop: _poller, _stack_poller, then _perf_mapping_poller. The BPF event and stack pollers therefore disconnect, fully drain, and join before the perf poller is dropped. PerfMappingPoller buffers mapping records and emits them during shutdown, after ordinary allocation/RSS/stack events have reached encode_events; encode_events preserves input order, so Mapping records are a terminal suffix in the one artifact stream.

This ordering is compatibility-critical. Mapping is a newer event variant; older stream consumers may treat the first unknown Mapping as EOF. Keeping it as the suffix lets those consumers process the complete memory timeline before stopping at that first unknown record. Do not reorder the poller fields or emit mapping records before shutdown.

> Note: the "on-demand attach" design in `.agents/docs/` (AttachWorker, `CODSPEED_MEMTRACK_ONDEMAND`, SIGSTOP/SIGCONT) is a **plan, not yet in source**. Current behavior is upfront attach + `sched_fork` auto-tracking.

## Key Directories

- `src/ebpf/` — BPF stack (feature-gated `ebpf`): `tracker.rs` (facade), `memtrack/` (libbpf-rs wrapper + generated skeleton, split into `mod.rs`/`macros.rs`/`maps.rs`/`allocator.rs`/`tracking.rs`), `poller.rs`, `events.rs`, `c/main.bpf.c` + `c/event.h` + `c/utils/*.h` + `c/allocator.h`.
- `src/ebpf/` — BPF stack (feature-gated `ebpf`): `tracker.rs` (facade), `memtrack/` (libbpf-rs wrapper + generated skeleton, split into `mod.rs`/`macros.rs`/`maps.rs`/`allocator.rs`/`tracking.rs`), `stacks/`, `poller.rs`, `events.rs`, `c/main.bpf.c` + `c/event.h` + `c/stack_capture.bpf.h` + `c/utils/*.h` + `c/allocator.h`.
- `src/perf_mappings.rs` — native per-CPU `PERF_RECORD_MMAP2` collector.
- `src/allocators/` — allocator classification: `mod.rs`, `dynamic.rs`, `static_linked.rs`.
- `tests/` — integration tests + `snapshots/` (insta).
- `testdata/` — allocation fixtures: `*.c` (gcc), `alloc_cpp/` (cmkr/CMake), `alloc_rust/` + `spawn_wrapper/` (standalone Cargo workspaces).
Expand Down Expand Up @@ -75,7 +84,32 @@ sudo -E cargo test --test c_tests -- --test-threads 1
- **Build toolchain:** `clang` + BTF/vmlinux headers, `libbpf-dev`, `zlib1g-dev`, `pkgconf`, `build-essential`; vendored libbpf also needs `autopoint`/`bison`/`flex`.
- `vmlinux.h` is pinned to a specific git rev; `libbpf-rs` uses the `vendored` feature (dist links `libbpf-rs/static`).

Env vars actually wired: `CODSPEED_MEMTRACK_BINARIES` (extra static-allocator binaries), `CODSPEED_LOG` (log filter, default `info`), `SUDO_UID`/`SUDO_GID` (privilege drop), `GITHUB_ACTIONS` (build rebuild trigger + test gate).
Env vars actually wired: `CODSPEED_MEMTRACK_BINARIES` (extra static-allocator binaries), `CODSPEED_MEMTRACK_TRACK_ALLOCATORS` (0/false disables), `CODSPEED_MEMTRACK_TRACK_PHYSICAL` (1 enables), `CODSPEED_MEMTRACK_CAPTURE_STACKS` (1 enables), `CODSPEED_MEMTRACK_STACK_BUDGET` (stack copy size in bytes, default 8192), `CODSPEED_LOG` (log filter, default `info`), `SUDO_UID`/`SUDO_GID` (privilege drop), `GITHUB_ACTIONS` (build rebuild trigger + test gate).

### Minimum kernel version

The BPF object needs **Linux >= 5.11** with `CONFIG_DEBUG_INFO_BTF` (vmlinux BTF). Nothing below is gated or `set_autoload(false)`-able, so the verifier rejects the whole skeleton on older kernels. Per feature:

| Feature | Used in | Kernel |
|---|---|---|
| `BPF_MAP_TYPE_STACK_TRACE`, `bpf_get_stackid` | `c/stack_capture.bpf.h` | 4.6 |
| `BPF_MAP_TYPE_LRU_HASH` | `c/utils/mm_ownership.h`, `c/rss.bpf.h` | 4.10 |
| `PERF_RECORD_MMAP2` with `clockid` | `src/perf_mappings.rs` | 4.1 |
| vmlinux BTF + CO-RE relocations | everything | 5.2 |
| `bpf_send_signal` | `c/attach.h` | 5.3 |
| `fentry/*`, `tp_btf/*` programs, `bpf_probe_read_user` | `c/attach.h`, `c/process_tracking.bpf.h`, `c/allocator.h` | 5.5 |
| `tracepoint/kmem/rss_stat` with `mm_id`/`curr` fields | `c/rss.bpf.h` | 5.5 |
| `bpf_get_ns_current_pid_tgid` | `c/utils/variant.h` | 5.7 |
| `BPF_MAP_TYPE_RINGBUF`, `bpf_ringbuf_*` | all event paths | 5.8 |
| `bpf_get_current_task_btf` (sets the floor) | `c/process_tracking.bpf.h`, `c/rss.bpf.h`, `c/rmap.bpf.h` | 5.11 |

Gated on top of the floor:

- rmap physical tracking (`src/ebpf/memtrack/rmap.rs`, `RmapSupport::for_version`): `Unsupported` < 6.8, PTE/PMD hooks 6.8–6.14, PUD hooks added at 6.15.
- `Token` variant (`uprobe_multi` links, `c/memtrack_token.bpf.c`): >= 6.6; BPF token delegation itself needs 6.9. The `Legacy` variant (`perf_event_open` uprobes) is the fallback.
- `folio` layout differences (`flags.f` >= 6.18, `_folio_order` < 6.6) are probed with `bpf_core_field_exists` in `c/utils/folio.h`.

Consequences: cgroup-based BPF memory accounting (5.11) means `bump_memlock_rlimit` failing is always harmless, and userspace syscalls newer than the floor (e.g. `close_range`, 5.9) need no `ENOSYS` fallback.

## Testing & QA

Expand Down
6 changes: 5 additions & 1 deletion crates/memtrack/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ required-features = ["ebpf"]

[features]
default = ["ebpf"]
ebpf = ["dep:libbpf-rs", "dep:libbpf-cargo", "dep:vmlinux"]
ebpf = ["dep:libbpf-rs", "dep:libbpf-cargo", "dep:vmlinux", "dep:mimalloc"]

[dependencies]
anyhow = { workspace = true }
Expand All @@ -34,9 +34,13 @@ itertools = { workspace = true }
paste = "1.0.15"
libbpf-rs = { version = "0.26", features = ["vendored"], optional = true }
object = { workspace = true }
perf-event2 = "0.7.4"
linux-perf-event-reader = { workspace = true }
byteorder = "1.5"
rayon = "1.12"
parking_lot = "0.12"
typed-builder = "0.23.2"
mimalloc = { version = "0.1", optional = true }

[build-dependencies]
libbpf-cargo = { version = "0.26", optional = true }
Expand Down
25 changes: 18 additions & 7 deletions crates/memtrack/src/ebpf/c/allocator.h
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
BPF_HASH_MAP(name##_arg, __u64, __u64, 10000); \
SEC(UPROBE_SEC) \
int uprobe_##name(struct pt_regs* ctx) { \
stash_stack_hash(capture_stack(ctx)); \
return store_param(&name##_arg, arg_expr); \
} \
SEC(URETPROBE_SEC) \
Expand All @@ -17,6 +18,7 @@
if (!arg_ptr) { \
return 0; \
} \
__u64 stack_hash = take_stack_hash(); \
__u64 ret_val = PT_REGS_RC(ctx); \
if (ret_val == 0) { \
return 0; \
Expand All @@ -32,6 +34,7 @@
if (arg0 == 0) { \
return 0; \
} \
__u64 stack_hash = capture_stack(ctx); \
submit_block; \
}

Expand All @@ -50,6 +53,8 @@
return 0; \
} \
\
stash_stack_hash(capture_stack(ctx)); \
\
struct name##_args_t args = {.arg0 = arg0_expr, .arg1 = arg1_expr}; \
\
bpf_map_update_elem(&name##_args, &tid, &args, BPF_ANY); \
Expand All @@ -63,6 +68,7 @@
if (!args) { \
return 0; \
} \
__u64 stack_hash = take_stack_hash(); \
\
struct name##_args_t a = *args; \
bpf_map_delete_elem(&name##_args, &tid); \
Expand All @@ -77,20 +83,22 @@
submit_block; \
}

UPROBE_ARG_RET(malloc, PT_REGS_PARM1(ctx), { return submit_alloc_event(arg0, ret_val); })
UPROBE_ARG_RET(malloc, PT_REGS_PARM1(ctx),
{ return submit_alloc_event(arg0, ret_val, stack_hash); })

UPROBE_RET(free, PT_REGS_PARM1(ctx), { return submit_free_event(arg0); })
UPROBE_RET(free, PT_REGS_PARM1(ctx), { return submit_free_event(arg0, stack_hash); })

UPROBE_ARG_RET(calloc, PT_REGS_PARM1(ctx) * PT_REGS_PARM2(ctx),
{ return submit_calloc_event(arg0, ret_val); })
{ return submit_calloc_event(arg0, ret_val, stack_hash); })

UPROBE_ARGS_RET(realloc, PT_REGS_PARM2(ctx), PT_REGS_PARM1(ctx),
{ return submit_realloc_event(arg1, ret_val, arg0); })
{ return submit_realloc_event(arg1, ret_val, arg0, stack_hash); })

UPROBE_ARG_RET(aligned_alloc, PT_REGS_PARM2(ctx),
{ return submit_aligned_alloc_event(arg0, ret_val); })
{ return submit_aligned_alloc_event(arg0, ret_val, stack_hash); })

UPROBE_ARG_RET(memalign, PT_REGS_PARM2(ctx), { return submit_aligned_alloc_event(arg0, ret_val); })
UPROBE_ARG_RET(memalign, PT_REGS_PARM2(ctx),
{ return submit_aligned_alloc_event(arg0, ret_val, stack_hash); })

/*
* posix_memalign(void** memptr, size_t alignment, size_t size)
Expand All @@ -115,6 +123,8 @@ int uprobe_posix_memalign(struct pt_regs* ctx) {
return 0;
}

stash_stack_hash(capture_stack(ctx));

struct posix_memalign_args_t args = {.memptr = PT_REGS_PARM1(ctx), .size = PT_REGS_PARM3(ctx)};
bpf_map_update_elem(&posix_memalign_args, &tid, &args, BPF_ANY);
return 0;
Expand All @@ -127,6 +137,7 @@ int uretprobe_posix_memalign(struct pt_regs* ctx) {
if (!args) {
return 0;
}
__u64 stack_hash = take_stack_hash();

struct posix_memalign_args_t a = *args;
bpf_map_delete_elem(&posix_memalign_args, &tid);
Expand All @@ -140,7 +151,7 @@ int uretprobe_posix_memalign(struct pt_regs* ctx) {
return 0;
}

return submit_aligned_alloc_event(a.size, addr);
return submit_aligned_alloc_event(a.size, addr, stack_hash);
}

#endif /* __ALLOCATOR_H__ */
Loading
Loading