diff --git a/.github/workflows/kernel-matrix.yml b/.github/workflows/kernel-matrix.yml new file mode 100644 index 0000000..94f875b --- /dev/null +++ b/.github/workflows/kernel-matrix.yml @@ -0,0 +1,226 @@ +name: kernel-matrix + +# Build every BPF object once per job, then boot a range of kernels and confirm +# each one's verifier accepts every program in bin/*.bpf.o. The check is the +# vendored static `veristat` (it loads each program and reports a verdict); +# kernels come from cilium's little-vm-helper (quay.io/lvh-images), booted under +# QEMU/KVM on the runner. Each job writes a detail table to its step summary and +# uploads its result; the final `matrix` job pivots them into one ✅/❌ grid. +# +# httpscope links one object per bpf// directory — socket, wire, walk, +# and the TLS taps (ssl, ssl_ex, gotls, gotls_read, rustls) — because a uprobe +# tap must load independently of the kernel-global probes. The matrix therefore +# has a row per (object, program): peer_sendmsg/peer_recvmsg recur in every TLS +# tap, so a bare program name would not be unique. +# +# Tune `matrix.kernel` to the kernel lines your script must support (`6.6`, +# `bpf-next`, …); available lines live at +# https://quay.io/repository/lvh-images/kind?tab=tags. Each line is resolved to +# a concrete image at run time rather than using the floating `-main` tag, +# which the action can't consume: little-vm-helper@v0.0.30 derives the VM image +# filename by stripping a trailing *numeric* build stamp, so a `-main` tag +# yields a name that doesn't match the file `lvh` actually unpacks and the run +# dies with "invalid reference format". So each job looks up the newest +# date-stamped tag (`-YYYYMMDD.HHMMSS`, which the action handles) — always +# tracking the latest build, with no tag to bump and immune to quay's pruning of +# old stamps. +# +# The floor is 6.6: the wire tap attaches with TCX, which 6.1 refuses at load +# (zero instructions verified), and every other object loads on 6.1 — the +# legacy socket tap names `iov_iter.__iov` (6.4+) but its reads are CO-RE +# guarded, so that 6.4 floor is for the build host only, where vmlinux.h is +# generated from the runner's own BTF by `make bpf`. So the gating lines are +# the LTS and stable kernels from 6.6 up. bpf-next is run but does not gate: +# it is a moving target, and what it shows tends to arrive in stable soon +# after. Case in point: the walk VM's three nested bpf_loops verified in 14k +# instructions until 7.2.7 backported a batch of verifier precision fixes +# for bpf_loop callbacks ("backtracking shouldn't clear outer frame R1-R5 +# for callbacks" and kin), after which the same object cost 850k in `step` +# alone and hit the 1M limit on 7.2.8 and on bpf-next. Its state now lives +# in a per-CPU map the verifier does not track, and it verifies in under 5k. + +on: + workflow_dispatch: + push: + branches: [master, main] + pull_request: + +permissions: + contents: read + +jobs: + verify: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + # Kernel lines to verify. Each is resolved to its newest date-stamped + # lvh image at run time (see the header). + kernel: + - '6.6' + - '6.12' + - '6.18' + - '7.2' + - 'bpf-next' + # bpf-next is informational: a failure there is recorded in the grid + # but does not fail the job, so the summary's gate ignores it. + include: + - kernel: 'bpf-next' + informational: true + name: kernel ${{ matrix.kernel }} + continue-on-error: ${{ matrix.informational == true }} + steps: + - uses: actions/checkout@v4 + + - name: Resolve newest lvh image tag + id: img + env: + KERNEL: ${{ matrix.kernel }} + run: | + set -euo pipefail + # Newest -YYYYMMDD.HHMMSS tag (date-stamps sort + # lexicographically, so tail -1 is the most recent build). + newest="$(curl -sf "https://quay.io/api/v1/repository/lvh-images/kind/tag/?onlyActiveTags=true&limit=100&filter_tag_name=like:${KERNEL}-" \ + | jq -r '.tags[].name' \ + | grep -E "^${KERNEL}-[0-9]{8}\.[0-9]+$" | sort | tail -1)" + [ -n "$newest" ] || { echo "::error::no date-stamped tag found for kernel line '${KERNEL}'"; exit 1; } + echo "resolved ${KERNEL} -> ${newest}" + echo "tag=${newest}" >> "$GITHUB_OUTPUT" + + - name: Build BPF objects + stage veristat + run: | + set -euo pipefail + # Builds every bin/.bpf.o with the vendored static toolchain, + # also populating the per-machine toolchain cache + # (clang/bpftool/veristat). + make bpf + + # Resolve the vendored static veristat the same way build/toolchain.mk + # does, and stage it into bin/ so the VM finds it under /host. It is + # fully static, so it runs in any kernel image's rootfs. + . build/toolchain.lock + arch="$(uname -m)"; [ "$arch" = arm64 ] && arch=aarch64 + cache="${XDG_CACHE_HOME:-$HOME/.cache}/yeet/toolchain/v${TOOLCHAIN_VERSION}/${arch}" + if [ ! -x "$cache/veristat" ]; then + echo "::error::veristat is not in the pinned toolchain (v${TOOLCHAIN_VERSION}). Bump build/toolchain.lock to a toolchain release that ships veristat." + exit 1 + fi + install -Dm755 "$cache/veristat" bin/veristat + file bin/veristat bin/*.bpf.o + + - name: Verify on kernel ${{ matrix.kernel }} + uses: cilium/little-vm-helper@v0.0.30 + with: + test-name: veristat-${{ matrix.kernel }} + image: kind + image-version: ${{ steps.img.outputs.tag }} + host-mount: ${{ github.workspace }} + install-dependencies: 'true' + cmd: | + cd /host + OUT_CSV=/host/.kmatrix/result.csv sh build/verify-kernel.sh + + - name: Render kernel summary + if: always() + env: + KVER: ${{ matrix.kernel }} + KCSV: ${{ github.workspace }}/.kmatrix/result.csv + run: | + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import csv, os + kver, path = os.environ["KVER"], os.environ["KCSV"] + if not os.path.exists(path): + print(f"### kernel `{kver}` — ⚠️ no result (build or boot failed)\n") + raise SystemExit + rows = list(csv.DictReader(open(path))) + mark = lambda v: "✅" if v == "success" else "❌" + ok = all(r["verdict"] == "success" for r in rows) + head = "✅ all programs loaded" if ok else "❌ verifier rejected a program" + print(f"### kernel `{kver}` — {head}\n") + print("| Object | Program | Verdict | Insns | States |") + print("|---|---|:---:|--:|--:|") + for r in rows: + print(f"| `{r['file_name']}` | `{r['prog_name']}` | {mark(r['verdict'])} | {r['total_insns']} | {r['total_states']} |") + print() + PY + + - name: Upload result + if: always() + uses: actions/upload-artifact@v4 + with: + name: kmatrix-${{ matrix.kernel }} + path: ${{ github.workspace }}/.kmatrix/result.csv + if-no-files-found: ignore + + matrix: + needs: verify + if: always() + runs-on: ubuntu-latest + name: matrix summary + steps: + - uses: actions/download-artifact@v4 + with: + path: results + pattern: kmatrix-* + + - name: Render matrix + run: | + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import csv, glob, os, re + + # One CSV per kernel under results/kmatrix-/result.csv. + # Rows are keyed by (object, program): peer_sendmsg/peer_recvmsg + # recur across the TLS taps, so a program name alone is not unique. + data, kernels, progs = {}, [], [] + for d in sorted(glob.glob("results/kmatrix-*")): + kver = os.path.basename(d)[len("kmatrix-"):] + f = os.path.join(d, "result.csv") + if not os.path.exists(f): + data[kver] = None + kernels.append(kver) + continue + data[kver] = {(r["file_name"], r["prog_name"]): r["verdict"] for r in csv.DictReader(open(f))} + kernels.append(kver) + for p in data[kver]: + if p not in progs: + progs.append(p) + + # Order kernels by version, bpf-next last. + def keyf(k): + m = re.match(r"(\d+)\.(\d+)", k) + return (1, 0, 0) if not m else (0, int(m.group(1)), int(m.group(2))) + kernels.sort(key=keyf) + short = lambda k: re.sub(r"-(main|\d{8}\.\d+)$", "", k) + + print("## 🐧 Kernel verification matrix\n") + if not progs: + print("⚠️ No results were produced — check the per-kernel job logs.\n") + raise SystemExit + print("| Object | Program | " + " | ".join(short(k) for k in kernels) + " |") + print("|---|---|" + "|".join(":-:" for _ in kernels) + "|") + fail = 0 + for obj, prog in progs: + cells = [] + for k in kernels: + d = data[k] + if d is None or (obj, prog) not in d: + cells.append("⚪") + elif d[(obj, prog)] == "success": + cells.append("✅") + else: + cells.append("❌"); fail += 1 + print(f"| `{obj}` | `{prog}` | " + " | ".join(cells) + " |") + print() + print("✅ accepted · ❌ rejected · ⚪ not run · bpf-next is informational and does not gate\n") + total = len(progs) * len([k for k in kernels if data[k] is not None]) + verb = "all programs loaded on every kernel" if fail == 0 else f"{fail} of {total} program×kernel checks failed" + print(f"**{len(progs)} program(s) × {len(kernels)} kernel(s) — {verb}.**") + PY + + - name: Gate on any rejection + run: | + # Fail the run if any per-kernel job failed (a rejection or a build/boot error). + if [ "${{ contains(needs.verify.result, 'failure') }}" = "true" ] || [ "${{ needs.verify.result }}" = "failure" ]; then + echo "::error::one or more kernels rejected a program (see the matrix summary)" + exit 1 + fi diff --git a/.gitignore b/.gitignore index 8eea11c..9be84cc 100644 --- a/.gitignore +++ b/.gitignore @@ -6,3 +6,6 @@ bin/*.bpf.o .build/ bpf/include/vmlinux.h .clangd + +# kernel-matrix run output (build/kernel-matrix.sh) +.kmatrix/ diff --git a/README.md b/README.md index 5ec3add..dec26b7 100644 --- a/README.md +++ b/README.md @@ -393,10 +393,18 @@ query overrides with `(kind)`. This is what tcpwalk2 does with a 54 KB schema table rendered at build time from one kernel's BTF — a table that, checked here, had `snd_cwnd` forty bytes from where this kernel keeps it. The kernel side is derived from tcpwalk2's VM with -one structural change: every op runs as a `bpf_loop` step, so the +two structural changes. Every op runs as a `bpf_loop` step, so the verifier walks the dispatch once per call site instead of once per -slot per section per field, which on this kernel was the difference -between a million-instruction rejection and a load. +slot per section per field. And the VM's mutable state lives in a +one-element per-CPU array, not on the stack, with the callbacks +reaching it by lookup and only the event pointer riding on the stack as +the callback context: a callback loop is verified once only when its +entry state converges with an earlier one, and the verifier does not +track map memory, so nothing the ops do can tell one iteration from the +last. With the state on the stack, the 7.2.7 verifier backports left +the object simulating every one of its 8 × 16 × 8 nested iterations and +hitting the million-instruction limit; with it in the map, the whole +object verifies in under 5k instructions on every kernel in the matrix. The event carries the socket's 4-tuple, state and the segment's sequence number and lengths, read through CO-RE in a fixed prologue, @@ -465,6 +473,28 @@ buffer held before — a response head where curl's body should be. - **Clean clone**: `git clone`, `npm install`, `make bpf`, `npm test`, `npx yeetkit build` all pass from a fresh checkout on this box. The yeetkit dependency is by absolute path for now. +- **Kernel matrix**: `.github/workflows/kernel-matrix.yml` (the one + `yeet new` scaffolds, adapted to this repo's per-directory objects) + builds every `bin/*.bpf.o` on the runner, boots 6.6, 6.12, 6.18, 7.2 + and bpf-next under cilium's little-vm-helper, and runs the vendored + static veristat in each VM via `build/verify-kernel.sh`. The summary + is a grid of (object, program) × kernel, since + `peer_sendmsg`/`peer_recvmsg` recur across the TLS taps. The floor is + 6.6: 6.1 refuses the wire tap's TCX attach type at load, and every + other object loads there (the socket tap's `iov_iter` reads are CO-RE + guarded, so the 6.4 note above is for the build host only). The first + run also caught the 6.6 verifier rejecting the wire tap's + `bpf_skb_load_bytes` size as possibly zero, since a verifier before + 6.9 does not narrow a register on a `!= 0` branch; the bound is now + rebuilt by arithmetic. The second catch was the walk VM: on this + host's 7.2.6 it verified in 14k instructions, but stable 7.2.7 + backported a batch of verifier precision fixes for `bpf_loop` + callbacks, after which the lvh 7.2.8 image and bpf-next ran it to the + one-million limit. Moving the VM's state into a per-CPU map (see the + walk VM section) brought it to under 5k everywhere. bpf-next runs but + does not gate. `make veristat-matrix` runs the same thing + locally with lvh + a static qemu (Linux, KVM, root for the VM), and + `make veristat` is the single-kernel check against this host. ### Known limits diff --git a/bpf/walk/walk.bpf.c b/bpf/walk/walk.bpf.c index 4f259d5..e89eb3c 100644 --- a/bpf/walk/walk.bpf.c +++ b/bpf/walk/walk.bpf.c @@ -161,46 +161,98 @@ struct { __uint(max_entries, 1 << 22); } events SEC(".maps"); -/* Everything the VM threads through its loops, on the tracepoint's - * frame: bpf_loop callbacks each get their own frame, and the whole - * chain shares 512 bytes, so state lives here and the callbacks hold - * one pointer to it. */ +/* The VM's mutable state lives in a per-CPU array element, not on the + * stack, and each callback reaches it with a lookup. Two reasons. The + * bpf_loop callbacks each take a frame of the 512-byte stack the whole + * chain shares, so there was little room for it there. And the + * verifier does not track the contents of map memory, so nothing the + * ops do to this state can tell one loop iteration from the last: a + * callback loop is verified once, when its entry state converges with + * an earlier one, and any scalar the verifier holds precise (a size it + * checked, a value it predicted a branch on) blocks that. With the + * state on the stack, 7.2.7's bpf_loop precision fixes left this + * object at the 1M instruction limit, every one of the 8 x 16 x 8 + * nested iterations simulated in full. raw_tp/tcp_probe cannot + * re-enter on a CPU (softirq, or process context with bottom halves + * off), so one element per CPU is enough. + * + * The one thing the callbacks need the verifier to keep typed is the + * event they write into: a ring buffer pointer kept in map memory + * would come back as a plain number. So it rides on the stack as the + * callback context, which bpf_loop requires to be a stack pointer. */ struct walk_state { __u64 regs[NREGS]; - struct walk_event *e; /* the field being run */ - struct walk_prog *p; __u32 f; + __u32 section; /* 0 = pre, terminals to nav_scratch; 1 = body, to the event */ + __u32 idx; /* the entry a body terminal fills: ENTRY_IDX(f, it) */ __u64 node; /* the section being run */ - struct walk_op *ops; __u32 n; __u64 cur; - __u8 *out; __u32 wrote; __u32 skip; __u32 live; __u64 scratch[NSCRATCH]; }; +struct { + __uint(type, BPF_MAP_TYPE_PERCPU_ARRAY); + __uint(max_entries, 1); + __type(key, __u32); + __type(value, struct walk_state); +} state SEC(".maps"); + +struct walk_ctx { + struct walk_event *e; +}; + +static __always_inline struct walk_state *vm_state(void) +{ + __u32 zero = 0; + return bpf_map_lookup_elem(&state, &zero); +} + +/* Throwaway terminal buffer for a navigation-only pre. Global, not + * stack: the frames of the nested loops share the 512-byte BPF stack. + * CPUs may race on it; the contents are never read. */ +static __u8 nav_scratch[ENTRY_CAP]; + +/* Where a terminal op writes: the field's entry in the event for a + * body section, the throwaway buffer for a pre. */ +static __always_inline __u8 *terminal_out(struct walk_ctx *c, struct walk_state *st) +{ + if (st->section) + return &c->e->edata[(st->idx & (MAX_FIELDS * MAX_ITERS - 1)) * ENTRY_CAP]; + return nav_scratch; +} + /* One op of a straight-line section, as a bpf_loop step. Every op is a * branch of one dispatch the verifier walks once per call site; run * unrolled, the same dispatch is walked for each of the eight slots of * each section of each field and the state budget is gone. Returns 1 * to stop the section: past its end, or a read failed. */ -static long step(__u32 i, void *pst) +static long step(__u32 i, void *pctx) { - struct walk_state *st = pst; + struct walk_ctx *c = pctx; + struct walk_state *st = vm_state(); + if (!st) + return 1; if (i >= st->n || !st->live) return 1; if (st->skip > 0) { st->skip--; return 0; } - struct walk_op *op = &st->ops[i & (MAX_STEP - 1)]; + __u32 key = st->f & (MAX_FIELDS - 1); + struct walk_prog *p = bpf_map_lookup_elem(&programs, &key); + if (!p) + return 1; + struct walk_op *op = st->section ? &p->body[i & (MAX_STEP - 1)] : &p->pre[i & (MAX_STEP - 1)]; __u32 code = op->code; __u32 arg = op->arg; __u64 cur = st->cur; + __u8 *out = terminal_out(c, st); if (code == WOP_LOADN) { __u64 v = 0; @@ -243,12 +295,12 @@ static long step(__u32 i, void *pst) __u32 sz = arg & (ENTRY_CAP - 1); if (sz == 0) sz = 1; - if (bpf_probe_read_kernel(st->out, sz, (void *) cur)) + if (bpf_probe_read_kernel(out, sz, (void *) cur)) st->live = 0; else st->wrote = sz; } else if (code == WOP_STR) { - long r = bpf_probe_read_kernel_str(st->out, ENTRY_CAP, (void *) cur); + long r = bpf_probe_read_kernel_str(out, ENTRY_CAP, (void *) cur); if (r < 0) st->live = 0; else @@ -257,7 +309,7 @@ static long step(__u32 i, void *pst) __u32 sz = arg; if (sz == 0 || sz > 8) sz = 8; - *(__u64 *) st->out = cur; /* all 8 written; the host reads `len` */ + *(__u64 *) out = cur; /* all 8 written; the host reads `len` */ st->wrote = sz; } @@ -265,40 +317,48 @@ static long step(__u32 i, void *pst) return st->live ? 0 : 1; } -/* Run a section: `ops`/`n` from cursor `start`, terminal bytes to - * `out`. Afterwards st->cur is where the cursor ended (a navigation-only - * pre hands its head onward) and st->wrote how much a terminal emitted. */ -static __always_inline void run_section(struct walk_state *st, struct walk_op *ops, __u32 n, __u64 start, __u8 *out) +/* Run a section: the field's pre (0) or body (1), `n` ops from cursor + * `start`. Afterwards st->cur is where the cursor ended (a + * navigation-only pre hands its head onward) and st->wrote how much a + * terminal emitted. */ +static __always_inline void run_section(struct walk_ctx *c, struct walk_state *st, __u32 section, __u32 n, __u64 start) { - st->ops = ops; + st->section = section; st->n = n > MAX_STEP ? MAX_STEP : n; st->cur = start; - st->out = out; st->wrote = 0; st->skip = 0; st->live = 1; for (int i = 0; i < NSCRATCH; i++) st->scratch[i] = 0; - bpf_loop(MAX_STEP, step, st, 0); + bpf_loop(MAX_STEP, step, c, 0); } /* One chain node: run body from the node, emit an entry, advance via * next_off. Returns 1 to stop (NULL, a failed read, or a scalar's one * entry), 0 to continue. */ -static long walk_node(__u32 it, void *pst) +static long walk_node(__u32 it, void *pctx) { - struct walk_state *st = pst; - struct walk_prog *p = st->p; + struct walk_ctx *c = pctx; + struct walk_state *st = vm_state(); + if (!st) + return 1; + __u32 key = st->f & (MAX_FIELDS - 1); + struct walk_prog *p = bpf_map_lookup_elem(&programs, &key); + if (!p) + return 1; if (p->next_off && st->node == 0) return 1; - __u32 idx = ENTRY_IDX(st->f, it); - __u8 *out = &st->e->edata[idx * ENTRY_CAP]; - run_section(st, p->body, p->n_body, st->node, out); + __u32 idx = ENTRY_IDX(key, it); + st->idx = idx; + run_section(c, st, 1, p->n_body, st->node); if (!st->live || st->wrote == 0) return 1; - st->e->elen[idx] = st->wrote; - st->e->fcount[st->f & (MAX_FIELDS - 1)] = it + 1; - st->e->ok |= (1u << (st->f & (MAX_FIELDS - 1))); + /* Masked again at each use: clang spills `key` as a 32-bit slot and + * the verifier forgets the bound on the fill. */ + c->e->elen[idx & (MAX_FIELDS * MAX_ITERS - 1)] = st->wrote; + c->e->fcount[key & (MAX_FIELDS - 1)] = it + 1; + c->e->ok |= (1u << (key & (MAX_FIELDS - 1))); if (p->next_off == 0) return 1; __u64 adv = 0; @@ -308,24 +368,21 @@ static long walk_node(__u32 it, void *pst) return 0; } -/* Throwaway terminal buffer for a navigation-only pre. Global, not - * stack: the frames of the nested loops share the 512-byte BPF stack. - * CPUs may race on it; the contents are never read. */ -static __u8 nav_scratch[ENTRY_CAP]; - /* One field: reach the chain head (or, for a scalar, leave the cursor * wherever pre lands), then walk. */ -static long run_field(__u32 f, void *pst) +static long run_field(__u32 f, void *pctx) { - struct walk_state *st = pst; + struct walk_ctx *c = pctx; + struct walk_state *st = vm_state(); + if (!st) + return 1; __u32 key = f & (MAX_FIELDS - 1); struct walk_prog *p = bpf_map_lookup_elem(&programs, &key); if (!p) return 0; - st->p = p; st->f = key; - run_section(st, p->pre, p->n_pre, 0, nav_scratch); + run_section(c, st, 0, p->n_pre, 0); st->node = st->cur; __u32 mi = p->max_iters; @@ -333,7 +390,7 @@ static long run_field(__u32 f, void *pst) mi = 1; if (mi > MAX_ITERS) mi = MAX_ITERS; - bpf_loop(mi, walk_node, st, 0); + bpf_loop(mi, walk_node, c, 0); return 0; } @@ -413,6 +470,9 @@ int on_tcp_probe(struct bpf_raw_tracepoint_args *ctx) return 0; last_ns = tnow; + struct walk_state *st = vm_state(); + if (!st) + return 0; struct walk_event *e = bpf_ringbuf_reserve(&events, sizeof(*e), 0); if (!e) { ring_full++; @@ -429,12 +489,11 @@ int on_tcp_probe(struct bpf_raw_tracepoint_args *ctx) e->plen = sg.plen; e->linear = sg.linear; - struct walk_state st = {}; - st.regs[0] = ctx->args[0]; /* struct sock * */ - st.regs[1] = ctx->args[1]; /* struct sk_buff * */ - st.regs[2] = sg.payload; - st.e = e; - bpf_loop(nf, run_field, &st, 0); + st->regs[0] = ctx->args[0]; /* struct sock * */ + st->regs[1] = ctx->args[1]; /* struct sk_buff * */ + st->regs[2] = sg.payload; + struct walk_ctx c = { .e = e }; + bpf_loop(nf, run_field, &c, 0); emitted++; bpf_ringbuf_submit(e, 0); diff --git a/bpf/wire/wire.bpf.c b/bpf/wire/wire.bpf.c index 3c570ca..6454d54 100644 --- a/bpf/wire/wire.bpf.c +++ b/bpf/wire/wire.bpf.c @@ -208,9 +208,20 @@ static __always_inline void emit(struct __sk_buff *skb, struct pkt *p, __u8 hook __u32 cap = len > off ? len - off : 0; if (cap > CAP_MASK) cap = CAP_MASK; + if (cap == 0) { + e->cap_len = 0; + bpf_ringbuf_submit(e, 0); + return; + } + /* bpf_skb_load_bytes refuses a size that may be zero, and a + * verifier before 6.9 does not narrow a register on the + * `!= 0` branch above (6.6 rejects: "R4 invalid zero-sized + * read: u64=[0,4094]"). Rebuild the bound by arithmetic it + * does track: [1, CAP_MASK + 1], an identity on [1, CAP_MASK]. + * The barrier keeps clang from folding it away. */ barrier_var(cap); - cap &= CAP_MASK; - if (cap && bpf_skb_load_bytes(skb, p->data_off + off, e->data, cap)) + cap = ((cap - 1) & CAP_MASK) + 1; + if (bpf_skb_load_bytes(skb, p->data_off + off, e->data, cap)) cap = 0; e->cap_len = cap; bpf_ringbuf_submit(e, 0); diff --git a/build/bpf.mk b/build/bpf.mk index d03e5d7..0dd268c 100644 --- a/build/bpf.mk +++ b/build/bpf.mk @@ -101,7 +101,7 @@ veristat: bpf | toolchain # quay.io/lvh-images/kind images with cilium's lvh + QEMU; pass kernels as # KERNELS="6.6-main bpf-next-main" or rely on the script's default spread. .PHONY: veristat-matrix -veristat-matrix: $(BPF_OUT) | toolchain +veristat-matrix: bpf | toolchain VERISTAT="$(VERISTAT)" sh build/kernel-matrix.sh $(KERNELS) # Write a local .clangd so the editor resolves vmlinux.h, the libbpf SDK diff --git a/build/kernel-matrix.sh b/build/kernel-matrix.sh new file mode 100755 index 0000000..58d2511 --- /dev/null +++ b/build/kernel-matrix.sh @@ -0,0 +1,172 @@ +#!/bin/sh +# Boot a matrix of kernels locally and run the verifier check (build/verify- +# kernel.sh) in each — the local counterpart to .github/workflows/kernel- +# matrix.yml. This is a TEST HARNESS, not part of the build, and needs a +# Linux host (ideally with /dev/kvm; without it QEMU falls back to slow TCG). +# +# build/kernel-matrix.sh [kernel ...] # default: 6.6 (the TCX floor) up, + bpf-next +# make veristat-matrix # same, via the Makefile +# +# It uses cilium's lvh + QEMU to boot quay.io/lvh-images/kind: images. +# Neither lvh nor qemu is part of the build toolchain (they're VM infra needing +# host KVM/root, not self-contained build tools), so both are fetched on demand: +# - lvh: an `lvh` on PATH, else extracted from the quay.io/lvh-images/lvh image. +# - qemu: the vendored static qemu-.tar.gz from the toolchain release +# (checksum-pinned in build/toolchain.lock), extracted to the toolchain +# cache and prepended to PATH; falls back to a system qemu-system. +# veristat is the vendored static binary, resolved like the build. + +set -eu + +KERNELS=${*:-"6.6-main 6.12-main 6.18-main 7.2-main bpf-next-main"} +LVH_VERSION="${LVH_VERSION:-v0.0.30}" +SSH_PORT="${SSH_PORT:-2222}" +MON_PORT="${MON_PORT:-45454}" +# Objects to verify. httpscope links one per bpf// directory, so the +# default is every bin/*.bpf.o (resolved after `make bpf` below). +OBJS="${OBJS:-}" + +case "$(uname -s)" in + Linux) ;; + *) echo "error: the kernel matrix needs a Linux host (QEMU/KVM); on $(uname -s) use the CI workflow instead." >&2; exit 1 ;; +esac + +# ARCH = uname machine (toolchain asset naming); QARCH = lvh/qemu platform name. +ARCH="$(uname -m)" +case "$ARCH" in + x86_64) QARCH=amd64 ;; + aarch64) QARCH=arm64 ;; + *) echo "error: unsupported arch '$ARCH'" >&2; exit 1 ;; +esac +QEMU="qemu-system-${ARCH}" + +need() { command -v "$1" >/dev/null 2>&1 || { echo "error: '$1' not found — $2" >&2; exit 1; }; } +sha256() { if command -v sha256sum >/dev/null 2>&1; then sha256sum "$1" | awk '{print $1}'; else shasum -a 256 "$1" | awk '{print $1}'; fi; } +need ssh "install openssh-client" + +# --- resolve qemu: prefer the vendored static qemu from the toolchain release, +# fall back to a system qemu-system on PATH. lvh finds qemu via PATH and the +# static qemu finds its firmware blobs relative to its own binary, so we just +# prepend the extracted bin dir to PATH — no -L plumbing into lvh needed. +if [ -f build/toolchain.lock ]; then + . ./build/toolchain.lock + qsha="$(eval "printf '%s' \"\${QEMU_SHA256_${ARCH}:-}\"")" + QDIR="${XDG_CACHE_HOME:-$HOME/.cache}/yeet/toolchain/v${TOOLCHAIN_VERSION}/${ARCH}/qemu" + if [ ! -x "$QDIR/bin/$QEMU" ] && [ -n "$qsha" ] && [ -n "${TOOLCHAIN_BASE_URL:-}" ]; then + echo ">> fetching static qemu (${ARCH}, v${TOOLCHAIN_VERSION})" + tmp="$(mktemp -d)" + if curl -fSL --retry 3 -o "$tmp/q.tgz" "${TOOLCHAIN_BASE_URL}/v${TOOLCHAIN_VERSION}/qemu-${ARCH}.tar.gz"; then + got="$(sha256 "$tmp/q.tgz")" + if [ "$got" = "$qsha" ]; then + mkdir -p "$QDIR"; tar xzf "$tmp/q.tgz" -C "$QDIR"; chmod +x "$QDIR/bin/$QEMU" 2>/dev/null || true + else + echo "warning: qemu checksum mismatch (got $got, want $qsha); using system qemu" >&2 + fi + else + echo "warning: could not download qemu; using system qemu" >&2 + fi + rm -rf "$tmp" + fi + if [ -x "$QDIR/bin/$QEMU" ]; then + PATH="$QDIR/bin:$PATH"; export PATH + echo ">> using vendored static qemu: $QDIR/bin/$QEMU" + fi +fi +need "$QEMU" "no vendored qemu in build/toolchain.lock and none on PATH — bump the lock to a toolchain that ships qemu, or install qemu-system" + +# --- resolve lvh (PATH, else extract from the OCI image with docker) ---------- +LVH="$(command -v lvh || true)" +if [ -z "$LVH" ]; then + need docker "needed to fetch lvh (or put an 'lvh' binary on PATH)" + echo ">> fetching lvh ${LVH_VERSION} from quay.io/lvh-images/lvh" + docker pull "quay.io/lvh-images/lvh:${LVH_VERSION}" >/dev/null + cid="$(docker create "quay.io/lvh-images/lvh:${LVH_VERSION}")" + LVH="$(mktemp -d)/lvh" + docker cp "$cid:/usr/bin/lvh" "$LVH" >/dev/null + docker rm "$cid" >/dev/null + chmod +x "$LVH" +fi + +[ -e /dev/kvm ] && ACCEL="--cpu-kind host" || { ACCEL="--no-hw-accel"; echo "note: /dev/kvm absent — running under TCG emulation (slow)"; } + +# --- build the object and stage the vendored static veristat ------------------ +echo ">> building bin/*.bpf.o" +make bpf >/dev/null +[ -n "$OBJS" ] || OBJS="$(ls bin/*.bpf.o 2>/dev/null || true)" +[ -n "$OBJS" ] || { echo "error: no BPF objects in bin/ after 'make bpf'" >&2; exit 1; } +VERISTAT="${VERISTAT:-}" +if [ -z "$VERISTAT" ] && [ -f build/toolchain.lock ]; then + . ./build/toolchain.lock + VERISTAT="${XDG_CACHE_HOME:-$HOME/.cache}/yeet/toolchain/v${TOOLCHAIN_VERSION}/${ARCH}/veristat" +fi +[ -n "$VERISTAT" ] && [ -x "$VERISTAT" ] || VERISTAT="$(command -v veristat || true)" +[ -n "$VERISTAT" ] && [ -x "$VERISTAT" ] || { echo "error: veristat not found — bump build/toolchain.lock to a toolchain that ships it, or install veristat" >&2; exit 1; } +install -Dm755 "$VERISTAT" bin/veristat + +# --- per-kernel: pull image, boot, verify, stop ------------------------------- +WORK="$(mktemp -d)" +OUTDIR="$PWD/.kmatrix"; rm -rf "$OUTDIR"; mkdir -p "$OUTDIR" +overall=0 + +stop_vm() { printf 'quit\n' | { nc -N 127.0.0.1 "$MON_PORT" 2>/dev/null || nc 127.0.0.1 "$MON_PORT" 2>/dev/null; } || true; sleep 1; } +trap 'stop_vm' EXIT INT TERM + +for k in $KERNELS; do + echo ">> ==== kernel $k ====" + imgdir="$WORK/img/$k"; mkdir -p "$imgdir" + "$LVH" images pull "quay.io/lvh-images/kind:$k" --dir "$imgdir" --platform "linux/$QARCH" + img="$(find "$imgdir" -name '*.qcow2' | head -1)" + [ -n "$img" ] || { echo "::skip:: no image for $k" >&2; overall=1; continue; } + + sudo "$LVH" run --image "$img" --host-mount "$PWD" --daemonize \ + -p "${SSH_PORT}:22" --serial-port 0 --qemu-monitor-port "$MON_PORT" \ + --console-log-file "$WORK/console-$k.log" --qemu-arch "$QARCH" $ACCEL + + # wait for sshd + n=0; up=0 + while [ "$n" -lt 120 ]; do + if ssh -p "$SSH_PORT" -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \ + -o ConnectTimeout=2 root@127.0.0.1 true 2>/dev/null; then up=1; break; fi + n=$((n+1)); sleep 1 + done + if [ "$up" = 0 ]; then echo "::error:: $k VM never came up"; cat "$WORK/console-$k.log" 2>/dev/null | tail -20; overall=1; stop_vm; continue; fi + + # run the gate in the VM; CSV lands in the host-mounted .kmatrix via /host + if ssh -p "$SSH_PORT" -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null root@127.0.0.1 \ + "cd /host && OUT_CSV=/host/.kmatrix/$k.csv sh build/verify-kernel.sh $OBJS"; then :; else overall=1; fi + stop_vm +done +trap - EXIT INT TERM + +# --- render a terminal matrix from the per-kernel CSVs ------------------------ +echo +KERNELS="$KERNELS" python3 - "$OUTDIR" <<'PY' || echo "(install python3 for the summary table; per-kernel CSVs are in .kmatrix/)" +import csv, glob, os, sys +outdir = sys.argv[1] +kernels = os.environ["KERNELS"].split() +data, progs = {}, [] +for k in kernels: + f = os.path.join(outdir, f"{k}.csv") + if not os.path.exists(f): + data[k] = None; continue + # Key by object and program: peer_sendmsg/peer_recvmsg recur across the TLS taps. + data[k] = {f'{r["file_name"]}:{r["prog_name"]}': r["verdict"] for r in csv.DictReader(open(f))} + for p in data[k]: + if p not in progs: progs.append(p) +short = lambda k: k.replace("-main", "") +if not progs: + print("no results — check the per-kernel output above"); raise SystemExit(0) +w = max(len(p) for p in progs) + 2 +cols = [short(k) for k in kernels] +print(" " * w + " ".join(f"{c:>9}" for c in cols)) +cell = {"success": " ok ", "failure": " FAIL "} +for p in progs: + row = [p.ljust(w)] + for k in kernels: + d = data[k] + row.append(" - " if d is None or p not in d else cell.get(d[p], " ? ")) + print(" ".join(row)) +PY + +[ "$overall" = 0 ] && echo ">> matrix: all programs loaded on every kernel" || echo ">> matrix: at least one kernel rejected a program (or failed to boot)" +exit "$overall" diff --git a/build/verify-kernel.sh b/build/verify-kernel.sh new file mode 100755 index 0000000..fa741bc --- /dev/null +++ b/build/verify-kernel.sh @@ -0,0 +1,79 @@ +#!/bin/sh +# CI helper — runs INSIDE a per-kernel VM. Loads every built BPF object with +# the vendored static veristat and fails if the running kernel's verifier +# rejects any program. Driven by .github/workflows/kernel-matrix.yml, which +# boots each kernel with cilium's little-vm-helper and mounts the project at +# /host; the workflow stages the static veristat into bin/ before booting. +# +# sh build/verify-kernel.sh [bpf-object ...] (default: bin/*.bpf.o) +# +# httpscope links one object per bpf// directory (socket, wire, walk, +# ssl, ssl_ex, gotls, gotls_read, rustls), so the default is the whole set: +# veristat takes several objects in one run and names the file in each row. +# +# Set OUT_CSV= to also write a machine-readable result (file,prog,verdict, +# insns,states) — the workflow points it at the mounted workspace so the runner +# can render a summary table from it after the VM exits. +# +# Why parse output instead of trusting the exit code: veristat returns 0 even +# when a program fails to load — a rejected program shows up as a VERDICT of +# "failure" in its table, not as a non-zero status. So the gate reads the verdict +# column. (veristat only exits non-zero on infra errors: missing file, OOM, etc.) + +set -eu + +if [ $# -gt 0 ]; then + OBJS="$*" +else + OBJS="$(ls bin/*.bpf.o 2>/dev/null || true)" +fi +VERISTAT="${VERISTAT:-./bin/veristat}" +# verdict LAST so the gate below can match it at end-of-line. veristat's CSV +# header uses each stat's canonical name, so the columns come out as +# file_name,prog_name,total_insns,total_states,verdict. +COLS="file,prog,insns,states,verdict" + +[ -x "$VERISTAT" ] || { echo "error: veristat not found/executable at $VERISTAT" >&2; exit 1; } +[ -n "$OBJS" ] || { echo "error: no BPF objects found (bin/*.bpf.o) — run 'make bpf' first" >&2; exit 1; } +for o in $OBJS; do + [ -f "$o" ] || { echo "error: BPF object not found at $o" >&2; exit 1; } +done + +KREL="$(uname -r)" +echo ">> kernel $KREL: loading $OBJS" + +# Human-readable table for the console log (full default columns). +# shellcheck disable=SC2086 +"$VERISTAT" $OBJS || true + +# Machine-readable pass: the verdict column is the gate; the rest feeds the +# workflow's summary table. +# shellcheck disable=SC2086 +csv="$("$VERISTAT" -o csv -e "$COLS" $OBJS)" +if [ -n "${OUT_CSV:-}" ]; then + mkdir -p "$(dirname "$OUT_CSV")" + printf '%s\n' "$csv" > "$OUT_CSV" +fi + +# Drop the header row; fail if any program's verdict is not "success". +# Before failing, reload just the rejected objects in verbose mode so the +# verifier's own explanation lands in the CI log — the first pass runs at +# log level 0 and only reports the verdict. The log is rotated to its tail +# (that is where the rejection reason is), bounded so a long program's +# trace does not swamp the job output. +failed="$(printf '%s\n' "$csv" | tail -n +2 | grep ',failure$' | cut -d, -f1 | sort -u || true)" +if [ -n "$failed" ]; then + objs="" + for f in $failed; do + for o in $OBJS; do + [ "$(basename "$o")" = "$f" ] && objs="$objs $o" + done + done + echo ">> verifier log for the rejected object(s):$objs" + # shellcheck disable=SC2086 + "$VERISTAT" -v -l1 --log-size=65536 $objs 2>&1 | grep -v '^\s*$' | tail -n 200 || true + echo "::error::BPF verifier rejected a program on kernel $KREL" >&2 + exit 1 +fi + +echo ">> all programs loaded on kernel $KREL"