From f32b964f3c281efce00126730e8a24e5bb96da36 Mon Sep 17 00:00:00 2001 From: diego Date: Thu, 17 Sep 2026 13:02:59 -0300 Subject: [PATCH 1/3] build: publish the shim and agent images on a tag, on both arches (#139) The images are built and gated in CI and nothing has ever pushed them, so both Kubernetes examples name images that do not exist. This adds the publish, and fixes the reason it could only ever have worked on one architecture. Dockerfile.shim hardcoded x86_64 twice. The include path becomes a glob over targets/*-linux, which needs no branching at all; the CUDA repo URL genuinely must branch, and picks sbsa on aarch64 -- the server-class ARM tree, not the Jetson one, because the target is a GH200 or Graviton node. An unknown machine now fails with a message rather than a 404 from wget. Verified on amd64 by building the image and running the repo's own gate against the shim it produced: exactly one export, no libstdc++, no RUNPATH, 14 USDT notes, glibc 2.34, "-> portable". The arm64 build cannot be verified here, which is why the CI container job is now matrixed over both native runners: the arm64 path is exercised on every PR instead of being discovered broken during a release. Both images carry the release tag, and that is load-bearing rather than tidy. The shim's USDT record layouts are frozen per version and the agent decodes them, so a shim from one release paired with an agent from another is a decode failure with nothing in the pod spec to reveal it. Stated in both the workflow and the README that this is a convention: nothing at runtime refuses a mismatched pair today. Nothing publishes until a v* tag is pushed. The example manifests now say so rather than implying the images are there. Claude-Session: https://claude.ai/code/session_01P5889hA6CrX8ysnQkvv6im --- .github/workflows/ci.yml | 12 ++- .github/workflows/release.yml | 80 +++++++++++++++++++ Dockerfile.shim | 10 ++- examples/kubernetes/README.md | 17 ++-- .../kubernetes/gpu-collector-daemonset.yaml | 5 +- 5 files changed, 113 insertions(+), 11 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 82d6aae..a7d4013 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -231,8 +231,16 @@ jobs: # Integration tests require root and are harder to run in CI # They can be run manually or in a different environment images: - name: Container images - runs-on: ubuntu-24.04 + name: Container images (${{ matrix.arch }}) + runs-on: ${{ matrix.runner }} + strategy: + fail-fast: false + matrix: + include: + - arch: amd64 + runner: ubuntu-24.04 + - arch: arm64 + runner: ubuntu-24.04-arm steps: - name: Checkout code uses: actions/checkout@v4 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 3821948..721087d 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -165,6 +165,86 @@ jobs: path: gpu-cuda-profile-linux-${{ matrix.goarch }} retention-days: 30 + # Both images carry the SAME tag as the binaries, and that is the point + # rather than tidiness: the shim's USDT record layouts are frozen per + # version and the consumer decodes them, so a shim from one release paired + # with an agent from another is a decode failure with nothing in the pod + # spec to reveal it. One tag, both images, or the pairing is guesswork. + # + # Note this is a convention, not an enforced guarantee -- nothing at + # runtime refuses a mismatched pair today. Issue #139. + images: + name: Publish images (${{ matrix.arch }}) + runs-on: ${{ matrix.runner }} + permissions: + contents: read + packages: write + strategy: + fail-fast: false + matrix: + include: + - arch: amd64 + runner: ubuntu-24.04 + - arch: arm64 + runner: ubuntu-24.04-arm + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Log in to ghcr.io + uses: docker/login-action@v3 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + # Built natively per architecture rather than under QEMU. The shim + # build installs the CUDA toolkit and compiles C++; emulating that is + # slow and is not the thing we want to have tested. + - name: Build and push the shim image + run: | + img=ghcr.io/${{ github.repository_owner }}/perf-agent-shim + docker build -f Dockerfile.shim \ + -t "$img:${{ github.ref_name }}-${{ matrix.arch }}" . + docker push "$img:${{ github.ref_name }}-${{ matrix.arch }}" + + - name: Build and push the agent image + run: | + img=ghcr.io/${{ github.repository_owner }}/perf-agent + docker build -f Dockerfile.agent \ + -t "$img:${{ github.ref_name }}-${{ matrix.arch }}" . + docker push "$img:${{ github.ref_name }}-${{ matrix.arch }}" + + # One manifest list per image so `:` and `:latest` resolve on either + # architecture. Separate job because it can only run once BOTH arches have + # pushed -- a manifest naming a tag that does not exist yet fails at create. + image-manifests: + name: Publish image manifests + needs: images + runs-on: ubuntu-24.04 + permissions: + contents: read + packages: write + steps: + - name: Log in to ghcr.io + uses: docker/login-action@v3 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + - name: Assemble multi-arch manifests + run: | + for name in perf-agent perf-agent-shim; do + img=ghcr.io/${{ github.repository_owner }}/$name + for tag in "${{ github.ref_name }}" latest; do + docker manifest create "$img:$tag" \ + --amend "$img:${{ github.ref_name }}-amd64" \ + --amend "$img:${{ github.ref_name }}-arm64" + docker manifest push "$img:$tag" + done + done + release: name: Create Release needs: build diff --git a/Dockerfile.shim b/Dockerfile.shim index ecb755c..a145221 100644 --- a/Dockerfile.shim +++ b/Dockerfile.shim @@ -28,8 +28,13 @@ ARG DEBIAN_FRONTEND=noninteractive # InitializeInjection and reports its absence instead of dying. RUN apt-get update -qq \ && apt-get install -y -qq --no-install-recommends g++ ca-certificates wget \ + && case "$(uname -m)" in \ + x86_64) CUDA_REPO_ARCH=x86_64 ;; \ + aarch64) CUDA_REPO_ARCH=sbsa ;; \ + *) echo "unsupported architecture $(uname -m) for the CUDA toolkit repo" >&2; exit 1 ;; \ + esac \ && wget -qO /tmp/keyring.deb \ - https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb \ + "https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/${CUDA_REPO_ARCH}/cuda-keyring_1.1-1_all.deb" \ && dpkg -i /tmp/keyring.deb \ && apt-get update -qq \ && apt-get install -y -qq --no-install-recommends \ @@ -45,7 +50,8 @@ COPY shim/ /src/shim/ # out rather than invoking make: this image must not depend on the host's # CUDA_HOME, which that target bind-mounts. RUN cd /src/shim \ - && CUDA_INC=$(ls -d /usr/local/cuda-*/targets/x86_64-linux/include | head -1) \ + && CUDA_INC=$(ls -d /usr/local/cuda-*/targets/*-linux/include | head -1) \ + && test -n "$CUDA_INC" || { echo "no CUDA include dir under /usr/local/cuda-*/targets/" >&2; exit 1; } \ && g++ -std=c++17 -O2 -Wall -Wextra -fPIC -fvisibility=hidden \ -mtls-dialect=gnu -pthread -I core -I "$CUDA_INC" \ -shared -o /libperfagent-gpu-nvidia.so \ diff --git a/examples/kubernetes/README.md b/examples/kubernetes/README.md index ad89ec7..c9ba254 100644 --- a/examples/kubernetes/README.md +++ b/examples/kubernetes/README.md @@ -5,16 +5,23 @@ Profile a PyTorch CUDA workload running in Kubernetes, without rebuilding it, relinking it, or changing a line of its code — then render the result as a flame graph. -> **Status: one blocker left — the shim image is not published.** +> **Status: runnable from the first release tag onward.** > > Every other directory under `examples/` runs end-to-end today. This one is > close, and the README says where it stops rather than letting you discover > it at `kubectl apply`. > -> **Still blocking: `ghcr.io/dpsoft/perf-agent-shim` does not exist** (#139). -> `Dockerfile.shim` builds it and the ELF gate runs inside that build, but -> nothing pushes it, so the `shim-installer` init container has no image to -> pull. Registry is decided; trigger, tagging and arch are not. +> **`ghcr.io/dpsoft/perf-agent-shim` does not exist until a `v*` tag is +> pushed** (#139). `release.yml` now builds both images natively on each +> architecture, pushes them tagged with the release version, and assembles +> multi-arch manifests for `:` and `:latest`. Nothing is published +> before then, so `kubectl apply` on these manifests will fail to pull until +> the first tag is cut. +> +> Both images carry the same tag deliberately: the shim's USDT record layouts +> are frozen per version and the agent decodes them, so a mismatched pair is a +> decode failure with nothing in the pod spec to reveal it. Note that this is a +> convention — nothing at runtime refuses a mismatched pair today. > > **Cleared since this was written:** > diff --git a/examples/kubernetes/gpu-collector-daemonset.yaml b/examples/kubernetes/gpu-collector-daemonset.yaml index fab207b..c54f7d4 100644 --- a/examples/kubernetes/gpu-collector-daemonset.yaml +++ b/examples/kubernetes/gpu-collector-daemonset.yaml @@ -1,8 +1,9 @@ # One agent per NODE. Contrast pytorch-gpu-profile.yaml, which puts one agent # in every profiled POD. # -# Status: the agent side of this is implemented (-mode=collector); the image -# below is not published yet, same blocker as the sidecar example (#139). +# Status: implemented (-mode=collector), and release.yml now publishes both +# images on a v* tag. Until a tag is cut, ghcr.io/dpsoft/perf-agent:latest +# does not exist yet and this will not pull (#139). # # ---------------------------------------------------------------- the idea -- # From 07f70b366f5ffe0e6e310b9393f62eb98c8bd202 Mon Sep 17 00:00:00 2001 From: diego Date: Thu, 17 Sep 2026 13:57:34 -0300 Subject: [PATCH 2/3] build: -mtls-dialect is spelled differently on aarch64 The arm64 image build failed the first time it ever ran: g++: error: unrecognized argument in option '-mtls-dialect=gnu' g++: note: valid arguments to '-mtls-dialect=' are: desc trad Not where I expected. The sbsa CUDA repo resolved, the packages installed and the include glob found its directory -- the compile itself rejected an x86-only flag. That flag selects TRADITIONAL TLS over TLS-descriptors, which is what keeps GLIBC_ABI_GNU2_TLS out of the object (#121). Both architectures have the choice; only the spelling differs, and the values are not interchangeable: x86-64 calls them gnu/gnu2, aarch64 trad/desc. So the fix is per-arch selection rather than dropping the flag, which would have reintroduced the marker that made the shim unloadable in the first place. Verified with a --no-cache amd64 build: the repo's own elfgate reports GLIBC_ABI_GNU2_TLS absent, glibc 2.34, one export, no libstdc++, no runpath, 14 USDT notes, portable. An earlier "successful" rebuild here had silently reused a cached layer, which is why this one is uncached. The explanation sits above the RUN rather than inside its line continuation, where a # is a parser footgun. Claude-Session: https://claude.ai/code/session_01P5889hA6CrX8ysnQkvv6im --- Dockerfile.shim | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/Dockerfile.shim b/Dockerfile.shim index a145221..89d8143 100644 --- a/Dockerfile.shim +++ b/Dockerfile.shim @@ -49,11 +49,22 @@ COPY shim/ /src/shim/ # The same command line as `make -C shim nvidia-portable`, deliberately spelled # out rather than invoking make: this image must not depend on the host's # CUDA_HOME, which that target bind-mounts. +# -mtls-dialect selects TRADITIONAL TLS over TLS-descriptors, which is what +# keeps GLIBC_ABI_GNU2_TLS out of the object (#121). The spelling is +# per-architecture and the values are NOT interchangeable: x86-64 calls the +# two dialects gnu/gnu2, aarch64 calls them trad/desc. Passing the x86 value +# on aarch64 is not a silently-ignored no-op -- g++ rejects it outright, which +# is how the arm64 image build failed the first time it ever ran. RUN cd /src/shim \ && CUDA_INC=$(ls -d /usr/local/cuda-*/targets/*-linux/include | head -1) \ && test -n "$CUDA_INC" || { echo "no CUDA include dir under /usr/local/cuda-*/targets/" >&2; exit 1; } \ + && case "$(uname -m)" in \ + x86_64) TLS_DIALECT=gnu ;; \ + aarch64) TLS_DIALECT=trad ;; \ + *) echo "no known traditional -mtls-dialect value for $(uname -m)" >&2; exit 1 ;; \ + esac \ && g++ -std=c++17 -O2 -Wall -Wextra -fPIC -fvisibility=hidden \ - -mtls-dialect=gnu -pthread -I core -I "$CUDA_INC" \ + -mtls-dialect=${TLS_DIALECT} -pthread -I core -I "$CUDA_INC" \ -shared -o /libperfagent-gpu-nvidia.so \ nvidia/cupti_adapter.cc core/batch.cc core/clock.cc core/cubin.cc \ core/cubinqueue.cc core/drain.cc core/enroll.cc core/sampler.cc \ From 8b848402ff7bf4791ee8e07d4c56fba74beb6152 Mon Sep 17 00:00:00 2001 From: diego Date: Thu, 17 Sep 2026 14:33:12 -0300 Subject: [PATCH 3/3] build: publish the shim amd64-only; the agent goes multi-arch The arm64 image build surfaced a source portability gap rather than a packaging one. shim/core/usdt_probe.h binds its probe arguments to rdi/rsi/rdx by name, with no architecture guards, so the shim does not compile on aarch64 at all. probe_args_test.cc already knew -- it skips with "not x86-64" -- but nothing built the shim there to find out. So the shim is built and published for amd64 only, and that is deliberate rather than a gap left open. CUDA injection fails open and silent, so an arm64 shim that compiled with wrong register bindings would yield an empty profile and no error anywhere. Not shipping one is safer than shipping one nobody can validate. Dockerfile.agent had its own hardcoding on the same theme: it downloaded go...linux-amd64.tar.gz unconditionally, so the agent image could only ever have been built on one architecture either. Now per-arch, and the agent publishes a real multi-arch manifest while the shim gets single-arch tags rather than a manifest list naming an image that was never built. Verified by rebuilding both images here: the shim passes elfgate (portable, GLIBC_ABI_GNU2_TLS absent, one export, 14 USDT notes) and the agent still refuses without capabilities, naming which, with file caps on the binary. The port itself is next, and is verifiable without a GPU: probe_args_test traps its own probe and reads the argument registers, so teaching it aarch64 proves the binding on a native arm64 runner. Claude-Session: https://claude.ai/code/session_01P5889hA6CrX8ysnQkvv6im --- .github/workflows/ci.yml | 9 +++++++++ .github/workflows/release.yml | 30 ++++++++++++++++++++++-------- Dockerfile.agent | 10 +++++++++- examples/kubernetes/README.md | 11 +++++++++++ 4 files changed, 51 insertions(+), 9 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a7d4013..f653cd1 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -250,10 +250,19 @@ jobs: # kills file capabilities -- and both are caught by gates inside the # Dockerfiles. A workflow that only ran hadolint would pass while # shipping neither. + # amd64 only: shim/core/usdt_probe.h binds the probe arguments to + # rdi/rsi/rdx by name, so the shim does not compile on aarch64 at all. + # That is a source portability gap, not a packaging one -- the arm64 + # image build is what surfaced it. Tracked for the port; until then an + # arm64 shim must not be built OR published, because CUDA injection + # fails open and silent and an untested one would produce empty + # profiles with no error anywhere. - name: Build the shim image + if: matrix.arch == 'amd64' run: docker build -f Dockerfile.shim -t perf-agent-shim:ci . - name: The shim image carries a loadable shim + if: matrix.arch == 'amd64' # elfgate already ran inside the build; this checks the FILE made it # into the final stage, which a bad COPY would silently skip. run: | diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 721087d..a5b6133 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -201,7 +201,13 @@ jobs: # Built natively per architecture rather than under QEMU. The shim # build installs the CUDA toolkit and compiles C++; emulating that is # slow and is not the thing we want to have tested. + # amd64 only, and deliberately: the shim binds its USDT probe arguments + # to x86-64 registers by name and does not compile on aarch64. Shipping + # an arm64 shim before the port would be worse than shipping none -- + # CUDA injection fails open and silent, so a wrong one yields an empty + # profile and no error. - name: Build and push the shim image + if: matrix.arch == 'amd64' run: | img=ghcr.io/${{ github.repository_owner }}/perf-agent-shim docker build -f Dockerfile.shim \ @@ -235,14 +241,22 @@ jobs: - name: Assemble multi-arch manifests run: | - for name in perf-agent perf-agent-shim; do - img=ghcr.io/${{ github.repository_owner }}/$name - for tag in "${{ github.ref_name }}" latest; do - docker manifest create "$img:$tag" \ - --amend "$img:${{ github.ref_name }}-amd64" \ - --amend "$img:${{ github.ref_name }}-arm64" - docker manifest push "$img:$tag" - done + # The agent is multi-arch. The shim is amd64 only until its USDT + # probes are ported, so it gets single-arch tags rather than a + # manifest list naming an arm64 image that was never built. + agent=ghcr.io/${{ github.repository_owner }}/perf-agent + for tag in "${{ github.ref_name }}" latest; do + docker manifest create "$agent:$tag" \ + --amend "$agent:${{ github.ref_name }}-amd64" \ + --amend "$agent:${{ github.ref_name }}-arm64" + docker manifest push "$agent:$tag" + done + + shim=ghcr.io/${{ github.repository_owner }}/perf-agent-shim + docker pull "$shim:${{ github.ref_name }}-amd64" + for tag in "${{ github.ref_name }}" latest; do + docker tag "$shim:${{ github.ref_name }}-amd64" "$shim:$tag" + docker push "$shim:$tag" done release: diff --git a/Dockerfile.agent b/Dockerfile.agent index 55998fe..6374777 100644 --- a/Dockerfile.agent +++ b/Dockerfile.agent @@ -17,7 +17,15 @@ RUN apt-get update -qq \ ca-certificates wget git build-essential libelf-dev zlib1g-dev \ && rm -rf /var/lib/apt/lists/* -RUN wget -qO /tmp/go.tgz "https://go.dev/dl/go${GO_VERSION}.linux-amd64.tar.gz" \ +# The Go tarball is per-architecture and the names differ from uname's: +# x86_64 -> amd64, aarch64 -> arm64. Hardcoding amd64 here is why this image +# could only ever have been built on one architecture. +RUN case "$(uname -m)" in \ + x86_64) GO_ARCH=amd64 ;; \ + aarch64) GO_ARCH=arm64 ;; \ + *) echo "no Go toolchain mapping for $(uname -m)" >&2; exit 1 ;; \ + esac \ + && wget -qO /tmp/go.tgz "https://go.dev/dl/go${GO_VERSION}.linux-${GO_ARCH}.tar.gz" \ && tar -C /usr/local -xzf /tmp/go.tgz && rm /tmp/go.tgz ENV PATH=/usr/local/go/bin:/root/.cargo/bin:$PATH diff --git a/examples/kubernetes/README.md b/examples/kubernetes/README.md index c9ba254..c1123ce 100644 --- a/examples/kubernetes/README.md +++ b/examples/kubernetes/README.md @@ -18,6 +18,17 @@ flame graph. > before then, so `kubectl apply` on these manifests will fail to pull until > the first tag is cut. > +> **The shim is published for amd64 only.** `shim/core/usdt_probe.h` binds its +> probe arguments to `rdi`/`rsi`/`rdx` by name, so it does not compile on +> aarch64 — a source portability gap, not a packaging one. GPU profiling is +> therefore x86-64 only today, on a Grace-Hopper or other ARM GPU node +> included. The agent image itself is multi-arch. +> +> This matters more than an unsupported-platform note usually would: CUDA +> injection **fails open and silent**, so an arm64 shim that compiled but bound +> the wrong registers would produce an empty profile and no error anywhere. +> Not shipping one is deliberate. +> > Both images carry the same tag deliberately: the shim's USDT record layouts > are frozen per version and the agent decodes them, so a mismatched pair is a > decode failure with nothing in the pod spec to reveal it. Note that this is a