Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
15 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
53 changes: 53 additions & 0 deletions .github/workflows/nvsnap-gpushare.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
# SPDX-FileCopyrightText: Copyright (c) NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Compile nvsnap's gpushare C code (libnvsnap_gpushare.so, nvsnap-gpu-suspend
# and its GPU tests) for amd64 and arm64 with warnings as errors. The Bazel row
# for nvsnap covers its Go code only, so without this job a change to the C
# code is first compiled when the agent base image is built. Runs only when the
# C code, its build or this workflow changes; no GPU is needed to compile.

name: nvsnap gpushare

on:
push:
branches: [main, 'release-**']
paths: &paths
- 'src/compute-plane-services/nvsnap/docker/agent/gpushare/**'
- 'src/compute-plane-services/nvsnap/tests/gpushare/**'
- 'src/compute-plane-services/nvsnap/docker/agent/Dockerfile.base'
- 'src/compute-plane-services/nvsnap/scripts/check-gpushare-build.sh'
- '.github/workflows/nvsnap-gpushare.yml'
pull_request:
branches: [main, 'release-**']
paths: *paths
workflow_dispatch:

permissions:
contents: read

concurrency:
group: nvsnap-gpushare-${{ github.ref }}
cancel-in-progress: true

jobs:
build:
name: build gpushare (${{ matrix.platform }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
platform: [linux/amd64, linux/arm64]
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false

- name: Set up QEMU
if: matrix.platform == 'linux/arm64'
uses: docker/setup-qemu-action@c7c53464625b32c7a7e944ae62b3e17d2b600130 # v3.7.0

- name: Compile with warnings as errors
env:
PLATFORMS: ${{ matrix.platform }}
run: src/compute-plane-services/nvsnap/scripts/check-gpushare-build.sh
6 changes: 5 additions & 1 deletion src/compute-plane-services/nvsnap/Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@ CONTAINER_TOOL ?= docker
GO_FLAGS ?= -ldflags="-s -w -X main.version=$(VERSION)"
BUILD_DIR ?= bin

.PHONY: all build agent nvsnap-server clean test test-coverage test-checkpoint test-server deploy install-crds generate
.PHONY: all build agent nvsnap-server clean test test-coverage test-checkpoint test-server deploy install-crds generate check-gpushare

all: build

Expand Down Expand Up @@ -70,6 +70,10 @@ test:

# Run all unit tests with coverage profile (used by CI).
# Coverage output lands in coverage/coverage.out for Codecov upload.
# Compile the gpushare C code for amd64 and arm64 (Docker, no GPU); same check as CI.
check-gpushare:
./scripts/check-gpushare-build.sh

test-coverage:
mkdir -p coverage
go test -v -race -coverprofile=coverage/coverage.out -covermode=atomic $(GO_TEST_PKGS)
Expand Down
13 changes: 13 additions & 0 deletions src/compute-plane-services/nvsnap/docker/agent/Dockerfile.base
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,16 @@ RUN gcc -O2 -o /src/nvsnap-cuda-checkpoint /src/nvsnap-cuda-checkpoint.c \
strings /src/nvsnap-cuda-checkpoint | grep -q "nvsnap-cuda-checkpoint" \
|| (echo "provenance marker missing" && exit 1)

# gpushare: libnvsnap_gpushare.so, an LD_PRELOAD shim that releases and
# re-maps GPU memory shared between a workload's processes (NCCL P2P and NVLS,
# CUDA IPC, page-locked host memory) around a CUDA checkpoint, and
# nvsnap-gpu-suspend, which drives it. Needs CUDA 13 headers
# (CUcheckpointGpuPair). Built on ubuntu 22.04: the shim is loaded into
# workload containers, which need glibc >= 2.35.
FROM nvidia/cuda:13.0.3-devel-ubuntu22.04 AS gpushare-builder
COPY gpushare /src/gpushare
RUN make -C /src/gpushare

FROM ubuntu:22.04 AS criu-builder

# Install build dependencies
Expand Down Expand Up @@ -165,6 +175,9 @@ COPY --from=cuda-cli-builder /src/nvsnap-cuda-checkpoint /criu-bundle/cuda-check
COPY cuda-checkpoint-wrapper.sh /criu-bundle/cuda-checkpoint
RUN chmod +x /criu-bundle/cuda-checkpoint

# gpushare (see gpushare-builder above).
COPY --from=gpushare-builder /src/gpushare/libnvsnap_gpushare.so /src/gpushare/nvsnap-gpu-suspend /criu-bundle/

# Make everything executable
RUN chmod +x /criu-bundle/*

Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# libnvsnap_gpushare.so: LD_PRELOAD shim that releases and re-maps GPU memory
# shared between processes around a CUDA checkpoint (see gpushare.h).
# nvsnap-gpu-suspend: drives it and the CUDA checkpoint API.
# Needs CUDA >= 13 headers (cuda.h: CUcheckpointGpuPair for --gpu-map).

CC ?= gcc
CUDA_INC ?= /usr/local/cuda/include
CFLAGS ?= -O2 -Wall -Wextra

all: libnvsnap_gpushare.so nvsnap-gpu-suspend

libnvsnap_gpushare.so: gpushare.c gpushare.h
$(CC) $(CFLAGS) -fPIC -shared -I. -I$(CUDA_INC) -o $@ gpushare.c -ldl -lpthread

nvsnap-gpu-suspend: nvsnap-gpu-suspend.c gpushare.h
$(CC) $(CFLAGS) -I. -I$(CUDA_INC) -o $@ nvsnap-gpu-suspend.c -ldl -lpthread

clean:
rm -f libnvsnap_gpushare.so nvsnap-gpu-suspend

.PHONY: all clean
Loading