diff --git a/runsc/boot/vfs.go b/runsc/boot/vfs.go index 34275e5de..489231653 100644 --- a/runsc/boot/vfs.go +++ b/runsc/boot/vfs.go @@ -1314,9 +1314,9 @@ func createDeviceFiles(ctx context.Context, creds *auth.Credentials, info *conta } } } - if info.conf.NVProxyDocker && specutils.GPUFunctionalityRequested(info.spec, info.conf) { - // In Docker mode, devices are not injected into spec.Linux.Devices. So - // manually create appropriate device files. + if specutils.GPUFunctionalityRequestedViaHook(info.spec, info.conf) { + // When using nvidia-container-runtime-hook, devices are not injected into + // spec.Linux.Devices. So manually create appropriate device files. mode := os.FileMode(0666) nvidiaDevs := []specs.LinuxDevice{ specs.LinuxDevice{Path: "/dev/nvidiactl", Type: "c", Major: nvgpu.NV_MAJOR_DEVICE_NUMBER, Minor: nvgpu.NV_CONTROL_DEVICE_MINOR, FileMode: &mode}, diff --git a/runsc/config/config.go b/runsc/config/config.go index dd1fd86ef..822bcd93b 100644 --- a/runsc/config/config.go +++ b/runsc/config/config.go @@ -310,9 +310,9 @@ type Config struct { // NVProxy enables support for Nvidia GPUs. NVProxy bool `flag:"nvproxy"` - // NVProxyDocker exposes GPUs to containers based on the - // NVIDIA_VISIBLE_DEVICES container environment variable, as requested by - // containers or set by `docker --gpus`. + // NVProxyDocker is deprecated. Please use nvidia-container-runtime or + // `docker run --gpus` directly. For backward compatibility, this has the + // effect of injecting nvidia-container-runtime-hook as a prestart hook. NVProxyDocker bool `flag:"nvproxy-docker"` // TPUProxy enables support for TPUs. diff --git a/runsc/config/flags.go b/runsc/config/flags.go index 68117afc4..fda4c44bc 100644 --- a/runsc/config/flags.go +++ b/runsc/config/flags.go @@ -127,7 +127,7 @@ func RegisterFlags(flagSet *flag.FlagSet) { // Flags that control sandbox runtime behavior: accelerator related. flagSet.Bool("nvproxy", false, "EXPERIMENTAL: enable support for Nvidia GPUs") - flagSet.Bool("nvproxy-docker", false, "Expose GPUs to containers based on NVIDIA_VISIBLE_DEVICES, as requested by the container or set by `docker --gpus`. Allows containers to self-serve GPU access and thus disabled by default for security. libnvidia-container must be installed on the host. No effect unless --nvproxy is enabled.") + flagSet.Bool("nvproxy-docker", false, "DEPRECATED: use nvidia-container-runtime or `docker run --gpus` directly. Or manually add nvidia-container-runtime-hook as a prestart hook and set up NVIDIA_VISIBLE_DEVICES container environment variable.") flagSet.Bool("tpuproxy", false, "EXPERIMENTAL: enable support for TPU device passthrough.") // Test flags, not to be used outside tests, ever. diff --git a/runsc/container/container.go b/runsc/container/container.go index 043087436..db8c8c8bb 100644 --- a/runsc/container/container.go +++ b/runsc/container/container.go @@ -297,8 +297,11 @@ func New(conf *config.Config, args Args) (*Container, error) { if err != nil { return nil, err } - if !goferConfs[0].ShouldUseLisafs() && conf.NVProxyDocker { - return nil, fmt.Errorf("--nvproxy-docker cannot be used together with non-lisafs backed root mount") + if !goferConfs[0].ShouldUseLisafs() && specutils.GPUFunctionalityRequestedViaHook(args.Spec, conf) { + // nvidia-container-runtime-hook attempts to populate the container + // rootfs with NVIDIA libraries and devices. With EROFS, spec.Root.Path + // points to an empty directory and populating that has no effect. + return nil, fmt.Errorf("nvidia-container-runtime-hook cannot be used together with non-lisafs backed root mount") } c.GoferMountConfs = goferConfs if err := nvProxyPreGoferHostSetup(args.Spec, conf); err != nil { @@ -1820,14 +1823,14 @@ func logIDMappings(mappings []specs.LinuxIDMapping, idType string) { } } -// nvProxyPreGoferHostSetup sets up nvproxy on the host. It runs before any -// Gofers start. +// nvProxyPreGoferHostSetup does host setup work so that `nvidia-container-cli +// configure` can be run in the future. It runs before any Gofers start. // It verifies that all the required dependencies are in place, loads kernel // modules, and ensures the correct device files exist and are accessible. // This should only be necessary once on the host. It should be run during the // root container setup sequence to make sure it has run at least once. func nvProxyPreGoferHostSetup(spec *specs.Spec, conf *config.Config) error { - if !conf.NVProxyDocker || !specutils.GPUFunctionalityRequested(spec, conf) { + if !specutils.GPUFunctionalityRequestedViaHook(spec, conf) { return nil } @@ -1924,7 +1927,7 @@ func nvproxyLoadKernelModules() { // construction. For this reason, we don't need to parse // NVIDIA_VISIBLE_DEVICES or pass --device to nvidia-container-cli. func nvproxySetupAfterGoferUserns(spec *specs.Spec, conf *config.Config, goferCmd *exec.Cmd, goferDonations *donation.Agency) (func() error, error) { - if !conf.NVProxyDocker || !specutils.GPUFunctionalityRequested(spec, conf) { + if !specutils.GPUFunctionalityRequestedViaHook(spec, conf) { return func() error { return nil }, nil } diff --git a/runsc/specutils/nvidia.go b/runsc/specutils/nvidia.go index 6dba7bf9f..7372a945d 100644 --- a/runsc/specutils/nvidia.go +++ b/runsc/specutils/nvidia.go @@ -52,6 +52,8 @@ func GPUFunctionalityRequested(spec *specs.Spec, conf *config.Config) bool { // nvproxy disabled. return false } + // In GKE, the nvidia_gpu device plugin injects NVIDIA devices into + // spec.Linux.Devices when GPUs are allocated to a container. if spec.Linux != nil { for _, dev := range spec.Linux.Devices { if dev.Path == "/dev/nvidiactl" { @@ -59,7 +61,25 @@ func GPUFunctionalityRequested(spec *specs.Spec, conf *config.Config) bool { } } } - if !conf.NVProxyDocker { + return gpuFunctionalityRequestedViaHook(spec, conf) +} + +// GPUFunctionalityRequestedViaHook returns true if the container should have +// access to GPU functionality configured via nvidia-container-runtime-hook. +// This hook is used by: +// - Docker when using `--gpus` flag from the CLI. +// - nvidia-container-runtime when using its legacy mode. +func GPUFunctionalityRequestedViaHook(spec *specs.Spec, conf *config.Config) bool { + if !NVProxyEnabled(spec, conf) { + // nvproxy disabled. + return false + } + return gpuFunctionalityRequestedViaHook(spec, conf) +} + +// Precondition: NVProxyEnabled(spec, conf). +func gpuFunctionalityRequestedViaHook(spec *specs.Spec, conf *config.Config) bool { + if !isNvidiaHookPresent(spec, conf) { return false } // In Docker mode, GPU access is only requested if NVIDIA_VISIBLE_DEVICES is @@ -73,6 +93,19 @@ func GPUFunctionalityRequested(spec *specs.Spec, conf *config.Config) bool { return nvd != "" && nvd != "void" } +func isNvidiaHookPresent(spec *specs.Spec, conf *config.Config) bool { + if conf.NVProxyDocker { + // This has the effect of injecting the nvidia-container-runtime-hook. + return true + } + for _, h := range spec.Hooks.Prestart { + if strings.HasSuffix(h.Path, "/nvidia-container-runtime-hook") { + return true + } + } + return false +} + // ParseNvidiaVisibleDevices parses NVIDIA_VISIBLE_DEVICES env var and returns // the devices specified in it. This can be passed to nvidia-container-cli. //