diff --git a/pkg/sentry/inet/namespace.go b/pkg/sentry/inet/namespace.go index d29e1db80..4a7971739 100644 --- a/pkg/sentry/inet/namespace.go +++ b/pkg/sentry/inet/namespace.go @@ -129,6 +129,12 @@ func (n *Namespace) RestoreRootStack(stack Stack) { n.stack = stack } +// ResetStack resets the stack in the network namespace to nil. This should +// only be called when restoring kernel. +func (n *Namespace) ResetStack() { + n.stack = nil +} + func (n *Namespace) init() { // Root network namespace will have stack assigned later. if n.isRoot { diff --git a/pkg/sentry/kernel/BUILD b/pkg/sentry/kernel/BUILD index a2147fdd0..968be2809 100644 --- a/pkg/sentry/kernel/BUILD +++ b/pkg/sentry/kernel/BUILD @@ -112,6 +112,13 @@ declare_mutex( prefix = "threadGroupTimer", ) +declare_mutex( + name = "cgroup_mounts_mutex", + out = "cgroup_mounts_mutex.go", + package = "kernel", + prefix = "cgroupMounts", +) + go_template_instance( name = "pending_signals_list", out = "pending_signals_list.go", @@ -230,6 +237,7 @@ go_library( "atomicptr_bucket_unsafe.go", "atomicptr_descriptor_unsafe.go", "cgroup.go", + "cgroup_mounts_mutex.go", "cgroup_mutex.go", "context.go", "cpu_clock_mutex.go", diff --git a/pkg/sentry/kernel/kernel.go b/pkg/sentry/kernel/kernel.go index 83f670842..199c90df5 100644 --- a/pkg/sentry/kernel/kernel.go +++ b/pkg/sentry/kernel/kernel.go @@ -110,6 +110,16 @@ func (uc *UserCounters) decRLimitNProc() { uc.rlimitNProc.Add(^uint64(0)) } +// CgroupMount contains the cgroup mount. These mounts are created for the root +// container by default and are stored in the kernel. +// +// +stateify savable +type CgroupMount struct { + Fs *vfs.Filesystem + Root *vfs.Dentry + Mount *vfs.Mount +} + // Kernel represents an emulated Linux kernel. It must be initialized by calling // Init() or LoadFrom(). // @@ -321,6 +331,13 @@ type Kernel struct { // the system. cgroupRegistry *CgroupRegistry + // cgroupMountsMap maps the cgroup controller names to the cgroup mounts + // created for the root container. These mounts are then bind mounted + // for other application containers by creating their own container + // directories. + cgroupMountsMap map[string]*CgroupMount + cgroupMountsMapMu cgroupMountsMutex `state:"nosave"` + // userCountersMap maps auth.KUID into a set of user counters. userCountersMap map[auth.KUID]*UserCounters userCountersMapMu userCountersMutex `state:"nosave"` @@ -1735,12 +1752,46 @@ func (k *Kernel) CgroupRegistry() *CgroupRegistry { return k.cgroupRegistry } +// AddCgroupMount adds the cgroup mounts to the cgroupMountsMap. These cgroup +// mounts are created during the creation of root container process and the +// reference ownership is transferred to the kernel. +func (k *Kernel) AddCgroupMount(ctl string, mnt *CgroupMount) { + k.cgroupMountsMapMu.Lock() + defer k.cgroupMountsMapMu.Unlock() + + if k.cgroupMountsMap == nil { + k.cgroupMountsMap = make(map[string]*CgroupMount) + } + k.cgroupMountsMap[ctl] = mnt +} + +// GetCgroupMount returns the cgroup mount for the given cgroup controller. +func (k *Kernel) GetCgroupMount(ctl string) *CgroupMount { + k.cgroupMountsMapMu.Lock() + defer k.cgroupMountsMapMu.Unlock() + + return k.cgroupMountsMap[ctl] +} + +// releaseCgroupMounts releases the cgroup mounts. +func (k *Kernel) releaseCgroupMounts(ctx context.Context) { + k.cgroupMountsMapMu.Lock() + defer k.cgroupMountsMapMu.Unlock() + + for _, m := range k.cgroupMountsMap { + m.Mount.DecRef(ctx) + m.Root.DecRef(ctx) + m.Fs.DecRef(ctx) + } +} + // Release releases resources owned by k. // // Precondition: This should only be called after the kernel is fully // initialized, e.g. after k.Start() has been called. func (k *Kernel) Release() { ctx := k.SupervisorContext() + k.releaseCgroupMounts(ctx) k.hostMount.DecRef(ctx) k.pipeMount.DecRef(ctx) k.nsfsMount.DecRef(ctx) diff --git a/runsc/boot/loader.go b/runsc/boot/loader.go index e20dc4919..a526400a1 100644 --- a/runsc/boot/loader.go +++ b/runsc/boot/loader.go @@ -170,10 +170,6 @@ type Loader struct { // apply to the entire pod. mountHints *PodMountHints - // cgroupMounts is a map of cgroup mounts that can be reused across - // containers. It is mapped by cgroup controller name. - cgroupMounts map[string]*cgroupMount - // productName is the value to show in // /sys/devices/virtual/dmi/id/product_name. productName string @@ -304,6 +300,33 @@ type Args struct { // make sure stdioFDs are always the same on initial start and on restore const startingStdioFD = 256 +func getRootCredentials(spec *specs.Spec, conf *config.Config, userNs *auth.UserNamespace) *auth.Credentials { + // Create capabilities. + caps, err := specutils.Capabilities(conf.EnableRaw, spec.Process.Capabilities) + if err != nil { + return nil + } + + // Convert the spec's additional GIDs to KGIDs. + extraKGIDs := make([]auth.KGID, 0, len(spec.Process.User.AdditionalGids)) + for _, GID := range spec.Process.User.AdditionalGids { + extraKGIDs = append(extraKGIDs, auth.KGID(GID)) + } + + if userNs == nil { + userNs = auth.NewRootUserNamespace() + } + // Create credentials. + creds := auth.NewUserCredentials( + auth.KUID(spec.Process.User.UID), + auth.KGID(spec.Process.User.GID), + extraKGIDs, + caps, + userNs) + + return creds +} + // New initializes a new kernel loader configured by spec. // New also handles setting up a kernel for restoring a container. func New(args Args) (*Loader, error) { @@ -414,26 +437,10 @@ func New(args Args) (*Loader, error) { return nil, fmt.Errorf("enabling strace: %w", err) } - // Create capabilities. - caps, err := specutils.Capabilities(args.Conf.EnableRaw, args.Spec.Process.Capabilities) - if err != nil { - return nil, fmt.Errorf("converting capabilities: %w", err) + creds := getRootCredentials(args.Spec, args.Conf, nil /* UserNamespace */) + if creds == nil { + return nil, fmt.Errorf("getting root credentials") } - - // Convert the spec's additional GIDs to KGIDs. - extraKGIDs := make([]auth.KGID, 0, len(args.Spec.Process.User.AdditionalGids)) - for _, GID := range args.Spec.Process.User.AdditionalGids { - extraKGIDs = append(extraKGIDs, auth.KGID(GID)) - } - - // Create credentials. - creds := auth.NewUserCredentials( - auth.KUID(args.Spec.Process.User.UID), - auth.KGID(args.Spec.Process.User.GID), - extraKGIDs, - caps, - auth.NewRootUserNamespace()) - // Create root network namespace/stack. netns, err := newRootNetworkNamespace(args.Conf, tk, k, creds.UserNamespace) if err != nil { @@ -625,11 +632,6 @@ func (l *Loader) Destroy() { for _, m := range l.sharedMounts { m.DecRef(ctx) } - for _, m := range l.cgroupMounts { - m.mount.DecRef(ctx) - m.root.DecRef(ctx) - m.fs.DecRef(ctx) - } // Stop the control server. This will indirectly stop any // long-running control operations that are in flight, e.g. @@ -849,12 +851,6 @@ func (l *Loader) createSubcontainer(cid string, tty *fd.FD) error { // the newly created process. Used FDs are either closed or released. It's safe // for the caller to close any remaining files upon return. func (l *Loader) startSubcontainer(spec *specs.Spec, conf *config.Config, cid string, stdioFDs, goferFDs, goferFilestoreFDs []*fd.FD, devGoferFD *fd.FD, goferMountConfs []GoferMountConf) error { - // Create capabilities. - caps, err := specutils.Capabilities(conf.EnableRaw, spec.Process.Capabilities) - if err != nil { - return fmt.Errorf("creating capabilities: %w", err) - } - l.mu.Lock() defer l.mu.Unlock() @@ -863,23 +859,14 @@ func (l *Loader) startSubcontainer(spec *specs.Spec, conf *config.Config, cid st return fmt.Errorf("trying to start a deleted container %q", cid) } - // Convert the spec's additional GIDs to KGIDs. - extraKGIDs := make([]auth.KGID, 0, len(spec.Process.User.AdditionalGids)) - for _, GID := range spec.Process.User.AdditionalGids { - extraKGIDs = append(extraKGIDs, auth.KGID(GID)) - } - // Create credentials. We reuse the root user namespace because the // sentry currently supports only 1 mount namespace, which is tied to a // single user namespace. Thus we must run in the same user namespace // to access mounts. - creds := auth.NewUserCredentials( - auth.KUID(spec.Process.User.UID), - auth.KGID(spec.Process.User.GID), - extraKGIDs, - caps, - l.k.RootUserNamespace()) - + creds := getRootCredentials(spec, conf, l.k.RootUserNamespace()) + if creds == nil { + return fmt.Errorf("getting root credentials") + } var pidns *kernel.PIDNamespace if ns, ok := specutils.GetNS(specs.PIDNamespace, spec); ok { if ns.Path != "" { @@ -912,6 +899,7 @@ func (l *Loader) startSubcontainer(spec *specs.Spec, conf *config.Config, cid st nvidiaUVMDevMajor: l.root.nvidiaUVMDevMajor, nvidiaDriverVersion: l.root.nvidiaDriverVersion, } + var err error info.procArgs, err = createProcessArgs(cid, spec, creds, l.k, pidns) if err != nil { return fmt.Errorf("creating new process: %w", err) @@ -1019,7 +1007,7 @@ func (l *Loader) createContainerProcess(info *containerInfo) (*kernel.ThreadGrou } // We can share l.sharedMounts with containerMounter since l.mu is locked. // Hence, mntr must only be used within this function (while l.mu is locked). - mntr := newContainerMounter(info, l.k, l.mountHints, l.sharedMounts, l.productName, l.sandboxID, l.cgroupMounts) + mntr := newContainerMounter(info, l.k, l.mountHints, l.sharedMounts, l.productName, l.sandboxID) if err := setupContainerVFS(ctx, info, mntr, &info.procArgs); err != nil { return nil, nil, err } diff --git a/runsc/boot/loader_test.go b/runsc/boot/loader_test.go index df4c3ec89..b32d966f4 100644 --- a/runsc/boot/loader_test.go +++ b/runsc/boot/loader_test.go @@ -476,7 +476,7 @@ func TestCreateMountNamespace(t *testing.T) { defer l.Destroy() defer loaderCleanup() - mntr := newContainerMounter(&l.root, l.k, l.mountHints, l.sharedMounts, "", l.sandboxID, l.cgroupMounts) + mntr := newContainerMounter(&l.root, l.k, l.mountHints, l.sharedMounts, "", l.sandboxID) ctx := l.k.SupervisorContext() creds := auth.NewRootCredentials(l.root.procArgs.Credentials.UserNamespace) mns, err := mntr.mountAll(ctx, creds, l.root.spec, l.root.conf, &l.root.procArgs) diff --git a/runsc/boot/restore.go b/runsc/boot/restore.go index 7790d554e..3b53d4a7e 100644 --- a/runsc/boot/restore.go +++ b/runsc/boot/restore.go @@ -18,7 +18,9 @@ import ( "fmt" "os" + "gvisor.dev/gvisor/pkg/sentry/inet" "gvisor.dev/gvisor/pkg/sentry/kernel" + "gvisor.dev/gvisor/pkg/sentry/socket/hostinet" "gvisor.dev/gvisor/pkg/sentry/socket/netstack" "gvisor.dev/gvisor/pkg/sentry/state" "gvisor.dev/gvisor/pkg/sentry/time" @@ -34,18 +36,50 @@ type restorer struct { deviceFile *os.File } -func (r *restorer) restore(l *Loader) error { +func createNetworkNamespaceForRestore(l *Loader) (*inet.Namespace, error) { + creds := getRootCredentials(l.root.spec, l.root.conf, nil /* UserNamespace */) + if creds == nil { + return nil, fmt.Errorf("getting root credentials") + } + // Save the current network stack to slap on top of the one that was restored. curNetwork := l.k.RootNetworkNamespace().Stack() - if eps, ok := curNetwork.(*netstack.Stack); ok { - stack.StackFromEnv = eps.Stack // FIXME(b/36201077) + eps, ok := curNetwork.(*netstack.Stack) + if !ok { + return inet.NewRootNamespace(hostinet.NewStack(), nil, creds.UserNamespace), nil } + stack.StackFromEnv = eps.Stack // FIXME(b/36201077) + creator := &sandboxNetstackCreator{ + clock: l.k.Timekeeper(), + uniqueID: l.k, + allowPacketEndpointWrite: l.root.conf.AllowPacketEndpointWrite, + } + return inet.NewRootNamespace(curNetwork, creator, creds.UserNamespace), nil +} + +func (r *restorer) restore(l *Loader) error { + // Create a new root network namespace with the network stack of the + // old kernel to preserve the exisiting network configuration. + netns, err := createNetworkNamespaceForRestore(l) + if err != nil { + return fmt.Errorf("creating network: %w", err) + } + + // Reset the network stack in the network namespace to nil before + // replacing the kernel. This will not free the network stack when this + // old kernel is released. + l.k.RootNetworkNamespace().ResetStack() + p, err := createPlatform(l.root.conf, r.deviceFile) if err != nil { return fmt.Errorf("creating platform: %v", err) } - // Replace the old kernel with a new one that will be restored into. + + // Release the kernel and replace it with a new one that will be restored into. + if l.k != nil { + l.k.Release() + } l.k = &kernel.Kernel{ Platform: p, } @@ -71,7 +105,7 @@ func (r *restorer) restore(l *Loader) error { // Set up the restore environment. ctx := l.k.SupervisorContext() // TODO(b/298078576): Need to process hints here probably - mntr := newContainerMounter(&l.root, l.k, l.mountHints, l.sharedMounts, l.productName, l.sandboxID, l.cgroupMounts) + mntr := newContainerMounter(&l.root, l.k, l.mountHints, l.sharedMounts, l.productName, l.sandboxID) ctx, err = mntr.configureRestore(ctx) if err != nil { return fmt.Errorf("configuring filesystem restore: %v", err) @@ -79,7 +113,7 @@ func (r *restorer) restore(l *Loader) error { // Load the state. loadOpts := state.LoadOpts{Source: r.stateFile} - if err := loadOpts.Load(ctx, l.k, nil, curNetwork, time.NewCalibratedClocks(), &vfs.CompleteRestoreOptions{}); err != nil { + if err := loadOpts.Load(ctx, l.k, nil, netns.Stack(), time.NewCalibratedClocks(), &vfs.CompleteRestoreOptions{}); err != nil { return err } diff --git a/runsc/boot/vfs.go b/runsc/boot/vfs.go index fe62558f7..34275e5de 100644 --- a/runsc/boot/vfs.go +++ b/runsc/boot/vfs.go @@ -405,23 +405,13 @@ type containerMounter struct { sandboxID string containerName string - // cgroupMounts is a map of cgroup mounts that can be reused across - // containers. Key is the cgroup controller name string. - cgroupMounts map[string]*cgroupMount - // cgroupsMounted indicates if cgroups are mounted in the container. // This is used to set the InitialCgroups before starting the container // process. cgroupsMounted bool } -type cgroupMount struct { - fs *vfs.Filesystem - root *vfs.Dentry - mount *vfs.Mount -} - -func newContainerMounter(info *containerInfo, k *kernel.Kernel, hints *PodMountHints, sharedMounts map[string]*vfs.Mount, productName string, sandboxID string, cgroupMounts map[string]*cgroupMount) *containerMounter { +func newContainerMounter(info *containerInfo, k *kernel.Kernel, hints *PodMountHints, sharedMounts map[string]*vfs.Mount, productName string, sandboxID string) *containerMounter { return &containerMounter{ root: info.spec.Root, mounts: compileMounts(info.spec, info.conf, info.procArgs.ContainerID), @@ -436,7 +426,6 @@ func newContainerMounter(info *containerInfo, k *kernel.Kernel, hints *PodMountH containerID: info.procArgs.ContainerID, sandboxID: sandboxID, containerName: info.containerName, - cgroupMounts: cgroupMounts, } } @@ -1107,10 +1096,9 @@ func (c *containerMounter) getSharedMount(ctx context.Context, spec *specs.Spec, } // mountCgroupMounts mounts the cgroups which are shared across all containers. -// Postcondition: Initialized l.cgroupMounts on success. +// Postcondition: Initialized k.cgroupMounts on success. func (l *Loader) mountCgroupMounts(conf *config.Config, creds *auth.Credentials) error { ctx := l.k.SupervisorContext() - cgroupMounts := make(map[string]*cgroupMount) for _, sopts := range kernel.CgroupCtrls { mopts := &vfs.MountOptions{ GetFilesystemOptions: vfs.GetFilesystemOptions{ @@ -1127,13 +1115,12 @@ func (l *Loader) mountCgroupMounts(conf *config.Config, creds *auth.Credentials) // Private so that mounts created by containers do not appear // in other container's cgroup paths. l.k.VFS().SetMountPropagation(mount, linux.MS_PRIVATE, false) - cgroupMounts[string(sopts)] = &cgroupMount{ - fs: fs, - root: root, - mount: mount, - } + l.k.AddCgroupMount(string(sopts), &kernel.CgroupMount{ + Fs: fs, + Root: root, + Mount: mount, + }) } - l.cgroupMounts = cgroupMounts log.Infof("created cgroup mounts for controllers %v", kernel.CgroupCtrls) return nil } @@ -1169,12 +1156,12 @@ func (c *containerMounter) mountCgroupSubmounts(ctx context.Context, spec *specs mountCtx := vfs.WithRoot(vfs.WithMountNamespace(ctx, mns), root) for _, ctrl := range kernel.CgroupCtrls { ctrlName := string(ctrl) - cgroupMnt, ok := c.cgroupMounts[ctrlName] - if !ok { + cgroupMnt := c.k.GetCgroupMount(ctrlName) + if cgroupMnt == nil { return fmt.Errorf("cgroup mount for controller %s not found", ctrlName) } - cgroupMntVD := vfs.MakeVirtualDentry(cgroupMnt.mount, cgroupMnt.root) + cgroupMntVD := vfs.MakeVirtualDentry(cgroupMnt.Mount, cgroupMnt.Root) sourcePop := vfs.PathOperation{ Root: cgroupMntVD, Start: cgroupMntVD,