Implement MS_SLAVE.

This change implements the MS_SLAVE flag as its described in
https://www.kernel.org/doc/Documentation/filesystems/sharedsubtree.txt.
In the sentry slaves are called followers and masters are called leaders.

PiperOrigin-RevId: 570866579
This commit is contained in:
Lucas Manning
2023-10-04 18:08:58 -07:00
committed by gVisor bot
parent e0bdd0d576
commit d493e93763
7 changed files with 1132 additions and 284 deletions
+3 -4
View File
@@ -45,8 +45,8 @@ func Mount(t *kernel.Task, sysno uintptr, args arch.SyscallArguments) (uintptr,
}
// Silently allow MS_NOSUID, since we don't implement set-id bits anyway.
const unsupported = linux.MS_REMOUNT | linux.MS_SLAVE |
linux.MS_UNBINDABLE | linux.MS_MOVE | linux.MS_REC | linux.MS_NODIRATIME
const unsupported = linux.MS_REMOUNT | linux.MS_UNBINDABLE | linux.MS_MOVE |
linux.MS_REC | linux.MS_NODIRATIME
// Linux just allows passing any flags to mount(2) - it won't fail when
// unknown or unsupported flags are passed. Since we don't implement
@@ -79,8 +79,7 @@ func Mount(t *kernel.Task, sysno uintptr, args arch.SyscallArguments) (uintptr,
return 0, nil, err
}
defer sourceTpop.Release(t)
_, err = t.Kernel().VFS().BindAt(t, creds, &sourceTpop.pop, &target.pop)
return 0, nil, err
return 0, nil, t.Kernel().VFS().BindAt(t, creds, &sourceTpop.pop, &target.pop)
}
const propagationFlags = linux.MS_SHARED | linux.MS_PRIVATE | linux.MS_SLAVE | linux.MS_UNBINDABLE
if propFlag := flags & propagationFlags; propFlag != 0 {
+13
View File
@@ -64,6 +64,18 @@ go_template_instance(
},
)
go_template_instance(
name = "mount_list",
out = "mount_list.go",
package = "vfs",
prefix = "follower",
template = "//pkg/ilist:generic_list",
types = {
"Element": "*Mount",
"Linker": "*Mount",
},
)
go_template_instance(
name = "event_list",
out = "event_list.go",
@@ -141,6 +153,7 @@ go_library(
"inotify_mutex.go",
"lock.go",
"mount.go",
"mount_list.go",
"mount_namespace_refs.go",
"mount_ring.go",
"mount_unsafe.go",
+192 -132
View File
@@ -23,6 +23,7 @@ import (
"gvisor.dev/gvisor/pkg/abi/linux"
"gvisor.dev/gvisor/pkg/atomicbitops"
"gvisor.dev/gvisor/pkg/cleanup"
"gvisor.dev/gvisor/pkg/context"
"gvisor.dev/gvisor/pkg/errors/linuxerr"
"gvisor.dev/gvisor/pkg/refs"
@@ -90,10 +91,19 @@ type Mount struct {
// isShared indicates this mount has the MS_SHARED propagation type.
isShared bool
// sharedEntry represents an entry in a circular list (ring) of mounts in a
// shared peer group.
// sharedEntry is an entry in a circular list (ring) of mounts in a shared
// peer group.
sharedEntry mountEntry
// followerList is a list of mounts which has this mount as its leader.
followerList followerList
// followerEntry is an entry in a followerList.
followerEntry
// leader is the mount that this mount receives propagation events from.
leader *Mount
// groupID is the ID for this mount's shared peer group. If the mount is not
// in a peer group, this is 0.
groupID uint32
@@ -166,15 +176,12 @@ func (mnt *Mount) MountFlags() uint64 {
return flags
}
func (mnt *Mount) generateOptionalTags() string {
mnt.vfs.lockMounts()
defer mnt.vfs.unlockMounts(context.Background())
// TODO(b/249777195): Support MS_SLAVE and MS_UNBINDABLE propagation types.
var optional string
if mnt.isShared {
optional = fmt.Sprintf("shared:%d", mnt.groupID)
}
return optional
func (mnt *Mount) isFollower() bool {
return mnt.leader != nil
}
func (mnt *Mount) neverConnected() bool {
return mnt.ns == nil
}
// coveringMount returns a mount that completely covers mnt if it exists and nil
@@ -234,6 +241,58 @@ func (vfs *VirtualFilesystem) MountDisconnected(ctx context.Context, creds *auth
return newMount(vfs, fs, root, nil /* mntns */, opts), nil
}
// attachMountLocked attaches mnt to vd and propagates the mount to vd.mount's
// peers and followers. This method is analogous to
// fs/namespace.c:attach_recursive_mnt() in Linux.
//
// +checklocks:vfs.mountMu
func (vfs *VirtualFilesystem) attachMountLocked(ctx context.Context, mnt *Mount, vd VirtualDentry) error {
vdCleanup := cleanup.Make(func() {
vd.DecRef(ctx)
})
defer vdCleanup.Clean()
// This is equivalent to checking for SB_NOUSER in Linux, which is set on all
// anon mounts and sentry-internal filesystems like pipefs.
if vd.mount.neverConnected() {
return linuxerr.EINVAL
}
if vd.mount.ns.mounts+1 > MountMax {
return linuxerr.ENOSPC
}
if vd.mount.isShared {
if err := vfs.allocMountGroupIDs(mnt, true); err != nil {
return err
}
}
propMnts, err := vfs.doPropagation(ctx, mnt, vd)
cleanup := cleanup.Make(func() {
// Checklocks can't understand lock state within closures, so we have to
// force.
vfs.freeMountGroupIDs(mnt.submountsLocked()) // +checklocksforce
for pmnt := range propMnts {
vfs.abortTree(ctx, pmnt) // +checklocksforce
}
})
defer cleanup.Clean()
if err != nil {
return err
}
if vd.mount.isShared {
for _, m := range mnt.submountsLocked() {
m.isShared = true
}
}
vdCleanup.Release()
if err := vfs.connectMountAtLocked(ctx, mnt, vd); err != nil {
return err
}
cleanup.Release()
for pmnt := range propMnts {
vfs.commitTree(ctx, pmnt)
}
return nil
}
// ConnectMountAt connects mnt at the path represented by target.
//
// Preconditions: mnt must be disconnected.
@@ -246,28 +305,7 @@ func (vfs *VirtualFilesystem) ConnectMountAt(ctx context.Context, creds *auth.Cr
}
vfs.lockMounts()
defer vfs.unlockMounts(ctx)
// This is equivalent to checking for SB_NOUSER in Linux, which is set on all
// anon mounts and sentry-internal filesystems like pipefs.
if vd.mount.ns == nil {
vfs.delayDecRef(vd)
return linuxerr.EINVAL
}
tree := vfs.preparePropagationTree(mnt, vd)
// Check if the new mount + all the propagation mounts puts us over the max.
if uint32(len(tree)+1)+vd.mount.ns.mounts > MountMax {
// We need to unlock mountMu first because DecRef takes a lock on the
// filesystem mutex in some implementations, which can lead to circular
// locking.
vfs.abortPropagationTree(ctx, tree)
vfs.delayDecRef(vd)
return linuxerr.ENOSPC
}
if err := vfs.connectMountAtLocked(ctx, mnt, vd); err != nil {
vfs.abortPropagationTree(ctx, tree)
return err
}
vfs.commitPropagationTree(ctx, tree)
return nil
return vfs.attachMountLocked(ctx, mnt, vd)
}
// connectMountAtLocked attaches mnt at vd. This method consumes a reference on
@@ -340,22 +378,22 @@ func (vfs *VirtualFilesystem) lockMountpoint(vd VirtualDentry) (VirtualDentry, e
}
// CloneMountAt returns a new mount with the same fs, specified root and
// mount options. If mnt's propagation type is shared the new mount is
// automatically made a peer of mnt. If mount options are nil, mnt's
// options are copied.
func (vfs *VirtualFilesystem) CloneMountAt(mnt *Mount, root *Dentry, mopts *MountOptions) *Mount {
// mount options. If mount options are nil, mnt's options are copied. The clone
// is added to mnt's peer group if mnt is shared. If not the clone is in a
// shared peer group by itself.
func (vfs *VirtualFilesystem) CloneMountAt(mnt *Mount, root *Dentry, mopts *MountOptions) (*Mount, error) {
vfs.lockMounts()
defer vfs.unlockMounts(context.Background())
clone := vfs.cloneMount(mnt, root, mopts)
return clone
return vfs.cloneMount(mnt, root, mopts, makeSharedClone)
}
// cloneMount returns a new mount with mnt.fs as the filesystem and root as the
// root. The returned mount has an extra reference.
// root, with a propagation type specified by cloneType. The returned mount has
// an extra reference. If mopts is nil, use the options found in mnt.
// This method is analogous to fs/namespace.c:clone_mnt() in Linux.
//
// +checklocks:vfs.mountMu
// +checklocksalias:mnt.vfs.mountMu=vfs.mountMu
func (vfs *VirtualFilesystem) cloneMount(mnt *Mount, root *Dentry, mopts *MountOptions) *Mount {
func (vfs *VirtualFilesystem) cloneMount(mnt *Mount, root *Dentry, mopts *MountOptions, cloneType int) (*Mount, error) {
opts := mopts
if opts == nil {
opts = &MountOptions{
@@ -364,10 +402,39 @@ func (vfs *VirtualFilesystem) cloneMount(mnt *Mount, root *Dentry, mopts *MountO
}
}
clone := vfs.NewDisconnectedMount(mnt.fs, root, opts)
if mnt.isShared {
vfs.addPeer(mnt, clone)
if cloneType&(makeFollowerClone|makePrivateClone|sharedToFollowerClone) != 0 {
clone.groupID = 0
} else {
clone.groupID = mnt.groupID
}
return clone
if cloneType&makeSharedClone != 0 && clone.groupID == 0 {
gid, err := vfs.allocateGroupID()
if err != nil {
vfs.delayDecRef(clone)
return nil, linuxerr.ENOSPC
}
clone.groupID = gid
}
clone.isShared = mnt.isShared
if cloneType&makeFollowerClone != 0 || (cloneType&sharedToFollowerClone != 0 && mnt.isShared) {
mnt.followerList.PushFront(clone)
clone.leader = mnt
clone.isShared = false
} else if cloneType&makePrivateClone == 0 {
if cloneType&makeSharedClone != 0 || mnt.isShared {
mnt.sharedEntry.Add(&clone.sharedEntry)
}
if mnt.isFollower() {
mnt.leader.followerList.InsertAfter(mnt, clone)
}
clone.leader = mnt.leader
} else {
clone.isShared = false
}
if cloneType&makeSharedClone != 0 {
clone.isShared = true
}
return clone, nil
}
type cloneTreeNode struct {
@@ -376,17 +443,16 @@ type cloneTreeNode struct {
}
// cloneMountTree creates a copy of mnt's tree with the specified root
// dentry at root. The new descendents are added to mnt's pending mount list.
// dentry at root. The new descendants are added to mnt's pending mount list.
// `cloneFunc` is a callback that is executed for each cloned mount.
// This method is analogous to fs/namespace.c:copy_tree() in Linux.
//
// +checklocks:vfs.mountMu
func (vfs *VirtualFilesystem) cloneMountTree(
ctx context.Context,
mnt *Mount,
root *Dentry,
cloneFunc func(ctx context.Context, oldmnt, newMnt *Mount),
) (*Mount, error) {
clone := vfs.cloneMount(mnt, root, nil)
func (vfs *VirtualFilesystem) cloneMountTree(ctx context.Context, mnt *Mount, root *Dentry, cloneType int, cloneFunc func(ctx context.Context, oldmnt, newMnt *Mount)) (*Mount, error) {
clone, err := vfs.cloneMount(mnt, root, nil, cloneType)
if err != nil {
return nil, err
}
if cloneFunc != nil {
cloneFunc(ctx, mnt, clone)
}
@@ -395,7 +461,14 @@ func (vfs *VirtualFilesystem) cloneMountTree(
p := queue[len(queue)-1]
queue = queue[:len(queue)-1]
for c := range p.prevMount.children {
m := vfs.cloneMount(c, c.root, nil)
if mp := c.getKey(); p.prevMount == mnt && !mp.mount.fs.Impl().IsDescendant(VirtualDentry{mnt, root}, mp) {
continue
}
m, err := vfs.cloneMount(c, c.root, nil, cloneType)
if err != nil {
vfs.abortTree(ctx, clone)
return nil, err
}
mp := VirtualDentry{
mount: p.parentMount,
dentry: c.point(),
@@ -419,42 +492,26 @@ func (vfs *VirtualFilesystem) cloneMountTree(
// path.
//
// TODO(b/249121230): Support recursive bind mounting.
func (vfs *VirtualFilesystem) BindAt(ctx context.Context, creds *auth.Credentials, source, target *PathOperation) (*Mount, error) {
func (vfs *VirtualFilesystem) BindAt(ctx context.Context, creds *auth.Credentials, source, target *PathOperation) error {
sourceVd, err := vfs.GetDentryAt(ctx, creds, source, &GetDentryOptions{})
if err != nil {
return nil, err
return err
}
defer sourceVd.DecRef(ctx)
targetVd, err := vfs.GetDentryAt(ctx, creds, target, &GetDentryOptions{})
if err != nil {
return nil, err
return err
}
vfs.lockMounts()
defer vfs.unlockMounts(ctx)
// This is equivalent to checking for SB_NOUSER in Linux, which is set on all
// anon mounts.
if targetVd.mount.ns == nil {
clone, err := vfs.cloneMount(sourceVd.mount, sourceVd.dentry, nil, 0)
if err != nil {
vfs.delayDecRef(targetVd)
return nil, linuxerr.EINVAL
return err
}
clone := vfs.cloneMount(sourceVd.mount, sourceVd.dentry, nil)
vfs.delayDecRef(clone)
tree := vfs.preparePropagationTree(clone, targetVd)
if uint32(1+len(tree))+targetVd.mount.ns.mounts > MountMax {
vfs.setPropagation(clone, linux.MS_PRIVATE)
vfs.abortPropagationTree(ctx, tree)
vfs.delayDecRef(targetVd)
return nil, linuxerr.ENOSPC
}
if err := vfs.connectMountAtLocked(ctx, clone, targetVd); err != nil {
vfs.setPropagation(clone, linux.MS_PRIVATE)
vfs.abortPropagationTree(ctx, tree)
return nil, err
}
vfs.commitPropagationTree(ctx, tree)
return clone, nil
return vfs.attachMountLocked(ctx, clone, targetVd)
}
// MountAt creates and mounts a Filesystem configured by the given arguments.
@@ -527,45 +584,30 @@ func (vfs *VirtualFilesystem) UmountAt(ctx context.Context, creds *auth.Credenti
}
}
umountTree := []*Mount{vd.mount}
parent, mountpoint := vd.mount.parent(), vd.mount.point()
if parent != nil && parent.isShared {
for peer := parent.sharedEntry.Next(); peer != parent; peer = peer.sharedEntry.Next() {
umountMnt := vfs.mounts.Lookup(peer, mountpoint)
// From https://www.kernel.org/doc/Documentation/filesystems/sharedsubtree.txt:
// If any peer has some child mounts, then that mount is not unmounted,
// but all other mounts are unmounted.
if umountMnt == nil {
continue
}
if len(umountMnt.children) == 0 || umountMnt.coveringMount() != nil {
umountTree = append(umountTree, umountMnt)
}
}
if opts.Flags&linux.MNT_DETACH == 0 && vfs.arePropMountsBusy(vd.mount) {
return linuxerr.EBUSY
}
// TODO(gvisor.dev/issue/1035): Linux special-cases umount of the caller's
// root, which we don't implement yet (we'll just fail it since the caller
// holds a reference on it).
vfs.mounts.seq.BeginWrite()
if opts.Flags&linux.MNT_DETACH == 0 {
if len(vd.mount.children) != 0 {
vfs.mounts.seq.EndWrite()
return linuxerr.EBUSY
}
// We are holding a reference on vd.mount.
expectedRefs := int64(1)
if !vd.mount.umounted {
expectedRefs = 2
}
if vd.mount.refs.Load()&^math.MinInt64 != expectedRefs { // mask out MSB
vfs.mounts.seq.EndWrite()
return linuxerr.EBUSY
propMounts := []*Mount{vd.mount}
if vd.mount.parent() != nil {
for m := nextPropMount(vd.mount.parent(), vd.mount.parent()); m != nil; m = nextPropMount(m, vd.mount.parent()) {
child := vfs.mounts.Lookup(m, vd.mount.point())
if child == nil {
continue
}
if len(child.children) != 0 && child.coveringMount() == nil {
continue
}
propMounts = append(propMounts, child)
}
}
for _, mnt := range umountTree {
vfs.umountRecursiveLocked(mnt, &umountRecursiveOptions{
vfs.mounts.seq.BeginWrite()
for _, m := range propMounts {
vfs.umountRecursiveLocked(m, &umountRecursiveOptions{
eager: opts.Flags&linux.MNT_DETACH == 0,
disconnectHierarchy: true,
})
@@ -574,6 +616,21 @@ func (vfs *VirtualFilesystem) UmountAt(ctx context.Context, creds *auth.Credenti
return nil
}
// mountHasExpectedRefs checks that mnt has the correct number of references
// before a umount. It is analogous to fs/pnode.c:do_refcount_check().
//
// +checklocks:vfs.mountMu
func (vfs *VirtualFilesystem) mountHasExpectedRefs(mnt *Mount) bool {
expectedRefs := int64(1)
if !mnt.umounted {
expectedRefs++
}
if mnt.coveringMount() != nil {
expectedRefs++
}
return mnt.refs.Load()&^math.MinInt64 == expectedRefs // mask out MSB
}
// +stateify savable
type umountRecursiveOptions struct {
// If eager is true, ensure that future calls to Mount.tryIncMountedRef()
@@ -617,9 +674,7 @@ func (vfs *VirtualFilesystem) umountRecursiveLocked(mnt *Mount, opts *umountRecu
if parent := mnt.parent(); parent != nil && (opts.disconnectHierarchy || !parent.umounted) {
vfs.delayDecRef(vfs.disconnectLocked(mnt))
}
if mnt.isShared {
vfs.setPropagation(mnt, linux.MS_PRIVATE)
}
vfs.setPropagation(mnt, linux.MS_PRIVATE)
}
if opts.eager {
for {
@@ -1255,7 +1310,7 @@ func (vfs *VirtualFilesystem) GenerateProcMountInfo(ctx context.Context, taskRoo
fmt.Fprintf(buf, "%s ", opts)
// (7) Optional fields: zero or more fields of the form "tag[:value]".
fmt.Fprintf(buf, "%s ", mnt.generateOptionalTags())
fmt.Fprintf(buf, "%s", vfs.generateOptionalTags(ctx, mnt, taskRootDir))
// (8) Separator: the end of the optional fields is marked by a single hyphen.
fmt.Fprintf(buf, "- ")
@@ -1311,25 +1366,30 @@ func superBlockOpts(mountPath string, mnt *Mount) string {
return opts
}
// allocateGroupID returns a new mount group id if one is available, and
// error otherwise. If the group ID bitmap is full, double the size of the
// bitmap before allocating the new group id.
//
// +checklocks:vfs.mountMu
func (vfs *VirtualFilesystem) allocateGroupID() (uint32, error) {
groupID, err := vfs.groupIDBitmap.FirstZero(1)
if err != nil {
if err := vfs.groupIDBitmap.Grow(uint32(vfs.groupIDBitmap.Size())); err != nil {
return 0, err
func (vfs *VirtualFilesystem) generateOptionalTags(ctx context.Context, mnt *Mount, root VirtualDentry) string {
vfs.lockMounts()
defer vfs.unlockMounts(ctx)
// TODO(b/249777195): Support MS_UNBINDABLE propagation type.
var optionalSb strings.Builder
if mnt.isShared {
optionalSb.WriteString(fmt.Sprintf("shared:%d ", mnt.groupID))
}
if mnt.isFollower() {
// Per man mount_namespaces(7), propagate_from should not be
// included in optional tags if the leader "is the immediate leader of the
// mount, or if there is no dominant peer group under the same root". A
// dominant peer group is the nearest reachable mount in the leader/follower
// chain.
optionalSb.WriteString(fmt.Sprintf("master:%d ", mnt.leader.groupID))
var dominant *Mount
for m := mnt.leader; m != nil; m = m.leader {
if dominant = vfs.peerUnderRoot(ctx, m, mnt.ns, root); dominant != nil {
break
}
}
if dominant != nil && dominant != mnt.leader {
optionalSb.WriteString(fmt.Sprintf("propagate_from:%d ", dominant.groupID))
}
}
vfs.groupIDBitmap.Add(groupID)
return groupID, nil
}
// freeGroupID marks a groupID as available for reuse.
//
// +checklocks:vfs.mountMu
func (vfs *VirtualFilesystem) freeGroupID(id uint32) {
vfs.groupIDBitmap.Remove(id)
return optionalSb.String()
}
+5 -2
View File
@@ -170,13 +170,16 @@ func (vfs *VirtualFilesystem) CloneMountNamespace(
vfs.lockMounts()
defer vfs.unlockMounts(ctx)
newRoot, err := vfs.cloneMountTree(ctx, ns.root, ns.root.root,
cloneType := 0
if ns.Owner != newns.Owner {
cloneType = sharedToFollowerClone
}
newRoot, err := vfs.cloneMountTree(ctx, ns.root, ns.root.root, cloneType,
func(ctx context.Context, src, dst *Mount) {
vfs.updateRootAndCWD(ctx, root, cwd, src, dst) // +checklocksforce: vfs.mountMu is locked.
})
if err != nil {
newns.DecRef(ctx)
vfs.abortTree(ctx, newRoot)
return nil, err
}
newns.root = newRoot
File diff suppressed because it is too large Load Diff
+2
View File
@@ -1410,6 +1410,7 @@ cc_binary(
"@com_google_absl//absl/time",
gtest,
"//test/util:eventfd_util",
"//test/util:logging",
"//test/util:mount_util",
"//test/util:multiprocess_util",
"//test/util:posix_error",
@@ -1418,6 +1419,7 @@ cc_binary(
"//test/util:test_main",
"//test/util:test_util",
"//test/util:thread_util",
"@com_google_absl//absl/strings:str_format",
],
)
File diff suppressed because it is too large Load Diff