Files

272 lines
9.0 KiB
Go
Raw Permalink Normal View History

// Copyright 2018 The gVisor Authors.
2018-04-27 10:37:02 -07:00
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//go:build linux
2018-04-27 10:37:02 -07:00
// +build linux
package ptrace
import (
"fmt"
"golang.org/x/sys/unix"
2019-06-13 16:49:09 -07:00
"gvisor.dev/gvisor/pkg/abi/linux"
"gvisor.dev/gvisor/pkg/bpf"
2023-02-03 17:45:36 -08:00
"gvisor.dev/gvisor/pkg/hosttid"
2019-06-13 16:49:09 -07:00
"gvisor.dev/gvisor/pkg/log"
"gvisor.dev/gvisor/pkg/seccomp"
"gvisor.dev/gvisor/pkg/sentry/arch"
2018-04-27 10:37:02 -07:00
)
const syscallEvent unix.Signal = 0x80
2018-04-27 10:37:02 -07:00
// createStub creates a fresh stub processes.
//
// Precondition: the runtime OS thread must be locked.
func createStub() (*thread, error) {
// The exact interactions of ptrace and seccomp are complex, and
// changed in recent kernel versions. Before commit 93e35efb8de45, the
// seccomp check is done before the ptrace emulation check. This means
// that any calls not matching this list will trigger the seccomp
// default action instead of notifying ptrace.
//
// After commit 93e35efb8de45, the seccomp check is done after the
// ptrace emulation check. This simplifies using SYSEMU, since seccomp
// will never run for emulation. Seccomp will only run for injected
// system calls, and thus we can use RET_KILL as our violation action.
2018-12-18 10:27:16 -08:00
var defaultAction linux.BPFAction
if probeSeccomp() {
log.Infof("Latest seccomp behavior found (kernel >= 4.8 likely)")
2018-12-18 10:27:16 -08:00
defaultAction = linux.SECCOMP_RET_KILL_THREAD
} else {
// We must rely on SYSEMU behavior; tracing with SYSEMU is broken.
log.Infof("Legacy seccomp behavior found (kernel < 4.8 likely)")
2018-12-18 10:27:16 -08:00
defaultAction = linux.SECCOMP_RET_ALLOW
}
// When creating the new child process, we specify SIGKILL as the
// signal to deliver when the child exits. We never expect a subprocess
// to exit; they are pooled and reused. This is done to ensure that if
// a subprocess is OOM-killed, this process (and all other stubs,
// transitively) will be killed as well. It's simply not possible to
// safely handle a single stub getting killed: the exact state of
// execution is unknown and not recoverable.
//
// In addition, we set the PTRACE_O_TRACEEXIT option to log more
// information about a stub process when it receives a fatal signal.
return attachedThread(uintptr(unix.SIGKILL)|unix.CLONE_FILES, defaultAction)
}
// attachedThread returns a new attached thread.
//
// Precondition: the runtime OS thread must be locked.
2018-12-18 10:27:16 -08:00
func attachedThread(flags uintptr, defaultAction linux.BPFAction) (*thread, error) {
// Create a BPF program that allows only the system calls needed by the
// stub and all its children. This is used to create child stubs
// (below), so we must include the ability to fork, but otherwise lock
// down available calls only to what is needed.
rules := []seccomp.RuleSet{}
2018-12-18 10:27:16 -08:00
if defaultAction != linux.SECCOMP_RET_ALLOW {
rules = append(rules, seccomp.RuleSet{
2023-10-10 14:07:40 -07:00
Rules: seccomp.MakeSyscallRules(map[uintptr]seccomp.SyscallRule{
unix.SYS_CLONE: seccomp.Or{
// Allow creation of new subprocesses (used by the master).
seccomp.PerArg{seccomp.EqualTo(unix.CLONE_FILES | unix.SIGKILL)},
2023-10-25 12:06:44 -07:00
// Allow creation of new threads within a single address space (used by address spaces).
seccomp.PerArg{
seccomp.EqualTo(
unix.CLONE_FILES |
unix.CLONE_FS |
unix.CLONE_SIGHAND |
unix.CLONE_THREAD |
unix.CLONE_PTRACE |
unix.CLONE_VM)},
},
// For the initial process creation.
unix.SYS_WAIT4: seccomp.MatchAll{},
unix.SYS_EXIT: seccomp.MatchAll{},
// For the stub prctl dance (all).
unix.SYS_PRCTL: seccomp.PerArg{seccomp.EqualTo(unix.PR_SET_PDEATHSIG), seccomp.EqualTo(unix.SIGKILL)},
unix.SYS_GETPPID: seccomp.MatchAll{},
// For the stub to stop itself (all).
unix.SYS_GETPID: seccomp.MatchAll{},
unix.SYS_KILL: seccomp.PerArg{seccomp.AnyValue{}, seccomp.EqualTo(unix.SIGSTOP)},
// Injected to support the address space operations.
unix.SYS_MMAP: seccomp.MatchAll{},
unix.SYS_MUNMAP: seccomp.MatchAll{},
2023-10-10 14:07:40 -07:00
}),
2018-12-18 10:27:16 -08:00
Action: linux.SECCOMP_RET_ALLOW,
})
}
rules = appendArchSeccompRules(rules, defaultAction)
instrs, _, err := seccomp.BuildProgram(rules, seccomp.ProgramOptions{
DefaultAction: defaultAction,
BadArchAction: defaultAction,
})
if err != nil {
return nil, err
}
return forkStub(flags, instrs)
}
// In the child, this function must not acquire any locks, because they might
// have been locked at the time of the fork. This means no rescheduling, no
// malloc calls, and no new stack segments. For the same reason compiler does
// not race instrument it.
//
//go:norace
func forkStub(flags uintptr, instrs []bpf.Instruction) (*thread, error) {
2018-04-27 10:37:02 -07:00
// Declare all variables up front in order to ensure that there's no
// need for allocations between beforeFork & afterFork.
var (
pid uintptr
ppid uintptr
errno unix.Errno
2018-04-27 10:37:02 -07:00
)
// Remember the current ppid for the pdeathsig race.
ppid, _, _ = unix.RawSyscall(unix.SYS_GETPID, 0, 0, 0)
2018-04-27 10:37:02 -07:00
// Among other things, beforeFork masks all signals.
beforeFork()
2018-06-26 16:53:48 -07:00
// Do the clone.
pid, _, errno = unix.RawSyscall6(unix.SYS_CLONE, flags, 0, 0, 0, 0, 0)
2018-04-27 10:37:02 -07:00
if errno != 0 {
afterFork()
return nil, errno
}
// Is this the parent?
if pid != 0 {
// Among other things, restore signal mask.
afterFork()
// Initialize the first thread.
t := &thread{
tgid: int32(pid),
tid: int32(pid),
cpu: ^uint32(0),
}
if sig := t.wait(stopped); sig != unix.SIGSTOP {
2018-04-27 10:37:02 -07:00
return nil, fmt.Errorf("wait failed: expected SIGSTOP, got %v", sig)
}
t.attach()
t.grabInitRegs()
2018-04-27 10:37:02 -07:00
return t, nil
}
// Move the stub to a new session (and thus a new process group). This
// prevents the stub from getting PTY job control signals intended only
// for the sentry process. We must call this before restoring signal
// mask.
if _, _, errno := unix.RawSyscall(unix.SYS_SETSID, 0, 0, 0); errno != 0 {
unix.RawSyscall(unix.SYS_EXIT, uintptr(errno), 0, 0)
}
2018-04-27 10:37:02 -07:00
// afterForkInChild resets all signals to their default dispositions
// and restores the signal mask to its pre-fork state.
afterForkInChild()
// Explicitly unmask all signals to ensure that the tracer can see
// them.
if errno := unmaskAllSignals(); errno != 0 {
unix.RawSyscall(unix.SYS_EXIT, uintptr(errno), 0, 0)
2018-04-27 10:37:02 -07:00
}
// Set an aggressive BPF filter for the stub and all it's children. See
// the description of the BPF program built above.
2021-07-08 18:55:56 -07:00
if errno := seccomp.SetFilterInChild(instrs); errno != 0 {
unix.RawSyscall(unix.SYS_EXIT, uintptr(errno), 0, 0)
}
// Enable cpuid-faulting.
enableCpuidFault()
2018-07-16 22:02:03 -07:00
2018-04-27 10:37:02 -07:00
// Call the stub; should not return.
stubCall(stubStart, ppid)
panic("unreachable")
}
// createStub creates a stub processes as a child of an existing subprocesses.
//
// Precondition: the runtime OS thread must be locked.
func (s *subprocess) createStub() (*thread, error) {
// There's no need to lock the runtime thread here, as this can only be
// called from a context that is already locked.
2023-02-03 17:45:36 -08:00
currentTID := int32(hosttid.Current())
2018-04-27 10:37:02 -07:00
t := s.syscallThreads.lookupOrCreate(currentTID, s.newThread)
// Pass the expected PPID to the child via R15.
regs := t.initRegs
initChildProcessPPID(&regs, t.tgid)
2018-04-27 10:37:02 -07:00
// Call fork in a subprocess.
//
// The new child must set up PDEATHSIG to ensure it dies if this
// process dies. Since this process could die at any time, this cannot
// be done via instrumentation from here.
//
// Instead, we create the child untraced, which will do the PDEATHSIG
// setup and then SIGSTOP itself for our attach below.
2018-06-26 16:53:48 -07:00
//
// See above re: SIGKILL.
2018-04-27 10:37:02 -07:00
pid, err := t.syscallIgnoreInterrupt(
&regs,
unix.SYS_CLONE,
arch.SyscallArgument{Value: uintptr(unix.SIGKILL | unix.CLONE_FILES)},
2018-04-27 10:37:02 -07:00
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: 0})
if err != nil {
return nil, fmt.Errorf("creating stub process: %v", err)
2018-04-27 10:37:02 -07:00
}
// Wait for child to enter group-stop, so we don't stop its
// bootstrapping work with t.attach below.
//
// We unfortunately don't have a handy part of memory to write the wait
// status. If the wait succeeds, we'll assume that it was the SIGSTOP.
// If the child actually exited, the attach below will fail.
_, err = t.syscallIgnoreInterrupt(
&t.initRegs,
unix.SYS_WAIT4,
2018-04-27 10:37:02 -07:00
arch.SyscallArgument{Value: uintptr(pid)},
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: unix.WALL | unix.WUNTRACED},
2018-04-27 10:37:02 -07:00
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: 0},
arch.SyscallArgument{Value: 0})
if err != nil {
return nil, fmt.Errorf("waiting on stub process: %v", err)
2018-04-27 10:37:02 -07:00
}
childT := &thread{
tgid: int32(pid),
tid: int32(pid),
cpu: ^uint32(0),
}
childT.attach()
return childT, nil
}