mirror of
https://github.com/netbirdio/gvisor.git
synced 2026-05-22 17:12:49 -07:00
make connect(2) fail when dest is unreachable
Previously, ICMP destination unreachable datagrams were ignored by TCP endpoints. This caused connect to hang when an intermediate router couldn't find a route to the host. This manifested as a Kokoro error when Docker IPv6 was enabled. The Ruby image test would try to install the sinatra gem and hang indefinitely attempting to use an IPv6 address. Fixes #3079.
This commit is contained in:
@@ -72,6 +72,7 @@ const (
|
||||
// Values for ICMP code as defined in RFC 792.
|
||||
const (
|
||||
ICMPv4TTLExceeded = 0
|
||||
ICMPv4HostUnreachable = 1
|
||||
ICMPv4PortUnreachable = 3
|
||||
ICMPv4FragmentationNeeded = 4
|
||||
)
|
||||
|
||||
@@ -110,9 +110,16 @@ const (
|
||||
ICMPv6RedirectMsg ICMPv6Type = 137
|
||||
)
|
||||
|
||||
// Values for ICMP code as defined in RFC 4443.
|
||||
// Values for ICMP destination unreachable code as defined in RFC 4443 section
|
||||
// 3.1.
|
||||
const (
|
||||
ICMPv6PortUnreachable = 4
|
||||
ICMPv6NetworkUnreachable = 0
|
||||
ICMPv6Prohibited = 1
|
||||
ICMPv6BeyondScope = 2
|
||||
ICMPv6AddressUnreachable = 3
|
||||
ICMPv6PortUnreachable = 4
|
||||
ICMPv6Policy = 5
|
||||
ICMPv6RejectRoute = 6
|
||||
)
|
||||
|
||||
// Type is the ICMP type field.
|
||||
|
||||
@@ -129,6 +129,9 @@ func (e *endpoint) handleICMP(r *stack.Route, pkt *stack.PacketBuffer) {
|
||||
|
||||
pkt.Data.TrimFront(header.ICMPv4MinimumSize)
|
||||
switch h.Code() {
|
||||
case header.ICMPv4HostUnreachable:
|
||||
e.handleControl(stack.ControlNoRoute, 0, pkt)
|
||||
|
||||
case header.ICMPv4PortUnreachable:
|
||||
e.handleControl(stack.ControlPortUnreachable, 0, pkt)
|
||||
|
||||
|
||||
@@ -128,6 +128,8 @@ func (e *endpoint) handleICMP(r *stack.Route, pkt *stack.PacketBuffer, hasFragme
|
||||
}
|
||||
pkt.Data.TrimFront(header.ICMPv6DstUnreachableMinimumSize)
|
||||
switch header.ICMPv6(hdr).Code() {
|
||||
case header.ICMPv6NetworkUnreachable:
|
||||
e.handleControl(stack.ControlNetworkUnreachable, 0, pkt)
|
||||
case header.ICMPv6PortUnreachable:
|
||||
e.handleControl(stack.ControlPortUnreachable, 0, pkt)
|
||||
}
|
||||
|
||||
@@ -52,8 +52,11 @@ type TransportEndpointID struct {
|
||||
type ControlType int
|
||||
|
||||
// The following are the allowed values for ControlType values.
|
||||
// TODO(http://gvisor.dev/issue/3210): Support time exceeded messages.
|
||||
const (
|
||||
ControlPacketTooBig ControlType = iota
|
||||
ControlNetworkUnreachable ControlType = iota
|
||||
ControlNoRoute
|
||||
ControlPacketTooBig
|
||||
ControlPortUnreachable
|
||||
ControlUnknown
|
||||
)
|
||||
|
||||
@@ -490,6 +490,9 @@ func (h *handshake) resolveRoute() *tcpip.Error {
|
||||
<-h.ep.undrain
|
||||
h.ep.mu.Lock()
|
||||
}
|
||||
if n¬ifyError != 0 {
|
||||
return h.ep.takeLastError()
|
||||
}
|
||||
}
|
||||
|
||||
// Wait for notification.
|
||||
@@ -616,6 +619,9 @@ func (h *handshake) execute() *tcpip.Error {
|
||||
<-h.ep.undrain
|
||||
h.ep.mu.Lock()
|
||||
}
|
||||
if n¬ifyError != 0 {
|
||||
return h.ep.takeLastError()
|
||||
}
|
||||
|
||||
case wakerForNewSegment:
|
||||
if err := h.processSegments(); err != nil {
|
||||
|
||||
@@ -1209,6 +1209,14 @@ func (e *endpoint) SetOwner(owner tcpip.PacketOwner) {
|
||||
e.owner = owner
|
||||
}
|
||||
|
||||
func (e *endpoint) takeLastError() *tcpip.Error {
|
||||
e.lastErrorMu.Lock()
|
||||
defer e.lastErrorMu.Unlock()
|
||||
err := e.lastError
|
||||
e.lastError = nil
|
||||
return err
|
||||
}
|
||||
|
||||
// Read reads data from the endpoint.
|
||||
func (e *endpoint) Read(*tcpip.FullAddress) (buffer.View, tcpip.ControlMessages, *tcpip.Error) {
|
||||
e.LockUser()
|
||||
@@ -1956,11 +1964,7 @@ func (e *endpoint) GetSockOptInt(opt tcpip.SockOptInt) (int, *tcpip.Error) {
|
||||
func (e *endpoint) GetSockOpt(opt interface{}) *tcpip.Error {
|
||||
switch o := opt.(type) {
|
||||
case tcpip.ErrorOption:
|
||||
e.lastErrorMu.Lock()
|
||||
err := e.lastError
|
||||
e.lastError = nil
|
||||
e.lastErrorMu.Unlock()
|
||||
return err
|
||||
return e.takeLastError()
|
||||
|
||||
case *tcpip.BindToDeviceOption:
|
||||
e.LockUser()
|
||||
@@ -2546,6 +2550,18 @@ func (e *endpoint) HandleControlPacket(id stack.TransportEndpointID, typ stack.C
|
||||
e.sndBufMu.Unlock()
|
||||
|
||||
e.notifyProtocolGoroutine(notifyMTUChanged)
|
||||
|
||||
case stack.ControlNoRoute:
|
||||
e.lastErrorMu.Lock()
|
||||
e.lastError = tcpip.ErrNoRoute
|
||||
e.lastErrorMu.Unlock()
|
||||
e.notifyProtocolGoroutine(notifyError)
|
||||
|
||||
case stack.ControlNetworkUnreachable:
|
||||
e.lastErrorMu.Lock()
|
||||
e.lastError = tcpip.ErrNetworkUnreachable
|
||||
e.lastErrorMu.Unlock()
|
||||
e.notifyProtocolGoroutine(notifyError)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -87,7 +87,6 @@ func (c *Container) doExec(ctx context.Context, r ExecOpts, args []string) (Proc
|
||||
execid: resp.ID,
|
||||
conn: hijack,
|
||||
}, nil
|
||||
|
||||
}
|
||||
|
||||
func (c *Container) execConfig(r ExecOpts, cmd []string) types.ExecConfig {
|
||||
|
||||
@@ -280,11 +280,13 @@ func TestOne(t *testing.T) {
|
||||
}
|
||||
|
||||
// Because the Linux kernel receives the SYN-ACK but didn't send the SYN it
|
||||
// will issue a RST. To prevent this IPtables can be used to filter out all
|
||||
// will issue an RST. To prevent this IPtables can be used to filter out all
|
||||
// incoming packets. The raw socket that packetimpact tests use will still see
|
||||
// everything.
|
||||
if logs, err := testbench.Exec(ctx, dockerutil.ExecOpts{}, "iptables", "-A", "INPUT", "-i", testNetDev, "-j", "DROP"); err != nil {
|
||||
t.Fatalf("unable to Exec iptables on container %s: %s, logs from testbench:\n%s", testbench.Name, err, logs)
|
||||
for _, bin := range []string{"iptables", "ip6tables"} {
|
||||
if logs, err := testbench.Exec(ctx, dockerutil.ExecOpts{}, bin, "-A", "INPUT", "-i", testNetDev, "-p", "tcp", "-j", "DROP"); err != nil {
|
||||
t.Fatalf("unable to Exec %s on container %s: %s, logs from testbench:\n%s", bin, testbench.Name, err, logs)
|
||||
}
|
||||
}
|
||||
|
||||
// FIXME(b/156449515): Some piece of the system has a race. The old
|
||||
|
||||
@@ -41,7 +41,8 @@ func portFromSockaddr(sa unix.Sockaddr) (uint16, error) {
|
||||
return 0, fmt.Errorf("sockaddr type %T does not contain port", sa)
|
||||
}
|
||||
|
||||
// pickPort makes a new socket and returns the socket FD and port. The domain should be AF_INET or AF_INET6. The caller must close the FD when done with
|
||||
// pickPort makes a new socket and returns the socket FD and port. The domain
|
||||
// should be AF_INET or AF_INET6. The caller must close the FD when done with
|
||||
// the port if there is no error.
|
||||
func pickPort(domain, typ int) (fd int, port uint16, err error) {
|
||||
fd, err = unix.Socket(domain, typ, 0)
|
||||
@@ -1061,3 +1062,58 @@ func (conn *UDPIPv6) Close() {
|
||||
func (conn *UDPIPv6) Drain() {
|
||||
conn.sniffer.Drain()
|
||||
}
|
||||
|
||||
// TCPIPv6 maintains the state for all the layers in a TCP/IPv6 connection.
|
||||
type TCPIPv6 Connection
|
||||
|
||||
// NewTCPIPv6 creates a new TCPIPv6 connection with reasonable defaults.
|
||||
func NewTCPIPv6(t *testing.T, outgoingTCP, incomingTCP TCP) TCPIPv6 {
|
||||
etherState, err := newEtherState(Ether{}, Ether{})
|
||||
if err != nil {
|
||||
t.Fatalf("can't make etherState: %s", err)
|
||||
}
|
||||
ipv6State, err := newIPv6State(IPv6{}, IPv6{})
|
||||
if err != nil {
|
||||
t.Fatalf("can't make ipv6State: %s", err)
|
||||
}
|
||||
tcpState, err := newTCPState(unix.AF_INET6, outgoingTCP, incomingTCP)
|
||||
if err != nil {
|
||||
t.Fatalf("can't make tcpState: %s", err)
|
||||
}
|
||||
injector, err := NewInjector(t)
|
||||
if err != nil {
|
||||
t.Fatalf("can't make injector: %s", err)
|
||||
}
|
||||
sniffer, err := NewSniffer(t)
|
||||
if err != nil {
|
||||
t.Fatalf("can't make sniffer: %s", err)
|
||||
}
|
||||
|
||||
return TCPIPv6{
|
||||
layerStates: []layerState{etherState, ipv6State, tcpState},
|
||||
injector: injector,
|
||||
sniffer: sniffer,
|
||||
t: t,
|
||||
}
|
||||
}
|
||||
|
||||
func (conn *TCPIPv6) SrcPort() uint16 {
|
||||
state := conn.layerStates[2].(*tcpState)
|
||||
return *state.out.SrcPort
|
||||
}
|
||||
|
||||
// ExpectData is a convenient method that expects a Layer and the Layer after
|
||||
// it. If it doens't arrive in time, it returns nil.
|
||||
func (conn *TCPIPv6) ExpectData(tcp *TCP, payload *Payload, timeout time.Duration) (Layers, error) {
|
||||
expected := make([]Layer, len(conn.layerStates))
|
||||
expected[len(expected)-1] = tcp
|
||||
if payload != nil {
|
||||
expected = append(expected, payload)
|
||||
}
|
||||
return (*Connection)(conn).ExpectFrame(expected, timeout)
|
||||
}
|
||||
|
||||
// Close frees associated resources held by the TCPIPv6 connection.
|
||||
func (conn *TCPIPv6) Close() {
|
||||
(*Connection)(conn).Close()
|
||||
}
|
||||
|
||||
@@ -805,7 +805,11 @@ func (l *ICMPv6) ToBytes() ([]byte, error) {
|
||||
// We need to search forward to find the IPv6 header.
|
||||
for prev := l.Prev(); prev != nil; prev = prev.Prev() {
|
||||
if ipv6, ok := prev.(*IPv6); ok {
|
||||
h.SetChecksum(header.ICMPv6Checksum(h, *ipv6.SrcAddr, *ipv6.DstAddr, buffer.VectorisedView{}))
|
||||
payload, err := payload(l)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
h.SetChecksum(header.ICMPv6Checksum(h, *ipv6.SrcAddr, *ipv6.DstAddr, payload))
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
@@ -219,6 +219,16 @@ packetimpact_go_test(
|
||||
],
|
||||
)
|
||||
|
||||
packetimpact_go_test(
|
||||
name = "tcp_network_unreachable",
|
||||
srcs = ["tcp_network_unreachable_test.go"],
|
||||
deps = [
|
||||
"//pkg/tcpip/header",
|
||||
"//test/packetimpact/testbench",
|
||||
"@org_golang_x_sys//unix:go_default_library",
|
||||
],
|
||||
)
|
||||
|
||||
packetimpact_go_test(
|
||||
name = "tcp_cork_mss",
|
||||
srcs = ["tcp_cork_mss_test.go"],
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
// Copyright 2020 The gVisor Authors.
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
package tcp_synsent_reset_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"flag"
|
||||
"net"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
"gvisor.dev/gvisor/pkg/tcpip/header"
|
||||
"gvisor.dev/gvisor/test/packetimpact/testbench"
|
||||
)
|
||||
|
||||
func init() {
|
||||
testbench.RegisterFlags(flag.CommandLine)
|
||||
}
|
||||
|
||||
// TestTCPSynSentUnreachable verifies that TCP connections fail immediately when
|
||||
// an ICMP destination unreachable message is sent in response to the inital
|
||||
// SYN.
|
||||
func TestTCPSynSentUnreachable(t *testing.T) {
|
||||
// Create the DUT and connection.
|
||||
dut := testbench.NewDUT(t)
|
||||
defer dut.TearDown()
|
||||
clientFD, clientPort := dut.CreateBoundSocket(unix.SOCK_STREAM|unix.SOCK_NONBLOCK, unix.IPPROTO_TCP, net.ParseIP(testbench.RemoteIPv4))
|
||||
port := uint16(9001)
|
||||
conn := testbench.NewTCPIPv4(t, testbench.TCP{SrcPort: &port, DstPort: &clientPort}, testbench.TCP{SrcPort: &clientPort, DstPort: &port})
|
||||
defer conn.Close()
|
||||
|
||||
// Bring the DUT to SYN-SENT state with a non-blocking connect.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), testbench.RPCTimeout)
|
||||
defer cancel()
|
||||
sa := unix.SockaddrInet4{Port: int(port)}
|
||||
copy(sa.Addr[:], net.IP(net.ParseIP(testbench.LocalIPv4)).To4())
|
||||
if _, err := dut.ConnectWithErrno(ctx, clientFD, &sa); err != syscall.Errno(unix.EINPROGRESS) {
|
||||
t.Errorf("expected connect to fail with EINPROGRESS, but got %v", err)
|
||||
}
|
||||
|
||||
// Get the SYN.
|
||||
tcpLayers, err := conn.ExpectData(&testbench.TCP{Flags: testbench.Uint8(header.TCPFlagSyn)}, nil, time.Second)
|
||||
if err != nil {
|
||||
t.Fatalf("expected SYN: %s", err)
|
||||
}
|
||||
|
||||
// Send a host unreachable message.
|
||||
rawConn := (*testbench.Connection)(&conn)
|
||||
layers := rawConn.CreateFrame(nil)
|
||||
layers = layers[:len(layers)-1]
|
||||
const ipLayer = 1
|
||||
const tcpLayer = ipLayer + 1
|
||||
ip, ok := tcpLayers[ipLayer].(*testbench.IPv4)
|
||||
if !ok {
|
||||
t.Fatalf("expected %s to be IPv4", tcpLayers[ipLayer])
|
||||
}
|
||||
tcp, ok := tcpLayers[tcpLayer].(*testbench.TCP)
|
||||
if !ok {
|
||||
t.Fatalf("expected %s to be TCP", tcpLayers[tcpLayer])
|
||||
}
|
||||
var icmpv4 testbench.ICMPv4 = testbench.ICMPv4{Type: testbench.ICMPv4Type(header.ICMPv4DstUnreachable), Code: testbench.Uint8(header.ICMPv4HostUnreachable)}
|
||||
layers = append(layers, &icmpv4, ip, tcp)
|
||||
rawConn.SendFrameStateless(layers)
|
||||
|
||||
if _, err = dut.ConnectWithErrno(ctx, clientFD, &sa); err != syscall.Errno(unix.EHOSTUNREACH) {
|
||||
t.Errorf("expected connect to fail with EHOSTUNREACH, but got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTCPSynSentUnreachable6 verifies that TCP connections fail immediately when
|
||||
// an ICMP destination unreachable message is sent in response to the inital
|
||||
// SYN.
|
||||
func TestTCPSynSentUnreachable6(t *testing.T) {
|
||||
// Create the DUT and connection.
|
||||
dut := testbench.NewDUT(t)
|
||||
defer dut.TearDown()
|
||||
clientFD, clientPort := dut.CreateBoundSocket(unix.SOCK_STREAM|unix.SOCK_NONBLOCK, unix.IPPROTO_TCP, net.ParseIP(testbench.RemoteIPv6))
|
||||
conn := testbench.NewTCPIPv6(t, testbench.TCP{DstPort: &clientPort}, testbench.TCP{SrcPort: &clientPort})
|
||||
defer conn.Close()
|
||||
|
||||
// Bring the DUT to SYN-SENT state with a non-blocking connect.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), testbench.RPCTimeout)
|
||||
defer cancel()
|
||||
sa := unix.SockaddrInet6{
|
||||
Port: int(conn.SrcPort()),
|
||||
ZoneId: uint32(testbench.RemoteInterfaceID),
|
||||
}
|
||||
copy(sa.Addr[:], net.IP(net.ParseIP(testbench.LocalIPv6)).To16())
|
||||
if _, err := dut.ConnectWithErrno(ctx, clientFD, &sa); err != syscall.Errno(unix.EINPROGRESS) {
|
||||
t.Errorf("expected connect to fail with EINPROGRESS, but got %v", err)
|
||||
}
|
||||
|
||||
// Get the SYN.
|
||||
tcpLayers, err := conn.ExpectData(&testbench.TCP{Flags: testbench.Uint8(header.TCPFlagSyn)}, nil, time.Second)
|
||||
if err != nil {
|
||||
t.Fatalf("expected SYN: %s", err)
|
||||
}
|
||||
|
||||
// Send a host unreachable message.
|
||||
rawConn := (*testbench.Connection)(&conn)
|
||||
layers := rawConn.CreateFrame(nil)
|
||||
layers = layers[:len(layers)-1]
|
||||
const ipLayer = 1
|
||||
const tcpLayer = ipLayer + 1
|
||||
ip, ok := tcpLayers[ipLayer].(*testbench.IPv6)
|
||||
if !ok {
|
||||
t.Fatalf("expected %s to be IPv6", tcpLayers[ipLayer])
|
||||
}
|
||||
tcp, ok := tcpLayers[tcpLayer].(*testbench.TCP)
|
||||
if !ok {
|
||||
t.Fatalf("expected %s to be TCP", tcpLayers[tcpLayer])
|
||||
}
|
||||
var icmpv6 testbench.ICMPv6 = testbench.ICMPv6{
|
||||
Type: testbench.ICMPv6Type(header.ICMPv6DstUnreachable),
|
||||
Code: testbench.Uint8(header.ICMPv6NetworkUnreachable),
|
||||
// Per RFC 4443 3.1, the payload contains 4 zeroed bytes.
|
||||
Payload: []byte{0, 0, 0, 0},
|
||||
}
|
||||
layers = append(layers, &icmpv6, ip, tcp)
|
||||
rawConn.SendFrameStateless(layers)
|
||||
|
||||
if _, err = dut.ConnectWithErrno(ctx, clientFD, &sa); err != syscall.Errno(unix.ENETUNREACH) {
|
||||
t.Errorf("expected connect to fail with ENETUNREACH, but got %v", err)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user