diff --git a/.gitignore b/.gitignore index 15f2f702..5498c8b1 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,8 @@ kernel/*.old test/shim/testdata/testbin test/stress/testdata/testbin .task + +# Rust build output. Cargo.lock is intentionally committed: crates/pause +# builds an executable, not a library. +crates/*/target/ +**/*.rs.bk diff --git a/Dockerfile b/Dockerfile index a456cb2a..fa339413 100644 --- a/Dockerfile +++ b/Dockerfile @@ -224,6 +224,34 @@ WORKDIR /usr/src/crun ARG TARGETARCH RUN mkdir /build && wget -O /build/crun https://github.com/containers/crun/releases/download/1.24/crun-1.24-linux-${TARGETARCH}-disable-systemd +# Anchor process for guest PID namespaces (see crates/pause). Built against +# musl so it is fully static: it runs as PID 1 of a namespace in a rootfs that +# carries no dynamic loader of its own. +FROM "${RUST_IMAGE}" AS pause-build +WORKDIR /usr/src/pause + +ARG TARGETARCH +RUN <&2; exit 1 ;; + esac + echo "${triple}" > /rust-target + rustup target add "${triple}" +EOT + +COPY crates/pause/ . +RUN --mount=type=cache,target=/usr/local/cargo/registry,id=pause-cargo-registry \ + --mount=type=cache,target=/usr/src/pause/target,id=pause-build-${TARGETARCH} < sysctls = 3; +} diff --git a/api/services/sharedresources/v1/sharedresources.pb.go b/api/services/sharedresources/v1/sharedresources.pb.go new file mode 100644 index 00000000..e94e8853 --- /dev/null +++ b/api/services/sharedresources/v1/sharedresources.pb.go @@ -0,0 +1,599 @@ +// +//Copyright The containerd Authors. +// +//Licensed under the Apache License, Version 2.0 (the "License"); +//you may not use this file except in compliance with the License. +//You may obtain a copy of the License at +// +//http://www.apache.org/licenses/LICENSE-2.0 +// +//Unless required by applicable law or agreed to in writing, software +//distributed under the License is distributed on an "AS IS" BASIS, +//WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +//See the License for the specific language governing permissions and +//limitations under the License. + +// Code generated by protoc-gen-go. DO NOT EDIT. +// versions: +// protoc-gen-go v1.28.1 +// protoc (unknown) +// source: proto/nerdbox/services/sharedresources/v1/sharedresources.proto + +package sharedresources + +import ( + protoreflect "google.golang.org/protobuf/reflect/protoreflect" + protoimpl "google.golang.org/protobuf/runtime/protoimpl" + reflect "reflect" + sync "sync" +) + +const ( + // Verify that this generated code is sufficiently up-to-date. + _ = protoimpl.EnforceVersion(20 - protoimpl.MinVersion) + // Verify that runtime/protoimpl is sufficiently up-to-date. + _ = protoimpl.EnforceVersion(protoimpl.MaxVersion - 20) +) + +// Type identifies a kind of guest-side shared resource: created once per +// (group id, type) and addressed by a guest path thereafter. Most values are +// Linux namespace types; TYPE_DEVSHM is not — see its own comment. +type Type int32 + +const ( + Type_TYPE_UNSPECIFIED Type = 0 + Type_TYPE_NAMESPACE_IPC Type = 1 + Type_TYPE_NAMESPACE_PID Type = 2 + Type_TYPE_NAMESPACE_NETWORK Type = 3 + // TYPE_DEVSHM is not an OCI namespace: it is a per-group tmpfs, + // mounted once and reused, that sharing containers bind-mount their + // /dev/shm onto so writes through it are real shared memory (guest + // RAM), not just a shared IPC namespace identifier. It reuses this + // same create-once/reuse/path-return mechanism because the shape of + // the problem — and the "only pay for it if actually requested" + // requirement — is identical to the namespace types. + Type_TYPE_DEVSHM Type = 4 + // TYPE_NAMESPACE_UTS is created the same way as TYPE_NAMESPACE_IPC + // (unshare + bind-mount, no anchor process). Callers do not need to + // coordinate a hostname through this API: an OCI runtime setting + // Spec.Hostname while joining an existing (not freshly created) UTS + // namespace calls sethostname(2) after joining it, updating the + // namespace for every container sharing it, and leaves it alone when + // Hostname is empty — so ordinary per-container spec fields already + // give "last write wins, empty means no opinion" for free. + Type_TYPE_NAMESPACE_UTS Type = 5 +) + +// Enum value maps for Type. +var ( + Type_name = map[int32]string{ + 0: "TYPE_UNSPECIFIED", + 1: "TYPE_NAMESPACE_IPC", + 2: "TYPE_NAMESPACE_PID", + 3: "TYPE_NAMESPACE_NETWORK", + 4: "TYPE_DEVSHM", + 5: "TYPE_NAMESPACE_UTS", + } + Type_value = map[string]int32{ + "TYPE_UNSPECIFIED": 0, + "TYPE_NAMESPACE_IPC": 1, + "TYPE_NAMESPACE_PID": 2, + "TYPE_NAMESPACE_NETWORK": 3, + "TYPE_DEVSHM": 4, + "TYPE_NAMESPACE_UTS": 5, + } +) + +func (x Type) Enum() *Type { + p := new(Type) + *p = x + return p +} + +func (x Type) String() string { + return protoimpl.X.EnumStringOf(x.Descriptor(), protoreflect.EnumNumber(x)) +} + +func (Type) Descriptor() protoreflect.EnumDescriptor { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_enumTypes[0].Descriptor() +} + +func (Type) Type() protoreflect.EnumType { + return &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_enumTypes[0] +} + +func (x Type) Number() protoreflect.EnumNumber { + return protoreflect.EnumNumber(x) +} + +// Deprecated: Use Type.Descriptor instead. +func (Type) EnumDescriptor() ([]byte, []int) { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP(), []int{0} +} + +// SharedResource is a created resource and the guest path it is pinned at. +type SharedResource struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + Type Type `protobuf:"varint,1,opt,name=type,proto3,enum=containerd.vminitd.services.sharedresources.v1.Type" json:"type,omitempty"` + // path is a guest-side path. For a namespace type, it is a bind-mount + // path suitable for use directly as an OCI runtime spec + // LinuxNamespace.Path. For TYPE_DEVSHM, it is a directory suitable for + // use as an OCI runtime spec Mount.Source with Type "bind". + Path string `protobuf:"bytes,2,opt,name=path,proto3" json:"path,omitempty"` +} + +func (x *SharedResource) Reset() { + *x = SharedResource{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[0] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *SharedResource) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*SharedResource) ProtoMessage() {} + +func (x *SharedResource) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[0] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use SharedResource.ProtoReflect.Descriptor instead. +func (*SharedResource) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP(), []int{0} +} + +func (x *SharedResource) GetType() Type { + if x != nil { + return x.Type + } + return Type_TYPE_UNSPECIFIED +} + +func (x *SharedResource) GetPath() string { + if x != nil { + return x.Path + } + return "" +} + +type CreateRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + // id groups the resources being created. Resources are created once + // per (id, type) and reused, so repeating a request returns the paths + // of the resources already created rather than creating new ones. + ID string `protobuf:"bytes,1,opt,name=id,proto3" json:"id,omitempty"` + // types are the resource types to create. Types already created for + // id are returned without being recreated. + Types []Type `protobuf:"varint,2,rep,packed,name=types,proto3,enum=containerd.vminitd.services.sharedresources.v1.Type" json:"types,omitempty"` + // dev_shm_size_bytes is the tmpfs size to use when creating a + // TYPE_DEVSHM resource for id for the first time. Ignored for every + // other type, and ignored if a TYPE_DEVSHM resource for id already + // exists (the size used by whichever caller created it first wins; + // callers within one sandbox are expected to agree on it, since + // containerd sends every member container of a pod the same /dev/shm + // size). A value that is zero or absent falls back to a guest-side + // default. + DevShmSizeBytes int64 `protobuf:"varint,3,opt,name=dev_shm_size_bytes,json=devShmSizeBytes,proto3" json:"dev_shm_size_bytes,omitempty"` +} + +func (x *CreateRequest) Reset() { + *x = CreateRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[1] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CreateRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CreateRequest) ProtoMessage() {} + +func (x *CreateRequest) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[1] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CreateRequest.ProtoReflect.Descriptor instead. +func (*CreateRequest) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP(), []int{1} +} + +func (x *CreateRequest) GetID() string { + if x != nil { + return x.ID + } + return "" +} + +func (x *CreateRequest) GetTypes() []Type { + if x != nil { + return x.Types + } + return nil +} + +func (x *CreateRequest) GetDevShmSizeBytes() int64 { + if x != nil { + return x.DevShmSizeBytes + } + return 0 +} + +type CreateResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + // resources contains one entry per requested type. + Resources []*SharedResource `protobuf:"bytes,1,rep,name=resources,proto3" json:"resources,omitempty"` +} + +func (x *CreateResponse) Reset() { + *x = CreateResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[2] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CreateResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CreateResponse) ProtoMessage() {} + +func (x *CreateResponse) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[2] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CreateResponse.ProtoReflect.Descriptor instead. +func (*CreateResponse) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP(), []int{2} +} + +func (x *CreateResponse) GetResources() []*SharedResource { + if x != nil { + return x.Resources + } + return nil +} + +type DeleteRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + // id is the group to delete resources from. + ID string `protobuf:"bytes,1,opt,name=id,proto3" json:"id,omitempty"` + // types are the resource types to delete. An empty list deletes every + // resource belonging to id. + // + // Deleting a namespace that containers are still using is a caller + // error: for a PID namespace it kills the anchor process, which makes + // the kernel tear the namespace down and kill everything in it. + Types []Type `protobuf:"varint,2,rep,packed,name=types,proto3,enum=containerd.vminitd.services.sharedresources.v1.Type" json:"types,omitempty"` +} + +func (x *DeleteRequest) Reset() { + *x = DeleteRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[3] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *DeleteRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*DeleteRequest) ProtoMessage() {} + +func (x *DeleteRequest) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[3] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use DeleteRequest.ProtoReflect.Descriptor instead. +func (*DeleteRequest) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP(), []int{3} +} + +func (x *DeleteRequest) GetID() string { + if x != nil { + return x.ID + } + return "" +} + +func (x *DeleteRequest) GetTypes() []Type { + if x != nil { + return x.Types + } + return nil +} + +type DeleteResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields +} + +func (x *DeleteResponse) Reset() { + *x = DeleteResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[4] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *DeleteResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*DeleteResponse) ProtoMessage() {} + +func (x *DeleteResponse) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[4] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use DeleteResponse.ProtoReflect.Descriptor instead. +func (*DeleteResponse) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP(), []int{4} +} + +var File_proto_nerdbox_services_sharedresources_v1_sharedresources_proto protoreflect.FileDescriptor + +var file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDesc = []byte{ + 0x0a, 0x3f, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x2f, 0x6e, 0x65, 0x72, 0x64, 0x62, 0x6f, 0x78, 0x2f, + 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2f, 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, + 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2f, 0x76, 0x31, 0x2f, 0x73, 0x68, 0x61, 0x72, + 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x70, 0x72, 0x6f, 0x74, + 0x6f, 0x12, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, 0x6d, + 0x69, 0x6e, 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, 0x73, + 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x76, + 0x31, 0x22, 0x6e, 0x0a, 0x0e, 0x53, 0x68, 0x61, 0x72, 0x65, 0x64, 0x52, 0x65, 0x73, 0x6f, 0x75, + 0x72, 0x63, 0x65, 0x12, 0x48, 0x0a, 0x04, 0x74, 0x79, 0x70, 0x65, 0x18, 0x01, 0x20, 0x01, 0x28, + 0x0e, 0x32, 0x34, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, + 0x6d, 0x69, 0x6e, 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, + 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, + 0x76, 0x31, 0x2e, 0x54, 0x79, 0x70, 0x65, 0x52, 0x04, 0x74, 0x79, 0x70, 0x65, 0x12, 0x12, 0x0a, + 0x04, 0x70, 0x61, 0x74, 0x68, 0x18, 0x02, 0x20, 0x01, 0x28, 0x09, 0x52, 0x04, 0x70, 0x61, 0x74, + 0x68, 0x22, 0x98, 0x01, 0x0a, 0x0d, 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x52, 0x65, 0x71, 0x75, + 0x65, 0x73, 0x74, 0x12, 0x0e, 0x0a, 0x02, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, + 0x02, 0x69, 0x64, 0x12, 0x4a, 0x0a, 0x05, 0x74, 0x79, 0x70, 0x65, 0x73, 0x18, 0x02, 0x20, 0x03, + 0x28, 0x0e, 0x32, 0x34, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, + 0x76, 0x6d, 0x69, 0x6e, 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, + 0x2e, 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, + 0x2e, 0x76, 0x31, 0x2e, 0x54, 0x79, 0x70, 0x65, 0x52, 0x05, 0x74, 0x79, 0x70, 0x65, 0x73, 0x12, + 0x2b, 0x0a, 0x12, 0x64, 0x65, 0x76, 0x5f, 0x73, 0x68, 0x6d, 0x5f, 0x73, 0x69, 0x7a, 0x65, 0x5f, + 0x62, 0x79, 0x74, 0x65, 0x73, 0x18, 0x03, 0x20, 0x01, 0x28, 0x03, 0x52, 0x0f, 0x64, 0x65, 0x76, + 0x53, 0x68, 0x6d, 0x53, 0x69, 0x7a, 0x65, 0x42, 0x79, 0x74, 0x65, 0x73, 0x22, 0x6e, 0x0a, 0x0e, + 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x5c, + 0x0a, 0x09, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x18, 0x01, 0x20, 0x03, 0x28, + 0x0b, 0x32, 0x3e, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, + 0x6d, 0x69, 0x6e, 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, + 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, + 0x76, 0x31, 0x2e, 0x53, 0x68, 0x61, 0x72, 0x65, 0x64, 0x52, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, + 0x65, 0x52, 0x09, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x22, 0x6b, 0x0a, 0x0d, + 0x44, 0x65, 0x6c, 0x65, 0x74, 0x65, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x12, 0x0e, 0x0a, + 0x02, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x02, 0x69, 0x64, 0x12, 0x4a, 0x0a, + 0x05, 0x74, 0x79, 0x70, 0x65, 0x73, 0x18, 0x02, 0x20, 0x03, 0x28, 0x0e, 0x32, 0x34, 0x2e, 0x63, + 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, 0x6d, 0x69, 0x6e, 0x69, 0x74, + 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, 0x73, 0x68, 0x61, 0x72, 0x65, + 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x76, 0x31, 0x2e, 0x54, 0x79, + 0x70, 0x65, 0x52, 0x05, 0x74, 0x79, 0x70, 0x65, 0x73, 0x22, 0x10, 0x0a, 0x0e, 0x44, 0x65, 0x6c, + 0x65, 0x74, 0x65, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x2a, 0x91, 0x01, 0x0a, 0x04, + 0x54, 0x79, 0x70, 0x65, 0x12, 0x14, 0x0a, 0x10, 0x54, 0x59, 0x50, 0x45, 0x5f, 0x55, 0x4e, 0x53, + 0x50, 0x45, 0x43, 0x49, 0x46, 0x49, 0x45, 0x44, 0x10, 0x00, 0x12, 0x16, 0x0a, 0x12, 0x54, 0x59, + 0x50, 0x45, 0x5f, 0x4e, 0x41, 0x4d, 0x45, 0x53, 0x50, 0x41, 0x43, 0x45, 0x5f, 0x49, 0x50, 0x43, + 0x10, 0x01, 0x12, 0x16, 0x0a, 0x12, 0x54, 0x59, 0x50, 0x45, 0x5f, 0x4e, 0x41, 0x4d, 0x45, 0x53, + 0x50, 0x41, 0x43, 0x45, 0x5f, 0x50, 0x49, 0x44, 0x10, 0x02, 0x12, 0x1a, 0x0a, 0x16, 0x54, 0x59, + 0x50, 0x45, 0x5f, 0x4e, 0x41, 0x4d, 0x45, 0x53, 0x50, 0x41, 0x43, 0x45, 0x5f, 0x4e, 0x45, 0x54, + 0x57, 0x4f, 0x52, 0x4b, 0x10, 0x03, 0x12, 0x0f, 0x0a, 0x0b, 0x54, 0x59, 0x50, 0x45, 0x5f, 0x44, + 0x45, 0x56, 0x53, 0x48, 0x4d, 0x10, 0x04, 0x12, 0x16, 0x0a, 0x12, 0x54, 0x59, 0x50, 0x45, 0x5f, + 0x4e, 0x41, 0x4d, 0x45, 0x53, 0x50, 0x41, 0x43, 0x45, 0x5f, 0x55, 0x54, 0x53, 0x10, 0x05, 0x32, + 0xa5, 0x02, 0x0a, 0x0f, 0x53, 0x68, 0x61, 0x72, 0x65, 0x64, 0x52, 0x65, 0x73, 0x6f, 0x75, 0x72, + 0x63, 0x65, 0x73, 0x12, 0x87, 0x01, 0x0a, 0x06, 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x12, 0x3d, + 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, 0x6d, 0x69, 0x6e, + 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, 0x73, 0x68, 0x61, + 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x76, 0x31, 0x2e, + 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, 0x3e, 0x2e, + 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, 0x6d, 0x69, 0x6e, 0x69, + 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, 0x73, 0x68, 0x61, 0x72, + 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x76, 0x31, 0x2e, 0x43, + 0x72, 0x65, 0x61, 0x74, 0x65, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x87, 0x01, + 0x0a, 0x06, 0x44, 0x65, 0x6c, 0x65, 0x74, 0x65, 0x12, 0x3d, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, + 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, 0x6d, 0x69, 0x6e, 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, + 0x72, 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, + 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x76, 0x31, 0x2e, 0x44, 0x65, 0x6c, 0x65, 0x74, 0x65, + 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, 0x3e, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, + 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x76, 0x6d, 0x69, 0x6e, 0x69, 0x74, 0x64, 0x2e, 0x73, 0x65, 0x72, + 0x76, 0x69, 0x63, 0x65, 0x73, 0x2e, 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, + 0x75, 0x72, 0x63, 0x65, 0x73, 0x2e, 0x76, 0x31, 0x2e, 0x44, 0x65, 0x6c, 0x65, 0x74, 0x65, 0x52, + 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x42, 0x4f, 0x5a, 0x4d, 0x67, 0x69, 0x74, 0x68, 0x75, + 0x62, 0x2e, 0x63, 0x6f, 0x6d, 0x2f, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, + 0x2f, 0x6e, 0x65, 0x72, 0x64, 0x62, 0x6f, 0x78, 0x2f, 0x61, 0x70, 0x69, 0x2f, 0x73, 0x65, 0x72, + 0x76, 0x69, 0x63, 0x65, 0x73, 0x2f, 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, 0x65, 0x73, 0x6f, + 0x75, 0x72, 0x63, 0x65, 0x73, 0x2f, 0x76, 0x31, 0x3b, 0x73, 0x68, 0x61, 0x72, 0x65, 0x64, 0x72, + 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x62, 0x06, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x33, +} + +var ( + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescOnce sync.Once + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescData = file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDesc +) + +func file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescGZIP() []byte { + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescOnce.Do(func() { + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescData = protoimpl.X.CompressGZIP(file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescData) + }) + return file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDescData +} + +var file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_enumTypes = make([]protoimpl.EnumInfo, 1) +var file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes = make([]protoimpl.MessageInfo, 5) +var file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_goTypes = []interface{}{ + (Type)(0), // 0: containerd.vminitd.services.sharedresources.v1.Type + (*SharedResource)(nil), // 1: containerd.vminitd.services.sharedresources.v1.SharedResource + (*CreateRequest)(nil), // 2: containerd.vminitd.services.sharedresources.v1.CreateRequest + (*CreateResponse)(nil), // 3: containerd.vminitd.services.sharedresources.v1.CreateResponse + (*DeleteRequest)(nil), // 4: containerd.vminitd.services.sharedresources.v1.DeleteRequest + (*DeleteResponse)(nil), // 5: containerd.vminitd.services.sharedresources.v1.DeleteResponse +} +var file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_depIdxs = []int32{ + 0, // 0: containerd.vminitd.services.sharedresources.v1.SharedResource.type:type_name -> containerd.vminitd.services.sharedresources.v1.Type + 0, // 1: containerd.vminitd.services.sharedresources.v1.CreateRequest.types:type_name -> containerd.vminitd.services.sharedresources.v1.Type + 1, // 2: containerd.vminitd.services.sharedresources.v1.CreateResponse.resources:type_name -> containerd.vminitd.services.sharedresources.v1.SharedResource + 0, // 3: containerd.vminitd.services.sharedresources.v1.DeleteRequest.types:type_name -> containerd.vminitd.services.sharedresources.v1.Type + 2, // 4: containerd.vminitd.services.sharedresources.v1.SharedResources.Create:input_type -> containerd.vminitd.services.sharedresources.v1.CreateRequest + 4, // 5: containerd.vminitd.services.sharedresources.v1.SharedResources.Delete:input_type -> containerd.vminitd.services.sharedresources.v1.DeleteRequest + 3, // 6: containerd.vminitd.services.sharedresources.v1.SharedResources.Create:output_type -> containerd.vminitd.services.sharedresources.v1.CreateResponse + 5, // 7: containerd.vminitd.services.sharedresources.v1.SharedResources.Delete:output_type -> containerd.vminitd.services.sharedresources.v1.DeleteResponse + 6, // [6:8] is the sub-list for method output_type + 4, // [4:6] is the sub-list for method input_type + 4, // [4:4] is the sub-list for extension type_name + 4, // [4:4] is the sub-list for extension extendee + 0, // [0:4] is the sub-list for field type_name +} + +func init() { file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_init() } +func file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_init() { + if File_proto_nerdbox_services_sharedresources_v1_sharedresources_proto != nil { + return + } + if !protoimpl.UnsafeEnabled { + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[0].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*SharedResource); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[1].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CreateRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[2].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CreateResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[3].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*DeleteRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes[4].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*DeleteResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + } + type x struct{} + out := protoimpl.TypeBuilder{ + File: protoimpl.DescBuilder{ + GoPackagePath: reflect.TypeOf(x{}).PkgPath(), + RawDescriptor: file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDesc, + NumEnums: 1, + NumMessages: 5, + NumExtensions: 0, + NumServices: 1, + }, + GoTypes: file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_goTypes, + DependencyIndexes: file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_depIdxs, + EnumInfos: file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_enumTypes, + MessageInfos: file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_msgTypes, + }.Build() + File_proto_nerdbox_services_sharedresources_v1_sharedresources_proto = out.File + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_rawDesc = nil + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_goTypes = nil + file_proto_nerdbox_services_sharedresources_v1_sharedresources_proto_depIdxs = nil +} diff --git a/api/services/sharedresources/v1/sharedresources_ttrpc.pb.go b/api/services/sharedresources/v1/sharedresources_ttrpc.pb.go new file mode 100644 index 00000000..2de36d74 --- /dev/null +++ b/api/services/sharedresources/v1/sharedresources_ttrpc.pb.go @@ -0,0 +1,60 @@ +// Code generated by protoc-gen-go-ttrpc. DO NOT EDIT. +// source: proto/nerdbox/services/sharedresources/v1/sharedresources.proto +package sharedresources + +import ( + context "context" + ttrpc "github.com/containerd/ttrpc" +) + +type TTRPCSharedResourcesService interface { + Create(context.Context, *CreateRequest) (*CreateResponse, error) + Delete(context.Context, *DeleteRequest) (*DeleteResponse, error) +} + +func RegisterTTRPCSharedResourcesService(srv *ttrpc.Server, svc TTRPCSharedResourcesService) { + srv.RegisterService("containerd.vminitd.services.sharedresources.v1.SharedResources", &ttrpc.ServiceDesc{ + Methods: map[string]ttrpc.Method{ + "Create": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req CreateRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.Create(ctx, &req) + }, + "Delete": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req DeleteRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.Delete(ctx, &req) + }, + }, + }) +} + +type ttrpcsharedresourcesClient struct { + client *ttrpc.Client +} + +func NewTTRPCSharedResourcesClient(client *ttrpc.Client) TTRPCSharedResourcesService { + return &ttrpcsharedresourcesClient{ + client: client, + } +} + +func (c *ttrpcsharedresourcesClient) Create(ctx context.Context, req *CreateRequest) (*CreateResponse, error) { + var resp CreateResponse + if err := c.client.Call(ctx, "containerd.vminitd.services.sharedresources.v1.SharedResources", "Create", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsharedresourcesClient) Delete(ctx context.Context, req *DeleteRequest) (*DeleteResponse, error) { + var resp DeleteResponse + if err := c.client.Call(ctx, "containerd.vminitd.services.sharedresources.v1.SharedResources", "Delete", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} diff --git a/api/types/doc.go b/api/types/doc.go new file mode 100644 index 00000000..67e5de49 --- /dev/null +++ b/api/types/doc.go @@ -0,0 +1,21 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +// Package types defines hand-maintained, wire-compatible subsets of +// external types nerdbox decodes but does not want to import wholesale +// (e.g. CRIPodConfig, a subset of k8s.io/cri-api's PodSandboxConfig). See +// CRIPodConfig's doc comment in sandbox.proto for why. +package types diff --git a/api/types/sandbox.pb.go b/api/types/sandbox.pb.go new file mode 100644 index 00000000..96babd91 --- /dev/null +++ b/api/types/sandbox.pb.go @@ -0,0 +1,391 @@ +// +//Copyright The containerd Authors. +// +//Licensed under the Apache License, Version 2.0 (the "License"); +//you may not use this file except in compliance with the License. +//You may obtain a copy of the License at +// +//http://www.apache.org/licenses/LICENSE-2.0 +// +//Unless required by applicable law or agreed to in writing, software +//distributed under the License is distributed on an "AS IS" BASIS, +//WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +//See the License for the specific language governing permissions and +//limitations under the License. + +// Code generated by protoc-gen-go. DO NOT EDIT. +// versions: +// protoc-gen-go v1.28.1 +// protoc (unknown) +// source: proto/nerdbox/types/sandbox.proto + +package types + +import ( + protoreflect "google.golang.org/protobuf/reflect/protoreflect" + protoimpl "google.golang.org/protobuf/runtime/protoimpl" + reflect "reflect" + sync "sync" +) + +const ( + // Verify that this generated code is sufficiently up-to-date. + _ = protoimpl.EnforceVersion(20 - protoimpl.MinVersion) + // Verify that runtime/protoimpl is sufficiently up-to-date. + _ = protoimpl.EnforceVersion(protoimpl.MaxVersion - 20) +) + +// CRIPodConfig is a hand-maintained, wire-compatible *subset* of +// k8s.io/cri-api's runtime.v1.PodSandboxConfig (the type containerd's CRI +// plugin marshals into CreateSandboxRequest.Options — see +// internal/shim/task/podconfig.go), covering only the fields the shim +// actually reads: pod hostname, DNS config, and Linux sysctls. +// +// This intentionally does NOT import k8s.io/cri-api: that package pulls in +// gRPC unconditionally (its generated *_grpc.pb.go carries no build tag), +// which defeats the shim's "no_grpc" build tag for the sake of decoding +// three scalar-ish fields out of an otherwise-unused message. Protobuf's +// wire format tolerates this: an unrecognized field number is simply +// skipped on decode, so a real CRI PodSandboxConfig — which carries many +// more fields than these — decodes cleanly into this subset, silently +// dropping everything this shim doesn't need. +// +// Field numbers below (and every message's nesting level) are copied +// verbatim from the upstream message (CRI v1, github.com/kubernetes/cri-api, +// pkg/apis/runtime/v1/api.proto) and MUST continue to match it exactly: +// this only decodes correctly because the wire bytes line up field-for-field +// with upstream at every level, not just the top one. CRI v1 is a frozen, +// GA, append-only-evolution API (see kubernetes/cri-api's own compatibility +// policy), so existing field numbers are not expected to ever be reused or +// renumbered. Field numbers NOT listed here (e.g. PodSandboxConfig's own +// 1, 3, 5, 6, 7, 9, 10) are other upstream fields this shim has no use for; +// they are simply omitted rather than declared, and are skipped like any +// other unrecognized field on decode. +// +// Deliberately named and packaged differently than upstream's +// runtime.v1.PodSandboxConfig, rather than reusing that exact +// fully-qualified proto name: registering a second, different message +// under upstream's own name would panic at process init if this binary +// ever also imported the real k8s.io/cri-api (protobuf-go's global type +// registry rejects duplicate registrations for the same full name). +type CRIPodConfig struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + // Hostname of the sandbox (upstream field 2). + Hostname string `protobuf:"bytes,2,opt,name=hostname,proto3" json:"hostname,omitempty"` + // DNS config for the sandbox (upstream field 4). + DnsConfig *CRIDNSConfig `protobuf:"bytes,4,opt,name=dns_config,json=dnsConfig,proto3" json:"dns_config,omitempty"` + // Linux-specific pod sandbox config (upstream field 8). + Linux *CRILinuxPodSandboxConfig `protobuf:"bytes,8,opt,name=linux,proto3" json:"linux,omitempty"` +} + +func (x *CRIPodConfig) Reset() { + *x = CRIPodConfig{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_types_sandbox_proto_msgTypes[0] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CRIPodConfig) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CRIPodConfig) ProtoMessage() {} + +func (x *CRIPodConfig) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_types_sandbox_proto_msgTypes[0] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CRIPodConfig.ProtoReflect.Descriptor instead. +func (*CRIPodConfig) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_types_sandbox_proto_rawDescGZIP(), []int{0} +} + +func (x *CRIPodConfig) GetHostname() string { + if x != nil { + return x.Hostname + } + return "" +} + +func (x *CRIPodConfig) GetDnsConfig() *CRIDNSConfig { + if x != nil { + return x.DnsConfig + } + return nil +} + +func (x *CRIPodConfig) GetLinux() *CRILinuxPodSandboxConfig { + if x != nil { + return x.Linux + } + return nil +} + +// CRIDNSConfig is a wire-compatible subset of runtime.v1.DNSConfig, +// covering every field it has today (unlike CRIPodConfig, this one just +// happens to be complete). See CRIPodConfig's doc comment for why this is +// a hand-maintained copy rather than an import. +type CRIDNSConfig struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + // List of DNS servers of the cluster (upstream field 1). + Servers []string `protobuf:"bytes,1,rep,name=servers,proto3" json:"servers,omitempty"` + // List of DNS search domains of the cluster (upstream field 2). + Searches []string `protobuf:"bytes,2,rep,name=searches,proto3" json:"searches,omitempty"` + // List of DNS options; see resolv.conf(5) (upstream field 3). + Options []string `protobuf:"bytes,3,rep,name=options,proto3" json:"options,omitempty"` +} + +func (x *CRIDNSConfig) Reset() { + *x = CRIDNSConfig{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_types_sandbox_proto_msgTypes[1] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CRIDNSConfig) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CRIDNSConfig) ProtoMessage() {} + +func (x *CRIDNSConfig) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_types_sandbox_proto_msgTypes[1] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CRIDNSConfig.ProtoReflect.Descriptor instead. +func (*CRIDNSConfig) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_types_sandbox_proto_rawDescGZIP(), []int{1} +} + +func (x *CRIDNSConfig) GetServers() []string { + if x != nil { + return x.Servers + } + return nil +} + +func (x *CRIDNSConfig) GetSearches() []string { + if x != nil { + return x.Searches + } + return nil +} + +func (x *CRIDNSConfig) GetOptions() []string { + if x != nil { + return x.Options + } + return nil +} + +// CRILinuxPodSandboxConfig is a wire-compatible subset of +// runtime.v1.LinuxPodSandboxConfig, covering only the field this shim +// reads (pod-level sysctls). See CRIPodConfig's doc comment for why this +// is a hand-maintained copy rather than an import. +type CRILinuxPodSandboxConfig struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + // Sysctls to apply within the pod's containers (upstream field 3). + // Fields NOT listed here (upstream 1, 2, 4, 5) are omitted; see + // CRIPodConfig's doc comment. + Sysctls map[string]string `protobuf:"bytes,3,rep,name=sysctls,proto3" json:"sysctls,omitempty" protobuf_key:"bytes,1,opt,name=key,proto3" protobuf_val:"bytes,2,opt,name=value,proto3"` +} + +func (x *CRILinuxPodSandboxConfig) Reset() { + *x = CRILinuxPodSandboxConfig{} + if protoimpl.UnsafeEnabled { + mi := &file_proto_nerdbox_types_sandbox_proto_msgTypes[2] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CRILinuxPodSandboxConfig) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CRILinuxPodSandboxConfig) ProtoMessage() {} + +func (x *CRILinuxPodSandboxConfig) ProtoReflect() protoreflect.Message { + mi := &file_proto_nerdbox_types_sandbox_proto_msgTypes[2] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CRILinuxPodSandboxConfig.ProtoReflect.Descriptor instead. +func (*CRILinuxPodSandboxConfig) Descriptor() ([]byte, []int) { + return file_proto_nerdbox_types_sandbox_proto_rawDescGZIP(), []int{2} +} + +func (x *CRILinuxPodSandboxConfig) GetSysctls() map[string]string { + if x != nil { + return x.Sysctls + } + return nil +} + +var File_proto_nerdbox_types_sandbox_proto protoreflect.FileDescriptor + +var file_proto_nerdbox_types_sandbox_proto_rawDesc = []byte{ + 0x0a, 0x21, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x2f, 0x6e, 0x65, 0x72, 0x64, 0x62, 0x6f, 0x78, 0x2f, + 0x74, 0x79, 0x70, 0x65, 0x73, 0x2f, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x70, 0x72, + 0x6f, 0x74, 0x6f, 0x12, 0x0d, 0x6e, 0x65, 0x72, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x74, 0x79, 0x70, + 0x65, 0x73, 0x22, 0xa5, 0x01, 0x0a, 0x0c, 0x43, 0x52, 0x49, 0x50, 0x6f, 0x64, 0x43, 0x6f, 0x6e, + 0x66, 0x69, 0x67, 0x12, 0x1a, 0x0a, 0x08, 0x68, 0x6f, 0x73, 0x74, 0x6e, 0x61, 0x6d, 0x65, 0x18, + 0x02, 0x20, 0x01, 0x28, 0x09, 0x52, 0x08, 0x68, 0x6f, 0x73, 0x74, 0x6e, 0x61, 0x6d, 0x65, 0x12, + 0x3a, 0x0a, 0x0a, 0x64, 0x6e, 0x73, 0x5f, 0x63, 0x6f, 0x6e, 0x66, 0x69, 0x67, 0x18, 0x04, 0x20, + 0x01, 0x28, 0x0b, 0x32, 0x1b, 0x2e, 0x6e, 0x65, 0x72, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x74, 0x79, + 0x70, 0x65, 0x73, 0x2e, 0x43, 0x52, 0x49, 0x44, 0x4e, 0x53, 0x43, 0x6f, 0x6e, 0x66, 0x69, 0x67, + 0x52, 0x09, 0x64, 0x6e, 0x73, 0x43, 0x6f, 0x6e, 0x66, 0x69, 0x67, 0x12, 0x3d, 0x0a, 0x05, 0x6c, + 0x69, 0x6e, 0x75, 0x78, 0x18, 0x08, 0x20, 0x01, 0x28, 0x0b, 0x32, 0x27, 0x2e, 0x6e, 0x65, 0x72, + 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x74, 0x79, 0x70, 0x65, 0x73, 0x2e, 0x43, 0x52, 0x49, 0x4c, 0x69, + 0x6e, 0x75, 0x78, 0x50, 0x6f, 0x64, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x43, 0x6f, 0x6e, + 0x66, 0x69, 0x67, 0x52, 0x05, 0x6c, 0x69, 0x6e, 0x75, 0x78, 0x22, 0x5e, 0x0a, 0x0c, 0x43, 0x52, + 0x49, 0x44, 0x4e, 0x53, 0x43, 0x6f, 0x6e, 0x66, 0x69, 0x67, 0x12, 0x18, 0x0a, 0x07, 0x73, 0x65, + 0x72, 0x76, 0x65, 0x72, 0x73, 0x18, 0x01, 0x20, 0x03, 0x28, 0x09, 0x52, 0x07, 0x73, 0x65, 0x72, + 0x76, 0x65, 0x72, 0x73, 0x12, 0x1a, 0x0a, 0x08, 0x73, 0x65, 0x61, 0x72, 0x63, 0x68, 0x65, 0x73, + 0x18, 0x02, 0x20, 0x03, 0x28, 0x09, 0x52, 0x08, 0x73, 0x65, 0x61, 0x72, 0x63, 0x68, 0x65, 0x73, + 0x12, 0x18, 0x0a, 0x07, 0x6f, 0x70, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x18, 0x03, 0x20, 0x03, 0x28, + 0x09, 0x52, 0x07, 0x6f, 0x70, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x22, 0xa6, 0x01, 0x0a, 0x18, 0x43, + 0x52, 0x49, 0x4c, 0x69, 0x6e, 0x75, 0x78, 0x50, 0x6f, 0x64, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x43, 0x6f, 0x6e, 0x66, 0x69, 0x67, 0x12, 0x4e, 0x0a, 0x07, 0x73, 0x79, 0x73, 0x63, 0x74, + 0x6c, 0x73, 0x18, 0x03, 0x20, 0x03, 0x28, 0x0b, 0x32, 0x34, 0x2e, 0x6e, 0x65, 0x72, 0x64, 0x62, + 0x6f, 0x78, 0x2e, 0x74, 0x79, 0x70, 0x65, 0x73, 0x2e, 0x43, 0x52, 0x49, 0x4c, 0x69, 0x6e, 0x75, + 0x78, 0x50, 0x6f, 0x64, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x43, 0x6f, 0x6e, 0x66, 0x69, + 0x67, 0x2e, 0x53, 0x79, 0x73, 0x63, 0x74, 0x6c, 0x73, 0x45, 0x6e, 0x74, 0x72, 0x79, 0x52, 0x07, + 0x73, 0x79, 0x73, 0x63, 0x74, 0x6c, 0x73, 0x1a, 0x3a, 0x0a, 0x0c, 0x53, 0x79, 0x73, 0x63, 0x74, + 0x6c, 0x73, 0x45, 0x6e, 0x74, 0x72, 0x79, 0x12, 0x10, 0x0a, 0x03, 0x6b, 0x65, 0x79, 0x18, 0x01, + 0x20, 0x01, 0x28, 0x09, 0x52, 0x03, 0x6b, 0x65, 0x79, 0x12, 0x14, 0x0a, 0x05, 0x76, 0x61, 0x6c, + 0x75, 0x65, 0x18, 0x02, 0x20, 0x01, 0x28, 0x09, 0x52, 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x3a, + 0x02, 0x38, 0x01, 0x42, 0x2f, 0x5a, 0x2d, 0x67, 0x69, 0x74, 0x68, 0x75, 0x62, 0x2e, 0x63, 0x6f, + 0x6d, 0x2f, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2f, 0x6e, 0x65, 0x72, + 0x64, 0x62, 0x6f, 0x78, 0x2f, 0x61, 0x70, 0x69, 0x2f, 0x74, 0x79, 0x70, 0x65, 0x73, 0x3b, 0x74, + 0x79, 0x70, 0x65, 0x73, 0x62, 0x06, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x33, +} + +var ( + file_proto_nerdbox_types_sandbox_proto_rawDescOnce sync.Once + file_proto_nerdbox_types_sandbox_proto_rawDescData = file_proto_nerdbox_types_sandbox_proto_rawDesc +) + +func file_proto_nerdbox_types_sandbox_proto_rawDescGZIP() []byte { + file_proto_nerdbox_types_sandbox_proto_rawDescOnce.Do(func() { + file_proto_nerdbox_types_sandbox_proto_rawDescData = protoimpl.X.CompressGZIP(file_proto_nerdbox_types_sandbox_proto_rawDescData) + }) + return file_proto_nerdbox_types_sandbox_proto_rawDescData +} + +var file_proto_nerdbox_types_sandbox_proto_msgTypes = make([]protoimpl.MessageInfo, 4) +var file_proto_nerdbox_types_sandbox_proto_goTypes = []interface{}{ + (*CRIPodConfig)(nil), // 0: nerdbox.types.CRIPodConfig + (*CRIDNSConfig)(nil), // 1: nerdbox.types.CRIDNSConfig + (*CRILinuxPodSandboxConfig)(nil), // 2: nerdbox.types.CRILinuxPodSandboxConfig + nil, // 3: nerdbox.types.CRILinuxPodSandboxConfig.SysctlsEntry +} +var file_proto_nerdbox_types_sandbox_proto_depIdxs = []int32{ + 1, // 0: nerdbox.types.CRIPodConfig.dns_config:type_name -> nerdbox.types.CRIDNSConfig + 2, // 1: nerdbox.types.CRIPodConfig.linux:type_name -> nerdbox.types.CRILinuxPodSandboxConfig + 3, // 2: nerdbox.types.CRILinuxPodSandboxConfig.sysctls:type_name -> nerdbox.types.CRILinuxPodSandboxConfig.SysctlsEntry + 3, // [3:3] is the sub-list for method output_type + 3, // [3:3] is the sub-list for method input_type + 3, // [3:3] is the sub-list for extension type_name + 3, // [3:3] is the sub-list for extension extendee + 0, // [0:3] is the sub-list for field type_name +} + +func init() { file_proto_nerdbox_types_sandbox_proto_init() } +func file_proto_nerdbox_types_sandbox_proto_init() { + if File_proto_nerdbox_types_sandbox_proto != nil { + return + } + if !protoimpl.UnsafeEnabled { + file_proto_nerdbox_types_sandbox_proto_msgTypes[0].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CRIPodConfig); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_proto_nerdbox_types_sandbox_proto_msgTypes[1].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CRIDNSConfig); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_proto_nerdbox_types_sandbox_proto_msgTypes[2].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CRILinuxPodSandboxConfig); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + } + type x struct{} + out := protoimpl.TypeBuilder{ + File: protoimpl.DescBuilder{ + GoPackagePath: reflect.TypeOf(x{}).PkgPath(), + RawDescriptor: file_proto_nerdbox_types_sandbox_proto_rawDesc, + NumEnums: 0, + NumMessages: 4, + NumExtensions: 0, + NumServices: 0, + }, + GoTypes: file_proto_nerdbox_types_sandbox_proto_goTypes, + DependencyIndexes: file_proto_nerdbox_types_sandbox_proto_depIdxs, + MessageInfos: file_proto_nerdbox_types_sandbox_proto_msgTypes, + }.Build() + File_proto_nerdbox_types_sandbox_proto = out.File + file_proto_nerdbox_types_sandbox_proto_rawDesc = nil + file_proto_nerdbox_types_sandbox_proto_goTypes = nil + file_proto_nerdbox_types_sandbox_proto_depIdxs = nil +} diff --git a/cmd/containerd-shim-nerdbox-v1/main.go b/cmd/containerd-shim-nerdbox-v1/main.go index de35a1d8..130c2ea2 100644 --- a/cmd/containerd-shim-nerdbox-v1/main.go +++ b/cmd/containerd-shim-nerdbox-v1/main.go @@ -18,16 +18,20 @@ package main import ( "context" + "os" "github.com/containerd/containerd/v2/pkg/shim" + "github.com/containerd/log" "github.com/containerd/nerdbox/pkg/logging" "github.com/containerd/nerdbox/pkg/shim/manager" + _ "github.com/containerd/nerdbox/plugins/sandbox" _ "github.com/containerd/nerdbox/plugins/shim/sandbox" _ "github.com/containerd/nerdbox/plugins/shim/streaming" _ "github.com/containerd/nerdbox/plugins/shim/task" _ "github.com/containerd/nerdbox/plugins/shim/transfer" + _ "github.com/containerd/nerdbox/plugins/task" _ "github.com/containerd/nerdbox/plugins/vm/libkrun" ) @@ -36,6 +40,18 @@ func init() { } func main() { + // Only ever set on the shim server child cloneMntNs itself launched + // (see manager.MountNSIsolatedEnv), so this never runs for containerd's + // own separate, direct "start"/"delete" invocations of this same + // binary. Must happen before any container-related mount, so as early + // in startup as possible. + if os.Getenv(manager.MountNSIsolatedEnv) == "1" { + if err := manager.IsolateMountPropagation(); err != nil { + log.G(context.Background()).WithError(err).Warn( + "failed to isolate shim mount namespace propagation; container rootfs mounts may leak into the host") + } + } + shim.RunShim(context.Background(), manager.New("io.containerd.nerdbox.v1"), func(c *shim.Config) { c.NoSetupLogger = true diff --git a/cmd/vminitd/main.go b/cmd/vminitd/main.go index 4af77b63..2e808cd4 100644 --- a/cmd/vminitd/main.go +++ b/cmd/vminitd/main.go @@ -28,6 +28,7 @@ import ( _ "github.com/containerd/nerdbox/plugins/services/bundle" _ "github.com/containerd/nerdbox/plugins/services/mount" + _ "github.com/containerd/nerdbox/plugins/services/sharedresources" _ "github.com/containerd/nerdbox/plugins/services/system" _ "github.com/containerd/nerdbox/plugins/services/transfer" diff --git a/crates/pause/Cargo.lock b/crates/pause/Cargo.lock new file mode 100644 index 00000000..fbb71341 --- /dev/null +++ b/crates/pause/Cargo.lock @@ -0,0 +1,16 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "nerdbox-pause" +version = "0.1.0" +dependencies = [ + "libc", +] diff --git a/crates/pause/Cargo.toml b/crates/pause/Cargo.toml new file mode 100644 index 00000000..3eec1092 --- /dev/null +++ b/crates/pause/Cargo.toml @@ -0,0 +1,38 @@ +# Copyright The containerd Authors. + +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at + +# http://www.apache.org/licenses/LICENSE-2.0 + +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +[package] +name = "nerdbox-pause" +version = "0.1.0" +description = "Anchor process holding a guest PID namespace open" +license = "Apache-2.0" +repository = "https://github.com/containerd/nerdbox" +homepage = "https://containerd.io" +edition = "2021" +publish = false + +[dependencies] +# default-features = false drops libc's "std" feature, which this crate does +# not have available. +libc = { version = "0.2", default-features = false } + +[profile.release] +# This binary is exec'd once per sandbox that shares a PID namespace, so what +# is being optimised for is process startup, not throughput. Keeping it small +# is what makes that startup cheap. +opt-level = "z" +lto = true +codegen-units = 1 +panic = "abort" +strip = true diff --git a/crates/pause/src/main.rs b/crates/pause/src/main.rs new file mode 100644 index 00000000..3e49293a --- /dev/null +++ b/crates/pause/src/main.rs @@ -0,0 +1,107 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +//! Anchor process holding a guest PID namespace open. +//! +//! A Linux PID namespace has no content of its own: the kernel destroys it, +//! killing everything inside, the moment its PID 1 exits, and no new process +//! can be created in it afterwards. Unlike a network or IPC namespace it +//! therefore cannot be kept alive by a bind mount, and since +//! `unshare(CLONE_NEWPID)` does not move the caller into the new namespace — +//! only the caller's next child becomes its PID 1 — a thread cannot stand in +//! either. Something has to be PID 1 and stay there, which is all this +//! program does. +//! +//! It is spawned by the guest's NamespaceManager (see +//! `internal/vminit/namespaces`) with `CLONE_NEWPID`, and lives until that +//! service kills it to tear the namespace down. +//! +//! This is `no_std` deliberately. The whole program is three signal +//! dispositions and a sleep, none of which needs anything from `std`, and +//! avoiding it keeps the binary small enough that exec'ing it costs +//! essentially nothing. + +#![no_std] +#![no_main] + +use core::ffi::{c_char, c_int}; +use core::panic::PanicInfo; +use core::ptr; + +// The libc crate is used without its "std" feature, which is also what +// normally arranges for libc itself to be linked. Request it explicitly: the +// C runtime provides this program's entry point (see `main` below) as well as +// the handful of functions it calls. +#[link(name = "c")] +extern "C" {} + +#[panic_handler] +fn panic(_info: &PanicInfo) -> ! { + // Nothing here can meaningfully panic, and there is no unwinding with + // panic = "abort". Failing loudly beats a PID 1 in an unknown state. + unsafe { libc::abort() } +} + +/// Installs `handler` for `signum` with `flags`, discarding the previous +/// disposition. +/// +/// # Safety +/// +/// `handler` must be a valid `sighandler_t` (`SIG_DFL`, `SIG_IGN`, or a +/// pointer to an async-signal-safe function). +unsafe fn set_disposition(signum: c_int, handler: libc::sighandler_t, flags: c_int) -> c_int { + let mut act: libc::sigaction = core::mem::zeroed(); + act.sa_sigaction = handler; + act.sa_flags = flags; + libc::sigemptyset(&mut act.sa_mask); + libc::sigaction(signum, &act, ptr::null_mut()) +} + +#[no_mangle] +pub extern "C" fn main(_argc: c_int, _argv: *const *const c_char) -> c_int { + unsafe { + // Reap anything reparented here. A process elsewhere in the shared + // namespace that outlives its original parent becomes our child, and + // left unreaped those would pile up as zombies for the sandbox's whole + // lifetime -- a PID 1 duty. SA_NOCLDWAIT has the kernel do it, so + // there is no handler to write and no wait loop to run; the + // alternative is a SIGCHLD handler calling waitpid(WNOHANG) until it + // drains, which achieves the same thing with more moving parts. + if set_disposition(libc::SIGCHLD, libc::SIG_DFL, libc::SA_NOCLDWAIT) != 0 { + return 1; + } + + // Refuse to die on anything catchable. PID 1 of a namespace is + // already protected from default-disposition signals raised inside + // that namespace, but not from ones sent by an ancestor namespace, so + // ignoring these explicitly is what makes SIGKILL -- which is how the + // namespace is deliberately torn down -- the only way out. + if set_disposition(libc::SIGINT, libc::SIG_IGN, 0) != 0 { + return 2; + } + if set_disposition(libc::SIGTERM, libc::SIG_IGN, 0) != 0 { + return 3; + } + + loop { + // Sleeps until a signal arrives. Ignored signals and SIGCHLD + // under SA_NOCLDWAIT do not wake it, so in practice this blocks + // forever; the loop is here so that a spurious wakeup cannot turn + // into an exit. + libc::pause(); + } + } +} diff --git a/docs/sandbox-architecture.md b/docs/sandbox-architecture.md new file mode 100644 index 00000000..4e98a595 --- /dev/null +++ b/docs/sandbox-architecture.md @@ -0,0 +1,829 @@ +# Sandbox Architecture + +This document describes how the nerdbox sandbox works — what lives on the +host, what lives in the VM, and how networking flows between them. + +## Overview + +A nerdbox **sandbox** is a single microVM that hosts one or more containers. +It maps directly to the Kubernetes pod model: one VM per pod, with all +containers in the pod sharing the VM's kernel, network stack, and IPC +facilities. + +``` +┌─────────────────────────────────────────────────────────────────────┐ +│ Host (Linux) │ +│ │ +│ ┌────────────────────────────────────┐ │ +│ │ containerd │ │ +│ │ ┌──────────────────────────────┐ │ │ +│ │ │ Sandbox Controller (shim) │ │ │ +│ │ │ • CreateSandbox │ │ │ +│ │ │ • StartSandbox │ │ │ +│ │ │ • Task.Create (per ctr) │ │ │ +│ │ └──────────────┬───────────────┘ │ │ +│ └─────────────────┼──────────────────┘ │ +│ │ TTRPC (vsock 1025) │ +│ │ │ +│ ┌─────────────────▼──────────────────────────────────────────┐ │ +│ │ VMM (libkrun) │ │ +│ │ │ │ +│ │ ┌─────────────────────────────────────────────────────┐ │ │ +│ │ │ vminitd (PID 1) │ │ │ +│ │ │ │ │ │ +│ │ │ ctr-A (runc) ctr-B (runc) ctr-C (runc) │ │ │ +│ │ │ ┌──────┐ ┌──────┐ ┌──────┐ │ │ │ +│ │ │ │ / │ │ / │ │ / │ │ │ │ +│ │ │ └──────┘ └──────┘ └──────┘ │ │ │ +│ │ │ shared, on demand: network, IPC, PID, /dev/shm │ │ │ +│ │ └─────────────────────────────────────────────────────┘ │ │ +│ └────────────────────────────────────────────────────────────┘ │ +└─────────────────────────────────────────────────────────────────────┘ +``` + +The runtime referenced above as "runc" is the OCI runtime interface. nerdbox +uses crun as its implementation, but the interface and container model follow +the runc specification. + +## Host / VM responsibility split + +Everything that needs to interact with the host OS — CNI plugins, image +snapshotters, volume mounts — is managed on the **host side** of the shim. +Everything that needs to interact with a running container process — +namespace setup, cgroup accounting, syscall filtering — is managed +**inside the VM** by vminitd. + +### What the host shim owns + +| Resource | Where it lives | Notes | +|---|---|---| +| VM lifecycle | Host shim process | libkrun starts/stops the VM; the shim holds the only reference | +| Container rootfs assembly | Host filesystem | Overlay / erofs layers mounted on host, exposed to VM via virtiofs | +| Bind mounts and volumes | Host filesystem | Resolved and mounted on the host inside the shim's mount namespace, exposed via the same virtiofs share | +| Network sandbox (netns path) | Host shim process | FD held open for the CNI lifetime — see [Networking](#networking) | +| Virtual NICs | Host VMM config | Configured before VM boot via libkrun; cannot be added after boot | +| Socket forwarding | Host shim | UNIX sockets forwarded host↔VM via the SocketForward TTRPC service | +| OCI bundle (config.json) | Host shim, pushed to guest | Assembled on host from snapshotter metadata, pushed to guest over the Bundle TTRPC service | + +### What the guest (vminitd) owns + +| Resource | Where it lives | Notes | +|---|---|---| +| Container process lifecycle | VM | runc creates/starts/stops containers | +| Mount namespaces | VM kernel | Each container gets its own mount namespace; rootfs is bind-mounted from the virtiofs share | +| cgroups (v2 unified) | VM kernel | One cgroup per container, under vminitd's cgroup tree | +| Network namespaces | VM kernel | A container with its own annotation-driven NIC gets a fresh namespace of its own; otherwise it stays in the VM's own network namespace, since that is the only one a real virtio-net interface is ever plumbed into (see [Sandbox networking summary](#sandbox-networking-summary)) | +| IPC namespace | VM kernel | Shared, created on demand, when CRI's pod-level IPC sharing is requested (see [Shared guest namespaces](#shared-guest-namespaces)); otherwise each container gets its own | +| PID namespace | VM kernel | Own PID namespace by default; joins a shared, on-demand PID namespace when CRI's pod-level PID sharing is requested (see [Shared guest namespaces](#shared-guest-namespaces)) | +| /dev/shm | VM kernel | A container sharing IPC bind-mounts a shared, size-limited guest tmpfs; otherwise its own private one — see [Shared `/dev/shm`](#shared-devshm) | +| UTS namespace | VM kernel | Shared, created on demand, when CRI's pod-level UTS sharing is requested (always, in practice — see [Shared guest namespaces](#shared-guest-namespaces)); otherwise each container gets its own | +| Hostname | Guest kernel, via each container's own OCI spec | Not tracked or coordinated by nerdbox at all: an OCI runtime calls `sethostname(2)` after joining the (now shared) UTS namespace when a container's spec has a non-empty hostname, which updates it for every container sharing that namespace — see [Shared guest namespaces](#shared-guest-namespaces) | + +## Container filesystem + +Each container's rootfs is assembled **on the host** inside the shim's +private mount namespace, then shared into the VM via a single persistent +virtiofs mount. + +``` +Host state directory: /vm/ +Virtiofs share root: /vm/containers/ ← tag "containers" +Guest mount point: /run/containers/ + +Per-container tree: + /run/containers//rootfs ← assembled from snapshotter mounts + /run/containers//volumes/0 ← first extra volume (if any) +``` + +The host-side assembly mounts the rootfs at the correct path, or fails the +container run if a required mount cannot be established. (An empty mount +list — legal for a from-scratch container with no image content — is not an +error: it simply leaves an empty directory in place.) The mount type is +determined by the snapshotter and containerd: + +- **Overlay mount** — overlayfs over multiple layer directories (the common + case with the native overlayfs snapshotter). +- **Bind mount** — a single pre-extracted directory bind-mounted read-only + (used by native snapshotter with a fully-extracted layer, or nydus). +- **FUSE mount** — a FUSE-based filesystem exposed by an external snapshotter + (e.g. stargz-snapshotter, nydus). + +If none of these mounts can be established, the container run fails. There is +no fallback to hard links or file copies — both would produce silent failures: +hard links can fail across filesystems, and copies accumulate dirty pages and +destroy filesystem metadata. + +After `Task.Delete`, `SharedFS.Unshare` removes the container's subtree from +the shared directory, unmounting any mounts and calling `os.RemoveAll` on the +directory entry. + +``` +┌── Host shim → vmm ────────────────────────────────────────────────┐ +│ │ +│ snapshotter mounts │ +│ ┌──────────────────┐ │ +│ │ erofs layer A │ │ +│ │ erofs layer B │ ──── mount (overlay/bind/fuse) ───► │ +│ │ ext4 upper │ │ │ +│ └──────────────────┘ ▼ │ +│ vm/containers//rootfs │ +│ │ │ +└─────────────────────────────────────────────┼─────────────────────┘ + │ virtiofs (tag "containers") + ▼ + vminitd: /run/containers//rootfs + │ + │ bind mount (by runc) + ▼ + Container rootfs in its mount namespace +``` + +### Bind mounts and volumes + +A member container's OCI "bind" mounts (Kubernetes `hostPath` volumes, and +CRI's own injected UDS/sandbox-file mounts) are handled by +`SharedFS.ShareVolume` rather than becoming a new virtiofs share: a sandbox +member container is created against an already-running VM, and virtio-fs +shares cannot be hot-added after boot, so each mount's host source is +instead bind-mounted directly into the container's own subtree of the +already-shared `containers` tree (`/volumes/`), which the +guest already sees with no new device and no extra guest-side mount step. + +This mechanism is oblivious to whether a given host source is exclusive to +one container or handed to several: each container that references the same +host path simply gets its own independent bind mount of that path. This is +what makes Kubernetes `emptyDir` volumes work transparently across multiple +containers in one pod — kubelet provisions a single host directory per +`emptyDir` volume and lists that same host path in every container's mount +spec that references it, so all of them end up bind-mounting identical +content, with no sandbox-specific "shared volume" logic required. + +### Shared `/dev/shm` + +CRI sends `/dev/shm` as an ordinary, independent tmpfs mount +(`{Type: "tmpfs", Source: "shm", Options: [..., "size=k"]}`) on **every** +container's spec — the mount itself carries no signal about whether the pod +actually wants it shared. That signal instead comes from the same place +namespace sharing does: `containerSharesIPC` (`internal/shim/task/devshm.go`) +checks for a non-empty-`Path` IPC namespace entry, the identical condition +[`sanitizeNamespaces`](#mechanism) uses to decide whether a container joins +the shared guest IPC namespace. + +When that signal is present, `shareDevShmMounter.FromBundle` rewrites the +`/dev/shm` mount from `tmpfs` to a `bind` mount pointing at a per-sandbox +tmpfs created on demand in the **guest**, by the same `SharedResources` +mechanism [Shared guest namespaces](#shared-guest-namespaces) uses for +network/IPC/PID (`internal/vminit/sharedresources.TypeDevShm`) — a real, +`size=`-limited tmpfs mounted once per sandbox under `/run/devshm/` in +vminitd's own root mount namespace, whose size comes from `devShmSize` +parsing the container's own CRI-provided `size=` option (64MiB fallback). +Every member container that shares IPC bind-mounts the *same* guest path +onto its own `/dev/shm`, so it is real, size-enforced guest RAM shared +across containers — not a host-backed directory standing in for one. When +the signal is absent, the mount is left untouched and each container keeps +its own independent, private tmpfs, matching CRI's non-shared default. + +This tmpfs is mounted **outside** the `containers` virtiofs tree entirely, +in vminitd's own root mount namespace, rather than living inside the +directory `ShareRootfs`/`ShareVolume` expose via virtiofs: a member +container's own crun bind-mounts `/run/devshm/` directly, the same way +it already joins `/run/ipcns/` for a shared IPC namespace, with no +virtiofs involved at any point. This also +means it is real guest RAM with a real, kernel-enforced size limit, rather +than something backed by host disk and reached over virtiofs. +`mmap(MAP_SHARED)` writes from one container are visible to +`mmap(MAP_SHARED)` reads from another through ordinary Linux page-cache +coherence on the shared inode — since every sharing container's bind mount +ultimately resolves to the same tmpfs inode in the same guest kernel, this +holds regardless of virtiofs's own cache behavior. + +## Networking + +Networking involves two independent layers that are often confused: + +1. **The host-side network sandbox** — a Linux network namespace on the host, + created and owned by the CRI layer (containerd), passed to the shim. +2. **The VM-side network stack** — the actual network interfaces the containers + use, configured inside the microVM. + +### Layer 1 — Host network sandbox (Linux netns) + +#### How the netns is created + +The CRI layer (containerd's CRI plugin, running in the containerd process) +creates the network namespace entirely by itself — no pause container is +involved. The mechanism is the long-standing CNI "persistent netns" technique: + +1. A dedicated goroutine calls `runtime.LockOSThread()` and never unlocks, + so Go retires the underlying OS thread when the goroutine exits (Go 1.10+). +2. On that locked thread, `unshare(CLONE_NEWNET)` creates a new, empty + network namespace for that thread only (containerd's `pkg/netns` + package). +3. The thread's netns is bind-mounted to a file under `/var/run/netns/` + (or the configured state dir) via `mount("/proc//task//ns/net", + "/var/run/netns/cni-", MS_BIND)`. + Here `` is the containerd process PID and `` is the + TID of the dedicated throwaway thread — `/proc/self/ns/net` cannot be used + because it always returns the thread-group-leader's namespace. +4. The bind-mount anchors the netns to the filesystem. The throwaway thread + exits but the namespace persists because the bind-mount still holds a + reference. **A netns persists with zero processes in it as long as the + bind-mount file exists.** + +#### Ordering: CNI runs before the sandbox + +``` +containerd CRI plugin (RunPodSandbox) + + 1. Create netns bind-mount at /var/run/netns/cni- ← unshare + bind + 2. Run CNI ADD against that empty netns ← configures IP/routes/etc + 3. CreateSandbox(netns_path=/var/run/netns/cni-) ← shim receives path + 4. StartSandbox ← shim boots VM +``` + +CNI **always runs before the sandbox is created**. CNI configures an empty, +process-less netns (which it can do because the bind-mount keeps it alive), +and the sandbox is later started knowing the fully-configured path. + +With the **shim sandboxer there is no pause container** — the shim receives +`netns_path` directly in `CreateSandboxRequest`. (The legacy `podsandbox` +controller creates a pause container which *joins* the pre-existing netns via +an OCI `LinuxNamespace{Type: network, Path: nsPath}`; the shim sandboxer skips +this entirely.) + +For host-network pods (`NamespaceMode_NODE`), no netns is created and +`netns_path` is empty. + +#### What the shim does with netns_path + +**At `CreateSandbox` time** the shim opens the path `O_RDONLY|O_CLOEXEC` and +holds the FD open. This second reference to the netns (alongside the +bind-mount) keeps it alive even if the bind-mount were removed prematurely, +and satisfies the CRI contract. The shim releases this FD as part of +handling `StopSandbox`, not afterward. + +``` +CRI layer nerdbox shim + │ │ + │── CreateSandbox(netns_path) ──►│ opens FD to netns_path + │ │ (secondary pin on the bind-mount) + │── StartSandbox ───────────────►│ setns into netns, just before boot + │ │ VM boots + │ │ FD remains open + │ [ pod running ] │ + │ │ + │── StopSandbox ────────────────►│ VM stops + │ │ FD closed + │ [ CNI DEL runs against netns_path ] + │── ShutdownSandbox ────────────►│ final cleanup +``` + +**At `StartSandbox` time** the netns path is stored on the `vmInstance` and +used just before boot (see [Layer 2](#layer-2--vm-network-stack) below): +`setns(2)` runs immediately before `krun_start_enter`, the last libkrun call +made on the boot path, so it takes effect right before libkrun opens any +host-side network resources. + +### Layer 2 — VM network stack + +#### Entering the pod netns before boot + +Configuring a VM context (`krun_create_ctx` through the calls that add +disks, filesystems, NICs, and CPU/memory) does not itself open any +network-namespace-sensitive host resources, and can run on any goroutine — +Go's scheduler is free to migrate it across OS threads. Only the final step, +`krun_start_enter`, matters for namespace placement: + +- `krun_start_enter` is what actually opens libkrun's host-side network + resources (NIC AF_UNIX sockets, TSI host sockets) and spawns libkrun's + internal worker threads (vCPU, virtio backends, TSI net workers) — those + workers inherit the network namespace of the thread that called it. +- So the calling thread's network namespace at the moment of that one call + is what determines which namespace all of libkrun's networking ends up in. + +`vmInstance.Start` accounts for this by running `krun_start_enter` inside a +dedicated goroutine that calls `runtime.LockOSThread()` and, when a +`netns_path` was configured, calls `setns(2)` into the pod netns +**immediately before** `krun_start_enter` — as the last thing that happens +before boot, not the first: + +``` +vmInstance.Start() + │ + └── goroutine: runtime.LockOSThread() + │ + ├── setns(pod netns) ← only if netns_path was set + │ + └── krun_start_enter ← blocks on this thread + │ + ├── vCPU thread ← inherits pod netns + ├── virtio workers ← inherits pod netns + └── TSI net workers ← inherits pod netns + │ + └── host connect(AF_INET, ...) ← in pod netns +``` + +`AddNIC` itself only registers the NIC's socket path with libkrun (the FD +field is left at -1); libkrun opens the actual AF_UNIX socket to that path +itself, from inside `krun_start_enter`, so it too lands in the pod netns by +the same mechanism. + +Control-plane goroutines (the shim TTRPC listener, vsock accept, vminitd +connection) operate over FD-based UDS/vsock connections established +independently of this goroutine and are unaffected by the namespace change. + +The in-process `setns` is sufficient on its own: a member container's +outbound traffic originates from the pinned pod netns, with no re-exec or +trampoline process required. + +#### TSI (Transparent Socket Impersonation) + +TSI is a compiled-in feature of the guest kernel (`CONFIG_TSI=y`, patches +`0011`–`0012` in `kernel/patches/`; `0009`–`0010` are generic vsock support +patches, not TSI-specific), gated at runtime by the `tsi_hijack` +kernel parameter, which defaults to **off**. The shim does not set it: libkrun +does, and only when no virtio-net interface has been attached. TSI and a NIC +are alternative ways to provide the same connectivity, so libkrun enables TSI +exactly when there is no NIC (`enable_tsi = net.list.is_empty() && ...` in its +`VsockConfig::Implicit` handling). libkrun gates its own host side to match, +rejecting proxy requests when the hijack is disabled. + +Inside the VM, the patched kernel intercepts `AF_INET` and `AF_INET6` socket +calls (`SOCK_STREAM`/`SOCK_DGRAM`, i.e. TCP/UDP). When a container opens a TCP connection, the kernel transparently +rewrites it to `AF_TSI` and proxies it over vsock to libkrun, which performs +the real `connect()` on the host — now inside the pod netns, since the +worker performing it descends from the `krun_start_enter` call made just +after `setns` (see [Layer 2](#layer-2--vm-network-stack)). + +``` +Container (guest) Host (pod netns) + ┌──────────────────────┐ + connect(AF_INET, 1.2.3.4:80) │ libkrun TSI worker │ + │ │ (descends from the │ + TSI kernel intercept │ krun_start_enter │ + │ │ thread, pod netns) │ + │ ── vsock ──────────────────►│ connect(1.2.3.4:80) │ + │ source: pod IP │ + └──────────────────────┘ +``` + +TSI covers TCP and UDP over both IPv4 (`AF_TSI`) and IPv6 (`AF_TSI6`). It does +not proxy ICMP or raw sockets, which stay in whatever network namespace the +container is in. + +#### DNS configuration + +Container resolv.conf content is resolved with the following priority: an +existing bundle mount, a per-container DNS annotation +(`io.containerd.nerdbox.ctr.dns`), the pod's CRI `DNSConfig`, and finally a +copy of the host's own resolv.conf. When falling back to the host's +resolv.conf, `addResolvConf` (`internal/shim/task/ctrnetworking.go`) +inspects the nameserver entries: only if **every** nameserver in +`/etc/resolv.conf` is a loopback address (the systemd-resolved stub +configuration, unreachable from inside the guest in the default +no-NIC/TSI configuration) does it substitute systemd-resolved's "full" +resolv.conf (`/run/systemd/resolve/resolv.conf`, listing the real upstream +nameservers) instead. A host not using systemd-resolved, or one with a mix +of loopback and real nameservers, is left as-is. + +#### TSI and guest network namespaces + +TSI's socket hijack operates on address family alone, before any +namespace-aware routing decision, and the resulting vsock channel to +`VMADDR_CID_HOST` is not real IP routing — so it is not scoped by, and +cannot be filtered via, guest-internal network namespaces. The +host-reachability boundary is established entirely on the host side: the pod +netns the shim pins and enters via `setns` just before boot (see +[Layer 1](#layer-1--host-network-sandbox-linux-netns) and +[Layer 2](#layer-2--vm-network-stack) above) determines which host network +TSI's proxied connections land in. + +Because of this, the shim does not create a guest network namespace at all +when TSI is carrying container traffic: one would be created, joined, and then +ignored. Containers instead stay in the VM's own network namespace, but that +sharing is largely incidental to cross-container connectivity: TSI hijacks +each `socket()` call before any netns-scoped routing decision is ever made, +so two sibling containers reaching each other over loopback are not really +using guest-kernel loopback routing at all — each side's hijacked socket is +proxied independently to the host, and they only rendezvous because both +proxied operations resolve to the same concrete host-side port. This +requires the guest and the host to agree on that port: an inbound bind on an +ephemeral port (`bind()`/`listen()` on port `0`) must forward the guest +kernel's *resolved* port to the host, not the literal `0` the application +requested — otherwise the host independently picks its own, unrelated +ephemeral port, and nothing is reachable at the port the application +believes it bound (see kernel patch +`0013-tsi-forward-the-resolved-port-for-ephemeral-binds.patch`). This in +turn means the guest's resolved port can now collide with something +already bound on the host — see +[Known limitations](#known-limitations) for what happens then. + +The decision needs nothing driver-specific. A network namespace exists to +scope in-guest networking, and a virtio-net interface is what creates that, +so the presence of a NIC *on this container* decides it: no NIC of its own +means nothing for a namespace to scope, whether or not the sandbox itself has +one, since a virtio-net interface is only ever plumbed into the VM's own +initial network namespace and never into a separate namespace a member +container without a NIC could meaningfully join. Where a container does have +its own NIC, the veth/bridge mechanisms in `internal/vminit/ctrnetworking` +provide container-to-container isolation. + +TSI proxies individual outbound `connect()`/`listen()` calls; it does not +mirror the host's own socket table into the guest, so introspection tools +like `netstat`/`ss` run inside a container only see the container's own +guest-side connections, not the host's. + +#### External NIC (explicit virtio-net) + +When the OCI spec annotations carry `io.containerd.nerdbox.network.*`, a +virtio-net NIC is attached to the VM. The NIC is backed by an AF_UNIX socket +(`krun_add_net_unixgram` or `krun_add_net_unixstream`) that connects libkrun +to an **externally-run** L2 network provider. + +Like the TSI host sockets, this AF_UNIX socket is opened by libkrun from +inside `krun_start_enter` (already in the pod netns by then), so the +connection to the external provider originates from the pod netns. + +Supported external providers: +- **passt** (unixgram mode) — passt-style helpers that exchange complete L2 + Ethernet frames as datagrams. +- **gvproxy / vfkit** (unixstream mode) — helpers that frame L2 packets over + a stream connection. + +The shim does **not** spawn the external provider. The user (or a future +shim enhancement) must run it out-of-band and pass its socket path via +annotation. Note: `krun_set_gvproxy_path` and `krun_set_net_mac` are declared +in the libkrun bindings but are currently unused. + +``` +External network provider nerdbox shim (pod netns) +(passt / gvproxy) │ + │ │ + │ AF_UNIX socket (L2 frames) │ + └────────────────────────────►│ libkrun: AddNIC(socket) + │ + ▼ + VM: virtio-net interface (eth0) + vminitd brings up eth0 with IP/routes + │ + ┌────────┴──────────┐ + │ │ + Container A Container B + (veth in its (shared eth0 or + own netns) own veth pair) +``` + +The NIC is configured before VM boot and cannot be changed while the VM runs +(libkrun does not support device hotplug). + +### What socketforward is not + +The socketforward service is a TTRPC service reached over the same control +channel as everything else (vsock port 1025 — see +[TTRPC communication](#ttrpc-communication)); it forwards **AF_UNIX domain +sockets** host↔guest. Only the forwarded payload itself — the data read from +and written to the forwarded UNIX sockets — is what rides the separate +streaming channel on port 1026. It is not IP networking: both ends are +`net.Listen("unix", ...)` / `net.Dial("unix", ...)`. AF_INET/TCP +networking is handled exclusively by TSI (default) or the virtio-net NIC +(opt-in). These three mechanisms are independent and must not be conflated. + +### Sandbox networking summary + +| Scenario | Host netns | VM network | Guest netns (per member container) | +|---|---|---|---| +| No annotation (default) | Pinned (FD); entered via `setns` just before boot | TSI — TCP/UDP through pod netns | None; container stays in the VM's own | +| `io.containerd.nerdbox.network.*` (sandbox-level NIC) | Pinned (FD); entered via `setns` just before boot | virtio-net NIC; AF_UNIX to external provider from pod netns | None; container stays in the VM's own, which is where the NIC actually lives | +| `io.containerd.nerdbox.ctr.network.*` (per-container NIC) | As above | As above, plus per-container veth/bridge wiring | Fresh namespace of its own | +| Kubernetes CRI pod | Created by containerd CRI (`unshare` + bind-mount); CNI ADD before sandbox | Either of the above, with full pod netns integration | As above, depending on per-container NIC annotations | +| `ctr run` (no sandbox) | No netns (legacy single-container path) | TSI or virtio in shim's own netns | n/a (legacy path) | +| Host-network pod (`NamespaceMode_NODE`) | Not created; `netns_path` is empty | TSI or virtio in shim's own netns | As above | + +Guest network namespaces follow *per-container* NIC presence alone, with no +per-driver behaviour: a container gets a fresh namespace of its own only when +it has its own annotation-driven NIC (`io.containerd.nerdbox.ctr.network.*`); +otherwise it stays in the VM's own network namespace. This is true whether or +not the sandbox itself has a NIC, since a virtio-net interface is only ever +plumbed into the VM's own initial network namespace, never into a separate +namespace member containers could join — a container without a NIC of its +own therefore has no in-guest networking for a namespace to scope in any +case, and reaches the host by other means (TSI) that no network namespace +can scope regardless. + +## Shared guest namespaces + +Kubernetes pods share an IPC namespace and a UTS namespace by default (there +is no per-pod option to turn either off), and can opt into sharing a PID +namespace (`shareProcessNamespace: true`) or the node's PID/IPC namespaces +(`hostPID`/`hostIPC: true`). containerd's `WithPodNamespaces` oci-spec opt +expresses all of these the same way: it sets a host path (e.g. +`/proc//ns/ipc`) on the relevant namespace entry of a member +container's OCI spec. That host path is meaningless in the guest — the +guest is a different kernel with its own, unrelated namespaces — so the shim +recognizes the request and substitutes a guest-side equivalent rather than +copying the host path verbatim. + +### Mechanism + +Guest namespaces are created **on demand** by a guest-side TTRPC service, +`SharedResources` (`internal/vminit/sharedresources`, registered as plugin +`sharedresources`). Resources are addressed by a group id — the sandbox ID — +plus a type, and are created once per `(id, type)` and reused thereafter. +The guest returns the path each resource is pinned at, so the host never +hardcodes a guest path. + +The same service also manages one resource that is not actually an OCI +namespace: the tmpfs backing a sandbox's shared `/dev/shm` (`TypeDevShm`; +see [Shared `/dev/shm`](#shared-devshm)). It is addressed, created, and +reused the same way — "create once per group id, return a guest path" is +the same problem either way — so `SharedResources` covers namespaces and +this kind of resource together rather than needing a separate service. + +Crucially, a caller requests **only the types it needs**, because the cost is +not uniform: + +- **Network**, **IPC**, and **UTS**: created by locking a goroutine to an OS + thread, calling `unshare(CLONE_NEWNET)` / `unshare(CLONE_NEWIPC)` / + `unshare(CLONE_NEWUTS)` (which, unlike `CLONE_NEWPID`, take effect on the + calling thread immediately), and bind-mounting the thread's namespace file + to `/run/netns/`, `/run/ipcns/`, or `/run/utsns/`. The bind + mount alone keeps the namespace alive, so the creating goroutine does not + need to stay running. Cheap. For network specifically, the fresh namespace + also has its loopback interface (`lo`) brought up before the bind-mount + step — a new network namespace's `lo` starts administratively down, and + without this step, container-to-container loopback traffic within the + shared namespace would fail. In practice the host currently never + requests network through this mechanism at all — see [Sandbox networking + summary](#sandbox-networking-summary) for why a shared guest network + namespace is not implemented, and why a container with its own NIC gets a + namespace crun creates fresh rather than one obtained here — but the + guest-side mechanism (and its own tests) exist independently of whether + the host currently calls into it for this type. +- **PID**: cannot work that way. `unshare(CLONE_NEWPID)` does not move the + caller into the new namespace — only the caller's *next child* becomes its + PID 1 — so a thread can never itself be PID 1, and + `/proc/self/ns/pid_for_children` has no value to bind-mount until that + first child exists. The kernel also destroys a PID namespace the instant + its PID 1 exits, after which no further process can be created in it, so a + bind mount cannot substitute for a live process the way it can for the + other types. The guest therefore starts a real anchor process, + `/sbin/nerdbox-pause` (a small `no_std` Rust binary, `crates/pause`), with + `SysProcAttr.Cloneflags: CLONE_NEWPID`, and bind-mounts + `/proc//ns/pid` to `/run/pidns/`. The anchor ignores + SIGINT and SIGTERM (`SIG_IGN`) so it cannot be torn down by a stray signal + delivered inside the shared namespace, and reaps reparented children via + `SA_NOCLDWAIT` (a PID-1-of-namespace duty) rather than an explicit + `wait()` loop. +- **DevShm**: like network/IPC, needs no anchor process — a plain + `mount("tmpfs", ...)` at `/run/devshm/` persists on its own for as + long as anything references it, with no separate process or bind-mount + step required to keep it alive. Cheap, and the same "only if actually + requested" reasoning applies: a pod that never shares IPC never causes + this tmpfs to be created either, since `shareDevShmMounter` only asks for + it when `containerSharesIPC` is true. + +Requesting only what is needed matters most for the PID namespace: Kubernetes +shares pod IPC by default but shares PID only when explicitly asked, so a +service that created both together would spawn an anchor process for +effectively every pod, whether or not anything used it. + +Sharing a PID namespace only changes what a container's processes can *see* +via `/proc` — it does not change what the shim's `Kill`/`Pids` TTRPC +handlers can *target*. Those requests identify a process by container ID +and exec ID, never by raw PID, and are resolved against that specific +container's own process table (`Kill`) or by running `crun ps +` scoped to that container's own cgroup (`Pids`). A signal +sent to one container therefore cannot land on a PID-namespace peer's +process, even though that peer's processes are visible to it. + +On the host side, `internal/shim/task/namespaces.go`'s `sanitizeNamespaces` +bundle transformer determines which namespaces a container needs, fetches +them from the guest in a single `SharedResources.Create` call (memoized per +`Task.Create` via `sharedResources`, in `internal/shim/task/sharedresources.go`), +and rewrites the spec's namespace paths to the returned guest paths. Any IPC, +UTS, or PID namespace entry with a non-empty incoming `Path` is treated as +"share within this sandbox". A container whose spec has no such entry at +all (the common case for PID: no pod-level sharing requested — IPC and UTS +always have one, per CRI's own defaults described above) never triggers the +guest RPC for that type, and therefore never causes the guest to create the +namespace on its behalf. + +Sharing a UTS namespace needs no extra coordination for the hostname value +itself, unlike IPC/PID which have no comparable per-container spec field +that could conflict. An OCI runtime setting a container's `Hostname` while +joining an existing (rather than freshly created) UTS namespace calls +`sethostname(2)` *after* joining it, which updates the namespace — and so +every container sharing it — rather than erroring; and leaves the namespace +alone when `Hostname` is empty. Since every member container of a pod +already carries the same CRI-provided hostname on its own spec, whichever +container's runtime happens to start last simply (re-)applies the same +value, giving "last write wins, empty means no opinion" behavior for free, +entirely through each container's own ordinary spec field — with nothing +for the shim or the guest's `SharedResources` service to track or +reconcile. (Verified directly against a real `crun`, not assumed from +reading the OCI runtime-spec.) + +### HostPID / HostIPC vs. PodPID + +containerd sets the *same* host path (derived from the sandbox's own PID) +for both `NamespaceMode_POD` (pod-level sharing) and `NamespaceMode_NODE` +(`hostPID`/`hostIPC: true`) — there is no data in the request that lets the +shim tell them apart, so both are treated identically: any non-empty +incoming `Path` is redirected to the pod's shared guest namespace. This +gives every member container of a pod a consistent, shared PID/IPC view +regardless of which CRI namespace mode requested it. + +## Sandbox lifecycle + +``` +containerd nerdbox shim VM + │ │ + │── CreateSandbox ──────────────►│ alloc state dir + │ (netns_path) │ create shared fs root + │ │ open netns FD (pin) + │ + │── StartSandbox ───────────────►│ add virtiofs "containers" share + │ │ parse bundle for resources/NICs + │ │ configure VM context (disks, FS, NICs) + │ │ [goroutine: LockOSThread, + │ │ setns into pod netns (if set), + │ │ krun_start_enter] + │ │ start VM ────────────────────►│ boot + │ │ │ vminitd starts + │ │◄── TTRPC connect (vsock 1025) ──│ + │ + │── Task.Create (ctr-A) ────────►│ ShareRootfs: mount rootfs + │ │ on host in shared dir + │ │ Bundle.Create ───────────────►│ + │ │ Mount.MountAll ──────────────►│ bind rootfs + │ │ Task.Create ─────────────────►│ runc create + │ + │── Task.Start (ctr-A) ─────────►│ Task.Start ──────────────────►│ runc start + │ │ │ container runs + │ + │── Task.Create (ctr-B) ────────►│ (same flow, same VM) + │── Task.Start (ctr-B) ─────────►│ + │ + │ [ pod running ] + │ + │── Task.Delete (ctr-A) ────────►│ Task.Delete ─────────────────►│ runc delete + │ │ SharedFS.Unshare(ctr-A) │ + │ │ unmount rootfs on host │ + │ │ remove shared dir entry │ + │ + │── StopSandbox ───────────────►│ SharedFS.UnshareAll + │ │ VM.Stop ──────────────────────►│ shutdown + │ │ netns FD closed (unpin) + │ + │ [ CNI DEL runs on host ] + │ + │── ShutdownSandbox ───────────►│ (idempotent stop if needed) +``` + +## TTRPC communication + +The host shim and vminitd communicate over two vsock channels: + +``` +Host shim vminitd (guest) + │ │ + │◄── vsock port 1025 (TTRPC) ──────►│ + │ Task, Bundle, Mount, │ + │ System, SocketForward, │ + │ Events, SharedResources, │ + │ Transfer services │ + │ │ + │◄── vsock port 1026 (streams) ────►│ + │ stdio (stdout/stderr/stdin) │ + │ transfer service data │ +``` + +vminitd **dials back** to the host on port 1025 (not the other way around), +which allows the host to accept the connection without needing to know the +guest CID in advance. + +## Security properties + +- The shim process runs in its own **mount namespace** (`CLONE_NEWNS`), plus a + **new user namespace** (`CLONE_NEWUSER`) when it is not already real root — + unprivileged callers gain CAP_SYS_ADMIN within that namespace to perform + rootfs mounts. When the shim is already real root (e.g. under `sudo`), + `CLONE_NEWUSER` is deliberately skipped: entering a *new* user namespace, + even one mapping root to root, demotes the process to a non-initial user + namespace, and the kernel restricts mounting real block-device-backed + filesystems (ext4, used for the sandbox scratch/overlay mounts) to the + initial user namespace regardless of capabilities held within a descendant + one. Whenever `CLONE_NEWNS` is actually applied (both the real-root branch + and a successful userns branch — not the apparmor-restricted fallback + described below, which sets no clone flags at all and shares the host's + mount namespace directly), the shim's new mount namespace is also remounted + `MS_REC | MS_SLAVE` on `/` before use. Without this, a mount namespace + created under a host root whose own `/` has `shared` propagation (the + common default) leaks every mount the shim makes back out into the host's + mount table; `MS_SLAVE` stops that leak in the host-visible direction while + still letting the shim's namespace receive host-side mount/unmount events. + Mounts made for container rootfs assembly are explicitly torn down by + `SharedFS.Unshare`/`UnshareAll` on `Task.Delete`/`StopSandbox` — nothing + relies on process exit to clean them up. This also holds when the shim + exits *without* running that cleanup — e.g. `SIGKILL`, a panic, or an OOM + kill — because `MS_SLAVE` means the host's mount table never had a copy of + the shim's mounts in the first place; the kernel discards them + unconditionally the moment the shim's mount namespace has no more + references, regardless of how the shim exited. (Confirmed directly: + killing a running shim with `SIGKILL` left zero entries for its bundle in + the host's own mount table, and the kernel promptly reused the freed mount + namespace's inode number for the next sandbox — evidence the namespace and + everything in it was fully reclaimed, not merely orphaned.) The one case + this doesn't cover is the apparmor-restricted fallback mentioned above: it + runs the shim directly in the host's own mount namespace, so a crash there + leaves real host mounts behind exactly as it would for any ordinary + process — see [Known limitations](#known-limitations). +- Container processes run inside the VM guest kernel. The guest kernel is a + different kernel instance from the host, providing strong isolation. +- The virtiofs share is writable (host-to-guest) but each container's subtree + is isolated: one container cannot see or modify another container's files + within the shared tree. +- The network sandbox FD is opened `O_RDONLY | O_CLOEXEC`. It exists solely + to pin the bind-mount for the CRI-managed netns lifetime; entering that + netns for VM boot (see [Layer 2](#layer-2--vm-network-stack)) opens its own, + separate FD on the `netns_path` rather than reusing this one. The shim's + control-plane goroutines remain in the shim's original network namespace. + +## Known limitations + +- **Shared `/dev/shm`'s size is set once, by whichever container asks + first.** The shared tmpfs (see [Shared `/dev/shm`](#shared-devshm)) is + created the first time any member container needing it is set up, sized + from that container's own CRI-provided `size=` mount option; a + later-created sibling with a *different* requested size does not resize + it — the guest's `SharedResources.Create` reuses the existing tmpfs + (matching every other resource type's "created once per id, reused + thereafter" contract) and logs a warning naming both sizes, but the call + itself still succeeds with the original size rather than failing. In + practice this is not expected to matter: CRI sends every member container + of a pod the same `/dev/shm` size, so there is normally nothing to + disagree about. +- **Mount-namespace crash safety does not cover the apparmor-restricted + fallback path.** As described in + [Security properties](#security-properties), a shim that cannot create a + user namespace due to `apparmor_restrict_unprivileged_userns=1` (and is + not already real root) runs with no mount namespace isolation at all — + its mounts are ordinary host mounts from the start. A crash in that + specific configuration leaves real, host-visible mounts behind, the same + as it would for any process outside of nerdbox; nothing currently scans + for and cleans up dangling mounts of this kind on shim startup. In + practice this only affects unprivileged shim invocations on + apparmor-restricted hosts — the shim is already skipping mount namespace + isolation entirely in that case, which is itself an existing, narrower + gap this doesn't change. +- **An inbound TSI ephemeral bind's resolved port can collide with something + already bound on the host.** Kernel patch + `0013-tsi-forward-the-resolved-port-for-ephemeral-binds.patch` (see + [TSI and guest network namespaces](#tsi-and-guest-network-namespaces)) + makes the guest forward its own resolved port to the host instead of the + literal `0` requested, so the host now attempts to bind that *specific* + port rather than picking its own free one. Two outcomes if it's already + taken, depending on how it's taken: + - **A normal bind held by something else on the host**: libkrun's TSI + proxy returns `EADDRINUSE`, and the guest kernel's own `tsi_listen` + fallback (pre-existing, previously essentially unreachable since a + host-side `bind(0)` could not fail) transparently switches to a real + in-guest `listen()` on the same socket. Host-external reachability at + that port is lost, but the application still gets a working listener, + and cross-container connectivity within the same VM is unaffected + (`tsi_connect` tries the in-guest socket first). + - **A listener from an unrelated VM's own TSI proxy on the same host + netns**: libkrun sets `SO_REUSEPORT` on TSI's host-side listening + sockets, so two different VMs' guest kernels — which cannot see each + other's port allocations and so can genuinely resolve the same + ephemeral port — both bind successfully, and the host kernel silently + load-balances inbound connections between two unrelated pods' listeners + with no error and no fallback triggered. This is only reachable when + multiple VMs' shims share a host network namespace (e.g. `ctr run` + without a CNI/pod netns); under CRI/Kubernetes each pod gets its own + netns containing only that pod's own shim, so this cannot occur in + practice today. Closing this properly means the host, not the guest, + should own ephemeral port allocation for TSI listeners — e.g. by + extending TSI's `tsi_listen_rsp` to report back the port the host + actually bound, and having the guest adopt it — which needs coordinated + kernel and libkrun changes; tracked as future work below rather than + attempted alongside the kernel-only kernel patch `0013` above. + +## Future work + +The following capabilities are planned but not yet implemented: + +- **Turnkey virtio networking** — have the shim spawn and manage a passt or + gvproxy process (inside the pod netns) rather than requiring a user-supplied + socket path via annotation. +- **Single ext4 upper layer** — a forthcoming containerd change will support + placing multiple container upper filesystems in one ext4 image, which can be + mounted upfront and eliminate per-container mount overhead on non-root hosts. +- **Host-authoritative TSI ephemeral port allocation** — extend libkrun's + `tsi_listen_rsp` to report back the concrete port the host actually bound + an inbound listener to, and have the guest kernel adopt it, rather than + the guest resolving its own ephemeral port and the host attempting to + match it (see [Known limitations](#known-limitations) for the + same-host-netns collision this can hit today). A libkrun bump is expected + soon regardless of this, which is the natural point to carry the + corresponding host-side change. diff --git a/go.mod b/go.mod index bfbbba86..82af6c4f 100644 --- a/go.mod +++ b/go.mod @@ -8,6 +8,7 @@ require ( github.com/containerd/console v1.0.5 github.com/containerd/containerd/api v1.11.1 github.com/containerd/containerd/v2 v2.3.5 + github.com/containerd/continuity v0.5.0 github.com/containerd/errdefs v1.0.0 github.com/containerd/errdefs/pkg v0.3.0 github.com/containerd/fifo v1.1.0 @@ -15,13 +16,14 @@ require ( github.com/containerd/log v0.2.0 github.com/containerd/otelttrpc v0.1.0 github.com/containerd/plugin v1.1.0 - github.com/containerd/shimtest v0.3.3 + github.com/containerd/shimtest v0.3.4-0.20260820001033-a0143efafccb github.com/containerd/ttrpc v1.2.9 github.com/containerd/typeurl/v2 v2.3.0 github.com/docker/go-events v0.1.0 github.com/ebitengine/purego v0.11.0 github.com/insomniacslk/dhcp v0.0.0-20260719225207-c76316d4aa82 github.com/mdlayher/vsock v1.3.0 + github.com/moby/sys/mountinfo v0.7.2 github.com/moby/sys/userns v0.2.1 github.com/opencontainers/runtime-spec v1.3.0 github.com/stretchr/testify v1.12.1 @@ -37,7 +39,6 @@ require ( github.com/Microsoft/hcsshim v0.15.0-rc.1 // indirect github.com/cespare/xxhash/v2 v2.3.0 // indirect github.com/cilium/ebpf v0.16.0 // indirect - github.com/containerd/continuity v0.5.0 // indirect github.com/containerd/platforms v1.0.0-rc.5 // indirect github.com/coreos/go-systemd/v22 v22.7.0 // indirect github.com/docker/go-units v0.5.0 // indirect @@ -50,7 +51,6 @@ require ( github.com/klauspost/compress v1.18.5 // indirect github.com/mdlayher/packet v1.1.2 // indirect github.com/mdlayher/socket v0.6.0 // indirect - github.com/moby/sys/mountinfo v0.7.2 // indirect github.com/opencontainers/go-digest v1.0.0 // indirect github.com/opencontainers/image-spec v1.1.1 // indirect github.com/pierrec/lz4/v4 v4.1.14 // indirect diff --git a/go.sum b/go.sum index 4fabf66c..81ff38e7 100644 --- a/go.sum +++ b/go.sum @@ -37,8 +37,8 @@ github.com/containerd/platforms v1.0.0-rc.5 h1:vXd569rDrz8LeMXzAnBsy6LADV5YtsD8o github.com/containerd/platforms v1.0.0-rc.5/go.mod h1:lKlMXyLybmBedS/JJm11uDofzI8L2v0J2ZbYvNsbq1A= github.com/containerd/plugin v1.1.0 h1:O+7lczNJVMy8rz0YNx3xGB8tTf5qY4i5abF041Ew19U= github.com/containerd/plugin v1.1.0/go.mod h1:qBTum+A8lJ6lO44A19Eo7y1OlcLj4OWFH1DA/vnHmcc= -github.com/containerd/shimtest v0.3.3 h1:n0bG1i5baAjrtFWtPySjIiDICq7Gmh50hhjvVkYUktU= -github.com/containerd/shimtest v0.3.3/go.mod h1:vT0DHiGsMJ6Hi56uGZLWS3o8gzRd0wSCC17Wen3XjP4= +github.com/containerd/shimtest v0.3.4-0.20260820001033-a0143efafccb h1:I+5Vl86FPBEBL5iIlFgrd1+RYpMuUoz1l0nb2/4XkWc= +github.com/containerd/shimtest v0.3.4-0.20260820001033-a0143efafccb/go.mod h1:vT0DHiGsMJ6Hi56uGZLWS3o8gzRd0wSCC17Wen3XjP4= github.com/containerd/ttrpc v1.2.9 h1:ha0ak962T0s3CA/RoZ6S6xiWZQF24GrBaEpiGX1uihg= github.com/containerd/ttrpc v1.2.9/go.mod h1:jjtQRwXm4DL3KsHKW8vDiUOV6wO0hi6IPhmJhxU7aEs= github.com/containerd/typeurl/v2 v2.3.0 h1:HZHPhRWo5XMy3QGQoPrUzbW/2ckwjfweHmOwlkIrPAQ= diff --git a/internal/mountutil/mount.go b/internal/mountutil/mount.go index a6da9d8f..1f89f28a 100644 --- a/internal/mountutil/mount.go +++ b/internal/mountutil/mount.go @@ -39,6 +39,27 @@ func All(ctx context.Context, rootfs, mdir string, mounts []*types.Mount) (retEr log.G(ctx).WithField("mounts", mounts).Debug("mounting rootfs components") active := []mount.ActiveMount{} + // Registered before the loop below (rather than after it, as originally + // written) so that it actually runs when the loop returns early on + // error: a defer only takes effect once the defer statement itself + // executes, and every error path inside the loop returns directly, + // never reaching a defer statement placed after the loop. Without this, + // a failure partway through left every mount already established by + // this call active and untracked by any caller. + defer func() { + if retErr != nil { + for i := len(active) - 1; i >= 0; i-- { + // TODO: delegate custom types to handlers + if active[i].Type == "mkdir" { + continue + } + if err := mount.UnmountAll(active[i].MountPoint, 0); err != nil { + log.G(ctx).WithError(err).WithField("mountpoint", active[i].MountPoint).Warn("failed to cleanup mount") + } + } + } + }() + // TODO: Use mount manager interface, mount temps to directory for i, m := range mounts { var target string @@ -130,19 +151,6 @@ func All(ctx context.Context, rootfs, mdir string, mounts []*types.Mount) (retEr active = append(active, am) } - defer func() { - if retErr != nil { - for i := len(active) - 1; i >= 0; i-- { - // TODO: delegate custom types to handlers - if active[i].Type == "mkdir" { - continue - } - if err := mount.UnmountAll(active[i].MountPoint, 0); err != nil { - log.G(ctx).WithError(err).WithField("mountpoint", active[i].MountPoint).Warn("failed to cleanup mount") - } - } - } - }() return nil } diff --git a/internal/shim/sandbox/networksandbox.go b/internal/shim/sandbox/networksandbox.go new file mode 100644 index 00000000..b6599328 --- /dev/null +++ b/internal/shim/sandbox/networksandbox.go @@ -0,0 +1,58 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +// NetworkSandbox represents the host-side network isolation resource +// associated with a sandbox. The concept is intentionally abstract so that +// it can be represented differently on each platform: +// +// - Linux: a bind-mounted network namespace file path. The caller (CRI) +// creates and owns the netns; the sandbox holds it open for the lifetime +// of the sandbox so that CNI and other host-side tools can inspect or +// manipulate it after the sandbox process has started. +// - Other platforms: the concept does not exist; the zero-value (NoNetworkSandbox) +// represents the absence of a network sandbox, which is also the host-network +// (no isolation) case on Linux. +// +// NetworkSandbox is used as the cross-platform public interface for the +// network sandbox lifecycle. Platform-specific implementations satisfy it. +type NetworkSandbox interface { + // Path returns the platform-specific path that identifies the network + // sandbox. On Linux this is the network namespace file path. + // Returns an empty string when there is no network sandbox (host network). + Path() string + + // Close releases any host-side resources held by the NetworkSandbox. + // Calling Close on a NoNetworkSandbox is a no-op. + Close() error +} + +// NoNetworkSandbox is a NetworkSandbox that represents the absence of any +// host-side network isolation — used for host-network pods or on platforms +// that do not support network namespaces. +type NoNetworkSandbox struct{} + +// Path returns an empty string (no network sandbox). +func (NoNetworkSandbox) Path() string { return "" } + +// Close is a no-op. +func (NoNetworkSandbox) Close() error { return nil } + +// openNetworkSandbox is the platform-specific factory. It is defined +// in networksandbox_linux.go (real netns FD) and +// networksandbox_other.go (no-op NoNetworkSandbox). +var openNetworkSandbox func(path string) (NetworkSandbox, error) diff --git a/internal/shim/sandbox/networksandbox_linux.go b/internal/shim/sandbox/networksandbox_linux.go new file mode 100644 index 00000000..0e78c690 --- /dev/null +++ b/internal/shim/sandbox/networksandbox_linux.go @@ -0,0 +1,95 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "fmt" + "os" + + "github.com/containerd/log" + "golang.org/x/sys/unix" +) + +// nsfsMagic is the filesystem magic number for Linux nsfs (the filesystem +// that backs namespace files under /proc/*/ns/). Mirrors the identical +// constant in internal/vm/libkrun/krun_linux.go, kept local to this +// package rather than shared to avoid a dependency between the two for a +// single well-known constant. +const nsfsMagic = 0x6e736673 + +func init() { + openNetworkSandbox = linuxOpenNetworkSandbox +} + +// linuxNetworkSandbox holds an open file descriptor to a Linux network +// namespace bind-mount. The open FD keeps the bind-mount alive for the +// lifetime of the sandbox, satisfying the CRI contract that the netns +// remains pinned while the sandbox is running — regardless of whether any +// process is actively in it. +type linuxNetworkSandbox struct { + path string + fd *os.File +} + +// linuxOpenNetworkSandbox opens the network namespace at path and returns a +// NetworkSandbox that holds the FD open. Returns NoNetworkSandbox when path +// is empty (host-network pod). +func linuxOpenNetworkSandbox(path string) (NetworkSandbox, error) { + if path == "" { + return NoNetworkSandbox{}, nil + } + + f, err := os.OpenFile(path, os.O_RDONLY|unix.O_CLOEXEC, 0) + if err != nil { + return nil, fmt.Errorf("open network sandbox %q: %w", path, err) + } + + // Verify path is actually backed by nsfs (the filesystem that exposes + // kernel namespace files under /proc/*/ns/), so that a bind-mounted + // netns is confirmed to be a real network namespace, not some other + // file that happens to sit at the given path. + // + // A plain regular file (not nsfs) is deliberately still accepted + // rather than rejected: shimtest's non-root test suite pins a plain + // file in place of a real netns specifically to avoid requiring + // CAP_SYS_ADMIN or kernel bind-mount support, matching the identical + // tolerance in internal/vm/libkrun's vmcontextSetNetns. + var sfs unix.Statfs_t + if err := unix.Fstatfs(int(f.Fd()), &sfs); err != nil { + f.Close() + return nil, fmt.Errorf("statfs network sandbox %q: %w", path, err) + } + if sfs.Type != nsfsMagic { + log.L.WithField("netns", path).Debug( + "network sandbox path is not an nsfs file; pinning it anyway (test or non-standard path)") + } + + return &linuxNetworkSandbox{path: path, fd: f}, nil +} + +// Path returns the network namespace file path. +func (n *linuxNetworkSandbox) Path() string { return n.path } + +// Close closes the held FD, releasing the pin on the network namespace. +func (n *linuxNetworkSandbox) Close() error { + if n.fd == nil { + return nil + } + err := n.fd.Close() + n.fd = nil + return err +} diff --git a/internal/shim/sandbox/networksandbox_linux_test.go b/internal/shim/sandbox/networksandbox_linux_test.go new file mode 100644 index 00000000..c6e02971 --- /dev/null +++ b/internal/shim/sandbox/networksandbox_linux_test.go @@ -0,0 +1,67 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "os" + "path/filepath" + "testing" +) + +// TestLinuxOpenNetworkSandbox_Empty verifies an empty path returns +// NoNetworkSandbox rather than attempting to open anything. +func TestLinuxOpenNetworkSandbox_Empty(t *testing.T) { + ns, err := linuxOpenNetworkSandbox("") + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if _, ok := ns.(NoNetworkSandbox); !ok { + t.Fatalf("expected NoNetworkSandbox, got %T", ns) + } +} + +// TestLinuxOpenNetworkSandbox_PlainFile verifies that a plain regular file +// (not backed by nsfs) is still accepted rather than rejected -- this is +// the technique shimtest's non-root suite uses to pin a fake network +// sandbox without requiring CAP_SYS_ADMIN or kernel bind-mount support. +func TestLinuxOpenNetworkSandbox_PlainFile(t *testing.T) { + path := filepath.Join(t.TempDir(), "network-sandbox") + f, err := os.OpenFile(path, os.O_CREATE|os.O_EXCL|os.O_RDONLY, 0o444) + if err != nil { + t.Fatalf("create test file: %v", err) + } + f.Close() + + ns, err := linuxOpenNetworkSandbox(path) + if err != nil { + t.Fatalf("unexpected error opening a plain-file network sandbox: %v", err) + } + defer ns.Close() + + if ns.Path() != path { + t.Fatalf("Path() = %q, want %q", ns.Path(), path) + } +} + +// TestLinuxOpenNetworkSandbox_Missing verifies that a nonexistent path +// returns an error rather than silently succeeding. +func TestLinuxOpenNetworkSandbox_Missing(t *testing.T) { + path := filepath.Join(t.TempDir(), "does-not-exist") + if _, err := linuxOpenNetworkSandbox(path); err == nil { + t.Fatalf("expected error for a nonexistent network sandbox path") + } +} diff --git a/internal/shim/sandbox/networksandbox_other.go b/internal/shim/sandbox/networksandbox_other.go new file mode 100644 index 00000000..ce46df52 --- /dev/null +++ b/internal/shim/sandbox/networksandbox_other.go @@ -0,0 +1,28 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +func init() { + // Non-Linux platforms do not have kernel network namespaces exposed + // as bind-mountable files. Always return NoNetworkSandbox regardless + // of the requested path. + openNetworkSandbox = func(_ string) (NetworkSandbox, error) { + return NoNetworkSandbox{}, nil + } +} diff --git a/internal/shim/sandbox/sandbox.go b/internal/shim/sandbox/sandbox.go index d6c18b56..fbd2eca6 100644 --- a/internal/shim/sandbox/sandbox.go +++ b/internal/shim/sandbox/sandbox.go @@ -75,6 +75,11 @@ type Options struct { InitArgs []string CPU uint8 Memory uint32 // in MiB + // NetnsPath is the host-side network namespace path (e.g. + // /var/run/netns/cni-) to enter on the libkrun FFI thread before + // any FFI calls are made. Empty means host-network (no namespace + // entry). + NetnsPath string } type Opt func(*Options) @@ -129,3 +134,12 @@ func WithResources(cpu uint8, memory uint32) Opt { o.Memory = memory } } + +// WithNetnsPath sets the host-side network namespace path that the VM's +// libkrun FFI thread will enter (via setns) before any context configuration +// calls. An empty path means host-network — no namespace entry. +func WithNetnsPath(path string) Opt { + return func(o *Options) { + o.NetnsPath = path + } +} diff --git a/internal/shim/sandbox/service.go b/internal/shim/sandbox/service.go new file mode 100644 index 00000000..c7e5deee --- /dev/null +++ b/internal/shim/sandbox/service.go @@ -0,0 +1,461 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "fmt" + "net" + "os" + "path/filepath" + "runtime" + "sync" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + "github.com/containerd/containerd/api/types" + "github.com/containerd/errdefs" + "github.com/containerd/errdefs/pkg/errgrpc" + "github.com/containerd/log" + "github.com/containerd/ttrpc" + "google.golang.org/protobuf/types/known/anypb" + "google.golang.org/protobuf/types/known/timestamppb" +) + +const ( + // sandboxStateReady is returned by SandboxStatus once StartSandbox has + // completed successfully. This must be exactly "SANDBOX_READY" — not a + // human-readable state name — because containerd's CRI layer + // (internal/cri/server/sandbox_status.go, toCRISandboxStatus) looks this + // string up in runtime.PodSandboxState_value, the CRI v1 + // PodSandboxState enum's name-to-value map, to derive + // PodSandboxStatus.State. Any string that isn't a name in that enum + // (including a more "sensible" one like "ready") silently falls back to + // SANDBOX_NOTREADY, so a real CRI client would see every sandbox as + // permanently not-ready even while StartSandbox has succeeded and + // containers are running in it. + sandboxStateReady = "SANDBOX_READY" + // sandboxStateStopped is returned after StopSandbox and before + // StartSandbox has completed. The CRI v1 PodSandboxState enum has only + // two values (ready / not ready) — there is no separate "stopped" vs + // "never started" state — so both map to the same string here. + sandboxStateStopped = "SANDBOX_NOTREADY" +) + +// StartOptionsFunc is a callback that the task service registers with the +// SandboxService to provide VM start options (networking, resources, init +// args) derived from the sandbox OCI bundle. It is called during StartSandbox +// before the VM boots. +// +// Using a callback avoids a circular import between the sandbox and task +// packages: the task package owns bundle parsing; the sandbox package owns +// VM lifecycle. +type StartOptionsFunc func(ctx context.Context, bundlePath string) ([]Opt, error) + +// SandboxService implements the containerd TTRPCSandboxService and is +// registered on the shim's TTRPC server alongside the Task service. It owns +// the VM lifecycle: CreateSandbox prepares the shared filesystem and VM +// configuration; StartSandbox boots the VM; StopSandbox/ShutdownSandbox tear +// it down. +// +// The task service acquires the already-running VM via the shared Sandbox +// interface and the SharedFS returned by FS(). +type SandboxService struct { + mu sync.Mutex + + sb Sandbox // underlying VM sandbox + sharedFS *SharedFS // shared host↔guest filesystem tree + + // networkSandbox pins the host-side network isolation resource + // (e.g. a Linux network namespace bind-mount) for the lifetime of the + // sandbox. It is set in CreateSandbox from CreateSandboxRequest.NetnsPath + // and released in StopSandbox/ShutdownSandbox. + networkSandbox NetworkSandbox + + // startOptsFn, if non-nil, is called in StartSandbox to get bundle-derived + // VM start options (networking, resources, init args). Set by the task + // plugin via RegisterStartOptions before Start is called. + startOptsFn StartOptionsFunc + + // lifecycle state + sandboxID string + bundlePath string + stateDir string + pid uint32 + createdAt time.Time + state string // "" | sandboxStateReady | sandboxStateStopped + exitCh chan struct{} + exitOnce sync.Once + + // options holds CreateSandboxRequest.Options verbatim: an opaque, + // caller-defined payload (in production CRI, a marshaled + // k8s.io/cri-api PodSandboxConfig — see internal/cri/server's + // sandbox_run.go, sandbox.WithOptions). The sandbox package + // deliberately does not interpret it: unmarshaling CRI-specific types + // is left to the task package (which already owns other CRI-shaped + // concerns like DNS annotations), keeping this package's API surface + // generic to the shim-v2 sandbox protocol rather than coupled to CRI. + options *anypb.Any +} + +var _ sandboxAPI.TTRPCSandboxService = (*SandboxService)(nil) + +// NewSandboxService creates a SandboxService backed by the given Sandbox. +func NewSandboxService(sb Sandbox) *SandboxService { + return &SandboxService{ + sb: sb, + exitCh: make(chan struct{}), + } +} + +// RegisterStartOptions installs a callback that SandboxService calls during +// StartSandbox to obtain bundle-derived VM options. The task plugin calls this +// at initialisation time before any sandbox RPCs arrive. +func (s *SandboxService) RegisterStartOptions(fn StartOptionsFunc) { + s.mu.Lock() + defer s.mu.Unlock() + s.startOptsFn = fn +} + +// FS returns the SharedFS associated with this sandbox, or nil if the sandbox +// has not been created yet. The task service uses this to share container +// rootfses into the VM. +func (s *SandboxService) FS() *SharedFS { + s.mu.Lock() + defer s.mu.Unlock() + return s.sharedFS +} + +// IsSandboxed returns true once CreateSandbox has been called. The task +// service uses this to distinguish the sandbox API path from the legacy +// single-container path. +func (s *SandboxService) IsSandboxed() bool { + s.mu.Lock() + defer s.mu.Unlock() + return s.sandboxID != "" +} + +// CreateSandbox is called by containerd right after the shim starts. It +// records the sandbox ID and bundle path, allocates the state directory, and +// creates the shared filesystem tree. The VM is NOT started here — that +// happens in StartSandbox. +func (s *SandboxService) CreateSandbox(ctx context.Context, req *sandboxAPI.CreateSandboxRequest) (*sandboxAPI.CreateSandboxResponse, error) { + s.mu.Lock() + defer s.mu.Unlock() + + log.G(ctx).WithField("sandboxID", req.SandboxID).Info("CreateSandbox") + + if s.sandboxID != "" { + return nil, errgrpc.ToGRPC(fmt.Errorf("sandbox already created: %w", errdefs.ErrAlreadyExists)) + } + + bundlePath := req.BundlePath + if bundlePath == "" { + // Fall back to the shim's current working directory. + var err error + bundlePath, err = os.Getwd() + if err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("getwd: %w", err)) + } + } + + // State lives under the shim working directory. + stateDir, err := filepath.Abs("vm") + if err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("abs vm state dir: %w", err)) + } + if err := os.MkdirAll(stateDir, 0o700); err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("create vm state dir: %w", err)) + } + + sharedFS, err := NewSharedFS(stateDir) + if err != nil { + return nil, errgrpc.ToGRPC(err) + } + + // Open the host-side network sandbox (e.g. a Linux netns bind-mount). + // The open FD pins the resource for the sandbox lifetime so that CNI + // and other host-side tools can reference it while the sandbox runs. + // An empty NetnsPath means host-network — openNetworkSandbox returns + // a no-op NoNetworkSandbox in that case. + ns, err := openNetworkSandbox(req.NetnsPath) + if err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("open network sandbox: %w", err)) + } + if ns.Path() != "" { + log.G(ctx).WithFields(log.Fields{ + "sandboxID": req.SandboxID, + "netns": ns.Path(), + }).Debug("network sandbox pinned") + } + + s.sandboxID = req.SandboxID + s.bundlePath = bundlePath + s.stateDir = stateDir + s.sharedFS = sharedFS + s.networkSandbox = ns + s.options = req.Options + s.state = "" + + return &sandboxAPI.CreateSandboxResponse{}, nil +} + +// Options returns CreateSandboxRequest.Options verbatim (nil if none was +// given, or CreateSandbox has not been called yet). See the field's doc +// comment on SandboxService for why this package does not interpret it. +func (s *SandboxService) Options() *anypb.Any { + s.mu.Lock() + defer s.mu.Unlock() + return s.options +} + +// SandboxID returns the ID of the sandbox this service manages, or the empty +// string if CreateSandbox has not been called yet. +func (s *SandboxService) SandboxID() string { + s.mu.Lock() + defer s.mu.Unlock() + return s.sandboxID +} + +// NetworkSandboxPath returns the host-side network sandbox path (e.g. a Linux netns) +func (s *SandboxService) NetworkSandboxPath() string { + s.mu.Lock() + defer s.mu.Unlock() + if s.networkSandbox != nil { + return s.networkSandbox.Path() + } + return "" +} + +// StartSandbox boots the VM. It calls the registered StartOptionsFunc (if +// any) to obtain bundle-derived options (networking, resources, init args), +// then adds the shared filesystem share and starts the VM. +func (s *SandboxService) StartSandbox(ctx context.Context, req *sandboxAPI.StartSandboxRequest) (*sandboxAPI.StartSandboxResponse, error) { + s.mu.Lock() + defer s.mu.Unlock() + + log.G(ctx).WithField("sandboxID", req.SandboxID).Info("StartSandbox") + + if s.sandboxID == "" { + return nil, errgrpc.ToGRPC(fmt.Errorf("sandbox not created: %w", errdefs.ErrFailedPrecondition)) + } + if s.state == sandboxStateReady { + return nil, errgrpc.ToGRPC(fmt.Errorf("sandbox already started: %w", errdefs.ErrAlreadyExists)) + } + // A stopped sandbox cannot be restarted: StopSandbox has already + // released s.networkSandbox (set to nil below) and closed s.exitCh, + // neither of which this function knows how to recreate. Restarting a + // stopped sandbox is not part of the shim-v2 sandbox lifecycle + // contract anyway — once stopped, a sandbox is torn down, not reused. + if s.state == sandboxStateStopped { + return nil, errgrpc.ToGRPC(fmt.Errorf("sandbox already stopped: %w", errdefs.ErrFailedPrecondition)) + } + + // Base options: state dir, the single shared virtiofs share, and the + // pod network namespace path (empty = host-network / no namespace entry). + opts := []Opt{ + WithStateDir(s.stateDir), + WithFS(SharedFSTag, s.sharedFS.Root(), false), + WithNetnsPath(s.networkSandbox.Path()), + } + + // Append bundle-derived options (networking, resources, init args). + if s.startOptsFn != nil { + bundleOpts, err := s.startOptsFn(ctx, s.bundlePath) + if err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("sandbox start options: %w", err)) + } + opts = append(opts, bundleOpts...) + } + + if err := s.sb.Start(ctx, opts...); err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("start VM: %w", err)) + } + + s.createdAt = time.Now() + s.pid = uint32(os.Getpid()) + s.state = sandboxStateReady + + return &sandboxAPI.StartSandboxResponse{ + Pid: s.pid, + CreatedAt: timestamppb.New(s.createdAt), + }, nil +} + +// Platform returns the platform the sandbox runs containers on. +func (s *SandboxService) Platform(_ context.Context, _ *sandboxAPI.PlatformRequest) (*sandboxAPI.PlatformResponse, error) { + return &sandboxAPI.PlatformResponse{ + Platform: &types.Platform{ + OS: "linux", + Architecture: runtime.GOARCH, + }, + }, nil +} + +// StopSandbox stops the VM. It cleans up all host-side container mounts +// before shutting down the VM to ensure clean state. +func (s *SandboxService) StopSandbox(ctx context.Context, req *sandboxAPI.StopSandboxRequest) (*sandboxAPI.StopSandboxResponse, error) { + log.G(ctx).WithField("sandboxID", req.SandboxID).Info("StopSandbox") + + s.mu.Lock() + defer s.mu.Unlock() + + if s.state != sandboxStateReady { + return &sandboxAPI.StopSandboxResponse{}, nil + } + + if s.sharedFS != nil { + if err := s.sharedFS.UnshareAll(ctx); err != nil { + log.G(ctx).WithError(err).Warn("failed to unshare all containers on stop") + } + } + + if err := s.sb.Stop(ctx); err != nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("stop VM: %w", err)) + } + + // Release the network sandbox pin after the VM stops so that CNI can + // run its teardown while the sandbox was still marked running. + if s.networkSandbox != nil { + if err := s.networkSandbox.Close(); err != nil { + log.G(ctx).WithError(err).Warn("failed to close network sandbox on stop") + } + s.networkSandbox = nil + } + + s.state = sandboxStateStopped + s.exitOnce.Do(func() { close(s.exitCh) }) + + return &sandboxAPI.StopSandboxResponse{}, nil +} + +// WaitSandbox blocks until the sandbox has exited. +func (s *SandboxService) WaitSandbox(ctx context.Context, req *sandboxAPI.WaitSandboxRequest) (*sandboxAPI.WaitSandboxResponse, error) { + log.G(ctx).WithField("sandboxID", req.SandboxID).Debug("WaitSandbox") + + select { + case <-s.exitCh: + case <-ctx.Done(): + return nil, errgrpc.ToGRPC(ctx.Err()) + } + + return &sandboxAPI.WaitSandboxResponse{ + ExitStatus: 0, + ExitedAt: timestamppb.Now(), + }, nil +} + +// SandboxStatus returns the current status of the sandbox. +func (s *SandboxService) SandboxStatus(_ context.Context, req *sandboxAPI.SandboxStatusRequest) (*sandboxAPI.SandboxStatusResponse, error) { + s.mu.Lock() + defer s.mu.Unlock() + + state := s.state + if state == "" { + // Created but not yet started: also not-ready, per the same + // SANDBOX_READY/SANDBOX_NOTREADY contract documented on + // sandboxStateReady above. + state = sandboxStateStopped + } + + // Populate the info map with observable sandbox metadata so that + // callers (CRI, tests) can inspect sandbox state without additional + // side-channel calls. + info := map[string]string{ + "state": state, + "pid": fmt.Sprintf("%d", s.pid), + } + if s.networkSandbox != nil && s.networkSandbox.Path() != "" { + info["networkSandboxPath"] = s.networkSandbox.Path() + } + + return &sandboxAPI.SandboxStatusResponse{ + SandboxID: req.SandboxID, + Pid: s.pid, + State: state, + Info: info, + CreatedAt: timestamppb.New(s.createdAt), + }, nil +} + +// PingSandbox is a lightweight liveness check. +func (s *SandboxService) PingSandbox(_ context.Context, _ *sandboxAPI.PingRequest) (*sandboxAPI.PingResponse, error) { + return &sandboxAPI.PingResponse{}, nil +} + +// ShutdownSandbox fully tears down the sandbox. containerd calls this after +// StopSandbox. +// +// Propagates a StopSandbox failure rather than reporting success +// unconditionally: a caller that only sees success has no reason to retry, +// so a failed VM shutdown, network sandbox release, or mount cleanup would +// otherwise go unnoticed and unretried, potentially leaking all three. A +// retry is safe: StopSandbox's own state check makes a second call a no-op +// once it has actually succeeded, and re-running it after a failure re-runs +// only the steps that did not complete (SharedFS.UnshareAll is safe to call +// again — see Unshare's own idempotency — and the underlying VM's Shutdown +// guards against being torn down twice). +func (s *SandboxService) ShutdownSandbox(ctx context.Context, req *sandboxAPI.ShutdownSandboxRequest) (*sandboxAPI.ShutdownSandboxResponse, error) { + log.G(ctx).WithField("sandboxID", req.SandboxID).Info("ShutdownSandbox") + + if _, err := s.StopSandbox(ctx, &sandboxAPI.StopSandboxRequest{SandboxID: req.SandboxID}); err != nil { + // Not re-wrapped: StopSandbox already returns an errgrpc.ToGRPC + // error carrying a gRPC status; wrapping it again risks losing + // that status across the TTRPC boundary depending on how the + // transport extracts it. + return nil, err + } + + return &sandboxAPI.ShutdownSandboxResponse{}, nil +} + +// SandboxMetrics returns metrics for the sandbox. +func (s *SandboxService) SandboxMetrics(_ context.Context, _ *sandboxAPI.SandboxMetricsRequest) (*sandboxAPI.SandboxMetricsResponse, error) { + return nil, errgrpc.ToGRPC(fmt.Errorf("metrics not implemented: %w", errdefs.ErrNotImplemented)) +} + +// ── Sandbox interface delegation ────────────────────────────────────────────── +// SandboxService implements the Sandbox interface by delegating to the inner +// sandbox VM. This allows the task service to accept a *SandboxService and +// use it both as a Sandbox (for VM communication) and as a SandboxService +// (for SharedFS access and lifecycle state). + +// Start implements Sandbox. The task service calls this on the legacy +// single-container path where no CreateSandbox/StartSandbox RPCs arrive. +func (s *SandboxService) Start(ctx context.Context, opts ...Opt) error { + return s.sb.Start(ctx, opts...) +} + +// Stop implements Sandbox. +func (s *SandboxService) Stop(ctx context.Context) error { + return s.sb.Stop(ctx) +} + +// Client implements Sandbox. Returns the TTRPC client connected to vminitd. +func (s *SandboxService) Client() (*ttrpc.Client, error) { + return s.sb.Client() +} + +// StartStream implements Sandbox. +func (s *SandboxService) StartStream(ctx context.Context, streamID string) (net.Conn, error) { + return s.sb.StartStream(ctx, streamID) +} + +// ReservedDisks implements Sandbox. +func (s *SandboxService) ReservedDisks() int { + return s.sb.ReservedDisks() +} diff --git a/internal/shim/sandbox/service_test.go b/internal/shim/sandbox/service_test.go new file mode 100644 index 00000000..6983067a --- /dev/null +++ b/internal/shim/sandbox/service_test.go @@ -0,0 +1,107 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "errors" + "net" + "strings" + "testing" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + "github.com/containerd/ttrpc" +) + +// fakeSandbox is a minimal Sandbox for exercising SandboxService's own +// lifecycle state machine independent of any real VM. The zero value never +// fails; set stopErr to make Stop fail, for testing error propagation out +// of StopSandbox/ShutdownSandbox. +type fakeSandbox struct { + stopErr error +} + +func (fakeSandbox) Start(context.Context, ...Opt) error { return nil } +func (f fakeSandbox) Stop(context.Context) error { return f.stopErr } +func (fakeSandbox) Client() (*ttrpc.Client, error) { return nil, nil } +func (fakeSandbox) StartStream(context.Context, string) (net.Conn, error) { return nil, nil } +func (fakeSandbox) ReservedDisks() int { return 0 } + +// newTestSandboxService builds a SandboxService in the "created" state +// without going through CreateSandbox, which does real filesystem I/O +// (creating a "vm" state dir relative to the process's current directory). +// Setting the fields directly is safe here: this test file is in the same +// package. +func newTestSandboxService(t *testing.T, sb Sandbox) *SandboxService { + t.Helper() + sharedFS, err := NewSharedFS(t.TempDir()) + if err != nil { + t.Fatalf("NewSharedFS: %v", err) + } + s := NewSandboxService(sb) + s.sandboxID = "test-sandbox" + s.stateDir = t.TempDir() + s.sharedFS = sharedFS + s.networkSandbox = NoNetworkSandbox{} + s.state = "" + return s +} + +// TestStartSandboxAfterStopSandbox verifies that StartSandbox rejects a +// sandbox that has already been stopped, rather than panicking on +// s.networkSandbox.Path() — StopSandbox sets networkSandbox to nil, and +// prior to this test's fix, StartSandbox only rejected the already-ready +// state, not the already-stopped one, so a restart attempt would +// dereference that nil. +func TestStartSandboxAfterStopSandbox(t *testing.T) { + ctx := context.Background() + s := newTestSandboxService(t, fakeSandbox{}) + + if _, err := s.StartSandbox(ctx, &sandboxAPI.StartSandboxRequest{SandboxID: s.sandboxID}); err != nil { + t.Fatalf("first StartSandbox: %v", err) + } + if _, err := s.StopSandbox(ctx, &sandboxAPI.StopSandboxRequest{SandboxID: s.sandboxID}); err != nil { + t.Fatalf("StopSandbox: %v", err) + } + + // This must return a clean error, not panic. + if _, err := s.StartSandbox(ctx, &sandboxAPI.StartSandboxRequest{SandboxID: s.sandboxID}); err == nil { + t.Fatal("StartSandbox after StopSandbox: got nil error, want one") + } +} + +// TestShutdownSandboxPropagatesStopError verifies that ShutdownSandbox +// surfaces a StopSandbox failure to the caller instead of unconditionally +// reporting success. A caller that only ever sees success has no signal to +// retry, so a failed VM shutdown would otherwise go unnoticed. +func TestShutdownSandboxPropagatesStopError(t *testing.T) { + ctx := context.Background() + wantErr := errors.New("vm stop failed") + s := newTestSandboxService(t, fakeSandbox{stopErr: wantErr}) + + if _, err := s.StartSandbox(ctx, &sandboxAPI.StartSandboxRequest{SandboxID: s.sandboxID}); err != nil { + t.Fatalf("StartSandbox: %v", err) + } + + _, err := s.ShutdownSandbox(ctx, &sandboxAPI.ShutdownSandboxRequest{SandboxID: s.sandboxID}) + if err == nil { + t.Fatal("ShutdownSandbox: got nil error, want one (Sandbox.Stop failed)") + } + if !strings.Contains(err.Error(), wantErr.Error()) { + t.Fatalf("ShutdownSandbox error = %v, want it to mention %q", err, wantErr) + } +} diff --git a/internal/shim/sandbox/sharedfs.go b/internal/shim/sandbox/sharedfs.go new file mode 100644 index 00000000..45b47492 --- /dev/null +++ b/internal/shim/sandbox/sharedfs.go @@ -0,0 +1,213 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "fmt" + "os" + "path" + "path/filepath" + "strings" + "sync" + + "github.com/containerd/containerd/api/types" +) + +// SharedFSTag is the virtiofs share tag used for the per-sandbox container +// filesystem tree. The guest mounts this at GuestContainersDir. +const SharedFSTag = "containers" + +// GuestContainersDir is the path in the guest where the shared filesystem +// is mounted. Per-container rootfs and volumes live under: +// +// /run/containers//rootfs +// /run/containers//volumes/ +// +// /run is backed by a tmpfs in the guest so the mount point is always +// writable even on the read-only erofs base rootfs. +const GuestContainersDir = "/run/containers" + +// SharedFS manages the host-side directory tree shared with the VM via a +// single virtiofs mount. It creates per-container subdirectories, assembles +// the container rootfs from snapshotter-provided mounts, and tears everything +// down on container delete. +// +// The root directory is /containers. It is added to the VM as +// a virtiofs share with tag "containers" before the VM starts and must not be +// modified until after the VM shuts down. +// +// Thread-safe: all exported methods may be called concurrently. +// +// ShareRootfs, ShareVolume, Unshare, and UnshareAll assemble the shared tree +// using real host-side mounts (bind/overlay/etc.) and are therefore only +// implemented on Linux today (see sharedfs_linux.go); sharedfs_other.go +// provides a not-supported stub for other platforms. +type SharedFS struct { + mu sync.Mutex + root string // host path of the shared dir + // mounts tracks the mount points we created per container so we can + // unmount them precisely on Unshare. + mounts map[string][]string // containerID -> ordered list of host mount points +} + +// NewSharedFS creates a SharedFS rooted at /containers. +// The directory is created if it does not exist. +func NewSharedFS(stateDir string) (*SharedFS, error) { + root := filepath.Join(stateDir, "containers") + if err := os.MkdirAll(root, 0o755); err != nil { + return nil, fmt.Errorf("create shared containers dir %s: %w", root, err) + } + return &SharedFS{ + root: root, + mounts: make(map[string][]string), + }, nil +} + +// Root returns the host-side root of the shared filesystem. This path is +// passed to the VM as the backing directory for the virtiofs share. +func (s *SharedFS) Root() string { + return s.root +} + +// GuestRootfsPath returns the in-guest path of the container's assembled +// rootfs, suitable for passing to the guest Task.Create as the rootfs source. +// +// This uses path.Join, not filepath.Join: the guest is always Linux +// regardless of the host OS this shim runs on, so the result must always +// use '/' separators, even on a Windows host (where filepath.Join would +// use '\' and produce a path the guest can't use). +func GuestRootfsPath(containerID string) string { + return path.Join(GuestContainersDir, containerID, "rootfs") +} + +// GuestVolumePath returns the in-guest path for volume mount n of the given +// container (0-indexed), suitable for bind-mounting into the container. +// See GuestRootfsPath for why this uses path.Join rather than filepath.Join. +func GuestVolumePath(containerID string, n int) string { + return path.Join(GuestContainersDir, containerID, "volumes", fmt.Sprintf("%d", n)) +} + +// RootfsHostPath returns the host-side path where ShareRootfs assembles the +// container's rootfs (the same directory GuestRootfsPath(containerID) +// exposes to the guest via the virtiofs share). Only meaningful after +// ShareRootfs has returned successfully for containerID: this is a pure +// path builder with no validation of its own (it has no error to report +// one with), so callers must only act on its result once containerID has +// already been accepted by ShareRootfs, ShareVolume, or Unshare. +func (s *SharedFS) RootfsHostPath(containerID string) string { + return filepath.Join(s.root, containerID, "rootfs") +} + +// validateContainerID rejects container IDs that are empty or that could +// escape s.root (on the host) or GuestContainersDir (in the guest) once +// joined onto it — e.g. "..", "../x", or an ID containing a path +// separator. containerID arrives over the shim-v2 TTRPC API from +// Task.Create and is never trusted before being used to build a +// filesystem path. Mirrors sharedresources.validateID's rules on the +// guest side, which the same untrusted-ID-over-RPC reasoning applies to. +func validateContainerID(id string) error { + if id == "" { + return fmt.Errorf("container id is required") + } + if id == "." || id == ".." || strings.ContainsRune(id, '/') || strings.ContainsRune(id, os.PathSeparator) || strings.ContainsRune(id, 0) { + return fmt.Errorf("invalid container id %q", id) + } + return nil +} + +// ShareRootfs resolves the container rootfs from the given containerd mount +// specs by executing them on the host inside the shim's mount namespace, and +// exposes the result in the shared filesystem tree so the guest can access it +// at GuestRootfsPath(containerID). +// +// The mounts parameter is exactly what containerd passes in the Task.Create +// request — the same set of specs the snapshotter would normally apply +// locally. We execute them here inside the shim's private mount namespace so +// that cleanup is automatic when the shim process exits. +// +// The rootfs is exposed at the correct guest path via a real mount: a kernel +// bind mount, an overlay mount, a FUSE mount, or any other type that +// mountutil.All can apply. If the mount cannot be established the container +// run fails — there is no fallback to file copies or hard links, which would +// silently produce incorrect behaviour (dirty-page accumulation, cross-device +// failures, and loss of file-system metadata). +// +// Returns the in-guest path where the assembled rootfs will be accessible. +func (s *SharedFS) ShareRootfs(ctx context.Context, containerID string, mounts []*types.Mount) (guestPath string, err error) { + if err := validateContainerID(containerID); err != nil { + return "", err + } + return s.shareRootfs(ctx, containerID, mounts) +} + +// ShareVolume bind-mounts hostSource (a host path from an OCI "bind" mount +// in a member container's spec) into the shared filesystem tree at +// GuestVolumePath(containerID, n), and returns that guest path. +// +// This exists because a member container's volume mounts cannot use the +// same mechanism as the legacy/plain-container path (internal/shim/task's +// bindMounter, which shares each bind mount as its own new virtiofs tag): +// by the time a member container is created the sandbox's VM is already +// running, and virtio-fs shares cannot be hot-added after boot. Instead, +// the host source is bind-mounted directly into the shared directory tree +// that is already exposed to the guest via the single, persistent, +// pre-boot "containers" virtiofs share (the same one ShareRootfs uses) — +// so the guest sees the volume's content immediately, with no new virtiofs +// device and no additional guest-side Mount.MountAll step required at all. +// +// isDir must reflect whether hostSource is a directory or a regular file: +// unlike a virtiofs share (which must be a directory), a plain bind mount +// can target either, but the mountpoint placeholder this function creates +// must match (a directory for a directory bind mount, an empty regular +// file for a file bind mount) or the mount(2) call fails. +// +// This mount is always read-write and recursive (rbind), regardless of +// what the container's OCI spec requests for the volume: it exists purely +// to expose hostSource's content (including any nested mounts under it) to +// the guest. The caller (sandboxVolumeMounter) only rewrites the spec's +// mount Source, leaving Options untouched, so the actual container-visible +// read-only/recursion semantics are enforced exactly once, by the guest's +// own OCI runtime (crun) performing its own bind mount from +// GuestVolumePath into the container using those original options. Making +// *this* mount read-only too would be actively wrong, not just redundant: +// a recursive read-only bind mount sets Linux's MNT_LOCKED on every mount +// in the hierarchy, and that lock cannot be undone by any later mount +// (including crun's, or a fresh mount the container creates inside it) — +// so a container-requested *non-recursive* read-only volume would become +// unintentionally, unremovably read-only all the way down. +func (s *SharedFS) ShareVolume(ctx context.Context, containerID string, n int, hostSource string, isDir bool) (guestPath string, err error) { + if err := validateContainerID(containerID); err != nil { + return "", err + } + return s.shareVolume(ctx, containerID, n, hostSource, isDir) +} + +// Unshare removes all host-side mounts created for containerID and deletes +// its subtree under the shared directory. It is idempotent. +func (s *SharedFS) Unshare(ctx context.Context, containerID string) error { + if err := validateContainerID(containerID); err != nil { + return err + } + return s.unshare(ctx, containerID) +} + +// UnshareAll removes all containers. Called on sandbox shutdown after the VM +// has stopped so host-side cleanup does not race live mounts. +func (s *SharedFS) UnshareAll(ctx context.Context) error { + return s.unshareAll(ctx) +} diff --git a/internal/shim/sandbox/sharedfs_linux.go b/internal/shim/sandbox/sharedfs_linux.go new file mode 100644 index 00000000..a6fdbb8d --- /dev/null +++ b/internal/shim/sandbox/sharedfs_linux.go @@ -0,0 +1,226 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "fmt" + "os" + "path/filepath" + "slices" + + "github.com/containerd/containerd/api/types" + "github.com/containerd/containerd/v2/core/mount" + "github.com/containerd/log" + "golang.org/x/sys/unix" + + "github.com/containerd/nerdbox/internal/mountutil" +) + +// shareRootfs is the Linux implementation backing SharedFS.ShareRootfs. See +// its doc comment in sharedfs.go for the full contract. +func (s *SharedFS) shareRootfs(ctx context.Context, containerID string, mounts []*types.Mount) (guestPath string, err error) { + hostRootfs := filepath.Join(s.root, containerID, "rootfs") + + if len(mounts) == 0 { + // No mounts: create an empty rootfs target directory. + if err := os.MkdirAll(hostRootfs, 0o755); err != nil { + return "", fmt.Errorf("create rootfs dir %s: %w", hostRootfs, err) + } + return GuestRootfsPath(containerID), nil + } + + if err := os.MkdirAll(hostRootfs, 0o755); err != nil { + return "", fmt.Errorf("create rootfs dir %s: %w", hostRootfs, err) + } + + // Intermediate directory for chained mounts (all but the last mount in + // the list are mounted under here; the last is mounted directly at + // hostRootfs). This mirrors the legacy/plain-container path in + // internal/shim/task/mount_linux.go, which uses mountutil.All the same + // way for the same reason: it, not the generic containerd mount.All, + // understands nerdbox's custom "format/" and "mkdir/" mount option + // prefixes (e.g. X-containerd.mkdir.path=...) used to build overlay + // upper/work directories before mounting. + lmounts := filepath.Join(s.root, containerID, "mnt") + if err := os.MkdirAll(lmounts, 0o755); err != nil { + return "", fmt.Errorf("create intermediate mount dir %s: %w", lmounts, err) + } + + log.G(ctx).WithFields(log.Fields{ + "container": containerID, + "mounts": mounts, + "target": hostRootfs, + }).Debug("assembling container rootfs on host") + + if err := mountutil.All(ctx, hostRootfs, lmounts, mounts); err != nil { + return "", fmt.Errorf("mount container rootfs for %s: %w", containerID, err) + } + + // mountutil.All mounts every entry in mounts: all but the last under + // lmounts/, and the last at hostRootfs. Track every mount point + // it created (not just hostRootfs) so Unshare tears all of them down — + // otherwise the intermediate lowerdir mounts backing the final overlay + // would leak. Order matters: hostRootfs (the outermost mount, depending + // on the others) must be unmounted before its lower layers, so it is + // appended last and Unshare's reverse-order unmount hits it first. + mountPts := make([]string, 0, len(mounts)) + for i := range mounts { + if i < len(mounts)-1 { + mountPts = append(mountPts, filepath.Join(lmounts, fmt.Sprintf("%d", i))) + } + } + mountPts = append(mountPts, hostRootfs) + + s.mu.Lock() + s.mounts[containerID] = append(s.mounts[containerID], mountPts...) + s.mu.Unlock() + + return GuestRootfsPath(containerID), nil +} + +// shareVolume is the Linux implementation backing SharedFS.ShareVolume. See +// its doc comment in sharedfs.go for the full contract. +func (s *SharedFS) shareVolume(ctx context.Context, containerID string, n int, hostSource string, isDir bool) (guestPath string, err error) { + target := filepath.Join(s.root, containerID, "volumes", fmt.Sprintf("%d", n)) + + if isDir { + if err := os.MkdirAll(target, 0o755); err != nil { + return "", fmt.Errorf("create volume dir %s: %w", target, err) + } + } else { + if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil { + return "", fmt.Errorf("create volume parent dir for %s: %w", target, err) + } + f, err := os.OpenFile(target, os.O_CREATE, 0o644) + if err != nil { + return "", fmt.Errorf("create volume file placeholder %s: %w", target, err) + } + f.Close() + } + + m := mount.Mount{Type: "bind", Source: hostSource, Options: []string{"rbind", "rw"}} + if err := m.Mount(target); err != nil { + return "", fmt.Errorf("bind mount volume %s -> %s: %w", hostSource, target, err) + } + + log.G(ctx).WithFields(log.Fields{ + "container": containerID, + "n": n, + "source": hostSource, + "target": target, + }).Debug("shared container volume mount") + + s.mu.Lock() + s.mounts[containerID] = append(s.mounts[containerID], target) + s.mu.Unlock() + + return GuestVolumePath(containerID, n), nil +} + +// unshare is the Linux implementation backing SharedFS.Unshare. See its doc +// comment in sharedfs.go for the full contract. +func (s *SharedFS) unshare(ctx context.Context, containerID string) error { + s.mu.Lock() + mountPts := s.mounts[containerID] + s.mu.Unlock() + + var errs []error + var remaining []string + + // Unmount in reverse order (deepest first). + for i := len(mountPts) - 1; i >= 0; i-- { + pt := mountPts[i] + log.G(ctx).WithFields(log.Fields{ + "container": containerID, + "target": pt, + }).Debug("unmounting container rootfs") + // MNT_DETACH performs a lazy unmount: the mount is detached from + // the filesystem hierarchy immediately even if the directory is + // still in use (e.g. while virtiofs is serving files from it). + // The mount is cleaned up when all references are dropped. + if err := mount.UnmountAll(pt, unix.MNT_DETACH); err != nil { + log.G(ctx).WithError(err).WithField("target", pt).Warn("failed to unmount rootfs") + errs = append(errs, fmt.Errorf("unmount %s: %w", pt, err)) + // Keep this target so a later retry (another Unshare or + // UnshareAll call) still knows about it. Without this, the + // bookkeeping delete below would drop it permanently after + // this one failed attempt, and it would never be retried. + remaining = append(remaining, pt) + } + } + + // Replace, rather than unconditionally delete, so a failed unmount's + // target survives for a later retry. remaining was built by appending + // while iterating mountPts in reverse (deepest/outermost first), so it + // is in the opposite order from mountPts's own outermost-last + // convention (see shareRootfs) — reverse it back before storing so a + // later retry's own reverse iteration unmounts outermost (e.g. + // hostRootfs) first again, not last. + s.mu.Lock() + if len(remaining) > 0 { + slices.Reverse(remaining) + s.mounts[containerID] = remaining + } else { + delete(s.mounts, containerID) + } + s.mu.Unlock() + + // Only remove the container subtree once every mount under it is + // confirmed gone: with a mount still active, RemoveAll could remove + // entries from a stale, about-to-be-detached view of the directory + // rather than its real (post-unmount) contents, or simply fail + // outright on a busy mount point — neither of which is better than + // leaving the directory for the next retry to clean up alongside the + // mount it still needs to unmount. + if len(remaining) == 0 { + ctrDir := filepath.Join(s.root, containerID) + if err := os.RemoveAll(ctrDir); err != nil && !os.IsNotExist(err) { + log.G(ctx).WithError(err).WithField("dir", ctrDir).Warn("failed to remove container shared dir") + } + } + + if len(errs) > 0 { + return fmt.Errorf("unshare %s: %w", containerID, errs[0]) + } + return nil +} + +// unshareAll is the Linux implementation backing SharedFS.UnshareAll. See +// its doc comment in sharedfs.go for the full contract. +func (s *SharedFS) unshareAll(ctx context.Context) error { + s.mu.Lock() + ids := make([]string, 0, len(s.mounts)) + for id := range s.mounts { + ids = append(ids, id) + } + s.mu.Unlock() + + var errs []error + for _, id := range ids { + if err := s.unshare(ctx, id); err != nil { + errs = append(errs, err) + } + } + + if len(errs) > 0 { + return fmt.Errorf("unshare all: %v", errs) + } + return nil +} diff --git a/internal/shim/sandbox/sharedfs_linux_test.go b/internal/shim/sandbox/sharedfs_linux_test.go new file mode 100644 index 00000000..1cd13662 --- /dev/null +++ b/internal/shim/sandbox/sharedfs_linux_test.go @@ -0,0 +1,73 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "os" + "path/filepath" + "testing" +) + +// TestUnshareRetainsFailedMountForRetry verifies that a mount point Unshare +// fails to unmount is kept in s.mounts, rather than discarded, so a later +// retry (another Unshare or UnshareAll call) still knows to try it again. +// Losing that bookkeeping would permanently strand the mount: nothing else +// records where it is. +// +// This runs only as non-root: as a regular user, unmount(2) on any target — +// mounted or not — fails with EPERM (lacking CAP_SYS_ADMIN), which +// mount.UnmountAll propagates as a real error. Run as root, unmounting an +// ordinary directory that was never mounted returns EINVAL, which +// mount.UnmountAll deliberately squelches to a nil (success) return — so +// this specific failure mode can't be exercised there without an actual +// mount to break. +func TestUnshareRetainsFailedMountForRetry(t *testing.T) { + if os.Geteuid() == 0 { + t.Skip("requires non-root: see doc comment") + } + + const containerID = "test-container" + root := t.TempDir() + target := filepath.Join(root, containerID, "rootfs") + if err := os.MkdirAll(target, 0o755); err != nil { + t.Fatalf("MkdirAll: %v", err) + } + + s := &SharedFS{ + root: root, + mounts: map[string][]string{containerID: {target}}, + } + + if err := s.unshare(context.Background(), containerID); err == nil { + t.Fatal("unshare: got nil error, want one (unmount as non-root should fail with EPERM)") + } + + s.mu.Lock() + got := s.mounts[containerID] + s.mu.Unlock() + if len(got) != 1 || got[0] != target { + t.Fatalf("s.mounts[%q] = %v, want [%q] (the failed mount should be retained for retry)", containerID, got, target) + } + + // The container directory must survive too: removing it while a mount + // under it is still (from our bookkeeping's point of view) unresolved + // would orphan whatever that mount was ultimately backing. + if _, err := os.Stat(filepath.Join(root, containerID)); err != nil { + t.Fatalf("container dir removed despite a retained failed mount: %v", err) + } +} diff --git a/internal/shim/sandbox/sharedfs_other.go b/internal/shim/sandbox/sharedfs_other.go new file mode 100644 index 00000000..f4531697 --- /dev/null +++ b/internal/shim/sandbox/sharedfs_other.go @@ -0,0 +1,63 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "fmt" + "runtime" + + "github.com/containerd/containerd/api/types" +) + +// errSharedFSUnsupported is returned by every SharedFS operation that +// requires assembling real host-side mounts (bind/overlay/etc.), which is +// only implemented on Linux today (see sharedfs_linux.go). The sandbox +// (multi-container-per-VM) shim API is Linux-only for now; see +// docs/sandbox-architecture.md. +var errSharedFSUnsupported = fmt.Errorf("sandbox shared filesystem not supported on %s", runtime.GOOS) + +// Every stub below takes the same lock the Linux implementation does +// (sharedfs_linux.go) even though there is nothing to mutate, purely so +// that SharedFS.mu has a use on every platform; leaving it genuinely +// unused here would fail the "unused" lint check on non-Linux builds. + +func (s *SharedFS) shareRootfs(context.Context, string, []*types.Mount) (string, error) { + s.mu.Lock() + defer s.mu.Unlock() + return "", errSharedFSUnsupported +} + +func (s *SharedFS) shareVolume(context.Context, string, int, string, bool) (string, error) { + s.mu.Lock() + defer s.mu.Unlock() + return "", errSharedFSUnsupported +} + +func (s *SharedFS) unshare(context.Context, string) error { + s.mu.Lock() + defer s.mu.Unlock() + return errSharedFSUnsupported +} + +func (s *SharedFS) unshareAll(context.Context) error { + s.mu.Lock() + defer s.mu.Unlock() + return errSharedFSUnsupported +} diff --git a/internal/shim/sandbox/sharedfs_test.go b/internal/shim/sandbox/sharedfs_test.go new file mode 100644 index 00000000..4b5d73e3 --- /dev/null +++ b/internal/shim/sandbox/sharedfs_test.go @@ -0,0 +1,96 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "context" + "strings" + "testing" +) + +// TestGuestRootfsPath_UsesForwardSlashes verifies GuestRootfsPath builds an +// in-guest (always Linux) path using '/' separators regardless of the host +// OS this test runs on. The guest is always Linux even when this shim runs +// on a Windows host, where filepath.Join would use '\' and produce a path +// the guest could never use. +func TestGuestRootfsPath_UsesForwardSlashes(t *testing.T) { + got := GuestRootfsPath("abc123") + want := "/run/containers/abc123/rootfs" + if got != want { + t.Fatalf("GuestRootfsPath() = %q, want %q", got, want) + } + if strings.ContainsRune(got, '\\') { + t.Fatalf("GuestRootfsPath() contains a backslash: %q", got) + } +} + +// TestGuestVolumePath_UsesForwardSlashes verifies GuestVolumePath builds an +// in-guest path using '/' separators. See TestGuestRootfsPath_UsesForwardSlashes. +func TestGuestVolumePath_UsesForwardSlashes(t *testing.T) { + got := GuestVolumePath("abc123", 2) + want := "/run/containers/abc123/volumes/2" + if got != want { + t.Fatalf("GuestVolumePath() = %q, want %q", got, want) + } + if strings.ContainsRune(got, '\\') { + t.Fatalf("GuestVolumePath() contains a backslash: %q", got) + } +} + +// TestValidateContainerID checks the specific inputs that would otherwise +// let a container ID escape SharedFS.root (on the host) or +// GuestContainersDir (in the guest) once joined onto it — see +// ShareRootfs/ShareVolume/Unshare, which all reject an ID via this +// function before constructing any path from it. +func TestValidateContainerID(t *testing.T) { + bad := []string{"", ".", "..", "../x", "a/../../b", "a/b", "/etc/passwd", "a\x00b"} + for _, id := range bad { + if err := validateContainerID(id); err == nil { + t.Errorf("validateContainerID(%q) = nil, want an error", id) + } + } + + good := []string{"abc123", "test-container_1", "a.b"} + for _, id := range good { + if err := validateContainerID(id); err != nil { + t.Errorf("validateContainerID(%q) = %v, want nil", id, err) + } + } +} + +// TestSharedFSRejectsInvalidContainerID verifies that ShareRootfs, +// ShareVolume, and Unshare all reject a malicious container ID before +// doing anything else — in particular, before ever reaching a platform +// implementation that would construct a filesystem path from it. Using an +// ID that would escape s.root if unchecked (rather than just checking the +// error's presence) makes this a regression test for the actual path +// traversal, not just for validateContainerID being called at all. +func TestSharedFSRejectsInvalidContainerID(t *testing.T) { + const evil = "../evil" + s := &SharedFS{root: t.TempDir(), mounts: make(map[string][]string)} + ctx := context.Background() + + if _, err := s.ShareRootfs(ctx, evil, nil); err == nil { + t.Error("ShareRootfs with a path-traversal container id: got nil error, want one") + } + if _, err := s.ShareVolume(ctx, evil, 0, "/tmp", true); err == nil { + t.Error("ShareVolume with a path-traversal container id: got nil error, want one") + } + if err := s.Unshare(ctx, evil); err == nil { + t.Error("Unshare with a path-traversal container id: got nil error, want one") + } +} diff --git a/internal/shim/sandbox/vm/vm.go b/internal/shim/sandbox/vm/vm.go index 8d686fd5..5b4799f9 100644 --- a/internal/shim/sandbox/vm/vm.go +++ b/internal/shim/sandbox/vm/vm.go @@ -127,6 +127,11 @@ func (s *localsandbox) Start(ctx context.Context, opts ...sandbox.Opt) error { if len(o.InitArgs) > 0 { startOpts = append(startOpts, vm.WithInitArgs(o.InitArgs...)) } + // The VM implementation is responsible for entering this network + // namespace (if non-empty) before creating any networking-related + // host resources or worker threads, so that VM traffic originates + // inside the pod netns. + startOpts = append(startOpts, vm.WithNetNS(o.NetnsPath)) if err := vmi.Start(ctx, startOpts...); err != nil { return err diff --git a/internal/shim/task/apparmor.go b/internal/shim/task/apparmor.go new file mode 100644 index 00000000..1c0225c3 --- /dev/null +++ b/internal/shim/task/apparmor.go @@ -0,0 +1,42 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// clearApparmorProfile strips Process.ApparmorProfile from the incoming OCI +// spec. CRI sets this field to the name of an AppArmor profile loaded on +// the host (e.g. via a pod's appArmorProfile field or the deprecated +// container.apparmor.security.beta.kubernetes.io annotation). That name is +// meaningless inside the VM guest: the guest kernel may not have AppArmor +// enabled at all, and even if it does, it never loaded a profile by that +// name. Left unmodified, the guest's crun invocation fails outright trying +// to apply an unknown profile. Clearing the field runs the container +// unconfined by AppArmor inside the guest, which is consistent with how +// this shim already handles other host-specific confinement it cannot +// honor in a nested kernel (see sanitizeNamespaces for the equivalent +// treatment of host namespace paths). +func clearApparmorProfile(_ context.Context, b *bundle.Bundle) error { + if b.Spec.Process != nil { + b.Spec.Process.ApparmorProfile = "" + } + return nil +} diff --git a/internal/shim/task/apparmor_test.go b/internal/shim/task/apparmor_test.go new file mode 100644 index 00000000..708cbaf5 --- /dev/null +++ b/internal/shim/task/apparmor_test.go @@ -0,0 +1,50 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "testing" + + specs "github.com/opencontainers/runtime-spec/specs-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +func TestClearApparmorProfile(t *testing.T) { + t.Run("nil Process is a no-op", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{}} + require.NoError(t, clearApparmorProfile(context.Background(), b)) + assert.Nil(t, b.Spec.Process) + }) + + t.Run("clears a host AppArmor profile", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{ + Process: &specs.Process{ApparmorProfile: "docker-default"}, + }} + require.NoError(t, clearApparmorProfile(context.Background(), b)) + assert.Empty(t, b.Spec.Process.ApparmorProfile) + }) + + t.Run("no-op when no profile was set", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{Process: &specs.Process{}}} + require.NoError(t, clearApparmorProfile(context.Background(), b)) + assert.Empty(t, b.Spec.Process.ApparmorProfile) + }) +} diff --git a/internal/shim/task/bundle/bundle.go b/internal/shim/task/bundle/bundle.go index 656d96ef..3debe38a 100644 --- a/internal/shim/task/bundle/bundle.go +++ b/internal/shim/task/bundle/bundle.go @@ -41,9 +41,26 @@ type Bundle struct { type Transformer func(ctx context.Context, b *Bundle) error -// Load loads an OCI bundle from the given path and apply a series of transformers -// to turn the host-side bundle into a VM-side bundle. +// Load loads a container's OCI bundle from the given path and applies a +// series of transformers to turn the host-side bundle into a VM-side +// bundle. A container bundle always has a Root, so its absence is an error; +// see LoadSandboxConfig for the sandbox-bundle case, which has no rootfs of +// its own. func Load(ctx context.Context, path string, transformers ...Transformer) (*Bundle, error) { + return load(ctx, path, true, transformers...) +} + +// LoadSandboxConfig loads the sandbox-level bundle at path — the one +// containerd's sandbox controller passes in CreateSandboxRequest.BundlePath, +// used only to derive VM start options (resources, networking) ahead of any +// container running in it — and applies transformers the same way Load +// does. Unlike a container bundle, a sandbox bundle has no rootfs of its +// own, so a missing Root in its config.json is expected, not an error. +func LoadSandboxConfig(ctx context.Context, path string, transformers ...Transformer) (*Bundle, error) { + return load(ctx, path, false, transformers...) +} + +func load(ctx context.Context, path string, rootRequired bool, transformers ...Transformer) (*Bundle, error) { specBytes, err := os.ReadFile(filepath.Join(path, "config.json")) if err != nil { return nil, err @@ -57,7 +74,7 @@ func Load(ctx context.Context, path string, transformers ...Transformer) (*Bundl return nil, err } - if err := resolveRootfsPath(ctx, &b); err != nil { + if err := resolveRootfsPath(ctx, &b, rootRequired); err != nil { return nil, err } @@ -87,9 +104,12 @@ func (b *Bundle) Files() (map[string][]byte, error) { return files, nil } -func resolveRootfsPath(ctx context.Context, b *Bundle) error { +func resolveRootfsPath(ctx context.Context, b *Bundle, required bool) error { if b.Spec.Root == nil { - return fmt.Errorf("root path not specified: %w", errdefs.ErrInvalidArgument) + if required { + return fmt.Errorf("root path not specified: %w", errdefs.ErrInvalidArgument) + } + return nil } if filepath.IsAbs(b.Spec.Root.Path) { diff --git a/internal/shim/task/ctrnetworking.go b/internal/shim/task/ctrnetworking.go index 68611e7d..953ca905 100644 --- a/internal/shim/task/ctrnetworking.go +++ b/internal/shim/task/ctrnetworking.go @@ -28,6 +28,7 @@ import ( "github.com/opencontainers/runtime-spec/specs-go" + "github.com/containerd/nerdbox/api/types" "github.com/containerd/nerdbox/internal/nwcfg" "github.com/containerd/nerdbox/internal/shim/task/bundle" ) @@ -161,7 +162,23 @@ func parseCtrNetwork(annotation string) (nwcfg.Network, error) { // addResolvConf adds a /etc/resolv.conf to the container, unless the // bundle already includes one. -func addResolvConf(ctx context.Context, b *bundle.Bundle, fallbackToHostRC bool) error { +// +// podDNS, if non-nil and non-empty, is the pod's CRI DNSConfig (from +// PodSandboxConfig.DnsConfig, threaded in from the sandbox's +// CreateSandboxRequest.Options — see podSandboxConfig). CRI's podsandbox +// controller writes a resolv.conf derived from this into a host file that +// every member container bind-mounts (internal/cri/server's +// linuxContainerMounts + podsandbox's setupSandboxFiles upstream); the +// shim sandboxer path this package implements gets no such file from +// containerd, so it must generate the same content itself. +// +// Priority, highest first: an existing bundle mount at /etc/resolv.conf +// (do nothing — some caller already handled it); the nerdbox-specific +// per-container annotation (pre-dates CRI support, kept for `ctr run` +// compatibility); podDNS; and finally, only when fallbackToHostRC is set +// (the container has no dedicated NIC, i.e. relies on TSI for +// connectivity), a copy of the host's own resolv.conf. +func addResolvConf(ctx context.Context, b *bundle.Bundle, fallbackToHostRC bool, podDNS *types.CRIDNSConfig) error { // If there's already a resolv.conf mount, don't do anything. if slices.ContainsFunc(b.Spec.Mounts, func(m specs.Mount) bool { return m.Destination == "/etc/resolv.conf" @@ -187,6 +204,8 @@ func addResolvConf(ctx context.Context, b *bundle.Bundle, fallbackToHostRC bool) _, _ = rcBuf.WriteRune('\n') } rcBytes = rcBuf.Bytes() + } else if podDNS != nil && (len(podDNS.GetServers()) > 0 || len(podDNS.GetSearches()) > 0 || len(podDNS.GetOptions()) > 0) { + rcBytes = []byte(formatPodDNSConfig(podDNS)) } else if fallbackToHostRC { // Try giving the VM a copy of the host's resolv.conf. if c, err := os.ReadFile(hostResolvConfPath()); err == nil { @@ -210,6 +229,25 @@ func addResolvConf(ctx context.Context, b *bundle.Bundle, fallbackToHostRC bool) return nil } +// formatPodDNSConfig renders a CRI DNSConfig as resolv.conf(5) content, +// matching the format used by containerd's own podsandbox controller +// (internal/cri/server/podsandbox's parseDNSOptions upstream): one +// "nameserver" line per server, a single "search" line listing every +// search domain, and a single "options" line listing every option. +func formatPodDNSConfig(dns *types.CRIDNSConfig) string { + var buf bytes.Buffer + for _, s := range dns.GetServers() { + fmt.Fprintf(&buf, "nameserver %s\n", s) + } + if searches := dns.GetSearches(); len(searches) > 0 { + fmt.Fprintf(&buf, "search %s\n", strings.Join(searches, " ")) + } + if opts := dns.GetOptions(); len(opts) > 0 { + fmt.Fprintf(&buf, "options %s\n", strings.Join(opts, " ")) + } + return buf.String() +} + // systemdResolvedFullRC is the "full" resolv.conf systemd-resolved maintains // alongside its stub file, listing the actual upstream DNS servers rather // than the stub's loopback listener. See resolv.conf(5) / diff --git a/internal/shim/task/ctrnetworking_test.go b/internal/shim/task/ctrnetworking_test.go index e06ae2e6..9c83b71c 100644 --- a/internal/shim/task/ctrnetworking_test.go +++ b/internal/shim/task/ctrnetworking_test.go @@ -29,6 +29,7 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/containerd/nerdbox/api/types" "github.com/containerd/nerdbox/internal/nwcfg" "github.com/containerd/nerdbox/internal/shim/task/bundle" ) @@ -240,7 +241,7 @@ func TestAddResolvConf(t *testing.T) { b := &bundle.Bundle{Spec: specs.Spec{Mounts: []specs.Mount{ {Destination: "/etc/resolv.conf", Type: "bind", Source: "/custom/resolv.conf"}, }}} - require.NoError(t, addResolvConf(context.Background(), b, true)) + require.NoError(t, addResolvConf(context.Background(), b, true, nil)) require.Len(t, b.Spec.Mounts, 1) assert.Equal(t, "/custom/resolv.conf", b.Spec.Mounts[0].Source) }) @@ -249,7 +250,7 @@ func TestAddResolvConf(t *testing.T) { b := loadTestBundle(t, specs.Spec{Annotations: map[string]string{ "io.containerd.nerdbox.ctr.dns": "nameserver=8.8.8.8,search=example.com", }}) - require.NoError(t, addResolvConf(context.Background(), b, false)) + require.NoError(t, addResolvConf(context.Background(), b, false, nil)) // Annotation is stripped after being consumed. _, hasAnnot := b.Spec.Annotations["io.containerd.nerdbox.ctr.dns"] @@ -273,7 +274,7 @@ func TestAddResolvConf(t *testing.T) { // /etc/resolv.conf; the source is "resolv.conf" (extra file) when the host // file was read, or the VM's own /etc/resolv.conf when it was not. b := loadTestBundle(t, specs.Spec{}) - require.NoError(t, addResolvConf(context.Background(), b, true)) + require.NoError(t, addResolvConf(context.Background(), b, true, nil)) require.Len(t, b.Spec.Mounts, 1) assert.Equal(t, "/etc/resolv.conf", b.Spec.Mounts[0].Destination) assert.Contains(t, []string{"resolv.conf", "/etc/resolv.conf"}, b.Spec.Mounts[0].Source) @@ -281,11 +282,54 @@ func TestAddResolvConf(t *testing.T) { t.Run("no annotation and fallback disabled defaults to VM resolv.conf", func(t *testing.T) { b := &bundle.Bundle{Spec: specs.Spec{}} - require.NoError(t, addResolvConf(context.Background(), b, false)) + require.NoError(t, addResolvConf(context.Background(), b, false, nil)) require.Len(t, b.Spec.Mounts, 1) assert.Equal(t, "/etc/resolv.conf", b.Spec.Mounts[0].Destination) assert.Equal(t, "/etc/resolv.conf", b.Spec.Mounts[0].Source) }) + + t.Run("pod DNSConfig generates resolv.conf content", func(t *testing.T) { + b := loadTestBundle(t, specs.Spec{}) + podDNS := &types.CRIDNSConfig{ + Servers: []string{"1.1.1.1", "8.8.8.8"}, + Searches: []string{"svc.cluster.local", "cluster.local"}, + Options: []string{"ndots:5"}, + } + require.NoError(t, addResolvConf(context.Background(), b, false, podDNS)) + + require.Len(t, b.Spec.Mounts, 1) + assert.Equal(t, "/etc/resolv.conf", b.Spec.Mounts[0].Destination) + assert.Equal(t, "resolv.conf", b.Spec.Mounts[0].Source) + + files, err := b.Files() + require.NoError(t, err) + content := string(files["resolv.conf"]) + assert.Contains(t, content, "nameserver 1.1.1.1\n") + assert.Contains(t, content, "nameserver 8.8.8.8\n") + assert.Contains(t, content, "search svc.cluster.local cluster.local\n") + assert.Contains(t, content, "options ndots:5\n") + }) + + t.Run("dns annotation takes priority over pod DNSConfig", func(t *testing.T) { + b := loadTestBundle(t, specs.Spec{Annotations: map[string]string{ + "io.containerd.nerdbox.ctr.dns": "nameserver=8.8.8.8", + }}) + podDNS := &types.CRIDNSConfig{Servers: []string{"1.1.1.1"}} + require.NoError(t, addResolvConf(context.Background(), b, false, podDNS)) + + files, err := b.Files() + require.NoError(t, err) + content := string(files["resolv.conf"]) + assert.Contains(t, content, "nameserver 8.8.8.8\n") + assert.NotContains(t, content, "1.1.1.1") + }) + + t.Run("empty pod DNSConfig falls through to fallback", func(t *testing.T) { + b := loadTestBundle(t, specs.Spec{}) + require.NoError(t, addResolvConf(context.Background(), b, false, &types.CRIDNSConfig{})) + require.Len(t, b.Spec.Mounts, 1) + assert.Equal(t, "/etc/resolv.conf", b.Spec.Mounts[0].Source) + }) } // TestOnlyLoopbackNameservers covers the resolv.conf parsing used to detect diff --git a/internal/shim/task/devshm.go b/internal/shim/task/devshm.go new file mode 100644 index 00000000..bbceac66 --- /dev/null +++ b/internal/shim/task/devshm.go @@ -0,0 +1,149 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "fmt" + "strconv" + "strings" + + specs "github.com/opencontainers/runtime-spec/specs-go" + + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// defaultDevShmSize is used when a /dev/shm mount's own options don't +// specify a size (or specify one this package can't parse). It matches the +// size Kubernetes itself defaults to when a pod sets no explicit limit. +const defaultDevShmSize = 64 * 1024 * 1024 + +// shareDevShmMounter is a bundle.Transformer for sandbox member containers. +// It gives member containers that share an IPC namespace a working +// /dev/shm: containerd sends every member container of a pod the same, +// independent `{Destination: "/dev/shm", Type: "tmpfs", Source: "shm"}` +// mount (confirmed via `crictl inspect` against a live CRI pod) — the same +// spec Kubernetes' non-VM runtimes rely on turning into a shared mount +// themselves, but on nerdbox's guest kernel it just makes each container's +// own crun create an independent, private tmpfs. Sidecars that share SysV +// IPC (via the pod's shared IPC namespace) but write POSIX shared memory +// through /dev/shm therefore don't actually see each other's writes. +// +// FromBundle rewrites that mount, when present on a container that is +// sharing IPC, into a "bind" mount of the sandbox's shared /dev/shm tmpfs +// (getDevShm, backed by the guest's SharedResources service — see +// internal/vminit/sharedresources.TypeDevShm). This is a real, sized tmpfs +// living entirely in the guest's own root mount namespace, not something +// shared through the host/virtiofs like sandboxVolumeMounter's ordinary +// bind mounts, so it is real guest RAM with a real, kernel-enforced size +// limit rather than something backed by host disk. Must therefore run +// after sandboxVolumeMounter.FromBundle, which only rewrites mounts that +// already have Type "bind" — this mount is still "tmpfs" until this +// transformer runs, so it is untouched by that pass, and the guest path +// this produces needs no further host-side sharing at all. +// +// A container with no shared IPC namespace (NamespaceMode_CONTAINER, the +// default for a pod that never asks for IPC sharing) is left alone: its +// /dev/shm mount passes through unchanged, and it gets the same private +// tmpfs it would have gotten anyway. +type shareDevShmMounter struct { + containerID string + // getDevShmFn is normally (*sharedResources).getDevShm; a field + // (rather than a *sharedResources) so tests can substitute a fake + // without needing a real guest TTRPC connection. + getDevShmFn func(ctx context.Context, sizeBytes int64) (string, error) +} + +// FromBundle implements the rewrite described in the type doc comment. +func (d *shareDevShmMounter) FromBundle(ctx context.Context, b *bundle.Bundle) error { + if !containerSharesIPC(b) { + return nil + } + + for i, m := range b.Spec.Mounts { + if m.Destination != "/dev/shm" || m.Type == "bind" { + continue + } + + guestPath, err := d.getDevShmFn(ctx, devShmSize(m.Options)) + if err != nil { + return fmt.Errorf("share /dev/shm: %w", err) + } + + // mode=/size=/nosuid/etc. are tmpfs-specific and meaningless (or + // invalid) on a bind mount; rbind is the only option this needs. + b.Spec.Mounts[i] = specs.Mount{ + Destination: m.Destination, + Type: "bind", + Source: guestPath, + Options: []string{"rbind"}, + } + } + + return nil +} + +// containerSharesIPC reports whether b's spec asks to join a shared IPC +// namespace — the same non-empty-Path test sanitizeNamespaces uses (see its +// doc comment) — regardless of whether sanitizeNamespaces has already run +// and rewritten that Path to the guest's shared namespace: either way, a +// present entry has a non-empty Path if and only if sharing was requested. +func containerSharesIPC(b *bundle.Bundle) bool { + if b.Spec.Linux == nil { + return false + } + for _, ns := range b.Spec.Linux.Namespaces { + if ns.Type == specs.IPCNamespace && ns.Path != "" { + return true + } + } + return false +} + +// devShmSize parses a tmpfs "size=" mount option (e.g. "size=65536k", the +// form containerd sends) into a byte count, falling back to +// defaultDevShmSize if opts has none or it can't be parsed. Only the +// suffixes tmpfs itself accepts (k/m/g, case-insensitive) are handled; a +// suffix this doesn't recognize (e.g. a raw percentage) falls back too, +// rather than guessing. +func devShmSize(opts []string) int64 { + for _, o := range opts { + v, ok := strings.CutPrefix(o, "size=") + if !ok { + continue + } + if n, err := strconv.ParseInt(v, 10, 64); err == nil { + return n + } + if len(v) < 2 { + continue + } + n, err := strconv.ParseInt(v[:len(v)-1], 10, 64) + if err != nil { + continue + } + switch v[len(v)-1] | 0x20 { // lowercase the suffix byte + case 'k': + return n * 1024 + case 'm': + return n * 1024 * 1024 + case 'g': + return n * 1024 * 1024 * 1024 + } + } + return defaultDevShmSize +} diff --git a/internal/shim/task/devshm_test.go b/internal/shim/task/devshm_test.go new file mode 100644 index 00000000..1d11ae7b --- /dev/null +++ b/internal/shim/task/devshm_test.go @@ -0,0 +1,213 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "testing" + + specs "github.com/opencontainers/runtime-spec/specs-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +func TestDevShmSize(t *testing.T) { + testcases := []struct { + name string + opts []string + want int64 + }{ + {name: "no options", opts: nil, want: defaultDevShmSize}, + {name: "no size option", opts: []string{"nosuid", "noexec"}, want: defaultDevShmSize}, + {name: "the exact form containerd sends", opts: []string{"nosuid", "noexec", "nodev", "mode=1777", "size=65536k"}, want: 65536 * 1024}, + {name: "kilobytes lowercase", opts: []string{"size=1024k"}, want: 1024 * 1024}, + {name: "kilobytes uppercase suffix", opts: []string{"size=1024K"}, want: 1024 * 1024}, + {name: "megabytes", opts: []string{"size=64m"}, want: 64 * 1024 * 1024}, + {name: "gigabytes", opts: []string{"size=1g"}, want: 1024 * 1024 * 1024}, + {name: "plain byte count, no suffix", opts: []string{"size=8192"}, want: 8192}, + {name: "unparseable value falls back", opts: []string{"size=50%"}, want: defaultDevShmSize}, + {name: "empty value falls back", opts: []string{"size="}, want: defaultDevShmSize}, + } + + for _, tc := range testcases { + t.Run(tc.name, func(t *testing.T) { + assert.Equal(t, tc.want, devShmSize(tc.opts)) + }) + } +} + +func TestContainerSharesIPC(t *testing.T) { + testcases := []struct { + name string + linux *specs.Linux + want bool + }{ + {name: "nil Linux", linux: nil, want: false}, + {name: "no namespaces", linux: &specs.Linux{}, want: false}, + { + name: "empty-Path IPC namespace (not sharing)", + linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace}, + }}, + want: false, + }, + { + name: "host-path IPC namespace (sharing requested)", + linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }}, + want: true, + }, + { + name: "already-rewritten guest IPC path still counts as sharing", + linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/run/ipcns/some-sandbox"}, + }}, + want: true, + }, + { + name: "other namespace types don't count", + linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: "/proc/12345/ns/net"}, + {Type: specs.PIDNamespace, Path: "/proc/12345/ns/pid"}, + }}, + want: false, + }, + } + + for _, tc := range testcases { + t.Run(tc.name, func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{Linux: tc.linux}} + assert.Equal(t, tc.want, containerSharesIPC(b)) + }) + } +} + +func TestShareDevShmMounter_FromBundle(t *testing.T) { + ctx := context.Background() + + t.Run("no IPC sharing: mount left alone, getDevShm not called", func(t *testing.T) { + var calls int + b := &bundle.Bundle{Spec: specs.Spec{ + Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace}, // empty Path: not sharing + }}, + Mounts: []specs.Mount{ + {Destination: "/dev/shm", Type: "tmpfs", Source: "shm", Options: []string{"size=65536k"}}, + }, + }} + + d := &shareDevShmMounter{containerID: "ctr-1"} + d.getDevShmFn = func(context.Context, int64) (string, error) { + calls++ + return "/should/not/be/used", nil + } + require.NoError(t, d.FromBundle(ctx, b)) + assert.Equal(t, 0, calls) + assert.Equal(t, "tmpfs", b.Spec.Mounts[0].Type, "mount must be left untouched when not sharing IPC") + }) + + t.Run("IPC sharing with a plain tmpfs /dev/shm: rewritten to bind", func(t *testing.T) { + var gotSize int64 + b := &bundle.Bundle{Spec: specs.Spec{ + Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }}, + Mounts: []specs.Mount{ + {Destination: "/proc", Type: "proc", Source: "proc"}, + {Destination: "/dev/shm", Type: "tmpfs", Source: "shm", Options: []string{"nosuid", "noexec", "nodev", "mode=1777", "size=65536k"}}, + }, + }} + + d := &shareDevShmMounter{containerID: "ctr-1"} + d.getDevShmFn = func(_ context.Context, sizeBytes int64) (string, error) { + gotSize = sizeBytes + return "/run/devshm/some-sandbox", nil + } + require.NoError(t, d.FromBundle(ctx, b)) + + assert.Equal(t, int64(65536*1024), gotSize) + assert.Equal(t, specs.Mount{Destination: "/proc", Type: "proc", Source: "proc"}, b.Spec.Mounts[0], + "unrelated mounts must be untouched") + assert.Equal(t, specs.Mount{ + Destination: "/dev/shm", + Type: "bind", + Source: "/run/devshm/some-sandbox", + Options: []string{"rbind"}, + }, b.Spec.Mounts[1]) + }) + + t.Run("already a bind mount: left alone (idempotent / caller-overridden)", func(t *testing.T) { + var calls int + b := &bundle.Bundle{Spec: specs.Spec{ + Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }}, + Mounts: []specs.Mount{ + {Destination: "/dev/shm", Type: "bind", Source: "/already/shared", Options: []string{"rbind"}}, + }, + }} + + d := &shareDevShmMounter{containerID: "ctr-1"} + d.getDevShmFn = func(context.Context, int64) (string, error) { + calls++ + return "/should/not/be/used", nil + } + require.NoError(t, d.FromBundle(ctx, b)) + assert.Equal(t, 0, calls) + assert.Equal(t, "/already/shared", b.Spec.Mounts[0].Source) + }) + + t.Run("no /dev/shm mount present: no-op", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{ + Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }}, + Mounts: []specs.Mount{ + {Destination: "/proc", Type: "proc", Source: "proc"}, + }, + }} + + d := &shareDevShmMounter{containerID: "ctr-1"} + d.getDevShmFn = func(context.Context, int64) (string, error) { + t.Fatal("must not be called: no /dev/shm mount present") + return "", nil + } + require.NoError(t, d.FromBundle(ctx, b)) + }) + + t.Run("getDevShm failure propagates", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{ + Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }}, + Mounts: []specs.Mount{ + {Destination: "/dev/shm", Type: "tmpfs", Source: "shm"}, + }, + }} + + d := &shareDevShmMounter{containerID: "ctr-1"} + d.getDevShmFn = func(context.Context, int64) (string, error) { + return "", assert.AnError + } + err := d.FromBundle(ctx, b) + require.Error(t, err) + assert.ErrorIs(t, err, assert.AnError) + }) +} diff --git a/internal/shim/task/namespaces.go b/internal/shim/task/namespaces.go new file mode 100644 index 00000000..f4cc7218 --- /dev/null +++ b/internal/shim/task/namespaces.go @@ -0,0 +1,186 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "fmt" + + specs "github.com/opencontainers/runtime-spec/specs-go" + + srAPI "github.com/containerd/nerdbox/api/services/sharedresources/v1" + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// sanitizeNamespaces is a bundle.Transformer for sandbox member containers. +// It has two jobs: +// +// 1. Strip host paths from the incoming OCI spec's Linux namespaces. In +// production CRI, a member container's spec sets the network, IPC, UTS, +// and (for pod- or node-level PID sharing) PID namespace entries' Path to +// a host path (e.g. "/proc//ns/net" — containerd's +// WithPodNamespaces), since that is meaningful to a normal (non-VM) OCI +// runtime running directly on the host. Copied verbatim into the guest, +// that path is meaningless (or, if it happens to collide with a real +// guest path, actively wrong) — the guest is a different kernel with an +// unrelated PID/namespace space entirely. +// +// 2. Ensure member containers of the same sandbox share the namespaces CRI +// actually asked them to share, by substituting guest-side equivalents +// obtained from getSharedResources. +// +// CRI's WithPodNamespaces sets a host Path on the IPC and UTS namespace +// entries unconditionally (Kubernetes shares both by default, and does not +// expose a per-pod option to turn either off), and on the PID namespace +// entry whenever the pod's PID sharing mode isn't NamespaceMode_CONTAINER +// (covering both NamespaceMode_POD, e.g. shareProcessNamespace: true, and +// NamespaceMode_NODE, e.g. hostPID: true). Since the shim reports its own +// host PID as the sandbox's PID for both of those modes, there is no data in +// the request that would let it tell them apart, so any non-empty incoming +// Path on any of these three types is treated as "share within this +// sandbox". A container with no such entry at all (NamespaceMode_CONTAINER, +// the default, applicable to PID only) keeps its own independent namespace. +// +// Sharing a UTS namespace needs no extra coordination for the hostname +// itself: an OCI runtime setting Spec.Hostname while joining an existing +// (rather than freshly created) UTS namespace calls sethostname(2) after +// joining it — updating the namespace, and so every container sharing it, +// rather than erroring — and leaves it alone when Hostname is empty. Every +// member container of a pod already carries the same CRI-provided hostname +// on its own spec, so this "last write wins, empty means no opinion" +// behavior is exactly what is wanted, entirely for free. +// +// Namespaces are requested from the guest in a single call, and only the +// types this container actually needs are asked for. That matters for the PID +// namespace in particular, which the guest can only provide by spawning a +// persistent anchor process. +// +// A network namespace exists to scope in-guest networking, and a virtio-net +// interface is what creates that, so NIC presence alone decides the outcome: +// +// - hasContainerNIC: the container has its own annotation-driven NIC +// (ctrNetConfig.Networks is non-empty), so it keeps a namespace of its own +// — an empty Path, which asks the runtime to create a fresh one. The +// per-container NIC/veth wiring in internal/vminit/ctrnetworking assumes +// each such container owns its namespace, so it must not be put in a +// shared one. +// - no container NIC: there is no in-guest networking for a namespace to +// scope, so the entry is dropped entirely, leaving the container in the +// VM's own network namespace. Container traffic in this configuration +// reaches the host by other means (the guest kernel proxying its IP +// sockets, i.e. TSI), which no network namespace can scope in any case. +// +// A shared guest network namespace for the case where the sandbox itself has +// a NIC but this container does not is deliberately not implemented: TSI +// hijacks a socket() call on address family alone, before any namespace or +// routing decision, so it is not scoped by a guest network namespace at all, +// and a real virtio-net interface is only ever plumbed into the VM's own +// initial network namespace (see internal/vminit/vmnetworking.SetupVM) — a +// second, separate network namespace created for member containers would +// contain nothing but loopback, cut those containers off from the sandbox's +// actual NIC entirely, and is not exercised by any test. If per-container +// sharing of a sandbox-level NIC is wanted in the future, the interface (or +// a veth peer of it) needs to be plumbed into the shared namespace itself, +// not just have containers join an empty one. +func sanitizeNamespaces(ctx context.Context, b *bundle.Bundle, hasContainerNIC bool, getSharedResources sharedResourceFunc) error { + if b.Spec.Linux == nil { + return nil + } + + dropNetwork := !hasContainerNIC + + // First pass: work out which shared namespaces this container needs, so + // they can all be requested from the guest in one call. The network + // namespace is never one of them — see the doc comment above for why + // there is no shared-network-namespace case at all. + var ( + wantIPC bool + wantUTS bool + wantPID bool + ) + for _, ns := range b.Spec.Linux.Namespaces { + switch ns.Type { + case specs.IPCNamespace: + wantIPC = wantIPC || ns.Path != "" + case specs.UTSNamespace: + wantUTS = wantUTS || ns.Path != "" + case specs.PIDNamespace: + wantPID = wantPID || ns.Path != "" + } + } + + var types []srAPI.Type + if wantIPC { + types = append(types, srAPI.Type_TYPE_NAMESPACE_IPC) + } + if wantUTS { + types = append(types, srAPI.Type_TYPE_NAMESPACE_UTS) + } + if wantPID { + types = append(types, srAPI.Type_TYPE_NAMESPACE_PID) + } + + var paths map[srAPI.Type]string + if len(types) > 0 { + var err error + if paths, err = getSharedResources(ctx, types); err != nil { + return fmt.Errorf("get shared namespaces: %w", err) + } + } + + // Second pass: rewrite the spec. Built as a new slice because the network + // namespace entry is dropped outright when this container has no NIC of + // its own. + out := make([]specs.LinuxNamespace, 0, len(b.Spec.Linux.Namespaces)) + for _, ns := range b.Spec.Linux.Namespaces { + switch ns.Type { + case specs.NetworkNamespace: + if dropNetwork { + continue + } + // hasContainerNIC: keep the entry but clear any host Path, so + // the guest runtime creates this container a fresh namespace + // of its own for internal/vminit/ctrnetworking's veth wiring + // to attach to. + ns.Path = "" + case specs.IPCNamespace: + if ns.Path != "" { + ns.Path = paths[srAPI.Type_TYPE_NAMESPACE_IPC] + } + case specs.UTSNamespace: + if ns.Path != "" { + ns.Path = paths[srAPI.Type_TYPE_NAMESPACE_UTS] + } + case specs.PIDNamespace: + if ns.Path != "" { + ns.Path = paths[srAPI.Type_TYPE_NAMESPACE_PID] + } + default: + // No other namespace type ever has a valid host Path in the + // guest. + ns.Path = "" + } + out = append(out, ns) + } + + if len(out) == 0 { + out = nil + } + b.Spec.Linux.Namespaces = out + + return nil +} diff --git a/internal/shim/task/namespaces_test.go b/internal/shim/task/namespaces_test.go new file mode 100644 index 00000000..c1745eeb --- /dev/null +++ b/internal/shim/task/namespaces_test.go @@ -0,0 +1,330 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "errors" + "reflect" + "testing" + + specs "github.com/opencontainers/runtime-spec/specs-go" + + srAPI "github.com/containerd/nerdbox/api/services/sharedresources/v1" + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// Stand-in guest paths. The real values are chosen by the guest and returned +// over the wire, so the host must never assume a particular layout; these +// only need to be distinguishable from each other. +const ( + fakeNetPath = "/run/netns/test-sandbox" + fakeIPCPath = "/run/ipcns/test-sandbox" + fakeUTSPath = "/run/utsns/test-sandbox" + fakePIDPath = "/run/pidns/test-sandbox" +) + +// recorder is a sharedResourceFunc that serves fixed paths and records +// every request made through it, so tests can assert not just the resulting +// spec but which namespace types were actually asked of the guest. +type recorder struct { + calls [][]srAPI.Type +} + +func (r *recorder) fn() sharedResourceFunc { + return func(_ context.Context, types []srAPI.Type) (map[srAPI.Type]string, error) { + r.calls = append(r.calls, types) + out := make(map[srAPI.Type]string, len(types)) + for _, t := range types { + switch t { + case srAPI.Type_TYPE_NAMESPACE_NETWORK: + out[t] = fakeNetPath + case srAPI.Type_TYPE_NAMESPACE_IPC: + out[t] = fakeIPCPath + case srAPI.Type_TYPE_NAMESPACE_UTS: + out[t] = fakeUTSPath + case srAPI.Type_TYPE_NAMESPACE_PID: + out[t] = fakePIDPath + } + } + return out, nil + } +} + +// requested flattens every recorded call into the list of types asked for. It +// also asserts the guest was called at most once, since sanitizeNamespaces is +// meant to batch its needs into a single request. +func (r *recorder) requested(t *testing.T) []srAPI.Type { + t.Helper() + if len(r.calls) > 1 { + t.Errorf("getSharedResources called %d times, want at most 1: %v", len(r.calls), r.calls) + } + if len(r.calls) == 0 { + return nil + } + return r.calls[0] +} + +func TestSanitizeNamespaces(t *testing.T) { + ctx := context.Background() + + testcases := []struct { + name string + linux *specs.Linux + hasContainerNIC bool + want []specs.LinuxNamespace + // wantRequested is the exact set of namespace types the guest must be + // asked for, in order. Nil means the guest must not be called at all. + wantRequested []srAPI.Type + }{ + { + name: "nil Linux is a no-op", + linux: nil, + want: nil, + }, + { + name: "container NIC: nothing added, guest not called", + linux: &specs.Linux{}, + hasContainerNIC: true, + want: nil, + }, + { + name: "no container NIC: host network namespace path dropped entirely", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.MountNamespace}, + {Type: specs.NetworkNamespace, Path: "/proc/12345/ns/net"}, + }, + }, + want: []specs.LinuxNamespace{ + {Type: specs.MountNamespace}, + }, + }, + { + name: "container NIC: existing network namespace path stripped (crun creates a fresh one)", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: "/proc/12345/ns/net"}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: ""}, + }, + }, + { + name: "host path on User namespace is stripped (no sharing mechanism for this type)", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.UserNamespace, Path: "/proc/12345/ns/user"}, + }, + }, + hasContainerNIC: true, // avoid also asserting the dropped network entry + want: []specs.LinuxNamespace{ + {Type: specs.UserNamespace, Path: ""}, + }, + }, + { + name: "host UTS namespace path redirected to the shared UTS namespace", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.UTSNamespace, Path: "/proc/12345/ns/uts"}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.UTSNamespace, Path: fakeUTSPath}, + }, + wantRequested: []srAPI.Type{srAPI.Type_TYPE_NAMESPACE_UTS}, + }, + { + name: "empty-Path UTS namespace (per-container mode) is left alone, guest not called", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.UTSNamespace}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.UTSNamespace}, + }, + }, + { + name: "no container NIC: empty-Path network namespace dropped too", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace}, + }, + }, + want: nil, + }, + { + name: "no container NIC: no network namespace added", + linux: &specs.Linux{}, + want: nil, + }, + { + name: "container NIC: empty-Path network namespace keeps its own namespace", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: ""}, + }, + }, + { + name: "host IPC namespace path redirected to the shared IPC namespace", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: fakeIPCPath}, + }, + wantRequested: []srAPI.Type{srAPI.Type_TYPE_NAMESPACE_IPC}, + }, + { + name: "host PID namespace path redirected to the shared PID namespace (covers both pod-level and node-level sharing)", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.PIDNamespace, Path: "/proc/12345/ns/pid"}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.PIDNamespace, Path: fakePIDPath}, + }, + wantRequested: []srAPI.Type{srAPI.Type_TYPE_NAMESPACE_PID}, + }, + { + name: "empty-Path IPC/PID namespaces (per-container mode) are left alone, guest not called", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace}, + {Type: specs.PIDNamespace}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace}, + {Type: specs.PIDNamespace}, + }, + }, + { + // A shared IPC namespace must not drag in a PID namespace. This + // is the common CRI shape: Kubernetes shares pod IPC by default + // but only shares PID when explicitly asked, and creating a PID + // namespace costs the guest a persistent anchor process. + name: "sharing IPC alone does not request a PID namespace", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + {Type: specs.PIDNamespace}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: fakeIPCPath}, + {Type: specs.PIDNamespace}, + }, + wantRequested: []srAPI.Type{srAPI.Type_TYPE_NAMESPACE_IPC}, + }, + { + name: "every shared namespace is requested in a single call", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: "/proc/12345/ns/net"}, + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + {Type: specs.UTSNamespace, Path: "/proc/12345/ns/uts"}, + {Type: specs.PIDNamespace, Path: "/proc/12345/ns/pid"}, + }, + }, + hasContainerNIC: true, + want: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: ""}, + {Type: specs.IPCNamespace, Path: fakeIPCPath}, + {Type: specs.UTSNamespace, Path: fakeUTSPath}, + {Type: specs.PIDNamespace, Path: fakePIDPath}, + }, + wantRequested: []srAPI.Type{ + srAPI.Type_TYPE_NAMESPACE_IPC, + srAPI.Type_TYPE_NAMESPACE_UTS, + srAPI.Type_TYPE_NAMESPACE_PID, + }, + }, + { + // Dropping the network namespace must not affect the others. + name: "no container NIC: IPC sharing still works", + linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace, Path: "/proc/12345/ns/net"}, + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }, + }, + want: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: fakeIPCPath}, + }, + wantRequested: []srAPI.Type{srAPI.Type_TYPE_NAMESPACE_IPC}, + }, + } + + for _, tc := range testcases { + t.Run(tc.name, func(t *testing.T) { + var rec recorder + + b := &bundle.Bundle{Spec: specs.Spec{Linux: tc.linux}} + if err := sanitizeNamespaces(ctx, b, tc.hasContainerNIC, rec.fn()); err != nil { + t.Fatalf("sanitizeNamespaces: %v", err) + } + + var got []specs.LinuxNamespace + if b.Spec.Linux != nil { + got = b.Spec.Linux.Namespaces + } + if !reflect.DeepEqual(got, tc.want) { + t.Errorf("namespaces = %+v, want %+v", got, tc.want) + } + if gotReq := rec.requested(t); !reflect.DeepEqual(gotReq, tc.wantRequested) { + t.Errorf("requested namespace types = %v, want %v", gotReq, tc.wantRequested) + } + }) + } +} + +// TestSanitizeNamespacesPropagatesSharedResourcesError verifies that a +// failure to obtain the shared namespaces (e.g. the guest RPC failing) is +// surfaced as an error, not silently ignored. +func TestSanitizeNamespacesPropagatesSharedResourcesError(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{Linux: &specs.Linux{ + Namespaces: []specs.LinuxNamespace{ + {Type: specs.IPCNamespace, Path: "/proc/12345/ns/ipc"}, + }, + }}} + wantErr := errors.New("guest unreachable") + err := sanitizeNamespaces(context.Background(), b, true, + func(context.Context, []srAPI.Type) (map[srAPI.Type]string, error) { + return nil, wantErr + }) + if err == nil || !errors.Is(err, wantErr) { + t.Errorf("sanitizeNamespaces error = %v, want wrapping %v", err, wantErr) + } +} diff --git a/internal/shim/task/podconfig.go b/internal/shim/task/podconfig.go new file mode 100644 index 00000000..c512699c --- /dev/null +++ b/internal/shim/task/podconfig.go @@ -0,0 +1,137 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "fmt" + "slices" + + specs "github.com/opencontainers/runtime-spec/specs-go" + "google.golang.org/protobuf/proto" + "google.golang.org/protobuf/types/known/anypb" + + "github.com/containerd/nerdbox/api/types" + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// podSandboxConfigTypeURL is the typeurl type URL containerd's CRI plugin +// uses for a marshaled k8s.io/cri-api runtime.v1.PodSandboxConfig (verified +// against containerd's RunPodSandbox, which passes the CRI-supplied +// *runtime.PodSandboxConfig to sandbox.WithOptions, ultimately reaching +// CreateSandboxRequest.Options as typeurl.MarshalAny(config) -- typeurl's +// bare-full-name convention, not a URL with a "type.googleapis.com/" +// scheme, for a type with no explicit typeurl.Register call). Checked here +// as a plain string rather than via typeurl's registry-based dispatch: +// types.CRIPodConfig is a distinct Go type from the real upstream message, +// deliberately not registered with typeurl under this name (typeurl.Register +// only panics if the same Go type is later registered under a different +// path -- it does not detect two different types sharing one path -- so it +// would not actually guard against a future k8s.io/cri-api reintroduction +// anyway; decoding goes straight through proto.Unmarshal below instead). +const podSandboxConfigTypeURL = "runtime.v1.PodSandboxConfig" + +// decodePodSandboxConfig unmarshals a sandbox's CreateSandboxRequest.Options +// into a types.CRIPodConfig. opts is exactly what SandboxService.Options() +// returns: nil for a sandbox created without one (a legacy/non-CRI +// caller), otherwise the opaque payload the sandbox package intentionally +// does not interpret itself (see that field's doc comment). +// +// Returns (nil, nil) for a nil opts — this is the common case for anything +// that isn't real CRI (e.g. shimtest's sandbox suite, `ctr` sandboxes) and +// must not be treated as an error. A non-nil error means opts was present +// but was not a PodSandboxConfig (wrong type URL) or failed to decode as +// one; callers should treat that as non-fatal too (log and continue +// without pod config) since a shim must never fail Task.Create over an +// optional, best-effort feature. +func decodePodSandboxConfig(opts *anypb.Any) (*types.CRIPodConfig, error) { + if opts == nil { + return nil, nil + } + if opts.GetTypeUrl() != podSandboxConfigTypeURL { + return nil, fmt.Errorf("unexpected sandbox options type %q, want %q", opts.GetTypeUrl(), podSandboxConfigTypeURL) + } + cfg := &types.CRIPodConfig{} + if err := proto.Unmarshal(opts.GetValue(), cfg); err != nil { + return nil, fmt.Errorf("unmarshal sandbox options as PodSandboxConfig: %w", err) + } + return cfg, nil +} + +// addHostname sets the container's hostname to match the pod's, mirroring +// what CRI's podsandbox controller does for the podsandbox path +// (Controller.setupSandboxFiles writing an /etc/hostname bind-mounted into +// every member container — internal/cri/server/podsandbox/sandbox_run_linux.go +// upstream). The shim sandboxer path this package implements gets no such +// file from containerd (only the podsandbox controller creates one), so +// the shim must generate it itself from the pod config it already has. +// +// hostname empty is a no-op: crun/the guest kernel's own default applies. +func addHostname(_ context.Context, b *bundle.Bundle, hostname string) error { + if hostname == "" { + return nil + } + + // The OCI runtime spec's own Hostname field is what actually sets the + // container's UTS hostname (crun calls sethostname() after + // establishing the UTS namespace). Setting this is enough on its own + // for anything using gethostname(2)/uname(2); the /etc/hostname file + // below additionally covers programs that read the file directly. + b.Spec.Hostname = hostname + + if slices.ContainsFunc(b.Spec.Mounts, func(m specs.Mount) bool { + return m.Destination == "/etc/hostname" + }) { + return nil + } + + b.AddExtraFile("hostname", []byte(hostname+"\n")) + b.Spec.Mounts = append(b.Spec.Mounts, specs.Mount{ + Destination: "/etc/hostname", + Type: "bind", + Source: "hostname", + Options: []string{"rbind", "rprivate"}, + }) + return nil +} + +// addSysctls merges the pod's CRI sysctls (PodSandboxConfig.Linux.Sysctls +// — CRI only carries sysctls at the pod level, not per-container) into +// the container's OCI spec, which crun applies inside the container's +// namespaces at start. Existing spec.Linux.Sysctl entries win on key +// collision (an explicit per-container value, however it got there, is +// assumed more specific than the pod default). +// +// A nil/empty sysctls map is a no-op. +func addSysctls(_ context.Context, b *bundle.Bundle, sysctls map[string]string) error { + if len(sysctls) == 0 { + return nil + } + if b.Spec.Linux == nil { + b.Spec.Linux = &specs.Linux{} + } + if b.Spec.Linux.Sysctl == nil { + b.Spec.Linux.Sysctl = make(map[string]string, len(sysctls)) + } + for k, v := range sysctls { + if _, exists := b.Spec.Linux.Sysctl[k]; exists { + continue + } + b.Spec.Linux.Sysctl[k] = v + } + return nil +} diff --git a/internal/shim/task/podconfig_test.go b/internal/shim/task/podconfig_test.go new file mode 100644 index 00000000..681514b8 --- /dev/null +++ b/internal/shim/task/podconfig_test.go @@ -0,0 +1,177 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "testing" + + specs "github.com/opencontainers/runtime-spec/specs-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "google.golang.org/protobuf/proto" + "google.golang.org/protobuf/types/known/anypb" + + "github.com/containerd/nerdbox/api/types" + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// marshalTestPodSandboxConfig marshals cfg with the real +// google.golang.org/protobuf/proto encoder, then wraps it with the actual +// type URL a real CreateSandboxRequest.Options carries, so these tests +// exercise the real decode path (decodePodSandboxConfig) exactly as a real +// CRI-originated payload would. +func marshalTestPodSandboxConfig(t *testing.T, cfg *types.CRIPodConfig) *anypb.Any { + t.Helper() + data, err := proto.Marshal(cfg) + require.NoError(t, err) + return &anypb.Any{TypeUrl: podSandboxConfigTypeURL, Value: data} +} + +func TestPodSandboxConfig(t *testing.T) { + t.Run("nil options is a no-op, not an error", func(t *testing.T) { + cfg, err := decodePodSandboxConfig(nil) + require.NoError(t, err) + assert.Nil(t, cfg) + }) + + t.Run("decodes a real PodSandboxConfig", func(t *testing.T) { + opts := marshalTestPodSandboxConfig(t, &types.CRIPodConfig{ + Hostname: "my-pod", + DnsConfig: &types.CRIDNSConfig{ + Servers: []string{"1.1.1.1"}, + }, + Linux: &types.CRILinuxPodSandboxConfig{ + Sysctls: map[string]string{"kernel.shm_rmid_forced": "1"}, + }, + }) + + cfg, err := decodePodSandboxConfig(opts) + require.NoError(t, err) + require.NotNil(t, cfg) + assert.Equal(t, "my-pod", cfg.GetHostname()) + assert.Equal(t, []string{"1.1.1.1"}, cfg.GetDnsConfig().GetServers()) + assert.Equal(t, "1", cfg.GetLinux().GetSysctls()["kernel.shm_rmid_forced"]) + }) + + t.Run("decodes repeated DNS fields and multiple sysctl entries", func(t *testing.T) { + opts := marshalTestPodSandboxConfig(t, &types.CRIPodConfig{ + DnsConfig: &types.CRIDNSConfig{ + Servers: []string{"1.1.1.1", "8.8.8.8"}, + Searches: []string{"svc.cluster.local", "cluster.local"}, + Options: []string{"ndots:5"}, + }, + Linux: &types.CRILinuxPodSandboxConfig{ + Sysctls: map[string]string{ + "kernel.shm_rmid_forced": "1", + "fs.mqueue.msg_max": "100", + }, + }, + }) + + cfg, err := decodePodSandboxConfig(opts) + require.NoError(t, err) + require.NotNil(t, cfg) + assert.Equal(t, []string{"1.1.1.1", "8.8.8.8"}, cfg.GetDnsConfig().GetServers()) + assert.Equal(t, []string{"svc.cluster.local", "cluster.local"}, cfg.GetDnsConfig().GetSearches()) + assert.Equal(t, []string{"ndots:5"}, cfg.GetDnsConfig().GetOptions()) + assert.Equal(t, map[string]string{ + "kernel.shm_rmid_forced": "1", + "fs.mqueue.msg_max": "100", + }, cfg.GetLinux().GetSysctls()) + }) + + t.Run("errors for a mismatched type URL", func(t *testing.T) { + data, err := proto.Marshal(&types.CRIDNSConfig{Servers: []string{"1.1.1.1"}}) + require.NoError(t, err) + opts := &anypb.Any{TypeUrl: "runtime.v1.DNSConfig", Value: data} + + _, err = decodePodSandboxConfig(opts) + assert.Error(t, err) + }) + + t.Run("errors for the right type URL but unparsable bytes", func(t *testing.T) { + opts := &anypb.Any{TypeUrl: podSandboxConfigTypeURL, Value: []byte("not a valid protobuf message")} + + _, err := decodePodSandboxConfig(opts) + assert.Error(t, err) + }) +} + +func TestAddHostname(t *testing.T) { + t.Run("empty hostname is a no-op", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{}} + require.NoError(t, addHostname(context.Background(), b, "")) + assert.Empty(t, b.Spec.Hostname) + assert.Empty(t, b.Spec.Mounts) + }) + + t.Run("sets spec.Hostname and adds an /etc/hostname mount", func(t *testing.T) { + b := loadTestBundle(t, specs.Spec{}) + require.NoError(t, addHostname(context.Background(), b, "my-pod")) + assert.Equal(t, "my-pod", b.Spec.Hostname) + + require.Len(t, b.Spec.Mounts, 1) + assert.Equal(t, "/etc/hostname", b.Spec.Mounts[0].Destination) + assert.Equal(t, "hostname", b.Spec.Mounts[0].Source) + + files, err := b.Files() + require.NoError(t, err) + assert.Equal(t, "my-pod\n", string(files["hostname"])) + }) + + t.Run("existing /etc/hostname mount is left untouched", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{Mounts: []specs.Mount{ + {Destination: "/etc/hostname", Type: "bind", Source: "/custom/hostname"}, + }}} + require.NoError(t, addHostname(context.Background(), b, "my-pod")) + // spec.Hostname is still set (it's a separate mechanism from the + // file mount and crun applies it regardless of /etc/hostname). + assert.Equal(t, "my-pod", b.Spec.Hostname) + require.Len(t, b.Spec.Mounts, 1) + assert.Equal(t, "/custom/hostname", b.Spec.Mounts[0].Source) + }) +} + +func TestAddSysctls(t *testing.T) { + t.Run("empty map is a no-op", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{}} + require.NoError(t, addSysctls(context.Background(), b, nil)) + assert.Nil(t, b.Spec.Linux) + }) + + t.Run("merges into a nil Linux/Sysctl", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{}} + require.NoError(t, addSysctls(context.Background(), b, map[string]string{ + "kernel.shm_rmid_forced": "1", + })) + require.NotNil(t, b.Spec.Linux) + assert.Equal(t, "1", b.Spec.Linux.Sysctl["kernel.shm_rmid_forced"]) + }) + + t.Run("existing per-container sysctl wins on collision", func(t *testing.T) { + b := &bundle.Bundle{Spec: specs.Spec{Linux: &specs.Linux{ + Sysctl: map[string]string{"kernel.shm_rmid_forced": "0"}, + }}} + require.NoError(t, addSysctls(context.Background(), b, map[string]string{ + "kernel.shm_rmid_forced": "1", + "fs.mqueue.msg_max": "100", + })) + assert.Equal(t, "0", b.Spec.Linux.Sysctl["kernel.shm_rmid_forced"]) + assert.Equal(t, "100", b.Spec.Linux.Sysctl["fs.mqueue.msg_max"]) + }) +} diff --git a/internal/shim/task/sandboxopts.go b/internal/shim/task/sandboxopts.go new file mode 100644 index 00000000..5a63a1b5 --- /dev/null +++ b/internal/shim/task/sandboxopts.go @@ -0,0 +1,96 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "fmt" + "os" + + "github.com/containerd/log" + + "github.com/containerd/nerdbox/internal/shim/sandbox" + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// SandboxStartOptions parses the sandbox OCI bundle at bundlePath to derive +// the VM start options: resources (CPU/mem), networking (NICs, init args), +// and resolv.conf injection. It is registered with the SandboxService as its +// StartOptionsFunc, allowing the sandbox service to boot the VM without +// importing the task package (avoiding a circular dependency). +// +// bundlePath is the path the containerd sandbox controller passed in +// CreateSandboxRequest.BundlePath. It may be the shim's working directory for +// the sandbox bundle. +func SandboxStartOptions(debug bool) sandbox.StartOptionsFunc { + return func(ctx context.Context, bundlePath string) ([]sandbox.Opt, error) { + var ( + nwpr networksProvider + resCfg resourceConfig + dumpInfoCfg dumpInfoConfig + ) + + _, err := bundle.LoadSandboxConfig(ctx, bundlePath, + nwpr.FromBundle, + resCfg.FromBundle, + dumpInfoCfg.FromBundle, + func(ctx context.Context, b *bundle.Bundle) error { + // No pod-level DNSConfig available here: this call only + // exists to populate nwpr/resCfg/dumpInfoCfg from the + // sandbox's own bundle, ahead of it being sent to the + // guest at all; the resulting *bundle.Bundle itself + // (and therefore addResolvConf's mutations to it) is + // discarded below. + return addResolvConf(ctx, b, len(nwpr.nws) == 0, nil) + }, + ) + if err != nil { + // A minimal sandbox bundle with no config.json at all (e.g. + // shimtest's sandbox suite, `ctr` sandboxes) is the one + // legitimate reason to fall back to defaults: LoadSandboxConfig's + // first step is reading config.json, and that specific + // failure is returned unwrapped, so os.IsNotExist still + // matches it. A sandbox bundle legitimately has no Root at + // all (it has no rootfs of its own), which + // LoadSandboxConfig already accounts for, so anything else + // here — a config.json that exists but fails to parse, a bad + // network/resource annotation, or an addResolvConf failure — + // means the caller's requested sandbox configuration could + // not be honored, and silently booting with 2 vCPU/2048MiB + // defaults instead would discard it without any indication + // anything went wrong. + if !os.IsNotExist(err) { + return nil, fmt.Errorf("load sandbox bundle: %w", err) + } + log.G(ctx).WithError(err).Debug("sandbox bundle has no config.json; using resource defaults") + return []sandbox.Opt{ + sandbox.WithResources(2, 2048), + }, nil + } + + var opts []sandbox.Opt + opts = append(opts, resCfg.SandboxOpts()...) + opts = append(opts, nwpr.SandboxOptions()...) + opts = append(opts, dumpInfoCfg.SandboxOpts()...) + if debug { + opts = append(opts, sandbox.WithInitArgs("-debug")) + } + opts = append(opts, sandbox.WithInitArgs(nwpr.InitArgs()...)) + + return opts, nil + } +} diff --git a/internal/shim/task/sandboxopts_test.go b/internal/shim/task/sandboxopts_test.go new file mode 100644 index 00000000..d4791528 --- /dev/null +++ b/internal/shim/task/sandboxopts_test.go @@ -0,0 +1,85 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "encoding/json" + "os" + "path/filepath" + "testing" + + specs "github.com/opencontainers/runtime-spec/specs-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestSandboxStartOptions(t *testing.T) { + t.Run("missing config.json falls back to resource defaults, not an error", func(t *testing.T) { + opts, err := SandboxStartOptions(false)(context.Background(), t.TempDir()) + require.NoError(t, err) + assert.NotEmpty(t, opts) + }) + + t.Run("config.json without Root (the real sandbox-bundle shape) is not an error", func(t *testing.T) { + // A sandbox bundle's config.json legitimately has no Root at all — + // a sandbox has no rootfs of its own — unlike a container bundle, + // where a missing Root is a real error. This is the actual shape + // containerd's sandbox controller and shimtest's sandbox suite + // both produce, so this case must go through LoadSandboxConfig + // successfully rather than being treated the same as a container + // bundle missing Root. + spec := specs.Spec{ + Annotations: map[string]string{ + "io.containerd.nerdbox.resources.cpu": "4", + "io.containerd.nerdbox.resources.memory": "4096", + }, + } + dir := t.TempDir() + data, err := json.Marshal(spec) + require.NoError(t, err) + require.NoError(t, os.WriteFile(filepath.Join(dir, "config.json"), data, 0o644)) + + opts, err := SandboxStartOptions(false)(context.Background(), dir) + require.NoError(t, err) + assert.NotEmpty(t, opts) + }) + + t.Run("malformed config.json is an error, not a silent fallback", func(t *testing.T) { + dir := t.TempDir() + require.NoError(t, os.WriteFile(filepath.Join(dir, "config.json"), []byte("not json"), 0o644)) + + _, err := SandboxStartOptions(false)(context.Background(), dir) + assert.Error(t, err) + }) + + t.Run("a transformer error (bad network annotation) is an error, not a silent fallback", func(t *testing.T) { + spec := specs.Spec{ + Root: &specs.Root{Path: "rootfs"}, + Annotations: map[string]string{ + "io.containerd.nerdbox.network.0": "not-a-valid-field", + }, + } + dir := t.TempDir() + data, err := json.Marshal(spec) + require.NoError(t, err) + require.NoError(t, os.WriteFile(filepath.Join(dir, "config.json"), data, 0o644)) + + _, err = SandboxStartOptions(false)(context.Background(), dir) + assert.Error(t, err) + }) +} diff --git a/internal/shim/task/sandboxvolumes.go b/internal/shim/task/sandboxvolumes.go new file mode 100644 index 00000000..3bb03525 --- /dev/null +++ b/internal/shim/task/sandboxvolumes.go @@ -0,0 +1,87 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "fmt" + "os" + + "github.com/containerd/log" + + "github.com/containerd/nerdbox/internal/shim/sandbox" + "github.com/containerd/nerdbox/internal/shim/task/bundle" +) + +// sandboxVolumeMounter is a bundle.Transformer for sandbox member +// containers that rewrites OCI "bind" mounts to reference the sandbox's +// shared filesystem tree instead of a new per-mount virtiofs share. +// +// This is the sandboxed-path counterpart to bindMounter (mount.go), which +// is used by the legacy/plain-container path: that path boots a fresh VM +// per container and can add a new virtiofs share before boot +// (sandbox.WithFS), so giving every bind mount its own virtiofs tag works +// fine there. A sandbox member container is created against an +// already-running VM, and virtio-fs shares cannot be hot-added after +// boot — asking the guest to mount a tag that was never wired up on the +// host/VMM side fails immediately (EINVAL). So instead, each bind mount's +// host source is itself bind-mounted (on the host, by +// sandbox.SharedFS.ShareVolume) into the sandbox's shared directory tree, +// which is already exposed to the guest via one persistent, pre-boot +// virtiofs share — the guest sees the content with no new device and no +// extra guest-side mount step at all. +type sandboxVolumeMounter struct { + fs *sandbox.SharedFS + containerID string + n int // next volume index to assign +} + +// FromBundle rewrites each "bind" mount's Source in the spec to the guest +// path where sandbox.SharedFS.ShareVolume exposes it. Must run after the +// bundle's rootfs mounts are known to fs (order relative to ShareRootfs +// does not matter: volumes live under a separate subtree), but before the +// spec is sent to the guest. +func (vm *sandboxVolumeMounter) FromBundle(ctx context.Context, b *bundle.Bundle) error { + for i, m := range b.Spec.Mounts { + if m.Type != "bind" { + continue + } + + log.G(ctx).WithField("mount", m).Debug("sharing bind mount volume via the sandbox virtiofs tree") + + fi, err := os.Stat(m.Source) + if err != nil { + return fmt.Errorf("failed to stat bind mount source %s: %w", m.Source, err) + } + + // Only Source changes here — Options (ro/rw, recursive or not, + // propagation) are left exactly as the spec requested, so crun's + // own bind mount from the returned guest path into the container + // is what actually enforces them. See ShareVolume's doc comment + // for why duplicating read-only enforcement at this layer would + // be actively wrong, not just redundant. + guestPath, err := vm.fs.ShareVolume(ctx, vm.containerID, vm.n, m.Source, fi.IsDir()) + if err != nil { + return fmt.Errorf("share volume mount %s: %w", m.Source, err) + } + vm.n++ + + b.Spec.Mounts[i].Source = guestPath + } + + return nil +} diff --git a/internal/shim/task/service.go b/internal/shim/task/service.go index 19466054..002792a6 100644 --- a/internal/shim/task/service.go +++ b/internal/shim/task/service.go @@ -51,10 +51,7 @@ import ( "github.com/containerd/nerdbox/internal/shim/task/bundle" ) -var ( - _ = shim.TTRPCService(&service{}) - empty = &ptypes.Empty{} -) +var empty = &ptypes.Empty{} // guestRuncOptions constructs a fresh runc Options message containing only the // fields that are meaningful inside the VM guest, and returns it as a @@ -120,7 +117,7 @@ func guestRuncOptions(ctx context.Context, opts *ptypes.Any) (*ptypes.Any, error } // NewTaskService creates a new instance of a task service -func NewTaskService(ctx context.Context, sb sandbox.Sandbox, publisher shim.Publisher, sd shutdown.Service) (taskAPI.TTRPCTaskService, error) { +func NewTaskService(ctx context.Context, svc *sandbox.SandboxService, publisher shim.Publisher, sd shutdown.Service) (taskAPI.TTRPCTaskService, error) { var debug bool if opts, ok := ctx.Value(shim.OptsKey{}).(shim.Opts); ok { debug = opts.Debug @@ -128,7 +125,8 @@ func NewTaskService(ctx context.Context, sb sandbox.Sandbox, publisher shim.Publ s := &service{ context: ctx, - sb: sb, + sb: svc, + svc: svc, events: make(chan any, 128), containers: make(map[string]*container), debug: debug, @@ -175,6 +173,10 @@ type container struct { execIODone map[string]<-chan struct{} // execStdinEOF holds the in-band stdin EOF sender per exec ID. execStdinEOF map[string]func() error + + // sharedFSID, when non-empty, is the container ID to unshare from the + // sandbox SharedFS on Delete. Set only on the sandboxed path. + sharedFSID string } // shutdown shuts down the container's IO streams, socket forwarding, and all @@ -203,25 +205,30 @@ func (c *container) shutdown(ctx context.Context) error { type service struct { mu sync.Mutex - // sb is the sandbox instance used to run the container + // sb is the sandbox instance used to run the container (VM lifecycle + + // TTRPC client). For the sandbox API path this is the SandboxService; + // for the legacy single-container path it is a plain vm sandbox. sb sandbox.Sandbox + // svc is the full SandboxService. It is non-nil when using the containerd + // sandbox API path, and nil on the legacy single-container path. + svc *sandbox.SandboxService + context context.Context events chan any containers map[string]*container + // eventStreamOnce ensures the VM event stream is started exactly once, + // regardless of how many containers are created in a sandboxed VM. + eventStreamOnce sync.Once + debug bool initiateShutdown func() initiateShutdownOnce sync.Once shutdownDone <-chan struct{} } -func (s *service) RegisterTTRPC(server *ttrpc.Server) error { - taskAPI.RegisterTTRPCTaskService(server, s) - return nil -} - func (s *service) shutdown(ctx context.Context) error { // Detach all containers from tracking under the lock, then shut them down // outside of it. Each container shutdown can block until its host-side @@ -241,7 +248,14 @@ func (s *service) shutdown(ctx context.Context) error { } } - if s.sb != nil { + // When using the containerd sandbox API (svc != nil and sandboxed), the + // SandboxService owns VM lifetime. ShutdownSandbox will be called by the + // sandbox controller, which triggers VM stop and SharedFS cleanup there. + // We only stop the VM ourselves on the legacy single-container path + // (svc == nil or not yet sandboxed via the API). + sandboxOwned := s.svc != nil && s.svc.IsSandboxed() + + if s.sb != nil && !sandboxOwned { // Unmount all block volumes inside the guest before stopping the VM, // to flush ext4 journals and dirty pages to the virtio-blk devices. // Best-effort with a short retry for transient EBUSY. @@ -309,6 +323,268 @@ func (s *service) Create(ctx context.Context, r *taskAPI.CreateTaskRequest) (_ * return nil, errgrpc.ToGRPC(fmt.Errorf("checkpoints not supported: %w", errdefs.ErrNotImplemented)) } + // When the containerd sandbox API is in use (svc.IsSandboxed()), the VM + // is already running (StartSandbox booted it). We skip VM boot and use + // the shared filesystem to serve the container rootfs. On the legacy + // single-container path, we boot the VM here as before. + if s.svc != nil && s.svc.IsSandboxed() { + return s.createSandboxedContainer(ctx, r) + } + return s.createLegacyContainer(ctx, r) +} + +// createSandboxedContainer handles Task.Create for a member container of an +// already-running sandbox VM. It resolves the rootfs on the host via the +// SharedFS (which exposes it into the VM over virtiofs), then drives the +// guest bundle/mount/task RPCs. +func (s *service) createSandboxedContainer(ctx context.Context, r *taskAPI.CreateTaskRequest) (_ *taskAPI.CreateTaskResponse, err error) { + presetup := time.Now() + + fs := s.svc.FS() + if fs == nil { + return nil, errgrpc.ToGRPC(fmt.Errorf("sandbox shared filesystem not initialised: %w", errdefs.ErrFailedPrecondition)) + } + + // Fetch the pod's CRI config (if any) once up front so the transformers + // below can use it. A nil/error result is never fatal here: pod config + // is a best-effort, CRI-specific enhancement (DNS, hostname), not + // something Task.Create can require — shimtest, `ctr` sandboxes, and + // any other non-CRI caller never provide one at all. + podCfg, err := decodePodSandboxConfig(s.svc.Options()) + if err != nil { + log.G(ctx).WithError(err).Warn("failed to parse sandbox options as PodSandboxConfig; continuing without pod-level DNS/hostname config") + } + + // Fetched here (rather than where the VM client is otherwise obtained + // further below) because sanitizeNamespaces, run as part of bundle.Load + // next, may need it to call the guest's SharedResources service if this + // container's spec asks to share any namespace. + vmc, err := s.sb.Client() + if err != nil { + return nil, errgrpc.ToGRPC(err) + } + sharedRes := &sharedResources{client: vmc, sandboxID: s.svc.SandboxID()} + + // Load the OCI bundle and apply per-container transformers. + var ( + ctrNetCfg ctrNetConfig + devShm = shareDevShmMounter{containerID: r.ID, getDevShmFn: sharedRes.getDevShm} + svm = sandboxVolumeMounter{fs: fs, containerID: r.ID} + blockM blockMounter + sfpr = socketForwardsProvider{containerID: r.ID} + ) + + // For the sandboxed path we use a dummy disk allocator since block + // devices cannot be hotplugged. ext4 volumes are still supported via + // the legacy path only. + da := newDiskAllocator(s.sb.ReservedDisks()) + + b, err := bundle.Load(ctx, r.Bundle, + svm.FromBundle, + // Must run after svm.FromBundle: svm only rewrites mounts that + // already have Type "bind", and a /dev/shm mount is still "tmpfs" + // until this runs, so the two never touch the same mount. + devShm.FromBundle, + ctrNetCfg.fromBundle, + sfpr.FromBundle, + func(ctx context.Context, b *bundle.Bundle) error { + // fallbackToHostRC (copy the host's own resolv.conf) only makes + // sense when this container has no dedicated NIC and so relies + // on TSI for connectivity: with a NIC, the container has its + // own real guest-side network stack and should get a + // network-appropriate resolver, not the host's. ctrNetCfg is + // already populated here — ctrNetCfg.fromBundle runs earlier in + // this same transformer chain. + return addResolvConf(ctx, b, len(ctrNetCfg.Networks) == 0, podCfg.GetDnsConfig()) + }, + func(ctx context.Context, b *bundle.Bundle) error { + return addHostname(ctx, b, podCfg.GetHostname()) + }, + func(ctx context.Context, b *bundle.Bundle) error { + return addSysctls(ctx, b, podCfg.GetLinux().GetSysctls()) + }, + func(ctx context.Context, b *bundle.Bundle) error { + return sanitizeNamespaces(ctx, b, len(ctrNetCfg.Networks) > 0, sharedRes.get) + }, + clearApparmorProfile, + ) + if err != nil { + return nil, errgrpc.ToGRPC(err) + } + + // UDS mounts are rewritten to bind mounts whose source is a socket + // file inside the VM and whose destination is a path in the container + // rootfs (e.g. /run/shared.sock). The OCI runtime requires the + // destination to already exist as a regular file. See + // udsPlaceholderSource's doc comment for why the correct target + // depends on the shape of r.Rootfs: a read-only bind needs its + // placeholder written to the still-writable Source before ShareRootfs + // mounts it read-only; anything else (in particular the common + // overlay/erofs assembly) needs it written to the assembled rootfs + // itself, which is only available after ShareRootfs runs. + placeholderSrc, placeholderBeforeAssembly := udsPlaceholderSource(r.Rootfs, fs.RootfsHostPath(r.ID)) + if placeholderBeforeAssembly { + sfpr.CreateRootfsPlaceholders(ctx, placeholderSrc) + } + + // Assemble the container rootfs on the host inside the shared dir. + guestRootfs, err := fs.ShareRootfs(ctx, r.ID, r.Rootfs) + if err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(fmt.Errorf("share rootfs for %s: %w", r.ID, err)) + } + + if !placeholderBeforeAssembly { + sfpr.CreateRootfsPlaceholders(ctx, placeholderSrc) + } + + nwJSON, err := json.Marshal(ctrNetCfg) + if err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(fmt.Errorf("marshal container network config: %w", err)) + } + b.AddExtraFile(nwcfg.Filename, nwJSON) + + // Process ext4 volume mounts in the OCI spec. Note: hotplug is not + // supported so ext4 volumes are not usable in sandboxed mode; FromBundle + // will return no-op if there are no ext4 mounts. + if err := blockM.FromBundle(ctx, b, r.ID, &da); err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + + // vmc was already fetched above (sharedRes needs it before bundle.Load + // runs). + + // Start the VM event stream exactly once for this sandbox (subsequent + // containers in the same VM reuse the same stream). + s.startVMEventStream(vmc) + + bundleFiles, err := b.Files() + if err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + + bundleService := bundleAPI.NewTTRPCBundleClient(vmc) + br, err := bundleService.Create(ctx, &bundleAPI.CreateRequest{ + ID: r.ID, + Files: bundleFiles, + }) + if err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + + // Tell the guest to bind-mount the assembled rootfs from the shared + // virtiofs into the bundle rootfs location. Bind-mount volumes need no + // entry here: sandboxVolumeMounter already exposed them inside the same + // "containers" virtiofs share that guestRootfs lives in, so the guest + // sees their content without any extra guest-side mount step. + var mountSpecs []*mountAPI.MountSpec + mountSpecs = append(mountSpecs, &mountAPI.MountSpec{ + Type: "bind", + Source: guestRootfs, + Target: br.Bundle + "/rootfs", + Options: []string{"rbind"}, + }) + for _, m := range blockM.VmMounts() { + mountSpecs = append(mountSpecs, &mountAPI.MountSpec{ + Type: m.Type, + Source: m.Source, + Target: m.Target, + Options: m.Options, + }) + } + + mc := mountAPI.NewTTRPCMountClient(vmc) + if _, err := mc.MountAll(ctx, &mountAPI.MountAllRequest{Mounts: mountSpecs}); err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(fmt.Errorf("guest MountAll: %w", err)) + } + + rio := stdio.Stdio{ + Stdin: r.Stdin, + Stdout: r.Stdout, + Stderr: r.Stderr, + Terminal: r.Terminal, + } + + cio, ioShutdown, initIODone, initStdinEOF, err := s.forwardIO(ctx, s.sb, r.ID, rio) + if err != nil { + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + + if err := bindSockets(ctx, s.sb, sfpr.entries); err != nil { + ioShutdown(ctx) //nolint:errcheck + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + + setupTime := time.Since(presetup) + preCreate := time.Now() + + c := &container{ + ioShutdown: ioShutdown, + ioDone: initIODone, + stdinEOF: initStdinEOF, + execShutdowns: make(map[string]func(context.Context) error), + execIODone: make(map[string]<-chan struct{}), + execStdinEOF: make(map[string]func() error), + sharedFSID: r.ID, // record for cleanup in Delete + } + + // For the sandboxed path the rootfs mount specs presented to the guest + // Task service are just a bind from the already-mounted shared path. + guestRootfsMounts := []*types.Mount{{ + Type: "bind", + Source: guestRootfs, + Options: []string{"rbind"}, + }} + + tc := taskAPI.NewTTRPCTaskClient(vmc) + resp, err := tc.Create(ctx, &taskAPI.CreateTaskRequest{ + ID: r.ID, + Bundle: br.Bundle, + Rootfs: guestRootfsMounts, + Terminal: cio.Terminal, + Stdin: cio.Stdin, + Stdout: cio.Stdout, + Stderr: cio.Stderr, + Options: r.Options, + }) + if err != nil { + log.G(ctx).WithError(err).Error("failed to create sandboxed task") + c.shutdown(ctx) //nolint:errcheck + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + + fwder, err := startSocketForwarding(context.Background(), s.sb, r.ID, sfpr.entries) + if err != nil { + log.G(ctx).WithError(err).Error("failed to start socket forwarding") + c.shutdown(ctx) //nolint:errcheck + fs.Unshare(ctx, r.ID) //nolint:errcheck + return nil, errgrpc.ToGRPC(err) + } + c.forwarder = fwder + + log.G(ctx).WithFields(log.Fields{ + "t_setup": setupTime, + "t_create": time.Since(preCreate), + }).Info("sandboxed task successfully created") + + s.mu.Lock() + s.containers[r.ID] = c + s.mu.Unlock() + + return &taskAPI.CreateTaskResponse{Pid: resp.Pid}, nil +} + +// createLegacyContainer is the original single-container path: boot a new VM +// per container. Preserved unchanged for non-sandboxed usage. +func (s *service) createLegacyContainer(ctx context.Context, r *taskAPI.CreateTaskRequest) (_ *taskAPI.CreateTaskResponse, err error) { presetup := time.Now() var ( @@ -330,8 +606,11 @@ func (s *service) Create(ctx context.Context, r *taskAPI.CreateTaskRequest) (_ * sfpr.FromBundle, func(ctx context.Context, b *bundle.Bundle) error { // If there are no VM networks, try falling back to host's resolv.conf (for TSI). - return addResolvConf(ctx, b, len(nwpr.nws) == 0) + // The legacy path has no sandbox/pod config concept, so there is + // no pod-level DNSConfig to consider. + return addResolvConf(ctx, b, len(nwpr.nws) == 0, nil) }, + clearApparmorProfile, ) if err != nil { return nil, errgrpc.ToGRPC(err) @@ -411,34 +690,10 @@ func (s *service) Create(ctx context.Context, r *taskAPI.CreateTaskRequest) (_ * return nil, errgrpc.ToGRPC(err) } - // Start forwarding events. - // Use the shim's long-lived context (not the RPC ctx) for the event - // stream. If the connection closes, ctx gets canceled, which causes - // RecvMsg to return without deleting the underlying ttrpc stream. The VM - // keeps sending events to that orphaned stream, which fills the stream's - // recv buffer and blocks the ttrpc receive loop — deadlocking all - // subsequent calls on the same ttrpc client. This needs a fix in ttrpc - // to avoid deadlock, but the stream should be consumed until the stream - // is done or the ttrpc connection closes. - sc, err := vmevents.NewTTRPCEventsClient(vmc).Stream(s.context, empty) - if err != nil { - return nil, errgrpc.ToGRPC(err) - } - ns, _ := namespaces.Namespace(ctx) - go func(ns string) { - for { - ev, err := sc.Recv() - if err != nil { - if errors.Is(err, io.EOF) || errors.Is(err, shutdown.ErrShutdown) || errors.Is(err, ttrpc.ErrClosed) { - log.G(ctx).Info("vm event stream closed") - } else { - log.G(ctx).WithError(err).Error("vm event stream error") - } - return - } - s.send(ev) - } - }(ns) + // Start forwarding events. Use the idempotent helper so the stream is + // started exactly once. On the legacy path there is always exactly one + // call, but using the same helper keeps the logic consistent. + s.startVMEventStream(vmc) bundleFiles, err := b.Files() if err != nil { @@ -556,26 +811,6 @@ func (s *service) Create(ctx context.Context, r *taskAPI.CreateTaskRequest) (_ * s.containers[r.ID] = c s.mu.Unlock() - // TODO: Forward events rather than generate here? - //s.send(&eventstypes.TaskCreate{ - // ContainerID: r.ID, - // Bundle: r.Bundle, - // Rootfs: r.Rootfs, - // IO: &eventstypes.TaskIO{ - // Stdin: r.Stdin, - // Stdout: r.Stdout, - // Stderr: r.Stderr, - // Terminal: r.Terminal, - // }, - // Pid: resp.Pid, - //}) - - // The following line cannot return an error as the only state in which that - // could happen would also cause the container.Pid() call above to - // nil-deference panic. - // proc, _ := container.Process("") - // handleStarted(container, proc) - return &taskAPI.CreateTaskResponse{ Pid: resp.Pid, }, nil @@ -646,6 +881,7 @@ func (s *service) Delete(ctx context.Context, r *taskAPI.DeleteRequest) (*taskAP // re-run the shutdown we are about to perform. s.mu.Lock() var shutdown func(context.Context) error + var sharedFSID string if c, ok := s.containers[r.ID]; ok { if r.ExecID != "" { if ioShutdown, ok := c.execShutdowns[r.ExecID]; ok { @@ -656,6 +892,7 @@ func (s *service) Delete(ctx context.Context, r *taskAPI.DeleteRequest) (*taskAP } } else { shutdown = c.shutdown + sharedFSID = c.sharedFSID delete(s.containers, r.ID) } } @@ -669,6 +906,17 @@ func (s *service) Delete(ctx context.Context, r *taskAPI.DeleteRequest) (*taskAP }).Error("failed to shutdown io after delete") } } + // Unshare the container's rootfs from the shared filesystem. This + // unmounts the host-side overlay/bind and removes the container + // subtree from /containers/. Only set for the + // init process on the sandboxed path. + if sharedFSID != "" && s.svc != nil { + if fs := s.svc.FS(); fs != nil { + if err := fs.Unshare(ctx, sharedFSID); err != nil { + log.G(ctx).WithError(err).WithField("id", sharedFSID).Warn("failed to unshare container rootfs on delete") + } + } + } } return resp, err } @@ -1007,6 +1255,36 @@ func (s *service) Stats(ctx context.Context, r *taskAPI.StatsRequest) (*taskAPI. return tc.Stats(ctx, r) } +// startVMEventStream starts forwarding guest VM events to the host event +// publisher. It is idempotent — the stream is started at most once per +// sandbox regardless of how many containers are created. On the legacy path +// this is called from createLegacyContainer; on the sandboxed path it is +// called from createSandboxedContainer via eventStreamOnce. +func (s *service) startVMEventStream(vmc *ttrpc.Client) { + s.eventStreamOnce.Do(func() { + ctx := s.context + sc, err := vmevents.NewTTRPCEventsClient(vmc).Stream(ctx, empty) + if err != nil { + log.G(ctx).WithError(err).Error("failed to start VM event stream") + return + } + go func() { + for { + ev, err := sc.Recv() + if err != nil { + if errors.Is(err, io.EOF) || errors.Is(err, shutdown.ErrShutdown) || errors.Is(err, ttrpc.ErrClosed) { + log.G(ctx).Info("vm event stream closed") + } else { + log.G(ctx).WithError(err).Error("vm event stream error") + } + return + } + s.send(ev) + } + }() + }) +} + func (s *service) send(evt interface{}) { s.events <- evt } diff --git a/internal/shim/task/sharedresources.go b/internal/shim/task/sharedresources.go new file mode 100644 index 00000000..647b9edd --- /dev/null +++ b/internal/shim/task/sharedresources.go @@ -0,0 +1,129 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "context" + "fmt" + "sync" + + "github.com/containerd/ttrpc" + + srAPI "github.com/containerd/nerdbox/api/services/sharedresources/v1" +) + +// sharedResourceFunc returns the guest paths of the sandbox's shared +// resources of the requested types, creating them on first use. It is +// called by sanitizeNamespaces at most once per container, and only if +// that container's spec actually asks to share something. +type sharedResourceFunc func(ctx context.Context, types []srAPI.Type) (map[srAPI.Type]string, error) + +// sharedResources calls the guest's SharedResources.Create the first time a +// given type is needed and memoizes the result per type. A value is +// created fresh per Task.Create call (see createSandboxedContainer), so a +// container that shares nothing never triggers the guest RPC at all — and +// therefore never causes the guest to create a namespace, or to spawn the +// PID namespace's anchor process, on its behalf. +type sharedResources struct { + client *ttrpc.Client // vminitd's TTRPC connection + sandboxID string // resource group id + + mu sync.Mutex + paths map[srAPI.Type]string +} + +// get implements sharedResourceFunc. Types already fetched are served from +// the memo; only the remainder is requested from the guest. +func (n *sharedResources) get(ctx context.Context, types []srAPI.Type) (map[srAPI.Type]string, error) { + n.mu.Lock() + defer n.mu.Unlock() + + var missing []srAPI.Type + for _, t := range types { + if _, ok := n.paths[t]; !ok { + missing = append(missing, t) + } + } + + if len(missing) > 0 { + c := srAPI.NewTTRPCSharedResourcesClient(n.client) + resp, err := c.Create(ctx, &srAPI.CreateRequest{ + ID: n.sandboxID, + Types: missing, + }) + if err != nil { + return nil, fmt.Errorf("guest resource create: %w", err) + } + if n.paths == nil { + n.paths = make(map[srAPI.Type]string, len(missing)) + } + for _, r := range resp.GetResources() { + n.paths[r.GetType()] = r.GetPath() + } + } + + out := make(map[srAPI.Type]string, len(types)) + for _, t := range types { + path, ok := n.paths[t] + if !ok { + return nil, fmt.Errorf("guest did not return a path for resource type %q", t) + } + out[t] = path + } + return out, nil +} + +// getDevShm returns the guest path of the sandbox's shared /dev/shm tmpfs, +// creating it (sized to sizeBytes) on first use. It shares this +// sharedResources instance's memo and guest connection with get, but is +// called independently by shareDevShmMounter rather than as part of +// sanitizeNamespaces's batched resource request, since a container may +// need one without the other. +// +// sizeBytes is only used the first time this sandbox creates its shared +// /dev/shm; see SharedResources's CreateRequest.dev_shm_size_bytes doc +// comment for why a later, different value is silently ignored. +func (n *sharedResources) getDevShm(ctx context.Context, sizeBytes int64) (string, error) { + n.mu.Lock() + defer n.mu.Unlock() + + const typ = srAPI.Type_TYPE_DEVSHM + if path, ok := n.paths[typ]; ok { + return path, nil + } + + c := srAPI.NewTTRPCSharedResourcesClient(n.client) + resp, err := c.Create(ctx, &srAPI.CreateRequest{ + ID: n.sandboxID, + Types: []srAPI.Type{typ}, + DevShmSizeBytes: sizeBytes, + }) + if err != nil { + return "", fmt.Errorf("guest devshm create: %w", err) + } + for _, r := range resp.GetResources() { + if r.GetType() != typ { + continue + } + if n.paths == nil { + n.paths = make(map[srAPI.Type]string, 1) + } + n.paths[typ] = r.GetPath() + return r.GetPath(), nil + } + return "", fmt.Errorf("guest did not return a path for devshm resource") +} diff --git a/internal/shim/task/socketforward.go b/internal/shim/task/socketforward.go index 4cf77057..3ebf4e22 100644 --- a/internal/shim/task/socketforward.go +++ b/internal/shim/task/socketforward.go @@ -23,8 +23,13 @@ import ( "fmt" "io" "net" + "os" + "path/filepath" + "slices" "strings" + "github.com/containerd/containerd/api/types" + "github.com/containerd/continuity/fs" "github.com/containerd/log" "github.com/opencontainers/runtime-spec/specs-go" @@ -128,6 +133,81 @@ func parseUDSMount(containerID string, m specs.Mount) (socketForwardEntry, error }, nil } +// CreateRootfsPlaceholders creates empty regular files for each UDS mount +// destination inside sourceRootfs. The OCI runtime requires the bind mount +// destination to already exist as a file; since the container's rootfs is +// mounted read-only, the placeholders must be present in the source before +// the mount is applied. +// +// entry.containerPath is an OCI mount destination and is normally absolute +// (e.g. "/run/shared.sock"); it may also contain ".." components. Both +// fs.RootPath (rather than a plain filepath.Join, which would resolve +// ".." components and could walk right out of sourceRootfs) and symlinks +// already present inside sourceRootfs are resolved safely so the +// placeholder can never be created outside sourceRootfs. +// +// Errors are logged but not returned: a missing placeholder will cause the +// OCI runtime to fail at container creation, which is reported there. +func (p *socketForwardsProvider) CreateRootfsPlaceholders(ctx context.Context, sourceRootfs string) { + for _, entry := range p.entries { + destInRootfs, err := fs.RootPath(sourceRootfs, entry.containerPath) + if err != nil { + log.G(ctx).WithError(err).WithField("path", entry.containerPath). + Warn("socketforward: failed to resolve UDS mount placeholder path") + continue + } + if err := os.MkdirAll(filepath.Dir(destInRootfs), 0o755); err != nil { + log.G(ctx).WithError(err).WithField("path", destInRootfs). + Warn("socketforward: failed to create parent dirs for UDS mount placeholder") + continue + } + f, err := os.OpenFile(destInRootfs, os.O_WRONLY|os.O_CREATE|os.O_EXCL, 0o644) + if err != nil && !os.IsExist(err) { + log.G(ctx).WithError(err).WithField("path", destInRootfs). + Warn("socketforward: failed to create UDS mount placeholder") + continue + } + if err == nil { + f.Close() + } + } +} + +// udsPlaceholderSource returns the writable directory where UDS mount +// placeholder files must be created for a container whose rootfs is +// assembled from rootfsMounts, and whether that directory is available +// before SharedFS.ShareRootfs assembles the rootfs (beforeAssembly) or only +// after (i.e. assembledRootfs, the host path SharedFS.RootfsHostPath +// returns once ShareRootfs has run). +// +// mountutil.All mounts every entry in rootfsMounts, but only the *last* +// entry ends up at the final assembled path — every other entry (lower +// layers, ext4 scratch devices, etc.) is mounted elsewhere purely to feed +// that last mount (e.g. as overlay lowerdir/upperdir sources). So the only +// mount spec that can tell us anything about the assembled rootfs itself is +// the last one: +// +// - If it is a plain "bind" mount with the "ro" option, ShareRootfs will +// mount its Source read-only at the assembled path, so placeholders +// must be written into that still-writable Source *before* ShareRootfs +// runs — writing into the assembled path afterward would fail with +// EROFS. +// - Otherwise — an overlay/erofs assembly with a writable upperdir, a +// plain writable bind, or anything else mountutil.All supports — the +// assembled path itself stays writable, and is in fact the *only* +// correct target: for a multi-entry rootfs (the common overlay/erofs +// case) no single entry's Source is the final tree, only the assembled +// mountpoint is. +func udsPlaceholderSource(rootfsMounts []*types.Mount, assembledRootfs string) (path string, beforeAssembly bool) { + if len(rootfsMounts) > 0 { + last := rootfsMounts[len(rootfsMounts)-1] + if last.Type == "bind" && last.Source != "" && slices.Contains(last.Options, "ro") { + return last.Source, true + } + } + return assembledRootfs, false +} + // bindSockets calls the Bind RPC on the VM to set up socket forward // listener sockets. This must be called before container creation so that // crun can bind-mount the listener sockets into the container. @@ -161,6 +241,21 @@ func bindSockets(ctx context.Context, sb sandbox.Sandbox, entries []socketForwar // socketForwarder manages active UDS socket forwarding for a single container. // It is started after the container is created and runs for the container // lifetime. +// +// TODO: the guest's Accept RPC this drives delivers connection +// notifications from a single, process-wide (per-VM, not per-container) +// channel (see the notify field on internal/vminit/socketforward.Service). +// Since this type and startSocketForwarding below are per-container, two +// sibling containers in the same VM that both use socket forwarding each +// start their own concurrent Accept stream, and the guest's shared channel +// can deliver a notification meant for one container's forward to the +// other's stream — which then fails to find it in its own entries map (see +// handleConnection's "unknown forward ID" case) and drops the connection. +// This is pre-existing code, unchanged by multi-container-per-VM support, +// but that support is what first makes more than one Accept stream possible +// at once and so makes this reachable. Fixing it belongs on the guest side +// (see the TODO there); nothing here can compensate for a misdelivered +// notification once the guest has already sent it to the wrong stream. type socketForwarder struct { sb sandbox.Sandbox containerID string diff --git a/internal/shim/task/socketforward_test.go b/internal/shim/task/socketforward_test.go index 7e660903..fae4094f 100644 --- a/internal/shim/task/socketforward_test.go +++ b/internal/shim/task/socketforward_test.go @@ -18,8 +18,11 @@ package task import ( "context" + "os" + "path/filepath" "testing" + "github.com/containerd/containerd/api/types" "github.com/opencontainers/runtime-spec/specs-go" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -134,3 +137,184 @@ func TestSocketForwardsProviderFromBundle(t *testing.T) { }) } } + +// TestCreateRootfsPlaceholders_ConfinesToRootfs verifies that a UDS mount +// destination containing ".." components cannot escape sourceRootfs when +// the placeholder file is created — a regression test for a path-traversal +// issue where a plain filepath.Join(sourceRootfs, containerPath) would +// resolve ".." segments right out of sourceRootfs. +func TestCreateRootfsPlaceholders_ConfinesToRootfs(t *testing.T) { + ctx := context.Background() + root := t.TempDir() + sourceRootfs := filepath.Join(root, "rootfs") + require.NoError(t, os.MkdirAll(sourceRootfs, 0o755)) + + p := &socketForwardsProvider{ + entries: []socketForwardEntry{ + {containerPath: "/../../etc/escaped.sock"}, + {containerPath: "/run/normal.sock"}, + }, + } + + p.CreateRootfsPlaceholders(ctx, sourceRootfs) + + // Neither placeholder should have escaped sourceRootfs: walk the + // entire temp dir tree and confirm every created file is contained + // within sourceRootfs. + err := filepath.Walk(root, func(path string, info os.FileInfo, err error) error { + require.NoError(t, err) + if info.IsDir() || path == sourceRootfs { + return nil + } + rel, err := filepath.Rel(sourceRootfs, path) + require.NoError(t, err) + assert.False(t, len(rel) >= 2 && rel[:2] == "..", + "file %q escaped sourceRootfs %q", path, sourceRootfs) + return nil + }) + require.NoError(t, err) + + // The well-behaved mount's placeholder must still be created normally. + assert.FileExists(t, filepath.Join(sourceRootfs, "run", "normal.sock")) +} + +// TestUDSPlaceholderSource covers the mount-shape decision at the heart of +// the sandboxed UDS placeholder fix: only a rootfs whose *last* mount spec +// (the one mountutil.All actually mounts at the assembled path) is a +// read-only bind needs its placeholder written to that mount's Source +// before ShareRootfs runs. Every other shape — in particular a multi-entry +// overlay/erofs assembly, which is what a real snapshotter or this repo's +// erofs layer format actually hands Task.Create — has no single mount +// whose Source is the final assembled tree, so the assembled rootfs path +// itself is the only correct, and only available-after-ShareRootfs, target. +func TestUDSPlaceholderSource(t *testing.T) { + const assembled = "/state/containers/ctr-1/rootfs" + + testcases := []struct { + name string + mounts []*types.Mount + wantPath string + wantBefore bool + wantPathReason string + }{ + { + name: "no mounts", + mounts: nil, + wantPath: assembled, + wantBefore: false, + wantPathReason: "empty rootfs still assembles an (empty) directory at the guest path", + }, + { + name: "read-only bind (non-root shimtest / a committed snapshot)", + mounts: []*types.Mount{ + {Type: "bind", Source: "/tmp/extracted-rootfs", Options: []string{"ro", "rbind"}}, + }, + wantPath: "/tmp/extracted-rootfs", + wantBefore: true, + wantPathReason: "ShareRootfs will mount this Source read-only at the assembled path", + }, + { + name: "writable bind (no ro option)", + mounts: []*types.Mount{ + {Type: "bind", Source: "/tmp/writable-rootfs", Options: []string{"rbind"}}, + }, + wantPath: assembled, + wantBefore: false, + wantPathReason: "the bind stays writable, so using the assembled path (equivalent content) after ShareRootfs is correct and simpler", + }, + { + name: "bind marked ro but missing Source", + mounts: []*types.Mount{ + {Type: "bind", Source: "", Options: []string{"ro"}}, + }, + wantPath: assembled, + wantBefore: false, + wantPathReason: "an empty Source can't be written to before assembly; fall back to the assembled path", + }, + { + name: "overlay/erofs multi-layer assembly (the common CRI/erofs shape)", + mounts: []*types.Mount{ + {Type: "ext4", Source: "/state/scratch.ext4", Options: []string{"rw", "loop"}}, + {Type: "erofs", Source: "/layers/base.erofs", Options: []string{"ro", "loop"}}, + { + Type: "format/mkdir/overlay", + Source: "overlay", + Options: []string{ + "workdir={{ mount 0 }}/work", + "upperdir={{ mount 0 }}/upper", + "lowerdir={{ mount 1 }}", + }, + }, + }, + wantPath: assembled, + wantBefore: false, + wantPathReason: "no single mount's Source is the assembled tree; the overlay's writable upperdir backs the assembled path itself", + }, + { + name: "single overlay mount with explicit upperdir", + mounts: []*types.Mount{ + { + Type: "overlay", + Source: "overlay", + Options: []string{"lowerdir=/l1:/l2", "upperdir=/upper", "workdir=/work"}, + }, + }, + wantPath: assembled, + wantBefore: false, + wantPathReason: "an overlay mount is never type \"bind\", so it must resolve to the assembled path", + }, + } + + for _, tc := range testcases { + t.Run(tc.name, func(t *testing.T) { + gotPath, gotBefore := udsPlaceholderSource(tc.mounts, assembled) + assert.Equal(t, tc.wantPath, gotPath, tc.wantPathReason) + assert.Equal(t, tc.wantBefore, gotBefore) + }) + } +} + +// TestCreateRootfsPlaceholders_OverlayShapedRootfs is an end-to-end +// regression test for the bug identified in review: previously, placeholder +// creation only ever scanned r.Rootfs for a "bind"-typed entry, so an +// overlay/erofs-shaped rootfs (no "bind" entry at all — the shape used by +// the erofs snapshotter and any real CRI overlay snapshotter) produced zero +// placeholders, leaving a UDS mount's rewritten bind destination missing. +// +// This test drives the same two-call sequence service.go's +// createSandboxedContainer uses (udsPlaceholderSource to pick a target, +// then CreateRootfsPlaceholders) against an overlay-shaped mount list and a +// writable directory standing in for the host path SharedFS.ShareRootfs +// would have assembled, and asserts the placeholder lands there. +func TestCreateRootfsPlaceholders_OverlayShapedRootfs(t *testing.T) { + ctx := context.Background() + assembledRootfs := t.TempDir() + + overlayShapedMounts := []*types.Mount{ + {Type: "ext4", Source: "/state/scratch.ext4", Options: []string{"rw", "loop"}}, + {Type: "erofs", Source: "/layers/base.erofs", Options: []string{"ro", "loop"}}, + {Type: "format/mkdir/overlay", Source: "overlay", Options: []string{ + "workdir={{ mount 0 }}/work", + "upperdir={{ mount 0 }}/upper", + "lowerdir={{ mount 1 }}", + }}, + } + + p := &socketForwardsProvider{ + entries: []socketForwardEntry{ + {containerPath: "/run/shared.sock"}, + }, + } + + placeholderSrc, beforeAssembly := udsPlaceholderSource(overlayShapedMounts, assembledRootfs) + require.False(t, beforeAssembly, "an overlay-shaped rootfs has no writable Source available before assembly") + require.Equal(t, assembledRootfs, placeholderSrc) + + // Mirror service.go: this call only happens after ShareRootfs would + // have assembled the rootfs (here, simply because assembledRootfs + // already exists and is writable). + p.CreateRootfsPlaceholders(ctx, placeholderSrc) + + assert.FileExists(t, filepath.Join(assembledRootfs, "run", "shared.sock"), + "UDS placeholder must be created in the assembled rootfs when no mount entry has a usable pre-assembly Source") +} diff --git a/internal/vm/libkrun/instance.go b/internal/vm/libkrun/instance.go index 158b48f9..01bf72c5 100644 --- a/internal/vm/libkrun/instance.go +++ b/internal/vm/libkrun/instance.go @@ -138,19 +138,22 @@ func (*vmManager) NewInstance(ctx context.Context, state string) (vm.Instance, e ret = lib.InitLog(os.Stderr.Fd(), uint32(warnLevel), 0, 0) }) if ret != 0 { + _ = dlClose(handler) return nil, fmt.Errorf("krun_init_log failed: %d", ret) } vmc, err := newvmcontext(lib) if err != nil { + _ = dlClose(handler) return nil, err } // Add the erofs rootfs as the first virtio-blk device so that it is - // always exposed as /dev/vda inside the guest. Container image disks - // are added later via AddDisk, which appends to the device list, so - // they receive /dev/vdb, /dev/vdc, … in order of addition. + // always exposed as /dev/vda inside the guest. Container-supplied + // disks are added later via AddDisk, which appends to the device + // list, so they receive /dev/vdb, /dev/vdc, … in order of addition. if err := vmc.AddDisk2("vmrootfs", rootfsPath, 0, true); err != nil { + _ = dlClose(handler) return nil, fmt.Errorf("failed to add VM rootfs disk %q: %w", rootfsPath, err) } @@ -177,10 +180,38 @@ type vmInstance struct { lib *libkrun handler uintptr + // netnsSet/netns record the pod network namespace requested by the + // first call to Start (successful or not), so that a subsequent Start + // attempt (e.g. a retry after a failed one) can be validated against + // it: a repeated request for the same (or no) namespace is a no-op, + // but a request for a different namespace is rejected outright rather + // than silently ignored, since that would hide a real caller bug. + netnsSet bool + netns string + client *ttrpc.Client conn net.Conn // underlying TTRPC connection; closed in Shutdown } +// resolveNetNS validates a Start-requested network namespace against the +// namespace recorded by an earlier Start attempt on this instance, if any +// (for example, a retry after a Start call that failed before reaching the +// network-namespace switch). The same namespace, or none at all, is a +// no-op; a genuinely different, non-empty namespace after one was already +// recorded is rejected rather than silently overriding the first request, +// since that would hide a caller bug. The caller must hold v.mu. +func (v *vmInstance) resolveNetNS(requested string) error { + if v.netnsSet { + if requested != "" && requested != v.netns { + return fmt.Errorf("cannot change VM netns after it was already set to %q: got %q", v.netns, requested) + } + return nil + } + v.netns = requested + v.netnsSet = true + return nil +} + func (v *vmInstance) AddFS(ctx context.Context, tag, mountPath string, opts ...vm.MountOpt) error { v.mu.Lock() defer v.mu.Unlock() @@ -279,6 +310,10 @@ func (v *vmInstance) Start(ctx context.Context, opts ...vm.StartOpt) (err error) o(&startOpts) } + if err := v.resolveNetNS(startOpts.NetNS); err != nil { + return err + } + if err := v.vmc.SetExec("/sbin/vminitd", startOpts.InitArgs, env); err != nil { return fmt.Errorf("failed to set exec: %w", err) } @@ -333,10 +368,44 @@ func (v *vmInstance) Start(ctx context.Context, opts ...vm.StartOpt) (err error) preVMStart := time.Now() - // Start it + // Start it. + // + // runtime.LockOSThread pins this goroutine to one OS thread for the + // VM's entire lifetime (krun_start_enter blocks until the VM shuts + // down). This is necessary for two reasons: + // 1. setns(2) affects only the calling OS thread; without + // LockOSThread the goroutine could migrate to a different + // thread and the setns would be lost before krun_start_enter is + // reached. + // 2. libkrun's worker threads (vCPU, virtio backends, vsock/TSI + // workers), which krun_start_enter spawns as descendants of the + // calling thread, inherit the netns of that thread. They must be + // created in the pod netns so that VM traffic (including TSI + // proxy sockets) lands there. + // + // We deliberately do NOT call runtime.UnlockOSThread. When a + // goroutine that holds a thread lock exits, the Go runtime retires + // the underlying OS thread (Go 1.10+), so there is no thread-pool + // "poisoning" concern, and the pod-netns thread is never returned to + // the pool where it could pollute the default netns. errC := make(chan error, 1) go func() { defer close(errC) + runtime.LockOSThread() + if v.netns != "" { + if err := vmcontextSetNetns(v.netns); err != nil { + errC <- fmt.Errorf("entering pod netns: %w", err) + return + } + // Log the resulting thread netns inode so it can be + // cross-checked against the pod netns inode when debugging + // connectivity issues. + inode, _ := os.Readlink("/proc/thread-self/ns/net") + log.G(ctx).WithFields(log.Fields{ + "netns_path": v.netns, + "netns_inode": inode, + }).Debug("VM start thread entered pod netns") + } if err := v.vmc.Start(); err != nil { errC <- err } @@ -469,8 +538,8 @@ func (v *vmInstance) Shutdown(ctx context.Context) error { } } - // On Unix, dlClose unloads the library after krun_free_ctx has joined all - // VM threads. On Windows it is a no-op (see dlfcn_windows.go). + // On Unix, dlClose unloads the library after krun_free_ctx has joined + // all VM threads. On Windows it is a no-op (see dlfcn_windows.go). if err := dlClose(v.handler); err != nil { return err } diff --git a/internal/vm/libkrun/instance_test.go b/internal/vm/libkrun/instance_test.go new file mode 100644 index 00000000..a85c04f8 --- /dev/null +++ b/internal/vm/libkrun/instance_test.go @@ -0,0 +1,90 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package libkrun + +import "testing" + +// TestResolveNetNS_FirstCallRecords verifies that the first call records +// whatever namespace (including empty, i.e. host-network) was requested. +func TestResolveNetNS_FirstCallRecords(t *testing.T) { + v := &vmInstance{} + if err := v.resolveNetNS("/run/netns/foo"); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !v.netnsSet || v.netns != "/run/netns/foo" { + t.Fatalf("netns not recorded: netnsSet=%v netns=%q", v.netnsSet, v.netns) + } +} + +// TestResolveNetNS_SameIsNoop verifies that repeating the same namespace +// after it was already recorded succeeds without changing anything. +func TestResolveNetNS_SameIsNoop(t *testing.T) { + v := &vmInstance{} + if err := v.resolveNetNS("/run/netns/foo"); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if err := v.resolveNetNS("/run/netns/foo"); err != nil { + t.Fatalf("repeating the same netns should be a no-op, got error: %v", err) + } + if v.netns != "/run/netns/foo" { + t.Fatalf("netns changed unexpectedly: %q", v.netns) + } +} + +// TestResolveNetNS_EmptyIsNoop verifies that an empty (host-network) request +// after a real namespace was already recorded is ignored rather than +// clearing the recorded namespace. +func TestResolveNetNS_EmptyIsNoop(t *testing.T) { + v := &vmInstance{} + if err := v.resolveNetNS("/run/netns/foo"); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if err := v.resolveNetNS(""); err != nil { + t.Fatalf("an empty netns request should be a no-op, got error: %v", err) + } + if v.netns != "/run/netns/foo" { + t.Fatalf("netns cleared unexpectedly: %q", v.netns) + } +} + +// TestResolveNetNS_ConflictErrors verifies that a genuinely different, +// non-empty namespace after one was already recorded is rejected. +func TestResolveNetNS_ConflictErrors(t *testing.T) { + v := &vmInstance{} + if err := v.resolveNetNS("/run/netns/foo"); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if err := v.resolveNetNS("/run/netns/bar"); err == nil { + t.Fatalf("expected error when requesting a different netns") + } + if v.netns != "/run/netns/foo" { + t.Fatalf("netns changed despite conflict: %q", v.netns) + } +} + +// TestResolveNetNS_EmptyFirstThenNonEmptyErrors verifies that a namespace +// requested after host-network was already recorded (the empty string) is +// treated as a genuine conflict, not a no-op. +func TestResolveNetNS_EmptyFirstThenNonEmptyErrors(t *testing.T) { + v := &vmInstance{} + if err := v.resolveNetNS(""); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if err := v.resolveNetNS("/run/netns/foo"); err == nil { + t.Fatalf("expected error when requesting a netns after host-network was recorded") + } +} diff --git a/internal/vm/libkrun/krun.go b/internal/vm/libkrun/krun.go index ef68e7d2..af07c3a2 100644 --- a/internal/vm/libkrun/krun.go +++ b/internal/vm/libkrun/krun.go @@ -57,16 +57,11 @@ type vmcontext struct { } func newvmcontext(lib *libkrun) (*vmcontext, error) { - // Start VM context - ctxId := lib.CreateCtx() - if ctxId < 0 { - return nil, fmt.Errorf("krun_create_ctx failed: %d", ctxId) + ctxID := lib.CreateCtx() + if ctxID < 0 { + return nil, fmt.Errorf("krun_create_ctx failed: %d", ctxID) } - - return &vmcontext{ - ctxID: uint32(ctxId), - lib: lib, - }, nil + return &vmcontext{lib: lib, ctxID: uint32(ctxID)}, nil } func (vmc *vmcontext) SetCPUAndMemory(cpu uint8, ram uint32) error { @@ -187,7 +182,6 @@ func (vmc *vmcontext) AddNIC(endpoint string, mac net.HardwareAddr, mode vm.Netw if vmc.lib.AddNetUnixgram == nil || vmc.lib.AddNetUnixstream == nil { return fmt.Errorf("libkrun not loaded") } - switch mode { case vm.NetworkModeUnixgram: ret := vmc.lib.AddNetUnixgram(vmc.ctxID, endpoint, -1, []uint8(mac), features, flags) @@ -202,10 +196,16 @@ func (vmc *vmcontext) AddNIC(endpoint string, mac net.HardwareAddr, mode vm.Netw default: return fmt.Errorf("invalid network mode: %d", mode) } - return nil } +// Start runs krun_start_enter on the calling goroutine. The caller is +// responsible for locking this goroutine to its OS thread (and, if a pod +// network namespace is required, entering it via setns) before calling +// Start — see vmInstance.Start in instance.go. krun_start_enter blocks for +// the entire VM lifetime, spawning all of the VM's worker threads (vCPU, +// virtio backends, vsock/TSI workers) as descendants of the calling thread; +// they inherit whatever network namespace that thread is in at the time. func (vmc *vmcontext) Start() error { if vmc.lib.StartEnter == nil { return fmt.Errorf("libkrun not loaded") @@ -217,6 +217,10 @@ func (vmc *vmcontext) Start() error { return nil } +// Shutdown calls krun_free_ctx. krun_free_ctx joins the VM's internal threads +// (vCPU, virtio workers) and can be called from any goroutine once +// krun_start_enter has returned — libkrun itself is thread-safe for this +// cross-thread teardown. func (vmc *vmcontext) Shutdown() error { if vmc.ctxID == 0 { return nil diff --git a/internal/vm/libkrun/krun_linux.go b/internal/vm/libkrun/krun_linux.go new file mode 100644 index 00000000..da435695 --- /dev/null +++ b/internal/vm/libkrun/krun_linux.go @@ -0,0 +1,83 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package libkrun + +import ( + "errors" + "fmt" + "os" + + "github.com/containerd/log" + "golang.org/x/sys/unix" +) + +// nsfsMagic is the filesystem magic number for Linux nsfs (the filesystem that +// backs namespace files under /proc/*/ns/). +const nsfsMagic = 0x6e736673 + +// vmcontextSetNetns enters the network namespace at path on the calling OS +// thread using setns(2). It must be called from the locked OS thread that +// is about to call krun_start_enter (see vmInstance.Start in instance.go) +// so that all worker threads libkrun spawns from that thread (vCPU, virtio +// backends, vsock/TSI workers) inherit the namespace. +// +// The file descriptor is opened O_RDONLY|O_CLOEXEC, used for setns, and then +// closed — the netns is pinned by the bind-mount at path (managed by the CRI +// layer), not by this FD. +// +// The function checks that path refers to a real network namespace file (nsfs +// magic). If not (e.g. a plain file used in tests), it returns nil without +// attempting setns. +// +// If setns fails with EPERM (the shim lacks CAP_SYS_ADMIN in the initial user +// namespace), the error is logged at warning level and the function returns +// nil. VM traffic will then originate from the shim's own network namespace +// rather than the pod netns, matching the previous behaviour. In production, +// containerd runs the shim as root, so setns succeeds. +func vmcontextSetNetns(path string) error { + f, err := os.OpenFile(path, os.O_RDONLY|unix.O_CLOEXEC, 0) + if err != nil { + return fmt.Errorf("open netns %q: %w", path, err) + } + defer f.Close() + + // Check whether path is a real nsfs file. A plain file (e.g. one + // created for testing) has a different filesystem magic and cannot be + // used with setns; skip silently in that case. + var sfs unix.Statfs_t + if err := unix.Fstatfs(int(f.Fd()), &sfs); err != nil { + return fmt.Errorf("statfs netns %q: %w", path, err) + } + if sfs.Type != nsfsMagic { + log.L.WithField("netns", path).Debug( + "netns path is not an nsfs file; skipping setns (test or non-standard path)") + return nil + } + + if err := unix.Setns(int(f.Fd()), unix.CLONE_NEWNET); err != nil { + if errors.Is(err, unix.EPERM) { + // Log and continue: shim lacks CAP_SYS_ADMIN; VM traffic will + // use the shim's own netns instead of the pod netns. + log.L.WithField("netns", path).Warn( + "setns into pod netns not permitted (shim not running as root); " + + "VM traffic will use shim netns") + return nil + } + return fmt.Errorf("setns %q: %w", path, err) + } + return nil +} diff --git a/internal/vm/libkrun/krun_other.go b/internal/vm/libkrun/krun_other.go new file mode 100644 index 00000000..ca01f332 --- /dev/null +++ b/internal/vm/libkrun/krun_other.go @@ -0,0 +1,23 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package libkrun + +// vmcontextSetNetns is a no-op on non-Linux platforms: network namespaces +// are a Linux-only concept. +func vmcontextSetNetns(_ string) error { return nil } diff --git a/internal/vminit/process/init.go b/internal/vminit/process/init.go index 4b96a7a2..fecc8ac8 100644 --- a/internal/vminit/process/init.go +++ b/internal/vminit/process/init.go @@ -297,6 +297,16 @@ func (p *Init) delete(ctx context.Context) error { err = fmt.Errorf("failed rootfs umount: %w", err2) } } + // Remove the bundle directory from the guest's /run/bundles/ tree. + // Once crun has deleted its container state and the rootfs mount has + // been unmounted, the bundle directory is no longer needed. Keeping + // it would cause /run (a size-limited tmpfs) to fill up over many + // container lifecycles. + if p.Bundle != "" { + if err2 := os.RemoveAll(p.Bundle); err2 != nil { + log.G(ctx).WithError(err2).WithField("bundle", p.Bundle).Warn("failed to remove guest bundle dir") + } + } return err } diff --git a/internal/vminit/runc/util.go b/internal/vminit/runc/util.go index 7a3d88df..f23c6da9 100644 --- a/internal/vminit/runc/util.go +++ b/internal/vminit/runc/util.go @@ -28,8 +28,23 @@ import ( "github.com/opencontainers/runtime-spec/specs-go" ) -// ShouldKillAllOnExit reads the bundle's OCI spec and returns true if -// there is an error reading the spec or if the container has a private PID namespace +// ShouldKillAllOnExit reads the bundle's OCI spec and reports whether the +// container's other processes need to be killed explicitly when its init +// process exits. +// +// It returns false only when the spec has a PID namespace entry with an +// empty Path — the container's own, private PID namespace (the default, +// and PID namespace sharing's opposite: see internal/shim/task/namespaces.go +// on the host side). In that case nothing else is needed: the kernel tears +// the namespace down and kills every process still in it the moment PID 1 +// exits, so by the time this is even checked, everything else already is +// gone. +// +// It returns true otherwise — including a shared PID namespace (a +// non-empty Path), no PID namespace entry in the spec at all (the host's +// PID namespace, which the kernel will not tear down on this container's +// account), and a failure to read the spec, which is treated as the safe +// default rather than silently skipping cleanup. func ShouldKillAllOnExit(ctx context.Context, bundlePath string) bool { spec, err := readSpec(bundlePath) if err != nil { diff --git a/internal/vminit/runc/util_test.go b/internal/vminit/runc/util_test.go new file mode 100644 index 00000000..f99fca96 --- /dev/null +++ b/internal/vminit/runc/util_test.go @@ -0,0 +1,92 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package runc + +import ( + "context" + "encoding/json" + "os" + "path/filepath" + "testing" + + specs "github.com/opencontainers/runtime-spec/specs-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func writeSpec(t *testing.T, spec *specs.Spec) string { + t.Helper() + dir := t.TempDir() + f, err := os.Create(filepath.Join(dir, "config.json")) + require.NoError(t, err) + defer f.Close() + require.NoError(t, json.NewEncoder(f).Encode(spec)) + return dir +} + +func TestShouldKillAllOnExit(t *testing.T) { + ctx := context.Background() + + testcases := []struct { + name string + spec *specs.Spec + want bool + }{ + { + name: "private PID namespace (empty Path): kernel already reaped everything", + spec: &specs.Spec{Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.PIDNamespace, Path: ""}, + }}}, + want: false, + }, + { + name: "shared PID namespace (non-empty Path): must kill explicitly", + spec: &specs.Spec{Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.PIDNamespace, Path: "/run/pidns/some-sandbox"}, + }}}, + want: true, + }, + { + name: "no PID namespace entry at all (host PID): must kill explicitly", + spec: &specs.Spec{Linux: &specs.Linux{Namespaces: []specs.LinuxNamespace{ + {Type: specs.NetworkNamespace}, + }}}, + want: true, + }, + { + name: "nil Linux: must kill explicitly", + spec: &specs.Spec{}, + want: true, + }, + } + + for _, tc := range testcases { + t.Run(tc.name, func(t *testing.T) { + dir := writeSpec(t, tc.spec) + assert.Equal(t, tc.want, ShouldKillAllOnExit(ctx, dir)) + }) + } +} + +// TestShouldKillAllOnExit_MissingSpec verifies the fail-safe default: a +// bundle whose config.json can't be read is treated the same as "must kill +// explicitly", not silently skipped. +func TestShouldKillAllOnExit_MissingSpec(t *testing.T) { + assert.True(t, ShouldKillAllOnExit(context.Background(), t.TempDir())) +} diff --git a/internal/vminit/sharedresources/manager_root_linux_test.go b/internal/vminit/sharedresources/manager_root_linux_test.go new file mode 100644 index 00000000..f87ae860 --- /dev/null +++ b/internal/vminit/sharedresources/manager_root_linux_test.go @@ -0,0 +1,423 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sharedresources + +import ( + "context" + "crypto/rand" + "encoding/hex" + "errors" + "fmt" + "os" + "path/filepath" + "runtime" + "testing" + "time" + + "golang.org/x/sys/unix" +) + +// requireNamespacePrivileges skips a test unless it can actually create and +// bind-mount namespaces. Creating them needs CAP_SYS_ADMIN, and the paths are +// absolute (/run/...), so these only run as real root. +func requireNamespacePrivileges(t *testing.T) { + t.Helper() + if os.Getuid() != 0 { + t.Skip("requires root to unshare and bind-mount namespaces") + } +} + +// useTestAnchor points the PID namespace anchor at a host binary that blocks +// forever, standing in for the real anchor (which only ships inside the guest +// rootfs). Any process that does not exit will do: what matters to the +// namespace is that its PID 1 stays alive. +func useTestAnchor(t *testing.T) { + t.Helper() + const sleep = "/bin/sleep" + if _, err := os.Stat(sleep); err != nil { + t.Skipf("no stand-in anchor binary available: %v", err) + } + saved := anchorCommand + anchorCommand = []string{sleep, "infinity"} + t.Cleanup(func() { anchorCommand = saved }) +} + +// isMountPoint reports whether path is a mount point, by comparing its device +// with that of the directory containing it. A pinned namespace is an nsfs +// mount, so once mounted its device always differs from the tmpfs directory it +// sits in. +func isMountPoint(t *testing.T, path string) bool { + t.Helper() + var st unix.Stat_t + if err := unix.Lstat(path, &st); err != nil { + return false + } + var dir unix.Stat_t + if err := unix.Lstat(filepath.Dir(path), &dir); err != nil { + t.Fatalf("lstat %s: %v", filepath.Dir(path), err) + } + return st.Dev != dir.Dev +} + +// TestManagerCreateDeleteRoundTrip exercises real namespace creation and +// deletion for every supported type: each must end up bind-mounted at the +// returned path, be returned again unchanged on a repeat request, and be fully +// gone after Delete. +// +// The PID namespace's anchor is substituted for a host stand-in, since the +// real one ships only in the guest rootfs; the anchor's liveness is asserted +// too, because that is what keeps the namespace usable. +func TestManagerCreateDeleteRoundTrip(t *testing.T) { + requireNamespacePrivileges(t) + useTestAnchor(t) + + ctx := context.Background() + id := "nerdbox-test-" + randomSuffix(t) + types := []Type{TypeNamespaceNetwork, TypeNamespaceIPC, TypeNamespaceUTS, TypeNamespacePID, TypeDevShm} + const devShmSize = 4 * 1024 * 1024 + + var m Manager + t.Cleanup(func() { + // Best effort, in case an assertion below fails before Delete runs. + _ = m.Delete(ctx, id, nil) + }) + + paths, err := m.Create(ctx, id, types, devShmSize) + if err != nil { + t.Fatalf("Create: %v", err) + } + if len(paths) != len(types) { + t.Fatalf("Create returned %d paths, want %d", len(paths), len(types)) + } + for _, typ := range types { + path, ok := paths[typ] + if !ok { + t.Fatalf("Create returned no path for %s", typ) + } + if !isMountPoint(t, path) { + t.Errorf("%s namespace at %s is not a mount point", typ, path) + } + } + + // A repeat request must reuse what already exists rather than creating + // anything new, and must report the same paths. A different size is + // passed here specifically to confirm it is ignored for an + // already-created devshm resource, per Create's doc comment. + again, err := m.Create(ctx, id, types, devShmSize*2) + if err != nil { + t.Fatalf("second Create: %v", err) + } + for _, typ := range types { + if again[typ] != paths[typ] { + t.Errorf("%s path changed across calls: %q then %q", typ, paths[typ], again[typ]) + } + } + + // Requesting a subset must not disturb the rest. + subset, err := m.Create(ctx, id, []Type{TypeNamespaceIPC}, 0) + if err != nil { + t.Fatalf("subset Create: %v", err) + } + if len(subset) != 1 || subset[TypeNamespaceIPC] != paths[TypeNamespaceIPC] { + t.Errorf("subset Create = %v, want just the IPC path %q", subset, paths[TypeNamespaceIPC]) + } + + // The PID namespace is only usable for as long as its anchor lives, so the + // anchor must still be running at this point. + anchor := anchorPID(t, &m, id) + if !processAlive(anchor) { + t.Errorf("PID namespace anchor (pid %d) is not running", anchor) + } + + if err := m.Delete(ctx, id, nil); err != nil { + t.Fatalf("Delete: %v", err) + } + for _, typ := range types { + path := paths[typ] + if isMountPoint(t, path) { + t.Errorf("%s namespace at %s is still mounted after Delete", typ, path) + } + if _, err := os.Lstat(path); !os.IsNotExist(err) { + t.Errorf("%s bind-mount target %s still exists after Delete (err=%v)", typ, path, err) + } + } + + // Deleting the PID namespace means killing its anchor, otherwise the + // process would outlive the sandbox. + deadline := time.Now().Add(5 * time.Second) + for processAlive(anchor) && time.Now().Before(deadline) { + time.Sleep(10 * time.Millisecond) + } + if processAlive(anchor) { + t.Errorf("PID namespace anchor (pid %d) still running after Delete", anchor) + } + + // Delete must be idempotent. + if err := m.Delete(ctx, id, nil); err != nil { + t.Errorf("second Delete: %v", err) + } +} + +// anchorPID returns the pid of the anchor process holding the group's PID +// namespace open. +func anchorPID(t *testing.T, m *Manager, id string) int { + t.Helper() + m.mu.Lock() + defer m.mu.Unlock() + e, ok := m.ns[key{id: id, typ: TypeNamespacePID}] + if !ok || e.anchor == nil { + t.Fatalf("no PID namespace anchor recorded for %q", id) + } + return e.anchor.Pid +} + +// processAlive reports whether pid is still running. The anchor is a child of +// this process, so once it has been killed and reaped the signal fails. +func processAlive(pid int) bool { + return unix.Kill(pid, 0) == nil +} + +// TestManagerCreateOnlyRequestedTypes verifies that asking for one type does +// not create the others. This is the guard for the PID namespace in +// particular, whose creation costs a persistent anchor process. +func TestManagerCreateOnlyRequestedTypes(t *testing.T) { + requireNamespacePrivileges(t) + + ctx := context.Background() + id := "nerdbox-test-" + randomSuffix(t) + + var m Manager + t.Cleanup(func() { _ = m.Delete(ctx, id, nil) }) + + if _, err := m.Create(ctx, id, []Type{TypeNamespaceIPC}, 0); err != nil { + t.Fatalf("Create: %v", err) + } + + for _, typ := range []Type{TypeNamespaceNetwork, TypeNamespaceUTS, TypeNamespacePID, TypeDevShm} { + dir, err := typ.dir() + if err != nil { + t.Fatal(err) + } + path := dir + "/" + id + if _, err := os.Lstat(path); !os.IsNotExist(err) { + t.Errorf("%s namespace at %s was created without being requested (err=%v)", typ, path, err) + } + } +} + +// TestManagerDevShmSizeEnforced verifies that the devshm resource is a real, +// size-limited tmpfs — not just a directory — by writing past the requested +// size and confirming the kernel itself rejects it with ENOSPC. +func TestManagerDevShmSizeEnforced(t *testing.T) { + requireNamespacePrivileges(t) + + ctx := context.Background() + id := "nerdbox-test-" + randomSuffix(t) + const size = 1 * 1024 * 1024 // 1MiB + + var m Manager + t.Cleanup(func() { _ = m.Delete(ctx, id, nil) }) + + paths, err := m.Create(ctx, id, []Type{TypeDevShm}, size) + if err != nil { + t.Fatalf("Create: %v", err) + } + path := paths[TypeDevShm] + + var st unix.Statfs_t + if err := unix.Statfs(path, &st); err != nil { + t.Fatalf("statfs %s: %v", path, err) + } + gotSize := int64(st.Blocks) * st.Bsize //nolint:unconvert // Bsize is int64 on some arches, int32 on others + if gotSize != size { + t.Errorf("tmpfs total size = %d bytes, want %d", gotSize, size) + } + + f, err := os.Create(filepath.Join(path, "toobig")) + if err != nil { + t.Fatalf("create file in devshm: %v", err) + } + defer f.Close() + + // Writing past the tmpfs's size must fail with ENOSPC. A plain + // directory (no size limit at all) would happily accept this. + buf := make([]byte, size*2) + _, err = f.Write(buf) + if !errors.Is(err, unix.ENOSPC) { + t.Errorf("write past tmpfs size = %v, want ENOSPC", err) + } +} + +// TestUTSNamespaceIsSharedAndHostnameIsLastWriterWins verifies the property +// the host relies on instead of any explicit coordination (see createUTS's +// doc comment and internal/shim/task/namespaces.go): joining the same +// TypeNamespaceUTS resource from two independent threads gives them a +// genuinely shared UTS namespace, and a hostname set from one is visible +// from the other, exactly as it would be for two sibling containers' +// crun-driven setns+sethostname sequences. +func TestUTSNamespaceIsSharedAndHostnameIsLastWriterWins(t *testing.T) { + requireNamespacePrivileges(t) + + ctx := context.Background() + id := "nerdbox-test-" + randomSuffix(t) + + var m Manager + t.Cleanup(func() { _ = m.Delete(ctx, id, nil) }) + + paths, err := m.Create(ctx, id, []Type{TypeNamespaceUTS}, 0) + if err != nil { + t.Fatalf("Create: %v", err) + } + path := paths[TypeNamespaceUTS] + + setHostnameInNamespace(t, path, "from-first-joiner") + if got := hostnameInNamespace(t, path); got != "from-first-joiner" { + t.Fatalf("hostname after first join = %q, want %q", got, "from-first-joiner") + } + + // A second, independent join must see the same namespace: setting the + // hostname again must overwrite what the first joiner set, and that + // change must in turn be visible through a third join. This is + // "joining", not "creating a similar but separate namespace" — if + // createUTS accidentally created a fresh namespace per bind mount + // reference this would not hold. + setHostnameInNamespace(t, path, "from-second-joiner") + if got := hostnameInNamespace(t, path); got != "from-second-joiner" { + t.Fatalf("hostname after second join = %q, want %q (namespace was not truly shared)", got, "from-second-joiner") + } +} + +// setHostnameInNamespace joins the UTS namespace pinned at path on a +// dedicated, locked OS thread and sets its hostname, mirroring what an OCI +// runtime does when a container's spec has a non-empty Hostname and a uts +// namespace entry with a non-empty Path. +func setHostnameInNamespace(t *testing.T, path, hostname string) { + t.Helper() + errCh := make(chan error, 1) + go func() { + runtime.LockOSThread() + defer runtime.UnlockOSThread() + + fd, err := unix.Open(path, unix.O_RDONLY, 0) + if err != nil { + errCh <- fmt.Errorf("open %s: %w", path, err) + return + } + defer unix.Close(fd) + + if err := unix.Setns(fd, unix.CLONE_NEWUTS); err != nil { + errCh <- fmt.Errorf("setns: %w", err) + return + } + errCh <- unix.Sethostname([]byte(hostname)) + }() + if err := <-errCh; err != nil { + t.Fatalf("setHostnameInNamespace(%q): %v", hostname, err) + } +} + +// hostnameInNamespace joins the UTS namespace pinned at path on a dedicated, +// locked OS thread and reads back its hostname. +func hostnameInNamespace(t *testing.T, path string) string { + t.Helper() + type result struct { + hostname string + err error + } + ch := make(chan result, 1) + go func() { + runtime.LockOSThread() + defer runtime.UnlockOSThread() + + fd, err := unix.Open(path, unix.O_RDONLY, 0) + if err != nil { + ch <- result{err: fmt.Errorf("open %s: %w", path, err)} + return + } + defer unix.Close(fd) + + if err := unix.Setns(fd, unix.CLONE_NEWUTS); err != nil { + ch <- result{err: fmt.Errorf("setns: %w", err)} + return + } + var uts unix.Utsname + if err := unix.Uname(&uts); err != nil { + ch <- result{err: fmt.Errorf("uname: %w", err)} + return + } + ch <- result{hostname: unix.ByteSliceToString(uts.Nodename[:])} + }() + r := <-ch + if r.err != nil { + t.Fatalf("hostnameInNamespace: %v", r.err) + } + return r.hostname +} + +// TestManagerDevShmSizeMismatchKeepsFirstSize verifies that a later Create +// call requesting a different devshm size than the one already in use does +// not resize the tmpfs (Create's doc comment is explicit that the first +// caller's size wins) and does not fail the call — it can only ever warn, +// never error, since this is a best-effort diagnostic, not a correctness +// requirement. +func TestManagerDevShmSizeMismatchKeepsFirstSize(t *testing.T) { + requireNamespacePrivileges(t) + + ctx := context.Background() + id := "nerdbox-test-" + randomSuffix(t) + const firstSize = 1 * 1024 * 1024 + const secondSize = 2 * 1024 * 1024 + + var m Manager + t.Cleanup(func() { _ = m.Delete(ctx, id, nil) }) + + paths, err := m.Create(ctx, id, []Type{TypeDevShm}, firstSize) + if err != nil { + t.Fatalf("first Create: %v", err) + } + path := paths[TypeDevShm] + + again, err := m.Create(ctx, id, []Type{TypeDevShm}, secondSize) + if err != nil { + t.Fatalf("second Create (different size): %v", err) + } + if again[TypeDevShm] != path { + t.Fatalf("path changed across calls: %q then %q", path, again[TypeDevShm]) + } + + var st unix.Statfs_t + if err := unix.Statfs(path, &st); err != nil { + t.Fatalf("statfs %s: %v", path, err) + } + gotSize := int64(st.Blocks) * st.Bsize //nolint:unconvert // Bsize is int64 on some arches, int32 on others + if gotSize != firstSize { + t.Errorf("tmpfs size after mismatched second Create = %d, want unchanged %d (first caller's size)", gotSize, firstSize) + } +} + +// randomSuffix keeps concurrent or repeated runs from colliding on the +// well-known, absolute paths namespaces are pinned at. +func randomSuffix(t *testing.T) string { + t.Helper() + var b [8]byte + if _, err := rand.Read(b[:]); err != nil { + t.Fatalf("read random bytes: %v", err) + } + return hex.EncodeToString(b[:]) +} diff --git a/internal/vminit/sharedresources/sharedresources_linux.go b/internal/vminit/sharedresources/sharedresources_linux.go new file mode 100644 index 00000000..f6dc3044 --- /dev/null +++ b/internal/vminit/sharedresources/sharedresources_linux.go @@ -0,0 +1,604 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +// Package sharedresources creates and deletes guest-side resources that +// containers sharing a sandbox use to share state with each other — Linux +// namespaces (IPC, PID, network) and other guest-side resources set up the +// same way (currently just a shared /dev/shm tmpfs) — and reports the guest +// paths at which they are pinned. It implements the SharedResources service +// declared in api/proto/nerdbox/services/sharedresources/v1; see that file +// for the contract and for why creation is per type rather than +// all-or-nothing. +package sharedresources + +import ( + "context" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "runtime" + "strings" + "sync" + "syscall" + + "github.com/containerd/log" + "github.com/vishvananda/netlink" + "github.com/vishvananda/netns" + "golang.org/x/sys/unix" +) + +// anchorBinary is the anchor executable in the guest rootfs. It exists only +// to be PID 1 of a namespace: see createPID for why that needs a process, and +// crates/pause for the implementation. +const anchorBinary = "/sbin/nerdbox-pause" + +// anchorCommand is the command createPID runs to anchor a PID namespace. It +// is a variable purely so tests can substitute a binary that exists on the +// host, since the real one only ships in the guest rootfs. +var anchorCommand = []string{anchorBinary} + +// Type identifies a kind of guest-side shared resource this package can +// manage. It mirrors the Type enum in the SharedResources API, kept as a +// separate domain type so this package does not depend on the generated +// protobuf bindings. Most values are Linux namespace types; TypeDevShm is +// not — see its own comment. +type Type int + +const ( + // TypeNamespaceIPC is an IPC namespace. + TypeNamespaceIPC Type = iota + 1 + // TypeNamespacePID is a PID namespace. + TypeNamespacePID + // TypeNamespaceNetwork is a network namespace. + TypeNamespaceNetwork + // TypeDevShm is not an OCI namespace: it is a per-group tmpfs that + // sharing containers bind-mount their /dev/shm onto. See the package + // doc comment for why it lives here. + TypeDevShm + // TypeNamespaceUTS is a UTS namespace. See createUTS for how a shared + // hostname falls out of this without any extra coordination here. + TypeNamespaceUTS +) + +// String implements fmt.Stringer. +func (t Type) String() string { + switch t { + case TypeNamespaceIPC: + return "ipc" + case TypeNamespacePID: + return "pid" + case TypeNamespaceNetwork: + return "network" + case TypeDevShm: + return "devshm" + case TypeNamespaceUTS: + return "uts" + default: + return fmt.Sprintf("unknown(%d)", int(t)) + } +} + +// dir returns the directory resources of this type are pinned in. The +// layout matches the convention used by iproute2 and containerd's CRI +// plugin for named network namespaces (/run/netns/), extended to the +// other types. +func (t Type) dir() (string, error) { + switch t { + case TypeNamespaceIPC: + return "/run/ipcns", nil + case TypeNamespacePID: + return "/run/pidns", nil + case TypeNamespaceNetwork: + return "/run/netns", nil + case TypeDevShm: + return "/run/devshm", nil + case TypeNamespaceUTS: + return "/run/utsns", nil + default: + return "", fmt.Errorf("unknown resource type %d: %w", int(t), errdefsInvalidArgument) + } +} + +// errdefsInvalidArgument is matched by the service layer to map validation +// failures onto an InvalidArgument status without importing errdefs here. +var errdefsInvalidArgument = errors.New("invalid argument") + +// ErrInvalidArgument is returned for a malformed group id or an unknown +// resource type. +var ErrInvalidArgument = errdefsInvalidArgument + +// key identifies one managed resource. +type key struct { + id string + typ Type +} + +// entry is the state of one managed resource. Once created, path is set and +// err is nil; if creation failed, err is set and is returned to every later +// caller rather than silently retrying a broken setup. +type entry struct { + path string + err error + // anchor is the process holding a PID namespace open. Only set for + // TypeNamespacePID; a PID namespace is destroyed by the kernel as soon + // as its PID 1 exits, so unlike the other types it cannot be kept + // alive by a bind mount alone. + anchor *os.Process + // sizeBytes is the effective tmpfs size a TypeDevShm resource was + // created with (after createDevShm's own <=0 fallback has already + // been applied). Only set for TypeDevShm; used purely to detect and + // warn about a later caller requesting a different size, not to + // change behavior — the first caller's size always wins. + sizeBytes int64 +} + +// Manager creates shared resources on demand and remembers them, so that +// repeated requests for the same (id, type) return the same path without +// doing the work again. A single Manager is meant to be shared for the +// lifetime of one vminitd process. +// +// Safe for concurrent use. +type Manager struct { + mu sync.Mutex + ns map[key]*entry +} + +// Create creates each of types for the group id that does not exist yet, and +// returns the guest path of every requested type. Duplicate types in the +// request are collapsed. On failure no partial result is returned, but any +// resources created earlier in the call are kept and will be reused by a +// later call. +// +// devShmSizeBytes is only consulted when types includes TypeDevShm and no +// devshm resource exists yet for id; see createDevShm. It is ignored +// otherwise, including when a devshm resource for id already exists — the +// size the first caller supplied wins. +func (m *Manager) Create(ctx context.Context, id string, types []Type, devShmSizeBytes int64) (map[Type]string, error) { + if err := validateID(id); err != nil { + return nil, err + } + wanted, err := dedupe(types) + if err != nil { + return nil, err + } + + m.mu.Lock() + defer m.mu.Unlock() + + paths := make(map[Type]string, len(wanted)) + for _, typ := range wanted { + e, err := m.ensureLocked(ctx, id, typ, devShmSizeBytes) + if err != nil { + return nil, err + } + paths[typ] = e.path + } + return paths, nil +} + +// Delete removes each of types for the group id. An empty types list deletes +// every resource belonging to id. Deleting something that does not exist is +// not an error. +func (m *Manager) Delete(ctx context.Context, id string, types []Type) error { + if err := validateID(id); err != nil { + return err + } + wanted, err := dedupe(types) + if err != nil { + return err + } + + m.mu.Lock() + defer m.mu.Unlock() + + if len(wanted) == 0 { + for k := range m.ns { + if k.id == id { + wanted = append(wanted, k.typ) + } + } + } + + var errs []error + for _, typ := range wanted { + if err := m.deleteLocked(ctx, id, typ); err != nil { + errs = append(errs, fmt.Errorf("delete %s resource: %w", typ, err)) + } + } + return errors.Join(errs...) +} + +// ensureLocked returns the entry for (id, typ), creating the resource if it +// does not exist. m.mu must be held. +func (m *Manager) ensureLocked(ctx context.Context, id string, typ Type, devShmSizeBytes int64) (*entry, error) { + k := key{id: id, typ: typ} + if e, ok := m.ns[k]; ok { + if e.err != nil { + return nil, e.err + } + // The tmpfs itself is never resized here — Create's own doc + // comment is explicit that the first caller's size wins — but a + // later caller silently getting a different size than it asked + // for is worth surfacing, since nothing else would ever tell it. + if typ == TypeDevShm && devShmSizeBytes > 0 && devShmSizeBytes != e.sizeBytes { + log.G(ctx).WithFields(log.Fields{ + "id": id, + "existing_size": e.sizeBytes, + "requested_size": devShmSizeBytes, + }).Warn("devshm resource already exists with a different size; keeping the existing size") + } + return e, nil + } + + dir, err := typ.dir() + if err != nil { + return nil, err + } + path := filepath.Join(dir, id) + + e := &entry{path: path} + switch typ { + case TypeNamespaceNetwork: + e.err = createNetwork(ctx, id, path) + case TypeNamespaceIPC: + e.err = createIPC(ctx, path) + case TypeNamespaceUTS: + e.err = createUTS(ctx, path) + case TypeNamespacePID: + e.anchor, e.err = createPID(ctx, path) + case TypeDevShm: + e.sizeBytes = effectiveDevShmSize(devShmSizeBytes) + e.err = createDevShm(path, e.sizeBytes) + default: + return nil, fmt.Errorf("unknown resource type %d: %w", int(typ), ErrInvalidArgument) + } + if e.err != nil { + e.err = fmt.Errorf("create %s resource %q: %w", typ, path, e.err) + } + + if m.ns == nil { + m.ns = make(map[key]*entry) + } + m.ns[k] = e + + if e.err != nil { + return nil, e.err + } + log.G(ctx).WithFields(log.Fields{ + "id": id, + "type": typ.String(), + "path": path, + }).Debug("created shared resource") + return e, nil +} + +// deleteLocked tears down the resource for (id, typ). m.mu must be held. +func (m *Manager) deleteLocked(ctx context.Context, id string, typ Type) error { + k := key{id: id, typ: typ} + e, ok := m.ns[k] + if !ok { + return nil + } + delete(m.ns, k) + + // A failed creation left nothing behind worth unmounting beyond the + // placeholder, which unpin handles. + if e.anchor != nil { + // Killing PID 1 is what actually destroys a PID namespace; the + // kernel then reaps everything else in it. Wait so the anchor does + // not linger as a zombie child of vminitd. + if err := e.anchor.Kill(); err != nil && !errors.Is(err, os.ErrProcessDone) { + log.G(ctx).WithError(err).WithField("id", id).Warn("failed to kill namespace anchor") + } + } + return unpin(e.path) +} + +// createNetwork creates a network namespace pinned at path and brings up its +// loopback interface. +// +// This is the "persistent namespace" technique containerd's CRI plugin uses +// on the host: a dedicated goroutine locks itself to an OS thread, unshares +// on that thread, and bind-mounts the thread's namespace file to a +// well-known path. The bind mount is what keeps the namespace alive, so the +// creating goroutine does not need to stay running afterwards. Go retires +// the locked OS thread when the goroutine exits (Go 1.10+), so leaving it +// locked does not poison the thread pool. +// +// netns.NewNamed pins at /run/netns/, which is exactly the layout +// Type.dir uses for TypeNamespaceNetwork, so id is passed as the name. +func createNetwork(ctx context.Context, id, path string) error { + errCh := make(chan error, 1) + go func() { + runtime.LockOSThread() + // Intentionally no UnlockOSThread: this thread's namespace has been + // replaced and must never be reused for unrelated work. + + nsh, err := netns.NewNamed(id) + if err != nil { + errCh <- fmt.Errorf("create named netns: %w", err) + return + } + // NewNamed returns an open handle to the new namespace. The bind + // mount it made is what keeps the namespace alive, so this + // descriptor is redundant and would otherwise be leaked for the + // lifetime of the process. + defer nsh.Close() + + // NewNamed leaves this locked thread inside the new namespace, so + // plain netlink calls operate on it without needing a NewHandleAt. + link, err := netlink.LinkByName("lo") + if err != nil { + errCh <- fmt.Errorf("lookup lo: %w", err) + return + } + if err := netlink.LinkSetUp(link); err != nil { + errCh <- fmt.Errorf("bring up lo: %w", err) + return + } + errCh <- nil + }() + if err := <-errCh; err != nil { + // netns.NewNamed may have created the bind-mount target before + // failing; make sure a later attempt is not blocked by it. + _ = unpin(path) + return err + } + return nil +} + +// createIPC creates an IPC namespace pinned at path, using the same +// locked-thread technique as createNetwork. unshare(CLONE_NEWIPC), unlike +// CLONE_NEWPID, takes effect on the calling thread immediately, so the +// thread's own namespace file is the one to bind-mount. +func createIPC(_ context.Context, path string) error { + if err := pin(path); err != nil { + return err + } + + errCh := make(chan error, 1) + go func() { + runtime.LockOSThread() + // Intentionally no UnlockOSThread: see createNetwork. + + if err := unix.Unshare(unix.CLONE_NEWIPC); err != nil { + errCh <- fmt.Errorf("unshare CLONE_NEWIPC: %w", err) + return + } + src := fmt.Sprintf("/proc/self/task/%d/ns/ipc", unix.Gettid()) + if err := unix.Mount(src, path, "", unix.MS_BIND, ""); err != nil { + errCh <- fmt.Errorf("bind mount %s: %w", src, err) + return + } + errCh <- nil + }() + if err := <-errCh; err != nil { + _ = unpin(path) + return err + } + return nil +} + +// createUTS creates a UTS namespace pinned at path, using the same +// locked-thread bind-mount technique as createIPC. +// +// Unlike the other namespace types, this package needs no extra +// coordination for the resource sharing containers actually care about: +// crun's own handling of a spec's Hostname field, applied after it joins +// this namespace via setns(2), calls sethostname(2) on it — which changes +// the hostname for every container sharing it — when Hostname is non-empty, +// and leaves it alone when Hostname is empty. So "last container to start +// with a non-empty hostname wins, empty means no opinion" already falls out +// of each container's own ordinary per-container spec field, with nothing +// for this API to track or reconcile. +func createUTS(_ context.Context, path string) error { + if err := pin(path); err != nil { + return err + } + + errCh := make(chan error, 1) + go func() { + runtime.LockOSThread() + // Intentionally no UnlockOSThread: see createNetwork. + + if err := unix.Unshare(unix.CLONE_NEWUTS); err != nil { + errCh <- fmt.Errorf("unshare CLONE_NEWUTS: %w", err) + return + } + src := fmt.Sprintf("/proc/self/task/%d/ns/uts", unix.Gettid()) + if err := unix.Mount(src, path, "", unix.MS_BIND, ""); err != nil { + errCh <- fmt.Errorf("bind mount %s: %w", src, err) + return + } + errCh <- nil + }() + if err := <-errCh; err != nil { + _ = unpin(path) + return err + } + return nil +} + +// createPID creates a PID namespace pinned at path and returns the anchor +// process holding it open. +// +// A PID namespace cannot be created the way createNetwork and createIPC +// create theirs. unshare(CLONE_NEWPID) does not move the caller into the new +// namespace; it only arranges for the caller's next child to become PID 1 of +// one. A thread can therefore never be the namespace's PID 1, and +// /proc/self/ns/pid_for_children has no value to bind-mount until that first +// child exists. Worse, the kernel destroys a PID namespace as soon as its +// PID 1 exits, and no further processes can be created in it after that, so +// a bind mount cannot substitute for a live process the way it can for the +// other types. Hence a real anchor process. +func createPID(ctx context.Context, path string) (*os.Process, error) { + if err := pin(path); err != nil { + return nil, err + } + + cmd := exec.Command(anchorCommand[0], anchorCommand[1:]...) //nolint:gosec // fixed command, not caller-controlled + cmd.SysProcAttr = &syscall.SysProcAttr{Cloneflags: syscall.CLONE_NEWPID} + if err := cmd.Start(); err != nil { + return nil, fmt.Errorf("start anchor: %w", err) + } + + src := fmt.Sprintf("/proc/%d/ns/pid", cmd.Process.Pid) + if err := unix.Mount(src, path, "", unix.MS_BIND, ""); err != nil { + // Nothing else will ever wait on or kill this process, so it would + // run for the rest of the VM's lifetime. Tear it down here, and do + // so synchronously so no goroutine is left blocked on a Wait that + // nothing else is coordinating with. + if killErr := cmd.Process.Kill(); killErr != nil { + log.G(ctx).WithError(killErr).Warn("failed to kill namespace anchor after mount failure") + } + if waitErr := cmd.Wait(); waitErr != nil { + log.G(ctx).WithError(waitErr).Debug("namespace anchor wait after mount failure") + } + return nil, fmt.Errorf("bind mount %s: %w", src, err) + } + + // Reap the anchor once it exits so it does not linger as a zombie child + // of vminitd. Normally it only exits when Delete kills it, or never. + // Started only after the bind mount succeeded, so the failure path above + // owns the Wait in that case. + go func() { + if err := cmd.Wait(); err != nil { + log.G(ctx).WithError(err).Debug("namespace anchor exited") + } + }() + + return cmd.Process, nil +} + +// defaultDevShmSizeBytes is used when devShmSizeBytes is not positive. The +// host-side caller always computes a real size from the sharing container's +// own CRI-provided /dev/shm mount options (falling back to its own default +// of the same value if unspecified), so this should not normally be +// reached; it exists purely as a defensive floor so a tmpfs is never +// created with a nonsensical size (in particular, "size=0" would make the +// tmpfs unusable — every write to it would fail with ENOSPC). +const defaultDevShmSizeBytes = 64 * 1024 * 1024 + +// effectiveDevShmSize applies the same non-positive-size fallback +// createDevShm itself applies, so ensureLocked can record the size a +// TypeDevShm entry was actually created with (for later mismatch warnings) +// before calling createDevShm. +func effectiveDevShmSize(sizeBytes int64) int64 { + if sizeBytes <= 0 { + return defaultDevShmSizeBytes + } + return sizeBytes +} + +// devShmMountFlags matches the nosuid/noexec/nodev flags CRI itself +// requests on a container's own (unshared) /dev/shm tmpfs mount, so sharing +// does not weaken those properties. +const devShmMountFlags = unix.MS_NOSUID | unix.MS_NOEXEC | unix.MS_NODEV + +// createDevShm creates a real, sized tmpfs pinned at path, for sharing +// containers to bind-mount their /dev/shm onto. +// +// Unlike the namespace types, this needs no bind-mount-of-a-namespace-file +// trick and no anchor process: it is a plain tmpfs mount, and the mount +// itself is what needs to stay alive, which the kernel already guarantees +// for as long as anything references it. It is mounted here, in the +// guest's own root mount namespace under /run, rather than inside the +// virtiofs-backed "containers" tree used for rootfs/volumes: that keeps +// this tmpfs real guest RAM with a real, kernel-enforced size limit, +// rather than something backed by host disk and reached over virtiofs. A +// member container's own crun bind-mounts this path directly, the same +// way it already joins /run/ipcns/ for a shared IPC namespace. +func createDevShm(path string, sizeBytes int64) error { + if sizeBytes <= 0 { + sizeBytes = defaultDevShmSizeBytes + } + + if err := os.MkdirAll(path, 0o755); err != nil { + return fmt.Errorf("create mount point: %w", err) + } + + data := fmt.Sprintf("size=%d,mode=1777", sizeBytes) + if err := unix.Mount("tmpfs", path, "tmpfs", devShmMountFlags, data); err != nil { + _ = os.Remove(path) + return fmt.Errorf("mount tmpfs: %w", err) + } + return nil +} + +// pin creates the empty file a namespace is bind-mounted onto, along with its +// parent directory. +func pin(path string) error { + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + return fmt.Errorf("create parent dir: %w", err) + } + f, err := os.OpenFile(path, os.O_RDONLY|os.O_CREATE|os.O_EXCL, 0o444) + if err != nil { + return fmt.Errorf("create bind-mount target: %w", err) + } + return f.Close() +} + +// unpin unmounts a pinned resource and removes its bind-mount target. It is +// idempotent: an already-unmounted or already-removed path is not an error. +func unpin(path string) error { + // The unmount result is deliberately ignored. There are several benign + // reasons it fails — the path was pinned but never mounted onto (EINVAL, + // or EPERM for an unprivileged caller), or it is already gone (ENOENT) — + // and distinguishing them from a real failure by errno alone is not + // reliable. The removal below is the actual check: if the resource is + // still mounted here, it fails with EBUSY and that is reported. + _ = unix.Unmount(path, 0) + + if err := os.Remove(path); err != nil && !errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("remove %s: %w", path, err) + } + return nil +} + +// validateID rejects group ids that are empty or that could escape the +// per-type directory once joined onto it. The id arrives over RPC and is +// used to build a filesystem path, so it is never trusted. +func validateID(id string) error { + if id == "" { + return fmt.Errorf("resource group id is required: %w", ErrInvalidArgument) + } + if id == "." || id == ".." || strings.ContainsRune(id, os.PathSeparator) || strings.ContainsRune(id, 0) { + return fmt.Errorf("invalid resource group id %q: %w", id, ErrInvalidArgument) + } + return nil +} + +// dedupe removes repeated types, preserving first-seen order, and rejects +// unknown ones. +func dedupe(types []Type) ([]Type, error) { + out := make([]Type, 0, len(types)) + seen := make(map[Type]struct{}, len(types)) + for _, typ := range types { + if _, err := typ.dir(); err != nil { + return nil, err + } + if _, ok := seen[typ]; ok { + continue + } + seen[typ] = struct{}{} + out = append(out, typ) + } + return out, nil +} diff --git a/internal/vminit/sharedresources/sharedresources_linux_test.go b/internal/vminit/sharedresources/sharedresources_linux_test.go new file mode 100644 index 00000000..0978c062 --- /dev/null +++ b/internal/vminit/sharedresources/sharedresources_linux_test.go @@ -0,0 +1,206 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sharedresources + +import ( + "context" + "errors" + "reflect" + "testing" +) + +// TestValidateID covers the group ids that must be rejected before being +// joined onto a directory to form a bind-mount path. The id arrives over RPC, +// so anything that could escape the per-type directory has to be refused +// rather than sanitized. +func TestValidateID(t *testing.T) { + testcases := []struct { + id string + wantErr bool + }{ + {id: "sandbox", wantErr: false}, + {id: "0f9d998d2b1c4e5a", wantErr: false}, + {id: "with-dashes_and_underscores.1", wantErr: false}, + {id: "..hidden", wantErr: false}, + + {id: "", wantErr: true}, + {id: ".", wantErr: true}, + {id: "..", wantErr: true}, + {id: "/", wantErr: true}, + {id: "a/b", wantErr: true}, + {id: "../etc/passwd", wantErr: true}, + {id: "/absolute", wantErr: true}, + {id: "trailing/", wantErr: true}, + {id: "nul\x00byte", wantErr: true}, + } + + for _, tc := range testcases { + t.Run(tc.id, func(t *testing.T) { + err := validateID(tc.id) + if tc.wantErr { + if err == nil { + t.Fatalf("validateID(%q) = nil, want error", tc.id) + } + if !errors.Is(err, ErrInvalidArgument) { + t.Errorf("validateID(%q) error = %v, want it to wrap ErrInvalidArgument", tc.id, err) + } + return + } + if err != nil { + t.Errorf("validateID(%q) = %v, want nil", tc.id, err) + } + }) + } +} + +// TestValidateIDRejectionsCannotEscape is a belt-and-braces check that every +// id validateID accepts stays inside its type directory once joined. +func TestCreateRejectsInvalidID(t *testing.T) { + var m Manager + if _, err := m.Create(context.Background(), "../escape", []Type{TypeNamespaceIPC}, 0); err == nil { + t.Fatal("Create with a traversing id = nil error, want failure") + } else if !errors.Is(err, ErrInvalidArgument) { + t.Errorf("Create error = %v, want it to wrap ErrInvalidArgument", err) + } + + // Nothing may have been recorded for a rejected request. + if len(m.ns) != 0 { + t.Errorf("manager recorded %d resources after a rejected request, want 0", len(m.ns)) + } +} + +func TestDedupe(t *testing.T) { + testcases := []struct { + name string + in []Type + want []Type + wantErr bool + }{ + { + name: "nil", + in: nil, + want: []Type{}, + }, + { + name: "order preserved", + in: []Type{TypeNamespaceNetwork, TypeNamespaceIPC, TypeNamespacePID}, + want: []Type{TypeNamespaceNetwork, TypeNamespaceIPC, TypeNamespacePID}, + }, + { + name: "duplicates collapsed, first-seen order kept", + in: []Type{TypeNamespacePID, TypeNamespaceIPC, TypeNamespacePID, TypeNamespaceIPC, TypeNamespacePID}, + want: []Type{TypeNamespacePID, TypeNamespaceIPC}, + }, + { + name: "unknown type rejected", + in: []Type{TypeNamespaceIPC, Type(99)}, + wantErr: true, + }, + { + name: "zero value is not a valid type", + in: []Type{Type(0)}, + wantErr: true, + }, + } + + for _, tc := range testcases { + t.Run(tc.name, func(t *testing.T) { + got, err := dedupe(tc.in) + if tc.wantErr { + if err == nil { + t.Fatalf("dedupe(%v) = nil error, want failure", tc.in) + } + if !errors.Is(err, ErrInvalidArgument) { + t.Errorf("dedupe error = %v, want it to wrap ErrInvalidArgument", err) + } + return + } + if err != nil { + t.Fatalf("dedupe(%v): %v", tc.in, err) + } + if !reflect.DeepEqual(got, tc.want) { + t.Errorf("dedupe(%v) = %v, want %v", tc.in, got, tc.want) + } + }) + } +} + +// TestTypeDir pins the guest path layout, since these paths are handed to the +// OCI runtime and are the only contract the host relies on. +func TestTypeDir(t *testing.T) { + testcases := []struct { + typ Type + want string + }{ + {typ: TypeNamespaceNetwork, want: "/run/netns"}, + {typ: TypeNamespaceIPC, want: "/run/ipcns"}, + {typ: TypeNamespaceUTS, want: "/run/utsns"}, + {typ: TypeNamespacePID, want: "/run/pidns"}, + } + for _, tc := range testcases { + t.Run(tc.typ.String(), func(t *testing.T) { + got, err := tc.typ.dir() + if err != nil { + t.Fatalf("dir(): %v", err) + } + if got != tc.want { + t.Errorf("dir() = %q, want %q", got, tc.want) + } + }) + } + + if _, err := Type(0).dir(); err == nil { + t.Error("dir() for the zero Type = nil error, want failure") + } +} + +// TestUnpinIsIdempotent verifies deleting a resource that was never created, +// or was already cleaned up, is not an error: Delete is documented as +// tolerating both. +func TestUnpinIsIdempotent(t *testing.T) { + path := t.TempDir() + "/never-existed" + if err := unpin(path); err != nil { + t.Errorf("unpin of a nonexistent path = %v, want nil", err) + } + + // A plain file that was pinned but never mounted onto must still be + // removed. + pinned := t.TempDir() + "/pinned" + if err := pin(pinned); err != nil { + t.Fatalf("pin: %v", err) + } + if err := unpin(pinned); err != nil { + t.Errorf("unpin of an unmounted pin = %v, want nil", err) + } + if err := unpin(pinned); err != nil { + t.Errorf("second unpin = %v, want nil", err) + } +} + +// TestDeleteUnknownIsNoop verifies Delete on a manager that never created +// anything succeeds rather than reporting a missing resource. +func TestDeleteUnknownIsNoop(t *testing.T) { + var m Manager + if err := m.Delete(context.Background(), "sandbox", []Type{TypeNamespaceIPC, TypeNamespacePID, TypeNamespaceNetwork}); err != nil { + t.Errorf("Delete of unknown resources = %v, want nil", err) + } + if err := m.Delete(context.Background(), "sandbox", nil); err != nil { + t.Errorf("Delete of an unknown group = %v, want nil", err) + } +} diff --git a/internal/vminit/socketforward/socketforward.go b/internal/vminit/socketforward/socketforward.go index c6b64775..18cf4e2c 100644 --- a/internal/vminit/socketforward/socketforward.go +++ b/internal/vminit/socketforward/socketforward.go @@ -67,6 +67,20 @@ type Service struct { // notify delivers ConnectRequest messages to the Accept stream so the // host shim can set up the vsock relay for each new connection. + // + // TODO: this channel, like the rest of Service, is process-wide (one + // per VM), but Accept below is a per-container stream on the host + // side (see socketForwarder in internal/shim/task/socketforward.go, + // "for a single container"). If more than one container in this VM + // binds a socket forward, each runs its own concurrent Accept RPC, + // and every one of those RPCs reads from this same channel — so a + // ConnectRequest can be delivered to a different container's Accept + // stream than the one whose forward it actually belongs to, which + // that container's host-side handleConnection then rejects as an + // "unknown forward ID". This was never reachable before multiple + // containers could share one VM. Fixing it means keying notify (and + // pending) by forward ID or container ID so a request only ever + // reaches the Accept stream that owns it. notify chan *socketforward.ConnectRequest } @@ -136,6 +150,11 @@ func (s *Service) Accept(ctx context.Context, srv socketforward.TTRPCSocketForwa } }() + // TODO: see the TODO on the notify field. Every concurrent caller of + // Accept (one per container sharing this VM) races here for the next + // value off the single shared notify channel, so a request meant for + // one container's forward can be delivered to a different + // container's stream instead. for { select { case req := <-s.notify: @@ -166,6 +185,16 @@ func (s *Service) bind(ctx context.Context, forwardID, socketPath string) error if err != nil { return fmt.Errorf("listening on %s: %w", socketPath, err) } + // Allow all processes (including those in user namespaces) to connect to + // this forwarded socket. Containers in user namespaces run as a mapped + // UID that is "other" from the VM init namespace's perspective, so they + // need write permission on the socket file to call connect(2). Execute + // bits are not meaningful for a UNIX socket, so 0o666 (rw for all) is + // sufficient; no need for 0o777. + if err := os.Chmod(socketPath, 0o666); err != nil { + l.Close() + return fmt.Errorf("chmod socket %s: %w", socketPath, err) + } s.listeners = append(s.listeners, l) diff --git a/internal/vminit/task/service.go b/internal/vminit/task/service.go index d6648ef7..3cea4ef6 100644 --- a/internal/vminit/task/service.go +++ b/internal/vminit/task/service.go @@ -572,7 +572,18 @@ func (s *service) Resume(ctx context.Context, r *taskAPI.ResumeRequest) (*ptypes return empty, nil } -// Kill a process with the provided signal +// Kill a process with the provided signal. +// +// KillRequest carries no PID at all — only a container ID and an ExecID — +// so a caller has no way to name a process by PID even if member containers +// of a sandbox share a PID namespace (see internal/shim/task/namespaces.go). +// getContainer resolves r.ID to this specific container's own tracked +// state, and container.Kill resolves r.ExecID within that container's own +// process map to the specific *runc.Container or *process.execProcess to +// signal; the numeric PID a shared PID namespace makes visible to other +// containers never enters the lookup. A signal request naming the wrong +// container or exec ID fails outright (getContainer/Process return an +// error) rather than silently landing on an unrelated process. func (s *service) Kill(ctx context.Context, r *taskAPI.KillRequest) (*ptypes.Empty, error) { started := time.Now() fields := log.Fields{ @@ -598,7 +609,13 @@ func (s *service) Kill(ctx context.Context, r *taskAPI.KillRequest) (*ptypes.Emp return empty, nil } -// Pids returns all pids inside the container +// Pids returns all pids inside the container. +// +// Like Kill, this is scoped by container ID: getContainerPids below calls +// `crun ps `, which crun resolves via the container's own +// cgroup, not by walking a (possibly shared) PID namespace. So a container +// that shares its PID namespace with sandbox peers still only ever reports +// its own processes here, never a peer's. func (s *service) Pids(ctx context.Context, r *taskAPI.PidsRequest) (*taskAPI.PidsResponse, error) { container, err := s.getContainer(r.ID) if err != nil { diff --git a/kernel/patches/0013-tsi-forward-the-resolved-port-for-ephemeral-binds.patch b/kernel/patches/0013-tsi-forward-the-resolved-port-for-ephemeral-binds.patch new file mode 100644 index 00000000..17ab1485 --- /dev/null +++ b/kernel/patches/0013-tsi-forward-the-resolved-port-for-ephemeral-binds.patch @@ -0,0 +1,90 @@ +From 5cee2292685440f69a2d3373cb9348b818577e9a Mon Sep 17 00:00:00 2001 +From: Derek McGowan +Date: Wed, 19 Aug 2026 00:44:13 -0700 +Subject: [PATCH] tsi: forward the resolved port for ephemeral bind()s + +tsi_bind() stored the caller's original sockaddr (e.g. port 0 for an +ephemeral bind) in tsk->bound_addr instead of the concrete port the +guest kernel's isocket->ops->bind() actually resolved it to. +tsi_listen() forwards tsk->bound_addr to the host over vsock, so with +an ephemeral bind the host picked its own independent ephemeral port +-- unrelated to the one tsi_getname() reports back to the application +(via the guest isocket, which does reflect the real resolved port). +Nothing ever listened on the application-visible port: not really +listening in the guest (tsi_listen only falls back to a real in-guest +listen on EPERM/EADDRINUSE/EADDRNOTAVAIL) and not the port the host +proxy actually bound either. A peer connecting to that port got +ECONNREFUSED. + +Fix: after the isocket bind succeeds, read back the resolved address +via getname and store that in tsk->bound_addr instead of the original +request, so the port forwarded to the host and the port reported to +the application always agree. Also switch tsk->bound_addr's allocation +to a fixed sizeof(struct sockaddr_storage) rather than the first +call's own addr_len, closing a latent overflow on rebind with a larger +address. + +Signed-off-by: Derek McGowan +--- + net/tsi/af_tsi.c | 31 +++++++++++++++++++++++++++---- + 1 file changed, 27 insertions(+), 4 deletions(-) + +diff --git a/net/tsi/af_tsi.c b/net/tsi/af_tsi.c +index bd3dc75b3..6ebe25cce 100644 +--- a/net/tsi/af_tsi.c ++++ b/net/tsi/af_tsi.c +@@ -10,6 +10,7 @@ + + #include + #include ++#include + #include + #include + #include +@@ -198,6 +199,8 @@ static int tsi_bind(struct socket *sock, struct sockaddr *addr, int addr_len) + struct socket *isocket; + struct socket *vsocket; + struct sockaddr_vm addr_vsock; ++ struct sockaddr_storage resolved_addr; ++ int resolved_len; + int err; + + lock_sock(sk); +@@ -245,10 +248,30 @@ static int tsi_bind(struct socket *sock, struct sockaddr *addr, int addr_len) + goto release; + } + +- if (!tsk->bound_addr) { +- tsk->bound_addr = kmalloc(addr_len, GFP_KERNEL); +- tsk->bound_addr_len = addr_len; +- } ++ /* ++ * addr may request an ephemeral port (port 0); isocket->ops->bind() ++ * above resolves that to a concrete port in the guest kernel, but ++ * does not mutate addr to reflect it. Read back the resolved ++ * address via getname so that what we forward to the host in ++ * tsi_listen() -- and what tsi_getname() reports back to the ++ * application -- agree on the same concrete port. Without this, the ++ * host independently resolves its own ephemeral port (there is no ++ * field in tsi_listen_rsp to report it back), so the ++ * application-visible port ends up bound and reachable nowhere: not ++ * really listening in the guest (tsi_listen only falls back to a ++ * real guest-side listen on EPERM/EADDRINUSE/EADDRNOTAVAIL), and not ++ * the port the host proxy actually bound either. ++ */ ++ resolved_len = isocket->ops->getname(isocket, ++ (struct sockaddr *)&resolved_addr, 0); ++ if (resolved_len > 0) { ++ addr = (struct sockaddr *)&resolved_addr; ++ addr_len = resolved_len; ++ } ++ ++ if (!tsk->bound_addr) ++ tsk->bound_addr = kmalloc(sizeof(struct sockaddr_storage), GFP_KERNEL); ++ tsk->bound_addr_len = addr_len; + memcpy(tsk->bound_addr, addr, addr_len); + + release: +-- +2.55.0 + diff --git a/pkg/shim/manager/manager.go b/pkg/shim/manager/manager.go index 6003566b..c4343cb3 100644 --- a/pkg/shim/manager/manager.go +++ b/pkg/shim/manager/manager.go @@ -26,6 +26,21 @@ import ( "github.com/containerd/containerd/v2/pkg/shim" ) +// MountNSIsolatedEnv is set (to "1") on the shim server child's own +// environment whenever cloneMntNs actually isolated its mount namespace +// (CLONE_NEWNS). The shim's own main, once running as that child, checks +// for this and calls IsolateMountPropagation if it's set — see that +// function's doc comment for why the isolation itself isn't enough +// without it. +// +// An env var rather than, say, a CLI flag or cloneMntNs's own return value: +// it needs to reach the child process specifically (not whichever +// invocation of the shim binary containerd made directly, e.g. its +// separate "start"/"delete" actions, which never go through cloneMntNs at +// all), and cloneMntNs's boolean return is already used by its caller for +// an unrelated purpose (whether to retry without a user namespace). +const MountNSIsolatedEnv = "NERDBOX_MOUNT_NS_ISOLATED" + // New returns a shim manager implementation that launches the nerdbox shim // process. The name is the runtime identifier reported to containerd (for // example "io.containerd.nerdbox.v1"). External callers building variants diff --git a/pkg/shim/manager/mount_linux.go b/pkg/shim/manager/mount_linux.go index 18e685f7..cb8b41a0 100644 --- a/pkg/shim/manager/mount_linux.go +++ b/pkg/shim/manager/mount_linux.go @@ -23,6 +23,8 @@ import ( "path/filepath" "strings" "syscall" + + "golang.org/x/sys/unix" ) // cloneMntNs configures the child command to start in a new mount @@ -53,13 +55,38 @@ import ( // container delete, and the VM itself performs all container-visible // filesystem setup. // +// The UID/GID mapping maps container-side 0 to the real host UID/GID, +// so the child appears as root *only inside its own, brand-new user +// namespace*. This grants no additional real host privilege: every +// interaction with a resource outside the namespace (files, sockets +// inherited across the namespace boundary, etc.) is still translated +// back through the mapping to the real, unprivileged host UID for +// permission checks. +// +// Mapping to UID 0 (rather than mapping the host UID to itself, which +// would leave euid non-zero inside the new namespace) matters because of +// how Linux computes capabilities across exec: a process whose effective +// UID is non-zero *within its own current user namespace* has its +// capability sets cleared to empty when it execs, even though the +// namespace's creator normally holds a full capability set in it. Since +// this child is always exec'd into the new namespace (see above), a +// non-zero-inside-its-own-namespace mapping would leave it with no +// capabilities at all afterward — unable to perform the bind mounts +// SharedFS needs, or even call getsockopt(2) on a listening-socket fd +// inherited across the namespace boundary (reproduced standalone by +// script/userns-check). Mapping to UID 0 keeps the child's effective UID +// zero *inside its own namespace* across exec, so the capability set is +// preserved and mount(2)/getsockopt(2) work as expected — with no change +// to what the process can do to real host resources, which remain gated +// by the real, unprivileged host UID/GID the mapping points at. +// // When the calling process already has real root (euid 0), we deliberately // skip CLONE_NEWUSER: entering a *new* user namespace — even one that maps -// a UID to itself — demotes the process to a non-initial user namespace, -// and the kernel restricts mounting real block-device-backed filesystems -// (e.g. ext4) to the initial user namespace regardless of the effective -// capabilities held within a descendant namespace. Real root gets -// CLONE_NEWNS alone, which still provides the mount-namespace +// UID 0 to the real root UID — demotes the process to a non-initial user +// namespace, and the kernel restricts mounting real block-device-backed +// filesystems (e.g. ext4) to the initial user namespace regardless of the +// effective capabilities held within a descendant namespace. Real root +// gets CLONE_NEWNS alone, which still provides the mount-namespace // isolation/cleanup-on-exit benefit without losing the ability to mount // real filesystems. // @@ -67,12 +94,21 @@ import ( // unprivileged user namespaces), the shim runs without mount isolation // and this function returns false. // cloneMntNs returns true if user namespace clone flags were set. +// +// Whenever CLONE_NEWNS is set (every case below except the AppArmor +// restriction), MountNSIsolatedEnv is also added to the child's +// environment, so the child itself knows to call IsolateMountPropagation +// early in its own startup — see that function's doc comment for why a +// new mount namespace alone does not stop mounts from leaking to the +// host, and MountNSIsolatedEnv's for why the child needs an explicit +// signal to know it should. func cloneMntNs(_ context.Context, cmd *exec.Cmd) bool { if os.Geteuid() == 0 { // Already real root: a plain mount namespace is enough, and // avoids demoting into a non-initial user namespace (which would // break mounts of real block-device filesystems). cmd.SysProcAttr.Cloneflags |= syscall.CLONE_NEWNS + cmd.Env = append(cmd.Env, MountNSIsolatedEnv+"=1") return false } @@ -90,14 +126,37 @@ func cloneMntNs(_ context.Context, cmd *exec.Cmd) bool { gid := os.Getgid() cmd.SysProcAttr.Cloneflags |= syscall.CLONE_NEWUSER | syscall.CLONE_NEWNS cmd.SysProcAttr.UidMappings = []syscall.SysProcIDMap{ - {ContainerID: uid, HostID: uid, Size: 1}, + {ContainerID: 0, HostID: uid, Size: 1}, } cmd.SysProcAttr.GidMappings = []syscall.SysProcIDMap{ - {ContainerID: gid, HostID: gid, Size: 1}, + {ContainerID: 0, HostID: gid, Size: 1}, } + cmd.Env = append(cmd.Env, MountNSIsolatedEnv+"=1") return true } +// IsolateMountPropagation detaches the calling process's entire mount tree +// from whatever propagation peer group it inherited. +// +// CLONE_NEWNS alone is not enough to stop the mounts cloneMntNs's child +// makes (container rootfs assembly: overlay/bind mounts under its +// bundle-specific state directory) from leaking to the host. The new +// namespace starts as a copy of the parent's mount table with each +// mount's propagation setting preserved, so a mount that was "shared" in +// the parent (the default on most systemd-managed hosts, including for +// "/") is still shared, in the same peer group, in the copy — meaning any +// submount made later inside the new namespace still propagates out to +// every other member of that peer group, including the host's own +// namespace. This must run before any such mount is made. +// +// MS_SLAVE (rather than MS_PRIVATE) is deliberate: it stops propagation +// in the leak-prone direction (out to the host) while leaving the +// process's own view still receiving mount/unmount events made by the +// host afterward, which nothing here needs to give up. +func IsolateMountPropagation() error { + return unix.Mount("", "/", "", unix.MS_REC|unix.MS_SLAVE, "") +} + // apparmorRestrictsUserns checks if the kernel sysctl // kernel.apparmor_restrict_unprivileged_userns is set to 1. // Returns (false, nil) when the sysctl does not exist (older kernels or diff --git a/pkg/shim/manager/mount_other.go b/pkg/shim/manager/mount_other.go index 30eccfae..dfc9f1b6 100644 --- a/pkg/shim/manager/mount_other.go +++ b/pkg/shim/manager/mount_other.go @@ -24,3 +24,9 @@ import ( ) func cloneMntNs(_ context.Context, _ *exec.Cmd) bool { return false } + +// IsolateMountPropagation is a no-op on this platform: MountNSIsolatedEnv +// is never set here, since cloneMntNs never sets CLONE_NEWNS, so nothing +// ever calls this. It exists so callers can build without a platform +// check. +func IsolateMountPropagation() error { return nil } diff --git a/pkg/vm/vm.go b/pkg/vm/vm.go index df13f67d..8a0e64a7 100644 --- a/pkg/vm/vm.go +++ b/pkg/vm/vm.go @@ -77,6 +77,14 @@ type StartOpts struct { // console output in addition to the implementation's default sink // (typically os.Stderr). Useful for capturing boot logs in tests. ConsoleWriter io.Writer + + // NetNS is the host-side network namespace path (e.g. + // "/var/run/netns/" or a bind-mount of /proc//ns/net) that + // the VM's networking should originate from. An empty value means + // host-network (no namespace switch). Implementations that support + // networking should enter this namespace before creating any + // networking-related host resources or worker threads. + NetNS string } // StartOpt mutates a [StartOpts] value. Options are applied in order. @@ -98,6 +106,14 @@ func WithConsoleWriter(w io.Writer) StartOpt { } } +// WithNetNS sets [StartOpts.NetNS] to path. An empty path is equivalent to +// not calling WithNetNS at all (host-network). +func WithNetNS(path string) StartOpt { + return func(o *StartOpts) { + o.NetNS = path + } +} + // MountConfig is the resolved configuration for a filesystem or block // device attachment, produced by applying [MountOpt] values. type MountConfig struct { @@ -182,6 +198,10 @@ type Instance interface { // must not be called after Start. Returns an error if the VM exits or // the guest fails to connect within an implementation-defined // timeout. + // + // If [WithNetNS] is used, implementations should enter that network + // namespace before creating any networking-related host resources or + // worker threads, so that VM traffic originates from it. Start(ctx context.Context, opts ...StartOpt) error // Client returns the TTRPC client connected to the guest agent. The diff --git a/pkg/vm/vm_test.go b/pkg/vm/vm_test.go new file mode 100644 index 00000000..618b283e --- /dev/null +++ b/pkg/vm/vm_test.go @@ -0,0 +1,38 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package vm + +import "testing" + +// TestWithNetNS verifies that WithNetNS sets StartOpts.NetNS. +func TestWithNetNS(t *testing.T) { + var o StartOpts + WithNetNS("/run/netns/foo")(&o) + if o.NetNS != "/run/netns/foo" { + t.Fatalf("NetNS = %q, want /run/netns/foo", o.NetNS) + } +} + +// TestWithNetNS_Empty verifies that WithNetNS accepts an empty path +// (host-network). +func TestWithNetNS_Empty(t *testing.T) { + o := StartOpts{NetNS: "should be overwritten"} + WithNetNS("")(&o) + if o.NetNS != "" { + t.Fatalf("NetNS = %q, want empty", o.NetNS) + } +} diff --git a/pkg/vminit/initd/containers_mount_linux.go b/pkg/vminit/initd/containers_mount_linux.go new file mode 100644 index 00000000..d42ffebc --- /dev/null +++ b/pkg/vminit/initd/containers_mount_linux.go @@ -0,0 +1,56 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package initd + +import ( + "context" + "os" + + "github.com/containerd/containerd/v2/core/mount" + "github.com/containerd/log" +) + +// mountContainersFS attempts to mount the "containers" virtiofs share at +// /run/containers. This share is added by the host shim when running in +// sandbox mode (CreateSandbox/StartSandbox) and exposes the assembled rootfs +// for each container under /run/containers//rootfs. +// +// /run is a tmpfs in the guest so the mount point directory can always be +// created, even though the erofs base rootfs is read-only. +// +// On the legacy single-container path the "containers" virtiofs tag is not +// registered by the host, so the mount will fail. We log that at debug +// level and continue — the legacy path does not use this mount. +func mountContainersFS() { + target := "/run/containers" + + // /run is a tmpfs so MkdirAll always succeeds here. + if err := os.MkdirAll(target, 0o755); err != nil { + log.G(context.Background()).WithError(err).Debug("failed to create /run/containers mountpoint") + return + } + + err := mount.All([]mount.Mount{{ + Type: "virtiofs", + Source: "containers", + Target: target, + }}, "/") + if err != nil { + // Expected on the legacy single-container path. + log.G(context.Background()).WithError(err).Debug("containers virtiofs share not available (expected on single-container path)") + } +} diff --git a/pkg/vminit/initd/initd.go b/pkg/vminit/initd/initd.go index ba7e92c0..f5c4c31c 100644 --- a/pkg/vminit/initd/initd.go +++ b/pkg/vminit/initd/initd.go @@ -194,6 +194,40 @@ func Run(ctx context.Context) error { func systemInit(ctx context.Context, config Config, shutdownSvc shutdown.Service) error { t := time.Now() + // Raise the open-file-descriptor limit for the init process and all + // children. The kernel default (1024) is too low for long-running + // sandbox sessions under sustained container churn. + // + // Each container's OOM monitor (oomv2.Add) creates one inotify FD via + // cgroup2.Manager.EventChan and spawns a short-lived goroutine that holds + // it until the goroutine is scheduled and completes (microseconds of work). + // Under heavy load with GOMAXPROCS=2, the Go scheduler may not immediately + // service these goroutines, allowing a burst of unscheduled goroutines to + // accumulate. Each holds one inotify FD until it runs. At ~36 container + // starts/second the burst can briefly hold hundreds of FDs before the + // scheduler catches up. 65536 gives ~1800 seconds of headroom at that + // rate — far beyond any scheduling stall in practice. + // + // Only ever raise the limit, never lower it: read the current soft/hard + // limits first and leave them untouched if either is already at or above + // the target, so we never clobber a higher value set by the guest kernel + // or anything that ran before us. + const wantNofile = 65536 + var nofileLimit unix.Rlimit + if err := unix.Getrlimit(unix.RLIMIT_NOFILE, &nofileLimit); err != nil { + log.G(ctx).WithError(err).Warn("failed to read RLIMIT_NOFILE; leaving it unchanged") + } else if nofileLimit.Cur < wantNofile || nofileLimit.Max < wantNofile { + if nofileLimit.Cur < wantNofile { + nofileLimit.Cur = wantNofile + } + if nofileLimit.Max < wantNofile { + nofileLimit.Max = wantNofile + } + if err := unix.Setrlimit(unix.RLIMIT_NOFILE, &nofileLimit); err != nil { + log.G(ctx).WithError(err).Warn("failed to raise RLIMIT_NOFILE; FD exhaustion may occur under sustained load") + } + } + if err := systemMounts(); err != nil { return err } @@ -223,7 +257,7 @@ func systemInit(ctx context.Context, config Config, shutdownSvc shutdown.Service } func systemMounts() error { - return mount.All([]mount.Mount{ + required := []mount.Mount{ { Type: "proc", Source: "proc", @@ -265,7 +299,25 @@ func systemMounts() error { }, // /dev is handled by the kernel via CONFIG_DEVTMPFS_MOUNT=y before // the init process starts; no explicit mount is needed here. - }, "/") + } + + if err := mount.All(required, "/"); err != nil { + return err + } + + // Mount the sandbox container-shared virtiofs at /run/containers. + // The host shim assembles each container's rootfs under + // /containers//rootfs and exposes it through this + // single share tagged "containers". This mount is optional: on the legacy + // single-container path the "containers" tag is not registered by the host + // and the mount will fail. We ignore the error so the legacy path is + // unaffected. + // + // /run/containers is created at runtime (under the /run tmpfs) so no + // change to the erofs rootfs image is required. + mountContainersFS() + + return nil } func setupCgroupControl() error { diff --git a/plugins/shim/sandbox/plugin.go b/plugins/sandbox/manager_plugin.go similarity index 85% rename from plugins/shim/sandbox/plugin.go rename to plugins/sandbox/manager_plugin.go index 6542ef94..7474d657 100644 --- a/plugins/shim/sandbox/plugin.go +++ b/plugins/sandbox/manager_plugin.go @@ -20,6 +20,7 @@ import ( "github.com/containerd/plugin" "github.com/containerd/plugin/registry" + "github.com/containerd/nerdbox/internal/shim/sandbox" vmsbox "github.com/containerd/nerdbox/internal/shim/sandbox/vm" "github.com/containerd/nerdbox/pkg/vm" "github.com/containerd/nerdbox/plugins" @@ -33,12 +34,13 @@ func init() { plugins.VMManagerPlugin, }, InitFn: func(ic *plugin.InitContext) (interface{}, error) { - // Only a single VM manager plugin is supported + // Only a single VM manager plugin is supported. vmm, err := ic.GetSingle(plugins.VMManagerPlugin) if err != nil { return nil, err } - return vmsbox.NewVMSandbox(vmm.(vm.Manager)), nil + sb := vmsbox.NewVMSandbox(vmm.(vm.Manager)) + return sandbox.NewSandboxService(sb), nil }, }) } diff --git a/plugins/services/mount/service.go b/plugins/services/mount/service.go index 47e2a3b3..0c098162 100644 --- a/plugins/services/mount/service.go +++ b/plugins/services/mount/service.go @@ -34,6 +34,7 @@ import ( "github.com/containerd/plugin" "github.com/containerd/plugin/registry" "github.com/containerd/ttrpc" + "github.com/moby/sys/mountinfo" api "github.com/containerd/nerdbox/api/services/mount/v1" ) @@ -43,7 +44,7 @@ func init() { Type: cplugins.TTRPCPlugin, ID: "mount", InitFn: func(ic *plugin.InitContext) (interface{}, error) { - return &service{}, nil + return &service{mounted: mountinfo.Mounted, doMount: doMount}, nil }, }) } @@ -51,6 +52,35 @@ func init() { type service struct { mu sync.Mutex mounts []*api.MountSpec // in-VM mounts, in mount order + + // mounted reports whether path is currently a real mount point. It is + // mountinfo.Mounted in production; tests substitute a fake to exercise + // the bookkeeping-reconcile logic in MountAll without needing a + // privileged real mount. + mounted func(path string) (bool, error) + + // doMount creates m.Target and performs the real mount. It is doMount + // in production; tests substitute a fake for the same reason as + // mounted above. + doMount func(m *api.MountSpec) error +} + +// doMount creates the mount point directory and performs the real mount +// described by m. +func doMount(m *api.MountSpec) error { + if err := os.MkdirAll(m.Target, 0700); err != nil { + return fmt.Errorf("failed to create mount target directory %s: %w", m.Target, err) + } + + if err := ctrMount.All([]ctrMount.Mount{{ + Type: m.Type, + Source: m.Source, + Target: m.Target, + Options: m.Options, + }}, "/"); err != nil { + return fmt.Errorf("failed to mount %s at %s: %w", m.Source, m.Target, err) + } + return nil } func (s *service) RegisterTTRPC(server *ttrpc.Server) error { @@ -72,24 +102,38 @@ func (s *service) MountAll(ctx context.Context, r *api.MountAllRequest) (*api.Mo i := slices.IndexFunc(s.mounts, func(e *api.MountSpec) bool { return e.Target == m.Target }) if i >= 0 { - if mountSpecsEqual(s.mounts[i], m) { - log.G(ctx).WithField("target", m.Target).Debug("mount already exists with matching spec; skipping") - continue + // Bookkeeping alone is not enough to skip the mount: nothing + // prevents the target from having been unmounted by other + // means since it was recorded (e.g. a container's rootfs + // cleanup racing a reused container ID, or the guest mount + // service having restarted). Reconcile against the real + // mount table before trusting the record, so a stale entry + // cannot cause a container to silently start with the wrong + // (or no) filesystem mounted at its target. + mounted, err := s.mounted(m.Target) + if err != nil && !os.IsNotExist(err) { + return nil, errgrpc.ToGRPC(fmt.Errorf("check mount state of %s: %w", m.Target, err)) } - return nil, errgrpc.ToGRPC(fmt.Errorf("target %s already mounted with a different spec: %w", m.Target, errdefs.ErrAlreadyExists)) - } - - if err := os.MkdirAll(m.Target, 0700); err != nil { - return nil, errgrpc.ToGRPC(fmt.Errorf("failed to create mount target directory %s: %w", m.Target, err)) + if mounted { + if mountSpecsEqual(s.mounts[i], m) { + log.G(ctx).WithField("target", m.Target).Debug("mount already exists with matching spec; skipping") + continue + } + return nil, errgrpc.ToGRPC(fmt.Errorf("target %s already mounted with a different spec: %w", m.Target, errdefs.ErrAlreadyExists)) + } + // The bookkeeping entry is stale: the target is not actually + // mounted (or no longer exists) despite our record saying + // otherwise. Drop it and fall through to mount fresh below. + // This can never disturb another container's mounts: mount + // targets are per-container by construction, so reconciling + // this entry away never touches state any other container + // depends on, and nothing here ever issues an unmount. + log.G(ctx).WithField("target", m.Target).Warn("bookkeeping said this target was mounted, but it is not; remounting") + s.mounts = slices.Delete(s.mounts, i, i+1) } - if err := ctrMount.All([]ctrMount.Mount{{ - Type: m.Type, - Source: m.Source, - Target: m.Target, - Options: m.Options, - }}, "/"); err != nil { - return nil, errgrpc.ToGRPC(fmt.Errorf("failed to mount %s at %s: %w", m.Source, m.Target, err)) + if err := s.doMount(m); err != nil { + return nil, errgrpc.ToGRPC(err) } s.mounts = append(s.mounts, m) @@ -107,7 +151,12 @@ func (s *service) Unmount(ctx context.Context, r *api.UnmountRequest) (*api.Unmo if i < 0 { return nil, errgrpc.ToGRPC(fmt.Errorf("cannot unmount %s: %w", r.Target, errdefs.ErrNotFound)) } - if err := ctrMount.Unmount(r.Target, 0); err != nil { + // ctrMount.Unmount already treats "not currently a mount point" as + // success; also tolerate the target directory itself having been + // removed already (e.g. by rootfs cleanup racing this call), so a + // caller retrying a partially-failed teardown does not get stuck on + // state that is already gone. + if err := ctrMount.Unmount(r.Target, 0); err != nil && !os.IsNotExist(err) { return nil, errgrpc.ToGRPC(fmt.Errorf("failed to unmount %s: %w", r.Target, err)) } s.mounts = slices.Delete(s.mounts, i, i+1) @@ -124,7 +173,10 @@ func (s *service) UnmountAll(ctx context.Context, _ *api.UnmountAllRequest) (*ap for i := len(s.mounts) - 1; i >= 0; i-- { target := s.mounts[i].Target log.G(ctx).WithField("target", target).Info("unmounting filesystem") - if err := ctrMount.Unmount(target, 0); err != nil { + // See the comment in Unmount: tolerate the target already being + // gone, on top of ctrMount.Unmount's own tolerance of "not + // currently a mount point". + if err := ctrMount.Unmount(target, 0); err != nil && !os.IsNotExist(err) { log.G(ctx).WithError(err).WithField("target", target).Warn("failed to unmount") errs = append(errs, fmt.Errorf("unmount %s: %w", target, err)) continue diff --git a/plugins/services/mount/service_test.go b/plugins/services/mount/service_test.go new file mode 100644 index 00000000..41a71ba2 --- /dev/null +++ b/plugins/services/mount/service_test.go @@ -0,0 +1,211 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package mount + +import ( + "context" + "errors" + "io/fs" + "strings" + "testing" + + api "github.com/containerd/nerdbox/api/services/mount/v1" +) + +// newTestService builds a service with fake mounted/doMount functions, so +// MountAll's reconcile logic can be exercised deterministically without a +// privileged real mount. +func newTestService(mounted func(string) (bool, error), doMount func(*api.MountSpec) error) *service { + if doMount == nil { + doMount = func(*api.MountSpec) error { return nil } + } + return &service{mounted: mounted, doMount: doMount} +} + +func mountSpec(target string) *api.MountSpec { + return &api.MountSpec{Type: "bind", Source: "/src", Target: target, Options: []string{"bind"}} +} + +// TestMountAllSkipsWhenReallyMounted verifies the common case: bookkeeping +// says the target is mounted with a matching spec, and it genuinely still +// is, so MountAll skips it without calling doMount again. +func TestMountAllSkipsWhenReallyMounted(t *testing.T) { + var doMountCalls int + s := newTestService( + func(string) (bool, error) { return true, nil }, + func(*api.MountSpec) error { doMountCalls++; return nil }, + ) + s.mounts = []*api.MountSpec{mountSpec("/target")} + + if _, err := s.MountAll(context.Background(), &api.MountAllRequest{Mounts: []*api.MountSpec{mountSpec("/target")}}); err != nil { + t.Fatalf("MountAll: %v", err) + } + if doMountCalls != 0 { + t.Errorf("doMount called %d times, want 0 (should have been skipped)", doMountCalls) + } + if len(s.mounts) != 1 { + t.Errorf("s.mounts = %v, want 1 entry retained", s.mounts) + } +} + +// TestMountAllReconcilesStaleBookkeeping verifies that when bookkeeping +// claims a target is mounted but it genuinely is not (e.g. unmounted by +// other means, or the guest mount service restarted), MountAll drops the +// stale entry and mounts fresh instead of trusting the record. +func TestMountAllReconcilesStaleBookkeeping(t *testing.T) { + var doMountCalls int + s := newTestService( + func(string) (bool, error) { return false, nil }, // not really mounted + func(*api.MountSpec) error { doMountCalls++; return nil }, + ) + s.mounts = []*api.MountSpec{mountSpec("/target")} + + if _, err := s.MountAll(context.Background(), &api.MountAllRequest{Mounts: []*api.MountSpec{mountSpec("/target")}}); err != nil { + t.Fatalf("MountAll: %v", err) + } + if doMountCalls != 1 { + t.Errorf("doMount called %d times, want 1 (stale entry should have been remounted)", doMountCalls) + } + if len(s.mounts) != 1 { + t.Errorf("s.mounts = %v, want exactly 1 entry after reconcile+remount", s.mounts) + } +} + +// TestMountAllReconcilesTargetGone verifies that mounted returning +// fs.ErrNotExist (the target directory itself no longer exists) is treated +// the same as "not mounted", not as a fatal error. +func TestMountAllReconcilesTargetGone(t *testing.T) { + var doMountCalls int + s := newTestService( + func(string) (bool, error) { return false, fs.ErrNotExist }, + func(*api.MountSpec) error { doMountCalls++; return nil }, + ) + s.mounts = []*api.MountSpec{mountSpec("/target")} + + if _, err := s.MountAll(context.Background(), &api.MountAllRequest{Mounts: []*api.MountSpec{mountSpec("/target")}}); err != nil { + t.Fatalf("MountAll: %v", err) + } + if doMountCalls != 1 { + t.Errorf("doMount called %d times, want 1", doMountCalls) + } +} + +// TestMountAllDifferentSpecStillMounted verifies that a target genuinely +// still mounted, but with a different spec than requested, is still +// rejected as a conflict rather than silently remounted or skipped. +func TestMountAllDifferentSpecStillMounted(t *testing.T) { + s := newTestService(func(string) (bool, error) { return true, nil }, nil) + s.mounts = []*api.MountSpec{mountSpec("/target")} + + different := mountSpec("/target") + different.Source = "/other-src" + + _, err := s.MountAll(context.Background(), &api.MountAllRequest{Mounts: []*api.MountSpec{different}}) + if err == nil { + t.Fatal("MountAll: got nil error, want a conflict error") + } + if !strings.Contains(err.Error(), "already mounted with a different spec") { + t.Errorf("MountAll error = %v, want a spec-conflict error", err) + } +} + +// TestMountAllReconcileErrorPropagates verifies that a real (non-"missing") +// error from the mounted check is surfaced, not silently treated as "not +// mounted". +func TestMountAllReconcileErrorPropagates(t *testing.T) { + wantErr := errors.New("mountinfo read failed") + s := newTestService(func(string) (bool, error) { return false, wantErr }, nil) + s.mounts = []*api.MountSpec{mountSpec("/target")} + + _, err := s.MountAll(context.Background(), &api.MountAllRequest{Mounts: []*api.MountSpec{mountSpec("/target")}}) + if err == nil || !strings.Contains(err.Error(), wantErr.Error()) { + t.Errorf("MountAll error = %v, want it to mention %q", err, wantErr) + } +} + +// TestMountAllNewTargetNoReconcile verifies a target with no bookkeeping +// entry at all is mounted directly, without ever consulting mounted. +func TestMountAllNewTargetNoReconcile(t *testing.T) { + var mountedCalls, doMountCalls int + s := newTestService( + func(string) (bool, error) { mountedCalls++; return true, nil }, + func(*api.MountSpec) error { doMountCalls++; return nil }, + ) + + if _, err := s.MountAll(context.Background(), &api.MountAllRequest{Mounts: []*api.MountSpec{mountSpec("/target")}}); err != nil { + t.Fatalf("MountAll: %v", err) + } + if mountedCalls != 0 { + t.Errorf("mounted called %d times, want 0 (no prior bookkeeping to reconcile)", mountedCalls) + } + if doMountCalls != 1 { + t.Errorf("doMount called %d times, want 1", doMountCalls) + } + if len(s.mounts) != 1 { + t.Errorf("s.mounts = %v, want 1 entry", s.mounts) + } +} + +// TestUnmountToleratesTargetAlreadyGone verifies that Unmount does not fail +// when the target directory no longer exists at all (e.g. removed by +// cleanup racing this call), since there is nothing left to unmount. +func TestUnmountToleratesTargetAlreadyGone(t *testing.T) { + // Use a target that both looks unmounted and does not exist, so the + // real ctrMount.Unmount call this test exercises (doMount/mounted + // fakes only affect MountAll) hits ENOENT rather than EINVAL. + target := t.TempDir() + "/does-not-exist" + + s := newTestService(nil, nil) + s.mounts = []*api.MountSpec{mountSpec(target)} + + if _, err := s.Unmount(context.Background(), &api.UnmountRequest{Target: target}); err != nil { + t.Fatalf("Unmount: %v", err) + } + if len(s.mounts) != 0 { + t.Errorf("s.mounts = %v, want empty after Unmount", s.mounts) + } +} + +// TestUnmountAllToleratesTargetAlreadyGone is the UnmountAll analogue of +// TestUnmountToleratesTargetAlreadyGone. +func TestUnmountAllToleratesTargetAlreadyGone(t *testing.T) { + target := t.TempDir() + "/does-not-exist" + + s := newTestService(nil, nil) + s.mounts = []*api.MountSpec{mountSpec(target)} + + if _, err := s.UnmountAll(context.Background(), &api.UnmountAllRequest{}); err != nil { + t.Fatalf("UnmountAll: %v", err) + } + if len(s.mounts) != 0 { + t.Errorf("s.mounts = %v, want empty after UnmountAll", s.mounts) + } +} + +// TestUnmountUnknownTarget verifies Unmount still rejects a target with no +// bookkeeping entry at all: "tolerant of already-gone" only applies to +// something this service actually thought it had mounted. +func TestUnmountUnknownTarget(t *testing.T) { + s := newTestService(nil, nil) + + _, err := s.Unmount(context.Background(), &api.UnmountRequest{Target: "/never-mounted"}) + if err == nil || !strings.Contains(err.Error(), "not found") { + t.Errorf("Unmount error = %v, want a not-found error", err) + } +} diff --git a/plugins/services/sharedresources/service.go b/plugins/services/sharedresources/service.go new file mode 100644 index 00000000..0c3350b6 --- /dev/null +++ b/plugins/services/sharedresources/service.go @@ -0,0 +1,147 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sharedresources + +import ( + "context" + "errors" + "fmt" + + "github.com/containerd/errdefs" + "github.com/containerd/errdefs/pkg/errgrpc" + "github.com/containerd/plugin" + "github.com/containerd/plugin/registry" + "github.com/containerd/ttrpc" + + api "github.com/containerd/nerdbox/api/services/sharedresources/v1" + "github.com/containerd/nerdbox/internal/vminit/sharedresources" + "github.com/containerd/nerdbox/plugins" +) + +var _ api.TTRPCSharedResourcesService = &service{} + +func init() { + registry.Register(&plugin.Registration{ + Type: plugins.TTRPCPlugin, + ID: "sharedresources", + InitFn: initFunc, + }) +} + +func initFunc(ic *plugin.InitContext) (interface{}, error) { + return &service{}, nil +} + +// service implements the SharedResources TTRPC service by delegating to a +// sharedresources.Manager, translating the generated protobuf types to and +// from that package's own domain types. +type service struct { + mgr sharedresources.Manager +} + +func (s *service) RegisterTTRPC(server *ttrpc.Server) error { + api.RegisterTTRPCSharedResourcesService(server, s) + return nil +} + +func (s *service) Create(ctx context.Context, r *api.CreateRequest) (*api.CreateResponse, error) { + types, err := fromAPITypes(r.GetTypes()) + if err != nil { + return nil, errgrpc.ToGRPC(err) + } + + paths, err := s.mgr.Create(ctx, r.GetID(), types, r.GetDevShmSizeBytes()) + if err != nil { + return nil, errgrpc.ToGRPC(toErrdefs(err)) + } + + // One entry per requested type, in request order. + resp := &api.CreateResponse{Resources: make([]*api.SharedResource, 0, len(types))} + for _, typ := range types { + path, ok := paths[typ] + if !ok { + return nil, errgrpc.ToGRPC(fmt.Errorf("no path for %s resource: %w", typ, errdefs.ErrFailedPrecondition)) + } + resp.Resources = append(resp.Resources, &api.SharedResource{ + Type: toAPIType(typ), + Path: path, + }) + } + return resp, nil +} + +func (s *service) Delete(ctx context.Context, r *api.DeleteRequest) (*api.DeleteResponse, error) { + types, err := fromAPITypes(r.GetTypes()) + if err != nil { + return nil, errgrpc.ToGRPC(err) + } + if err := s.mgr.Delete(ctx, r.GetID(), types); err != nil { + return nil, errgrpc.ToGRPC(toErrdefs(err)) + } + return &api.DeleteResponse{}, nil +} + +// fromAPITypes converts requested wire types to domain types, rejecting +// unspecified or unrecognized values. +func fromAPITypes(in []api.Type) ([]sharedresources.Type, error) { + out := make([]sharedresources.Type, 0, len(in)) + for _, t := range in { + switch t { + case api.Type_TYPE_NAMESPACE_IPC: + out = append(out, sharedresources.TypeNamespaceIPC) + case api.Type_TYPE_NAMESPACE_PID: + out = append(out, sharedresources.TypeNamespacePID) + case api.Type_TYPE_NAMESPACE_NETWORK: + out = append(out, sharedresources.TypeNamespaceNetwork) + case api.Type_TYPE_DEVSHM: + out = append(out, sharedresources.TypeDevShm) + case api.Type_TYPE_NAMESPACE_UTS: + out = append(out, sharedresources.TypeNamespaceUTS) + default: + return nil, fmt.Errorf("unsupported resource type %q: %w", t, errdefs.ErrInvalidArgument) + } + } + return out, nil +} + +func toAPIType(t sharedresources.Type) api.Type { + switch t { + case sharedresources.TypeNamespaceIPC: + return api.Type_TYPE_NAMESPACE_IPC + case sharedresources.TypeNamespacePID: + return api.Type_TYPE_NAMESPACE_PID + case sharedresources.TypeNamespaceNetwork: + return api.Type_TYPE_NAMESPACE_NETWORK + case sharedresources.TypeDevShm: + return api.Type_TYPE_DEVSHM + case sharedresources.TypeNamespaceUTS: + return api.Type_TYPE_NAMESPACE_UTS + default: + return api.Type_TYPE_UNSPECIFIED + } +} + +// toErrdefs maps the Manager's validation failures onto an errdefs error so +// the caller sees InvalidArgument rather than Unknown. +func toErrdefs(err error) error { + if errors.Is(err, sharedresources.ErrInvalidArgument) { + return fmt.Errorf("%s: %w", err.Error(), errdefs.ErrInvalidArgument) + } + return err +} diff --git a/plugins/shim/sandbox/ttrpc_plugin.go b/plugins/shim/sandbox/ttrpc_plugin.go new file mode 100644 index 00000000..5c5edd40 --- /dev/null +++ b/plugins/shim/sandbox/ttrpc_plugin.go @@ -0,0 +1,69 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox + +import ( + "fmt" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + "github.com/containerd/plugin" + "github.com/containerd/plugin/registry" + "github.com/containerd/ttrpc" + + "github.com/containerd/nerdbox/plugins" +) + +func init() { + registry.Register(&plugin.Registration{ + Type: plugins.TTRPCPlugin, + ID: "sandbox", + Requires: []plugin.Type{ + plugins.SandboxPlugin, + }, + InitFn: func(ic *plugin.InitContext) (interface{}, error) { + sbPlugin, err := ic.GetSingle(plugins.SandboxPlugin) + if err != nil { + return nil, err + } + + sm, ok := sbPlugin.(sandboxAPI.TTRPCSandboxService) + if !ok { + return nil, fmt.Errorf("unexpected sandbox plugin implementation %T", sbPlugin) + } + return &sbService{srv: sm}, nil + }, + }) +} + +// sbService adapts a sandboxAPI.TTRPCSandboxService to shim.TTRPCService, +// so that the "sandbox" TTRPCPlugin registration above (rather than the +// SandboxPlugin "manager" registration, which other plugins such as +// streaming/transfer depend on as a plain sandbox.Sandbox) is the one the +// shim framework calls RegisterTTRPC on. Without this indirection, the +// framework would either not find a RegisterTTRPC method at all, or (if +// SandboxService implemented it directly) call it a second time when it +// scans the "manager" plugin's own instance, double-registering the +// service. +type sbService struct { + srv sandboxAPI.TTRPCSandboxService +} + +// RegisterTTRPC registers the sandbox service on the TTRPC server. +func (s *sbService) RegisterTTRPC(server *ttrpc.Server) error { + sandboxAPI.RegisterTTRPCSandboxService(server, s.srv) + return nil +} diff --git a/plugins/shim/task/plugin.go b/plugins/shim/task/plugin.go index 5b897a19..268b5a26 100644 --- a/plugins/shim/task/plugin.go +++ b/plugins/shim/task/plugin.go @@ -17,14 +17,13 @@ package task import ( - "github.com/containerd/containerd/v2/pkg/shim" - "github.com/containerd/containerd/v2/pkg/shutdown" - cplugins "github.com/containerd/containerd/v2/plugins" + "fmt" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" "github.com/containerd/plugin" "github.com/containerd/plugin/registry" + "github.com/containerd/ttrpc" - "github.com/containerd/nerdbox/internal/shim/sandbox" - "github.com/containerd/nerdbox/internal/shim/task" "github.com/containerd/nerdbox/plugins" ) @@ -33,25 +32,29 @@ func init() { Type: plugins.TTRPCPlugin, ID: "task", Requires: []plugin.Type{ - cplugins.EventPlugin, - cplugins.InternalPlugin, - plugins.SandboxPlugin, + plugins.TaskPlugin, }, InitFn: func(ic *plugin.InitContext) (interface{}, error) { - pp, err := ic.GetByID(cplugins.EventPlugin, "publisher") - if err != nil { - return nil, err - } - ss, err := ic.GetByID(cplugins.InternalPlugin, "shutdown") + tPlugin, err := ic.GetSingle(plugins.TaskPlugin) if err != nil { return nil, err } - sb, err := ic.GetSingle(plugins.SandboxPlugin) - if err != nil { - return nil, err + + tm, ok := tPlugin.(taskAPI.TTRPCTaskService) + if !ok { + return nil, fmt.Errorf("unexpected task plugin implementation %T", tPlugin) } - return task.NewTaskService(ic.Context, sb.(sandbox.Sandbox), pp.(shim.Publisher), ss.(shutdown.Service)) + + return taskService{srv: tm}, nil }, }) +} + +type taskService struct { + srv taskAPI.TTRPCTaskService +} +func (s taskService) RegisterTTRPC(server *ttrpc.Server) error { + taskAPI.RegisterTTRPCTaskService(server, s.srv) + return nil } diff --git a/plugins/task/manager_plugin.go b/plugins/task/manager_plugin.go new file mode 100644 index 00000000..14d9b150 --- /dev/null +++ b/plugins/task/manager_plugin.go @@ -0,0 +1,75 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package task + +import ( + "fmt" + + "github.com/containerd/containerd/v2/pkg/shim" + "github.com/containerd/containerd/v2/pkg/shutdown" + cplugins "github.com/containerd/containerd/v2/plugins" + "github.com/containerd/plugin" + "github.com/containerd/plugin/registry" + + intsandbox "github.com/containerd/nerdbox/internal/shim/sandbox" + "github.com/containerd/nerdbox/internal/shim/task" + "github.com/containerd/nerdbox/plugins" +) + +func init() { + registry.Register(&plugin.Registration{ + Type: plugins.TaskPlugin, + ID: "manager", + Requires: []plugin.Type{ + cplugins.EventPlugin, + cplugins.InternalPlugin, + plugins.SandboxPlugin, + }, + InitFn: func(ic *plugin.InitContext) (interface{}, error) { + pp, err := ic.GetByID(cplugins.EventPlugin, "publisher") + if err != nil { + return nil, err + } + ss, err := ic.GetByID(cplugins.InternalPlugin, "shutdown") + if err != nil { + return nil, err + } + sbRaw, err := ic.GetSingle(plugins.SandboxPlugin) + if err != nil { + return nil, err + } + + svc, ok := sbRaw.(*intsandbox.SandboxService) + if !ok { + return nil, fmt.Errorf("unexpected SandboxPlugin implementation %T", sbRaw) + } + + // Determine debug flag from shim opts stored in context. + debug := false + if opts, ok := ic.Context.Value(shim.OptsKey{}).(shim.Opts); ok { + debug = opts.Debug + } + + // Wire the bundle-derived VM start options callback into the + // SandboxService so that StartSandbox can boot the VM with the + // correct resources and networking without importing the task package. + svc.RegisterStartOptions(task.SandboxStartOptions(debug)) + + return task.NewTaskService(ic.Context, svc, pp.(shim.Publisher), ss.(shutdown.Service)) + }, + }) +} diff --git a/plugins/types.go b/plugins/types.go index 1d546111..58eaed6f 100644 --- a/plugins/types.go +++ b/plugins/types.go @@ -31,6 +31,9 @@ const ( // StreamingPlugin implements a stream manager StreamingPlugin plugin.Type = "nerdbox.streaming.v1" + + // TaskPlugin implements the task interface + TaskPlugin plugin.Type = "nerdbox.task.v1" ) const ( diff --git a/script/userns-check/main.go b/script/userns-check/main.go index 14b27053..7cbe86a7 100644 --- a/script/userns-check/main.go +++ b/script/userns-check/main.go @@ -20,11 +20,15 @@ // getsockopt(SO_TYPE) returns EACCES when a unix socket fd is inherited // by a child spawned with CLONE_NEWUSER + a UID mapping + exec. // -// This reproduces the exact failure path in the nerdbox shim where +// This reproduces the failure path in the nerdbox shim where // net.FileListener calls getsockopt(fd, SOL_SOCKET, SO_TYPE) and gets EACCES. // // The exec is critical: it triggers capability recomputation. With euid != 0 -// in the new userns, caps drop to zero, and cross-userns socket access fails. +// in the new userns, caps drop to zero, and cross-userns socket access +// fails. This script maps container UID 0 to the host UID, the same +// mapping pkg/shim/manager.cloneMntNs uses (see that function's doc +// comment for the full explanation), to test whether user namespaces +// work at all in the current environment. // // Exit codes: // @@ -109,7 +113,8 @@ func parentMain() int { // Re-exec ourselves as "--child" with CLONE_NEWUSER|CLONE_NEWNS. // This is the same clone+exec pattern Go's ForkExec uses when // SysProcAttr.Cloneflags is set — which triggers cap recomputation. - // The UID/GID mappings mirror the shim's cloneMntNs implementation. + // The UID/GID mappings mirror the shim's cloneMntNs implementation + // (container UID/GID 0 mapped to the real host UID/GID). cmd := exec.Command("/proc/self/exe", "--child") cmd.Stdout = os.Stdout cmd.Stderr = os.Stderr @@ -117,10 +122,10 @@ func parentMain() int { cmd.SysProcAttr = &syscall.SysProcAttr{ Cloneflags: syscall.CLONE_NEWUSER | syscall.CLONE_NEWNS, UidMappings: []syscall.SysProcIDMap{ - {ContainerID: uid, HostID: uid, Size: 1}, + {ContainerID: 0, HostID: uid, Size: 1}, }, GidMappings: []syscall.SysProcIDMap{ - {ContainerID: gid, HostID: gid, Size: 1}, + {ContainerID: 0, HostID: gid, Size: 1}, }, } diff --git a/test/critest/.gitignore b/test/critest/.gitignore new file mode 100644 index 00000000..6ae4a15f --- /dev/null +++ b/test/critest/.gitignore @@ -0,0 +1,2 @@ +/.work/ +/dummy-pause.tar diff --git a/test/critest/README.md b/test/critest/README.md new file mode 100644 index 00000000..28f7884f --- /dev/null +++ b/test/critest/README.md @@ -0,0 +1,246 @@ +# CRI conformance harness (critest) + +This directory drives a dedicated containerd instance, configured with a +**RuntimeClass-style runtime handler** that uses this shim through +containerd's built-in **shim sandboxer** (`sandboxer = "shim"`, *not* the +`podsandbox` controller), through smoke tests and the full +[critest](https://github.com/kubernetes-sigs/cri-tools) (CRI conformance) +suite. + +See `docs/sandbox-architecture.md` for background on the shim sandboxer vs. +podsandbox distinction, and why this matters for the nerdbox shim. + +## Why a runtime handler + shim sandboxer, and not the default podsandbox path + +containerd's CRI plugin supports two ways to run a pod sandbox: + +- **podsandbox** (default): containerd's CRI layer builds the sandbox's OCI + spec itself and runs a real "pause" container for it via the ordinary + shim-v2 task API. +- **shim** (what this harness configures): containerd hands the sandbox + lifecycle entirely to the shim's own TTRPC sandbox controller + (`CreateSandbox`/`StartSandbox`/`StopSandbox`/`ShutdownSandbox`). This is + the API nerdbox actually implements (`internal/shim/sandbox/service.go`) — + one VM per pod, with member containers created afterward over the shim-v2 + task API on the same TTRPC connection. + +The runtime handler is configured with `sandboxer = "shim"` in +`run-critest.sh`'s generated `config.toml`. + +## Prerequisites + +- Linux host with `/dev/kvm` accessible. +- The nerdbox artifacts built into `_output/` at the repo root: + `containerd-shim-nerdbox-v1`, `nerdbox-kernel-x86_64`, + `nerdbox-rootfs.erofs`, `libkrun.so`. Build with: + ``` + task build:shim + DESTDIR=_output docker buildx bake kernel rootfs libkrun + ``` +- A **containerd binary built from source at the version pinned in + `go.mod`** (v2.3.2 as of writing) — not a distro package, and not an + older prebuilt release: the CRI plugin's config schema + (`[plugins.'io.containerd.cri.v1.runtime']`, split from + `io.containerd.cri.v1.images`) is version-specific. + ``` + git clone --branch v2.3.2 https://github.com/containerd/containerd.git + cd containerd && make binaries # produces bin/containerd, bin/ctr + ``` +- `crictl` and `critest` from + [cri-tools](https://github.com/kubernetes-sigs/cri-tools), built at the + version containerd itself pins for testing + (`script/setup/critools-version` in the containerd source, v1.35.0 as of + writing): + ``` + git clone --branch v1.35.0 https://github.com/kubernetes-sigs/cri-tools.git + cd cri-tools && make binaries # produces build/bin/linux/amd64/{crictl,critest} + ``` +- Standard CNI plugins (`bridge`, `loopback`, `host-local`, `portmap`) — + typically already present at `/opt/cni/bin` on a host that has ever run + Kubernetes or a CNI-based container runtime. Get them from + [containernetworking/plugins](https://github.com/containernetworking/plugins) + releases otherwise. +- `jq` (used by the smoke test to inspect pod status JSON). + +## Usage + +Point the script at your built tools via env vars (or put them on `PATH`), +then run one of the subcommands: + +```sh +export CONTAINERD_BIN=/path/to/containerd/bin/containerd +export CTR_BIN=/path/to/containerd/bin/ctr +export CRICTL_BIN=/path/to/cri-tools/build/bin/linux/amd64/crictl +export CRITEST_BIN=/path/to/cri-tools/build/bin/linux/amd64/critest + +sudo -E env PATH="$PATH" \ + CONTAINERD_BIN="$CONTAINERD_BIN" CTR_BIN="$CTR_BIN" \ + CRICTL_BIN="$CRICTL_BIN" CRITEST_BIN="$CRITEST_BIN" \ + ./run-critest.sh smoke # quick end-to-end sanity check +``` + +```sh +# same env, then: +./run-critest.sh critest # full CRI conformance suite +./run-critest.sh up # start containerd and leave it running +./run-critest.sh shell # start containerd, drop into a shell to poke at it with crictl +./run-critest.sh down # stop whatever "up" started +``` + +`critest` accepts extra args, passed straight through to the critest +binary's own (ginkgo/go test) flag parser — do not prepend a literal `--` +separator; ginkgo's flag parser itself treats a bare `--` as "stop parsing +flags", which silently disables everything after it, including +`--ginkgo.focus`/`--ginkgo.skip`: + +```sh +./run-critest.sh critest --ginkgo.focus="HostPID" # correct +./run-critest.sh critest -- --ginkgo.focus="HostPID" # wrong: focus silently ignored +``` + +By default, `critest` skips the 4 specs described in "Known conformance +gaps" below (permanent architectural limitations, not bugs). Pass +`--no-skip` to run the full, unfiltered suite and see them fail: + +```sh +./run-critest.sh critest --no-skip +``` + +Note: if you also pass your own `--ginkgo.focus` and it happens to match +one of the 4 default-skipped specs, you need `--no-skip` too, or it will +match zero specs. + +`sudo` is required: containerd's default root/state dirs and the CNI +bridge setup need it, matching how `crictl`/CRI integration tests are +normally run (see containerd's own `script/critest.sh` / +`script/test/cri-integration.sh` for the same pattern). + +Everything scratch-state lives under `test/critest/.work/` (gitignored): +`config.toml`, containerd's `root`/`state`, the containerd log, the dummy +pause image tar, CNI conf, and (for `smoke`) captured pod/container status +JSON. Inspect `.work/containerd.log` and `.work/critest-report/` after a +run. + +## The dummy pause image + +CRI's `RunPodSandbox` unconditionally calls `ensurePauseImageExists()` +before starting the sandbox, regardless of which sandboxer is configured. +On the shim-sandboxer path, however, the pause image is never actually +used: containerd's CRI `sandbox_run.go` only calls +`sandbox.WithOptions`/`WithNetNSPath` when creating the sandbox, never +`WithRootFS`, so the pause image only needs to *resolve* in containerd's +image store — it is never pulled by weight, unpacked, or run. + +`build-dummy-pause.sh` builds a deliberately non-functional OCI image (a +valid manifest + config, but an empty layer — no `/pause` binary, nothing +to execute) and `run-critest.sh` imports it under a pinned CRI +`sandbox_image` ref. Using a non-functional image is intentional: if +anything ever did try to actually run it, it would fail loudly instead of +silently working, which is a running proof that this shim's sandbox path +truly doesn't depend on it. The smoke test asserts this explicitly (no +snapshot is ever created for the dummy image, and the pod sandbox status +reports an empty `snapshotter`/`snapshotKey`). + +## Known conformance gaps + +A first full `critest` run found and fixed two real shim bugs blocking CRI +use entirely (see git history for `pkg/shim/manager` and +`internal/shim/sandbox/service.go` around this harness's introduction: a +missing-`config.json` crash at shim `Start`, and `SandboxStatus.State` not +matching the CRI `PodSandboxState` enum names), then a further round fixed +host bind-mount volumes for member containers, DNS config, hostname, and +sysctls (see git history for `internal/shim/sandbox/sharedfs.go`'s +`ShareVolume`, `internal/shim/task/sandboxvolumes.go`, and +`internal/shim/task/podconfig.go`), then a further round added pod-level +PID and IPC namespace sharing between member containers (see git history +for `internal/vminit/sharedresources` and `internal/shim/task/namespaces.go`'s +`sanitizeNamespaces`). + +**Current status (`--no-skip`, the full unfiltered suite): 85 passed / 4 +failed / 24 skipped.** (With the default skip list applied: 85 passed / 0 +failed / 28 skipped.) All 4 remaining failures are **genuine architectural +limitations** of the current design, not bugs, and are not expected to be +fixed without a fundamentally different sharing mechanism: + +- **`mount with 'rshared' should support propagation from host to + container and vice versa`**: this test checks *two* directions, and only + one of them actually fails. **Host→container works**: the test's own + setup (`createHostPathForMountPropagation`) explicitly bind-mounts the + volume's host source onto itself and marks it `MS_SHARED`, and per + `mount_namespaces(7)`, a later bind mount taken *from* an already-shared + mount joins the same peer group — which is exactly what + `SharedFS.ShareVolume`'s own (plain, non-private) bind mount does. So a + mount created on the host under the volume's source dir *after* the + container starts lands in the same host-kernel peer group as our + virtiofs-shared copy, and virtiofs (a live FUSE content server, not a + point-in-time snapshot) simply serves the now-updated content — this is + confirmed by the sibling test `mount with 'rslave' should support + propagation from host to container`, which tests only this direction + and **passes**. **Container→host is what actually fails**: a mount the + *container* creates (`mount --bind /etc containerMntPoint`, run inside + the guest) is a guest-kernel-internal operation. Virtio-fs's protocol + has no message for "a mount happened" — it only relays file/directory + *content* operations (open/read/readdir/etc.) — so there is no path by + which a guest-side `mount(2)` syscall could ever be observed by the host + kernel, regardless of any peer-group configuration on the host side. + This is a one-way, permanent limitation of the container→host direction + specifically, not of virtiofs-based propagation as a whole. +- **`should support non-recursive readonly mounts`**: this test mounts a + *separate, real* tmpfs on the host, nested inside a volume's source + directory, *before* the container starts (not a live-propagation + scenario — the nested mount already exists when `ShareVolume`'s + recursive (`rbind`) host-side bind mount runs), and expects the OCI + runtime to recognize the nested mount as a distinct kernel object and + leave its own read-write flag alone when the container's own bind mount + is non-recursive. `rbind` does duplicate the nested tmpfs as its own + mount object in our host-side copy — but virtiofs (like most tree-share + protocols) does not cross mount points while serving a shared directory + to the guest, so the guest simply sees `/mnt/tmpfs` as an ordinary + (flattened) subdirectory of `/mnt`, with no mount boundary at all. From + crun's point of view inside the guest there is only one mount to apply + non-recursive-readonly to, so `/mnt/tmpfs` inherits it along with + everything else. Related to, but distinct from, the `rshared` case + above: that one is about a guest-created mount never reaching the host; + this one is about a host-side nested mount boundary never reaching the + guest as a distinct mount object in the first place. +- **`runtime should support HostNetwork is true`**: this test runs + `netstat -ln` inside the container and expects the *host's own listening + socket* to literally appear in the output — true, introspectable network + stack sharing (the container sees the same socket table as the host), + not just outbound reachability. TSI (this shim's default outbound + networking — see docs/sandbox-architecture.md) proxies individual + outbound connections over vsock; it does not mirror the host's socket + table into the guest, so nothing the shim does with network namespaces + can satisfy this specific check. +- **`runtime should support HostIpc is true`**: this test creates a SysV + shared memory segment directly on the machine running `critest` (the + *real* host), before creating the pod sandbox, then expects a container + with `HostIpc: true` to see it. This shim runs every sandbox inside a + VM, so "the host" from the guest kernel's point of view is the guest's + own root IPC namespace — a different kernel instance entirely from the + machine `critest` is actually creating shm segments on. No IPC namespace + configuration inside the guest can make a segment that only exists in + the real host kernel visible there; it is the same category of + limitation as `HostNetwork is true` above (the guest is not the literal + host), just for SysV IPC instead of the socket table. Pod-level IPC + sharing *between member containers of the same sandbox* — the far more + common Kubernetes use case (pods share IPC by default) — works + correctly and is covered by shimtest's `MemberContainersShareIPC`. + +These 4 are wired into `run-critest.sh`'s `DEFAULT_SKIP_SPECS`, which +`critest` applies by default (pass `--no-skip` to see them fail) — see +"Usage" above. Keep that list and this section in sync if either changes; +ask before assuming any *other* failure is out of scope for follow-up +work. + +For comparison: Kata Containers, the most mature production VM-isolated +CRI runtime, does not run the upstream `critest` `[k8s.io]` validation +suite in CI at all. Its containerd `cri-integration` job uses an explicit +*allowlist* of the handful of Go tests it knows pass +(`FOCUS="^(TestContainerStats|TestImageLoad|...)$"`), each exclusion +documented inline with its own rationale (e.g. its `TestContainerRestart` +exclusion notes that starting a new container in an already-torn-down +sandbox VM "has never been supported by kata-containers"). That is the +same category of reasoning as the 4 specs here: tests that assume +shared-kernel/host-visibility semantics no VM-isolated runtime can +provide, excluded and documented rather than chased as bugs. diff --git a/test/critest/build-dummy-pause.sh b/test/critest/build-dummy-pause.sh new file mode 100755 index 00000000..ceca1653 --- /dev/null +++ b/test/critest/build-dummy-pause.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash + +# Copyright The containerd Authors. + +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at + +# http://www.apache.org/licenses/LICENSE-2.0 + +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# build-dummy-pause.sh builds a deliberately non-functional OCI image and +# writes it as an importable tar (OCI image layout) to $OUT (default: +# ./dummy-pause.tar next to this script). +# +# Why a dummy image at all: CRI's RunPodSandbox unconditionally calls +# ensurePauseImageExists() before starting the sandbox, regardless of which +# sandboxer is configured. On the *shim* sandboxer path (which is what +# nerdbox uses — see docs/sandbox-architecture.md), the pause image is never +# actually mounted or run: CRI's sandbox_run.go only calls +# sandbox.WithOptions/WithNetNSPath when creating the sandbox, never +# WithRootFS, so CreateSandboxRequest.Rootfs arrives empty at the shim. +# ensurePauseImageExists only needs the ref to *resolve locally* in +# containerd's image store (a manifest + config blob reachable in the +# content store) — it does not need to be pulled, unpacked, or runnable. +# +# Why deliberately non-functional (no /pause binary, empty layer): if +# anything ever DID try to actually run this image, it would fail loudly +# instead of silently working — proof that nerdbox's shim-sandbox path +# truly does not depend on the pause image. +# +# Usage: build-dummy-pause.sh [output-tar-path] [image-ref] +set -euo pipefail + +SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" > /dev/null 2>&1; pwd -P)" +OUT="${1:-${SCRIPT_DIR}/dummy-pause.tar}" +REF="${2:-nerdbox.local/dummy-pause:1}" + +WORK="$(mktemp -d)" +trap 'rm -rf "${WORK}"' EXIT + +OCIDIR="${WORK}/oci" +BLOBS="${OCIDIR}/blobs/sha256" +mkdir -p "${BLOBS}" + +echo '{"imageLayoutVersion": "1.0.0"}' > "${OCIDIR}/oci-layout" + +# --- Empty layer: a well-formed, but empty, tar+gzip. Real tar/gzip tools +# so the blob is format-valid (in case any tooling ever inspects it), while +# containing zero files — nothing to unpack, nothing to execute. +LAYER_TAR="${WORK}/layer.tar" +tar --create --file="${LAYER_TAR}" --files-from=/dev/null +DIFF_ID="sha256:$(sha256sum "${LAYER_TAR}" | awk '{print $1}')" + +LAYER_GZ="${WORK}/layer.tar.gz" +gzip -n -c "${LAYER_TAR}" > "${LAYER_GZ}" +LAYER_DIGEST="$(sha256sum "${LAYER_GZ}" | awk '{print $1}')" +LAYER_SIZE="$(stat -c%s "${LAYER_GZ}")" +cp "${LAYER_GZ}" "${BLOBS}/${LAYER_DIGEST}" + +# --- Image config: minimal valid OCI image config. No Entrypoint/Cmd — +# there is nothing in the (empty) rootfs to exec anyway. +CONFIG_JSON="${WORK}/config.json" +cat > "${CONFIG_JSON}" < "${MANIFEST_JSON}" < "${INDEX_JSON}" < crictl lifecycle smoke test -> down (always) +# run-critest.sh critest [ARGS] # up -> critest --runtime-handler=nerdbox [ARGS] -> down (always) +# run-critest.sh shell # up, then drop into a shell with env set for manual crictl use +# +# ARGS are passed straight through to the critest binary's own (ginkgo/go +# test) flag parser — do NOT prepend a literal "--" separator: a bare "--" +# is itself consumed by that parser as "stop parsing flags", which silently +# disables every flag after it (including --ginkgo.focus/--ginkgo.skip). +# e.g.: run-critest.sh critest --ginkgo.focus="HostPID" +# +# By default, "critest" skips a small, fixed set of specs that are known, +# permanent architectural limitations of running each sandbox in its own VM +# (not implementation bugs) — see README.md's "Known conformance gaps" for +# what they are and why. Pass --no-skip to run the full, unfiltered suite +# and see them fail. Note --no-skip and --ginkgo.focus/--ginkgo.skip compose +# via ginkgo's normal flag semantics: if ARGS also specifies --ginkgo.skip, +# that value applies (skipping is not additionally layered in that case). +# +# Env vars (all optional, defaults shown): +# NERDBOX_OUTPUT_DIR repo _output/ dir (shim, kernel, rootfs, libkrun.so) [/_output] +# CONTAINERD_BIN path to a containerd binary (built from source) [containerd on PATH] +# CTR_BIN path to ctr [ctr on PATH] +# CRICTL_BIN path to crictl [crictl on PATH] +# CRITEST_BIN path to critest [critest on PATH] +# CNI_BIN_DIR directory with bridge/loopback/host-local/portmap [/opt/cni/bin] +# RUNTIME_HANDLER CRI runtime handler to exercise [nerdbox] +# WORK_DIR scratch dir for root/state/socket/logs/CNI conf [/.work] +# KEEP_WORK_DIR if set to 1, don't delete WORK_DIR content on "down" +set -euo pipefail + +SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" > /dev/null 2>&1; pwd -P)" +REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../.." > /dev/null 2>&1; pwd -P)" + +NERDBOX_OUTPUT_DIR="${NERDBOX_OUTPUT_DIR:-${REPO_ROOT}/_output}" +CONTAINERD_BIN="${CONTAINERD_BIN:-containerd}" +CTR_BIN="${CTR_BIN:-ctr}" +CRICTL_BIN="${CRICTL_BIN:-crictl}" +CRITEST_BIN="${CRITEST_BIN:-critest}" +CNI_BIN_DIR="${CNI_BIN_DIR:-/opt/cni/bin}" +RUNTIME_HANDLER="${RUNTIME_HANDLER:-nerdbox}" +WORK_DIR="${WORK_DIR:-${SCRIPT_DIR}/.work}" +KEEP_WORK_DIR="${KEEP_WORK_DIR:-0}" + +SOCK="${WORK_DIR}/c.sock" +PIDFILE="${WORK_DIR}/containerd.pid" +CONFIG="${WORK_DIR}/config.toml" +CNI_CONF_DIR="${WORK_DIR}/cni/net.d" +DUMMY_PAUSE_TAR="${WORK_DIR}/dummy-pause.tar" +DUMMY_PAUSE_REF="nerdbox.local/dummy-pause:1" + +log() { echo "[run-critest] $*" >&2; } +die() { log "ERROR: $*"; exit 1; } + +require_bin() { + local name="$1" path="$2" + if [[ "${path}" == */* ]]; then + [[ -x "${path}" ]] || die "$name not found or not executable: ${path}" + else + command -v "${path}" > /dev/null 2>&1 || die "$name not found on PATH: ${path} (set ${name^^}_BIN or add it to PATH)" + fi +} + +check_prereqs() { + require_bin containerd "${CONTAINERD_BIN}" + require_bin ctr "${CTR_BIN}" + require_bin crictl "${CRICTL_BIN}" + require_bin jq jq + [[ -c /dev/kvm ]] || die "/dev/kvm not found; the nerdbox shim needs KVM" + for f in containerd-shim-nerdbox-v1 nerdbox-kernel-x86_64 nerdbox-rootfs.erofs libkrun.so; do + [[ -e "${NERDBOX_OUTPUT_DIR}/${f}" ]] || die "missing ${f} in NERDBOX_OUTPUT_DIR=${NERDBOX_OUTPUT_DIR} (build it first, see README.md)" + done + for p in bridge loopback host-local portmap; do + [[ -x "${CNI_BIN_DIR}/${p}" ]] || die "missing CNI plugin ${p} in CNI_BIN_DIR=${CNI_BIN_DIR}" + done +} + +gen_cni_conf() { + mkdir -p "${CNI_CONF_DIR}" + cp "${SCRIPT_DIR}/cni-net.conflist" "${CNI_CONF_DIR}/10-nerdbox-critest.conflist" +} + +gen_dummy_pause() { + [[ -f "${DUMMY_PAUSE_TAR}" ]] || "${SCRIPT_DIR}/build-dummy-pause.sh" "${DUMMY_PAUSE_TAR}" "${DUMMY_PAUSE_REF}" +} + +gen_config() { + mkdir -p "${WORK_DIR}/root" "${WORK_DIR}/state" + cat > "${CONFIG}" </dev/null && die "containerd already running (pid $(cat "${PIDFILE}")); run 'down' first" + + mkdir -p "${WORK_DIR}" + check_prereqs + gen_cni_conf + gen_dummy_pause + gen_config + + log "starting containerd (log: ${WORK_DIR}/containerd.log)" + # PATH must carry the nerdbox artifacts (shim binary, libkrun.so, kernel, + # rootfs) so internal/vm/libkrun's PATH/LIBKRUN_PATH search finds them — + # see internal/vm/libkrun/instance.go. + PATH="${NERDBOX_OUTPUT_DIR}:${PATH}" \ + setsid "${CONTAINERD_BIN}" --config "${CONFIG}" \ + > "${WORK_DIR}/containerd.log" 2>&1 < /dev/null & + echo $! > "${PIDFILE}" + disown || true + + log "waiting for ${SOCK}" + for _ in $(seq 1 100); do + [[ -S "${SOCK}" ]] && "${CTR_BIN}" --address "${SOCK}" version > /dev/null 2>&1 && break + sleep 0.2 + done + "${CTR_BIN}" --address "${SOCK}" version > /dev/null 2>&1 || die "containerd did not become ready; see ${WORK_DIR}/containerd.log" + + log "importing dummy pause image into k8s.io namespace" + "${CTR_BIN}" --address "${SOCK}" -n k8s.io images import "${DUMMY_PAUSE_TAR}" > /dev/null + + log "up: pid=$(cat "${PIDFILE}") sock=${SOCK}" +} + +stop_containerd() { + if [[ -f "${PIDFILE}" ]]; then + local pid + pid="$(cat "${PIDFILE}")" + if kill -0 "${pid}" 2>/dev/null; then + log "stopping containerd (pid ${pid})" + kill "${pid}" 2>/dev/null || true + for _ in $(seq 1 50); do + kill -0 "${pid}" 2>/dev/null || break + sleep 0.2 + done + kill -9 "${pid}" 2>/dev/null || true + fi + rm -f "${PIDFILE}" + fi + # Best-effort: reap any leaked nerdbox shim/VM processes from this run. + pkill -9 -f "containerd-shim-nerdbox-v1.*${SOCK}" 2>/dev/null || true + + if [[ "${KEEP_WORK_DIR}" != "1" ]]; then + rm -rf "${WORK_DIR}/root" "${WORK_DIR}/state" + fi +} + +crictl_() { + "${CRICTL_BIN}" --runtime-endpoint "unix://${SOCK}" --image-endpoint "unix://${SOCK}" "$@" +} + +cmd_smoke() { + local sandbox_json container_json podid cid out + + sandbox_json="${WORK_DIR}/smoke-sandbox.json" + container_json="${WORK_DIR}/smoke-container.json" + + cat > "${sandbox_json}" < "${container_json}" <<'EOF' +{ + "metadata": {"name": "smoke"}, + "image": {"image": "docker.io/library/busybox:latest"}, + "command": ["sleep", "3600"], + "log_path": "smoke.log" +} +EOF + + # --with-pull: the image is pulled as part of CreateContainer, scoped to + # this pod's sandbox (podid) — which resolves to the "nerdbox" runtime's + # snapshotter (erofs) via CRIImageService.RuntimeSnapshotter, the same + # path container_create.go uses for the real snapshot/mount setup. No + # separate "crictl pull" step (and no --runtime-platform flag, which + # doesn't exist) is needed. + log "CreateContainer (pulls busybox via the nerdbox/erofs snapshotter)" + cid="$(crictl_ create --with-pull "${podid}" "${container_json}" "${sandbox_json}")" + log "container: ${cid}" + + log "StartContainer" + crictl_ start "${cid}" + + log "ExecSync" + out="$(crictl_ exec "${cid}" echo smoke-ok)" + [[ "${out}" == *smoke-ok* ]] || die "unexpected exec output: ${out}" + log "exec output: ${out}" + + log "container/pod status" + crictl_ inspect "${cid}" > "${WORK_DIR}/smoke-container-inspect.json" + crictl_ inspectp "${podid}" > "${WORK_DIR}/smoke-pod-inspect.json" + + # --- Verify the shim-sandboxer path was actually used, and that the + # dummy pause image is exactly as inert as intended: CRI's + # ensurePauseImageExists only needs it to *resolve*, and the shim + # sandboxer never mounts a sandbox rootfs at all (see + # build-dummy-pause.sh and docs/sandbox-architecture.md). Confirm both: + local pod_snapshotter pod_snapshot_key + pod_snapshotter="$(jq -r '.info.snapshotter // ""' < "${WORK_DIR}/smoke-pod-inspect.json")" + pod_snapshot_key="$(jq -r '.info.snapshotKey // ""' < "${WORK_DIR}/smoke-pod-inspect.json")" + if [[ -n "${pod_snapshotter}" || -n "${pod_snapshot_key}" ]]; then + die "pod sandbox unexpectedly has a snapshotter/rootfs (snapshotter=${pod_snapshotter} snapshotKey=${pod_snapshot_key}); expected empty on the shim-sandboxer path" + fi + log "confirmed: pod sandbox has no snapshotter/rootfs (shim-sandboxer path, not podsandbox)" + + if "${CTR_BIN}" --address "${SOCK}" -n k8s.io snapshots --snapshotter erofs ls 2>/dev/null | grep -q "${DUMMY_PAUSE_REF}"; then + die "dummy pause image was unexpectedly unpacked (a snapshot exists for it)" + fi + log "confirmed: dummy pause image was never unpacked (no snapshot exists for it)" + + log "StopContainer / RemoveContainer / StopPodSandbox / RemovePodSandbox" + crictl_ stop "${cid}" + crictl_ rm "${cid}" + crictl_ stopp "${podid}" + crictl_ rmp "${podid}" + + log "SMOKE TEST PASSED" +} + +# DEFAULT_SKIP_SPECS are critest specs that are known, permanent +# architectural limitations of running each sandbox in its own VM kernel — +# not implementation bugs — so they are skipped by default. Each one's +# setup mutates state on the literal machine `critest` runs on (the real +# host) and then expects a container running inside a *different* kernel +# (the guest) to observe that mutation, which a VM-isolated runtime cannot +# ever do without abandoning that isolation. See README.md's "Known +# conformance gaps" for the detailed root-cause analysis of each one. +DEFAULT_SKIP_SPECS=( + "runtime should support HostIpc is true" + "runtime should support HostNetwork is true" + "mount with 'rshared' should support propagation from host to container and vice versa" + "should support non-recursive readonly mounts" +) + +join_regex() { + local IFS='|' + echo "(${*})" +} + +cmd_critest() { + local no_skip=0 + local args=() + for a in "$@"; do + if [[ "${a}" == "--no-skip" ]]; then + no_skip=1 + else + args+=("${a}") + fi + done + + local skip_flag=() + if [[ "${no_skip}" != "1" ]]; then + skip_flag=(--ginkgo.skip="$(join_regex "${DEFAULT_SKIP_SPECS[@]}")") + log "skipping ${#DEFAULT_SKIP_SPECS[@]} known architectural-limitation specs (see README.md; pass --no-skip to run them anyway)" + fi + + log "running critest --runtime-handler=${RUNTIME_HANDLER}" + "${CRITEST_BIN}" \ + --runtime-endpoint "unix://${SOCK}" \ + --image-endpoint "unix://${SOCK}" \ + --runtime-handler "${RUNTIME_HANDLER}" \ + --report-dir "${WORK_DIR}/critest-report" \ + "${skip_flag[@]}" \ + "${args[@]}" +} + +main() { + local sub="${1:-}" + [[ $# -gt 0 ]] && shift || true + + case "${sub}" in + up) + start_containerd + ;; + down) + stop_containerd + ;; + smoke) + trap stop_containerd EXIT + start_containerd + cmd_smoke + ;; + critest) + trap stop_containerd EXIT + start_containerd + cmd_critest "$@" + ;; + shell) + start_containerd + log "environment ready; sock=${SOCK}" + log "example: crictl --runtime-endpoint unix://${SOCK} --image-endpoint unix://${SOCK} info" + CRICTL_SOCK="${SOCK}" bash -i + ;; + *) + die "usage: $0 {up|down|smoke|critest [ARGS]|shell}" + ;; + esac +} + +main "$@" diff --git a/test/shim/shim_test.go b/test/shim/shim_test.go index 4c726156..c51351d0 100644 --- a/test/shim/shim_test.go +++ b/test/shim/shim_test.go @@ -79,6 +79,7 @@ func TestMain(m *testing.M) { // // -run TestShim/Exec // -run TestShim/Lifecycle +// -run TestShim/Sandbox // // LayersSuite (HundredLayers) packs 101 erofs layers into a single // GPT-partitioned VMDK, consuming only one virtio-blk device regardless @@ -89,6 +90,10 @@ func TestMain(m *testing.M) { // to provide it. This is the regression guard for TSI (Transparent Socket // Impersonation), the default connectivity path for containers started // without any network configuration. +// +// SandboxSuite verifies the containerd sandbox shim API contract +// (runtime/sandbox/v1): lifecycle, platform, ping, single and multiple +// member containers, per-container independence, and error cases. func TestShim(t *testing.T) { cfg := shimConfig() shimtest.NewRunSuite(cfg).Run(t) @@ -98,6 +103,26 @@ func TestShim(t *testing.T) { shimtest.NewUDSSuite(cfg).Run(t) shimtest.NewLayersSuite(cfg).Run(t) shimtest.NewNetworkSuite(cfg).Run(t) + shimtest.NewSandboxSuite(cfg).Run(t) +} + +// BenchmarkShim runs the shimtest benchmark suites against the nerdbox shim. +// Run individual benchmarks with -bench, e.g.: +// +// go test -bench 'BenchmarkShim/Lifecycle' -benchtime 5x ./test/shim/... +// go test -bench 'BenchmarkShim/ContainerCreate' -benchtime 5x ./test/shim/... +// +// ContainerCreate benchmarks the per-container create/start/wait/delete cycle +// inside a shared sandbox VM; Lifecycle benchmarks the full shim-start + +// single-container cycle. Comparing their ms/create and ms/total metrics +// shows the marginal cost of adding a container to an existing sandbox versus +// booting a fresh VM. +func BenchmarkShim(b *testing.B) { + cfg := shimConfig() + shimtest.NewRunSuite(cfg).Bench(b) + shimtest.NewExecSuite(cfg).Bench(b) + shimtest.NewLayersSuite(cfg).Bench(b) + shimtest.NewSandboxSuite(cfg).Bench(b) } // FuzzTransferMissing exercises the transfer service with arbitrary paths @@ -109,21 +134,13 @@ func FuzzTransferMissing(f *testing.F) { // shimPath returns a PATH value that prepends candidate _output directories // to the current PATH. The local module _output/ is highest priority, followed -// by sibling worktree _output/ directories (to find kernel/initrd/libkrun built -// in another branch worktree). +// by sibling worktree _output/ directories that do NOT contain a libkrun.so — +// those are included for kernel/rootfs/vminitd assets only. Sibling _output +// dirs that carry a libkrun.so are skipped to prevent the shim from resolving +// a stale libkrun built in another worktree. func shimPath() string { root := moduleRoot() current := os.Getenv("PATH") - - // Build the final PATH as an ordered, deduplicated list: - // 1. local _output (always first, re-anchored even if already present) - // 2. sibling worktree _output dirs (fallback for kernel/initrd/libkrun) - // 3. everything already in PATH, minus any entries already added above - // - // The local _output must be unconditionally first: shimtest helpers call - // os.Setenv to inject it into the test-process PATH between tests, so by - // the time shimPath is called again it may already be present — but - // sibling dirs may also have been added and could sort ahead of it. localOutput := filepath.Join(root, "_output") parent := filepath.Dir(root) @@ -133,7 +150,13 @@ func shimPath() string { if !e.IsDir() || e.Name() == filepath.Base(root) { continue } - siblingOutputs = append(siblingOutputs, filepath.Join(parent, e.Name(), "_output")) + dir := filepath.Join(parent, e.Name(), "_output") + // Skip sibling _output dirs that have their own libkrun.so; + // using a stale libkrun can cause symbol-not-found crashes. + if _, err := os.Stat(filepath.Join(dir, "libkrun.so")); err == nil { + continue + } + siblingOutputs = append(siblingOutputs, dir) } } @@ -147,17 +170,17 @@ func shimPath() string { } } - // 1. Local _output first (exists check; silently skip if missing). + // 1. Local _output first. if _, err := os.Stat(localOutput); err == nil { add(localOutput) } - // 2. Sibling _output dirs that exist and haven't been added yet. + // 2. Sibling _output dirs without libkrun.so (kernel/rootfs fallback). for _, dir := range siblingOutputs { if _, err := os.Stat(dir); err == nil { add(dir) } } - // 3. Retain existing PATH entries not already included above. + // 3. Retain existing PATH entries. for _, dir := range filepath.SplitList(current) { add(dir) } diff --git a/test/stress/stress_test.go b/test/stress/stress_test.go index fc4a85c1..070bf669 100644 --- a/test/stress/stress_test.go +++ b/test/stress/stress_test.go @@ -76,61 +76,86 @@ func TestMain(m *testing.M) { } // TestShimStress runs the shimtest stress suites against the nerdbox shim. -// Each subtest (Lifecycle, Exec, Transfer) runs until one minute before the -// -test.timeout deadline. Select individual subtests with -run: +// Each subtest (Lifecycle, Exec, Transfer, Sandbox) runs until one minute +// before the -test.timeout deadline. Select individual subtests with -run: // // -run TestShimStress/Lifecycle // -run TestShimStress/Exec // -run TestShimStress/Transfer +// -run TestShimStress/Sandbox func TestShimStress(t *testing.T) { shimtest.NewStressSuite(shimConfig(), shimtest.StressOptions{ Transfer: true, + Sandbox: true, + // The sandbox stress test creates thousands of container lifecycles + // inside a single VM (default 2 GiB guest RAM). The host shim's RSS + // grows as the VM progressively faults in guest pages and the Go + // runtime's heap settles at a high watermark. This one-time step + // saturates well below 2 GiB (the full guest RAM) and is not a leak. + // + // Observed nerdbox data over a 21-minute / 45K-container run: + // RSS before: ~112 MiB (just sandbox booted) + // RSS after: ~1270 MiB (~20 min) + // Growth from guest RAM faults saturates; the rate drops after the + // first few minutes. 3 GiB accommodates this one-time step with + // headroom; a true per-container leak at the observed ~3 KiB/iter + // rate would cross 3 GiB only after ~1 million containers. + SandboxRSSGrowthOverride: 3 * 1024 * 1024 * 1024, // 3 GiB }).Run(t) } // shimPath returns a PATH value that prepends candidate _output directories -// to the current PATH. The local module _output/ is highest priority, followed -// by sibling worktree _output/ directories (to find kernel/initrd/libkrun built -// in another branch worktree). +// to the current PATH. The local module _output/ is always first. Sibling +// worktree _output/ directories are included for kernel/rootfs/vminitd +// fallback, but any sibling that contains its own libkrun.so is skipped: +// using a stale libkrun from another worktree can cause symbol-not-found +// crashes (e.g. missing krun_add_virtiofs3). func shimPath() string { root := moduleRoot() current := os.Getenv("PATH") + localOutput := filepath.Join(root, "_output") - var candidates []string - candidates = append(candidates, filepath.Join(root, "_output")) - - // Walk sibling worktrees: the parent of root is the common worktree parent. parent := filepath.Dir(root) + var siblingOutputs []string if entries, err := os.ReadDir(parent); err == nil { for _, e := range entries { if !e.IsDir() || e.Name() == filepath.Base(root) { continue } - candidates = append(candidates, filepath.Join(parent, e.Name(), "_output")) + dir := filepath.Join(parent, e.Name(), "_output") + // Skip sibling _output dirs that carry their own libkrun.so. + if _, err := os.Stat(filepath.Join(dir, "libkrun.so")); err == nil { + continue + } + siblingOutputs = append(siblingOutputs, dir) } } - // Build a set of existing PATH elements for exact membership tests. - existing := make(map[string]bool) - for _, e := range filepath.SplitList(current) { - existing[e] = true + seen := make(map[string]bool) + var result []string + add := func(dir string) { + if !seen[dir] { + seen[dir] = true + result = append(result, dir) + } } - var prepend []string - for _, dir := range candidates { - if _, err := os.Stat(dir); err != nil { - continue - } - if existing[dir] { - continue + // 1. Local _output first. + if _, err := os.Stat(localOutput); err == nil { + add(localOutput) + } + // 2. Sibling _output dirs without libkrun.so (kernel/rootfs fallback). + for _, dir := range siblingOutputs { + if _, err := os.Stat(dir); err == nil { + add(dir) } - prepend = append(prepend, dir) } - if len(prepend) == 0 { - return current + // 3. Retain existing PATH entries. + for _, dir := range filepath.SplitList(current) { + add(dir) } - return strings.Join(prepend, string(os.PathListSeparator)) + - string(os.PathListSeparator) + current + + return strings.Join(result, string(os.PathListSeparator)) } // moduleRoot returns the absolute path to the module root directory. diff --git a/test/testbin/main.go b/test/testbin/main.go index 22273bf9..969097f6 100644 --- a/test/testbin/main.go +++ b/test/testbin/main.go @@ -1,3 +1,5 @@ +//go:build linux + /* Copyright The containerd Authors. @@ -21,7 +23,11 @@ // // This main is intentionally a one-liner: all logic lives in the importable // github.com/containerd/shimtest/testbin package so it is covered by -// go mod vendor and stays in sync with the vendored shimtest version. +// go mod vendor and stays in sync with the vendored shimtest version. That +// package is itself linux-only (it exercises linux-specific syscalls with +// no portable equivalents), so this file must be restricted to linux too -- +// otherwise cross-platform builds and lint runs that typecheck the whole +// module fail trying to compile it for other GOOS values. package main import "github.com/containerd/shimtest/testbin" diff --git a/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/doc.go b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/doc.go new file mode 100644 index 00000000..f960350c --- /dev/null +++ b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/doc.go @@ -0,0 +1,17 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package sandbox diff --git a/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox.pb.go b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox.pb.go new file mode 100644 index 00000000..38c4239d --- /dev/null +++ b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox.pb.go @@ -0,0 +1,1647 @@ +// +//Copyright The containerd Authors. +// +//Licensed under the Apache License, Version 2.0 (the "License"); +//you may not use this file except in compliance with the License. +//You may obtain a copy of the License at +// +//http://www.apache.org/licenses/LICENSE-2.0 +// +//Unless required by applicable law or agreed to in writing, software +//distributed under the License is distributed on an "AS IS" BASIS, +//WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +//See the License for the specific language governing permissions and +//limitations under the License. + +// Code generated by protoc-gen-go. DO NOT EDIT. +// versions: +// protoc-gen-go v1.28.1 +// protoc (unknown) +// source: runtime/sandbox/v1/sandbox.proto + +package sandbox + +import ( + types "github.com/containerd/containerd/api/types" + protoreflect "google.golang.org/protobuf/reflect/protoreflect" + protoimpl "google.golang.org/protobuf/runtime/protoimpl" + anypb "google.golang.org/protobuf/types/known/anypb" + timestamppb "google.golang.org/protobuf/types/known/timestamppb" + reflect "reflect" + sync "sync" +) + +const ( + // Verify that this generated code is sufficiently up-to-date. + _ = protoimpl.EnforceVersion(20 - protoimpl.MinVersion) + // Verify that runtime/protoimpl is sufficiently up-to-date. + _ = protoimpl.EnforceVersion(protoimpl.MaxVersion - 20) +) + +type CreateSandboxRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` + BundlePath string `protobuf:"bytes,2,opt,name=bundle_path,json=bundlePath,proto3" json:"bundle_path,omitempty"` + Rootfs []*types.Mount `protobuf:"bytes,3,rep,name=rootfs,proto3" json:"rootfs,omitempty"` + Options *anypb.Any `protobuf:"bytes,4,opt,name=options,proto3" json:"options,omitempty"` + NetnsPath string `protobuf:"bytes,5,opt,name=netns_path,json=netnsPath,proto3" json:"netns_path,omitempty"` + Annotations map[string]string `protobuf:"bytes,6,rep,name=annotations,proto3" json:"annotations,omitempty" protobuf_key:"bytes,1,opt,name=key,proto3" protobuf_val:"bytes,2,opt,name=value,proto3"` +} + +func (x *CreateSandboxRequest) Reset() { + *x = CreateSandboxRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[0] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CreateSandboxRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CreateSandboxRequest) ProtoMessage() {} + +func (x *CreateSandboxRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[0] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CreateSandboxRequest.ProtoReflect.Descriptor instead. +func (*CreateSandboxRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{0} +} + +func (x *CreateSandboxRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +func (x *CreateSandboxRequest) GetBundlePath() string { + if x != nil { + return x.BundlePath + } + return "" +} + +func (x *CreateSandboxRequest) GetRootfs() []*types.Mount { + if x != nil { + return x.Rootfs + } + return nil +} + +func (x *CreateSandboxRequest) GetOptions() *anypb.Any { + if x != nil { + return x.Options + } + return nil +} + +func (x *CreateSandboxRequest) GetNetnsPath() string { + if x != nil { + return x.NetnsPath + } + return "" +} + +func (x *CreateSandboxRequest) GetAnnotations() map[string]string { + if x != nil { + return x.Annotations + } + return nil +} + +type CreateSandboxResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields +} + +func (x *CreateSandboxResponse) Reset() { + *x = CreateSandboxResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[1] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *CreateSandboxResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*CreateSandboxResponse) ProtoMessage() {} + +func (x *CreateSandboxResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[1] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use CreateSandboxResponse.ProtoReflect.Descriptor instead. +func (*CreateSandboxResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{1} +} + +type StartSandboxRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` +} + +func (x *StartSandboxRequest) Reset() { + *x = StartSandboxRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[2] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *StartSandboxRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StartSandboxRequest) ProtoMessage() {} + +func (x *StartSandboxRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[2] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StartSandboxRequest.ProtoReflect.Descriptor instead. +func (*StartSandboxRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{2} +} + +func (x *StartSandboxRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +type StartSandboxResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + Pid uint32 `protobuf:"varint,1,opt,name=pid,proto3" json:"pid,omitempty"` + CreatedAt *timestamppb.Timestamp `protobuf:"bytes,2,opt,name=created_at,json=createdAt,proto3" json:"created_at,omitempty"` + Spec *anypb.Any `protobuf:"bytes,3,opt,name=spec,proto3" json:"spec,omitempty"` +} + +func (x *StartSandboxResponse) Reset() { + *x = StartSandboxResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[3] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *StartSandboxResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StartSandboxResponse) ProtoMessage() {} + +func (x *StartSandboxResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[3] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StartSandboxResponse.ProtoReflect.Descriptor instead. +func (*StartSandboxResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{3} +} + +func (x *StartSandboxResponse) GetPid() uint32 { + if x != nil { + return x.Pid + } + return 0 +} + +func (x *StartSandboxResponse) GetCreatedAt() *timestamppb.Timestamp { + if x != nil { + return x.CreatedAt + } + return nil +} + +func (x *StartSandboxResponse) GetSpec() *anypb.Any { + if x != nil { + return x.Spec + } + return nil +} + +type PlatformRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` +} + +func (x *PlatformRequest) Reset() { + *x = PlatformRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[4] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *PlatformRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*PlatformRequest) ProtoMessage() {} + +func (x *PlatformRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[4] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use PlatformRequest.ProtoReflect.Descriptor instead. +func (*PlatformRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{4} +} + +func (x *PlatformRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +type PlatformResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + Platform *types.Platform `protobuf:"bytes,1,opt,name=platform,proto3" json:"platform,omitempty"` +} + +func (x *PlatformResponse) Reset() { + *x = PlatformResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[5] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *PlatformResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*PlatformResponse) ProtoMessage() {} + +func (x *PlatformResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[5] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use PlatformResponse.ProtoReflect.Descriptor instead. +func (*PlatformResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{5} +} + +func (x *PlatformResponse) GetPlatform() *types.Platform { + if x != nil { + return x.Platform + } + return nil +} + +type StopSandboxRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` + TimeoutSecs uint32 `protobuf:"varint,2,opt,name=timeout_secs,json=timeoutSecs,proto3" json:"timeout_secs,omitempty"` +} + +func (x *StopSandboxRequest) Reset() { + *x = StopSandboxRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[6] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *StopSandboxRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StopSandboxRequest) ProtoMessage() {} + +func (x *StopSandboxRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[6] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StopSandboxRequest.ProtoReflect.Descriptor instead. +func (*StopSandboxRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{6} +} + +func (x *StopSandboxRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +func (x *StopSandboxRequest) GetTimeoutSecs() uint32 { + if x != nil { + return x.TimeoutSecs + } + return 0 +} + +type StopSandboxResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields +} + +func (x *StopSandboxResponse) Reset() { + *x = StopSandboxResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[7] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *StopSandboxResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StopSandboxResponse) ProtoMessage() {} + +func (x *StopSandboxResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[7] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StopSandboxResponse.ProtoReflect.Descriptor instead. +func (*StopSandboxResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{7} +} + +type UpdateSandboxRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` + Resources *anypb.Any `protobuf:"bytes,2,opt,name=resources,proto3" json:"resources,omitempty"` + Annotations map[string]string `protobuf:"bytes,3,rep,name=annotations,proto3" json:"annotations,omitempty" protobuf_key:"bytes,1,opt,name=key,proto3" protobuf_val:"bytes,2,opt,name=value,proto3"` +} + +func (x *UpdateSandboxRequest) Reset() { + *x = UpdateSandboxRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[8] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *UpdateSandboxRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*UpdateSandboxRequest) ProtoMessage() {} + +func (x *UpdateSandboxRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[8] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use UpdateSandboxRequest.ProtoReflect.Descriptor instead. +func (*UpdateSandboxRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{8} +} + +func (x *UpdateSandboxRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +func (x *UpdateSandboxRequest) GetResources() *anypb.Any { + if x != nil { + return x.Resources + } + return nil +} + +func (x *UpdateSandboxRequest) GetAnnotations() map[string]string { + if x != nil { + return x.Annotations + } + return nil +} + +type WaitSandboxRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` +} + +func (x *WaitSandboxRequest) Reset() { + *x = WaitSandboxRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[9] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *WaitSandboxRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*WaitSandboxRequest) ProtoMessage() {} + +func (x *WaitSandboxRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[9] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use WaitSandboxRequest.ProtoReflect.Descriptor instead. +func (*WaitSandboxRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{9} +} + +func (x *WaitSandboxRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +type WaitSandboxResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + ExitStatus uint32 `protobuf:"varint,1,opt,name=exit_status,json=exitStatus,proto3" json:"exit_status,omitempty"` + ExitedAt *timestamppb.Timestamp `protobuf:"bytes,2,opt,name=exited_at,json=exitedAt,proto3" json:"exited_at,omitempty"` +} + +func (x *WaitSandboxResponse) Reset() { + *x = WaitSandboxResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[10] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *WaitSandboxResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*WaitSandboxResponse) ProtoMessage() {} + +func (x *WaitSandboxResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[10] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use WaitSandboxResponse.ProtoReflect.Descriptor instead. +func (*WaitSandboxResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{10} +} + +func (x *WaitSandboxResponse) GetExitStatus() uint32 { + if x != nil { + return x.ExitStatus + } + return 0 +} + +func (x *WaitSandboxResponse) GetExitedAt() *timestamppb.Timestamp { + if x != nil { + return x.ExitedAt + } + return nil +} + +type UpdateSandboxResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields +} + +func (x *UpdateSandboxResponse) Reset() { + *x = UpdateSandboxResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[11] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *UpdateSandboxResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*UpdateSandboxResponse) ProtoMessage() {} + +func (x *UpdateSandboxResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[11] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use UpdateSandboxResponse.ProtoReflect.Descriptor instead. +func (*UpdateSandboxResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{11} +} + +type SandboxStatusRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` + Verbose bool `protobuf:"varint,2,opt,name=verbose,proto3" json:"verbose,omitempty"` +} + +func (x *SandboxStatusRequest) Reset() { + *x = SandboxStatusRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[12] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *SandboxStatusRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*SandboxStatusRequest) ProtoMessage() {} + +func (x *SandboxStatusRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[12] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use SandboxStatusRequest.ProtoReflect.Descriptor instead. +func (*SandboxStatusRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{12} +} + +func (x *SandboxStatusRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +func (x *SandboxStatusRequest) GetVerbose() bool { + if x != nil { + return x.Verbose + } + return false +} + +type SandboxStatusResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` + Pid uint32 `protobuf:"varint,2,opt,name=pid,proto3" json:"pid,omitempty"` + State string `protobuf:"bytes,3,opt,name=state,proto3" json:"state,omitempty"` + Info map[string]string `protobuf:"bytes,4,rep,name=info,proto3" json:"info,omitempty" protobuf_key:"bytes,1,opt,name=key,proto3" protobuf_val:"bytes,2,opt,name=value,proto3"` + CreatedAt *timestamppb.Timestamp `protobuf:"bytes,5,opt,name=created_at,json=createdAt,proto3" json:"created_at,omitempty"` + ExitedAt *timestamppb.Timestamp `protobuf:"bytes,6,opt,name=exited_at,json=exitedAt,proto3" json:"exited_at,omitempty"` + Extra *anypb.Any `protobuf:"bytes,7,opt,name=extra,proto3" json:"extra,omitempty"` +} + +func (x *SandboxStatusResponse) Reset() { + *x = SandboxStatusResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[13] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *SandboxStatusResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*SandboxStatusResponse) ProtoMessage() {} + +func (x *SandboxStatusResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[13] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use SandboxStatusResponse.ProtoReflect.Descriptor instead. +func (*SandboxStatusResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{13} +} + +func (x *SandboxStatusResponse) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +func (x *SandboxStatusResponse) GetPid() uint32 { + if x != nil { + return x.Pid + } + return 0 +} + +func (x *SandboxStatusResponse) GetState() string { + if x != nil { + return x.State + } + return "" +} + +func (x *SandboxStatusResponse) GetInfo() map[string]string { + if x != nil { + return x.Info + } + return nil +} + +func (x *SandboxStatusResponse) GetCreatedAt() *timestamppb.Timestamp { + if x != nil { + return x.CreatedAt + } + return nil +} + +func (x *SandboxStatusResponse) GetExitedAt() *timestamppb.Timestamp { + if x != nil { + return x.ExitedAt + } + return nil +} + +func (x *SandboxStatusResponse) GetExtra() *anypb.Any { + if x != nil { + return x.Extra + } + return nil +} + +type PingRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` +} + +func (x *PingRequest) Reset() { + *x = PingRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[14] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *PingRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*PingRequest) ProtoMessage() {} + +func (x *PingRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[14] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use PingRequest.ProtoReflect.Descriptor instead. +func (*PingRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{14} +} + +func (x *PingRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +type PingResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields +} + +func (x *PingResponse) Reset() { + *x = PingResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[15] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *PingResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*PingResponse) ProtoMessage() {} + +func (x *PingResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[15] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use PingResponse.ProtoReflect.Descriptor instead. +func (*PingResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{15} +} + +type ShutdownSandboxRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` +} + +func (x *ShutdownSandboxRequest) Reset() { + *x = ShutdownSandboxRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[16] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *ShutdownSandboxRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*ShutdownSandboxRequest) ProtoMessage() {} + +func (x *ShutdownSandboxRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[16] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use ShutdownSandboxRequest.ProtoReflect.Descriptor instead. +func (*ShutdownSandboxRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{16} +} + +func (x *ShutdownSandboxRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +type ShutdownSandboxResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields +} + +func (x *ShutdownSandboxResponse) Reset() { + *x = ShutdownSandboxResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[17] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *ShutdownSandboxResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*ShutdownSandboxResponse) ProtoMessage() {} + +func (x *ShutdownSandboxResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[17] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use ShutdownSandboxResponse.ProtoReflect.Descriptor instead. +func (*ShutdownSandboxResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{17} +} + +type SandboxMetricsRequest struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + SandboxID string `protobuf:"bytes,1,opt,name=sandbox_id,json=sandboxId,proto3" json:"sandbox_id,omitempty"` +} + +func (x *SandboxMetricsRequest) Reset() { + *x = SandboxMetricsRequest{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[18] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *SandboxMetricsRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*SandboxMetricsRequest) ProtoMessage() {} + +func (x *SandboxMetricsRequest) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[18] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use SandboxMetricsRequest.ProtoReflect.Descriptor instead. +func (*SandboxMetricsRequest) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{18} +} + +func (x *SandboxMetricsRequest) GetSandboxID() string { + if x != nil { + return x.SandboxID + } + return "" +} + +type SandboxMetricsResponse struct { + state protoimpl.MessageState + sizeCache protoimpl.SizeCache + unknownFields protoimpl.UnknownFields + + Metrics *types.Metric `protobuf:"bytes,1,opt,name=metrics,proto3" json:"metrics,omitempty"` +} + +func (x *SandboxMetricsResponse) Reset() { + *x = SandboxMetricsResponse{} + if protoimpl.UnsafeEnabled { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[19] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +} + +func (x *SandboxMetricsResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*SandboxMetricsResponse) ProtoMessage() {} + +func (x *SandboxMetricsResponse) ProtoReflect() protoreflect.Message { + mi := &file_runtime_sandbox_v1_sandbox_proto_msgTypes[19] + if protoimpl.UnsafeEnabled && x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use SandboxMetricsResponse.ProtoReflect.Descriptor instead. +func (*SandboxMetricsResponse) Descriptor() ([]byte, []int) { + return file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP(), []int{19} +} + +func (x *SandboxMetricsResponse) GetMetrics() *types.Metric { + if x != nil { + return x.Metrics + } + return nil +} + +var File_runtime_sandbox_v1_sandbox_proto protoreflect.FileDescriptor + +var file_runtime_sandbox_v1_sandbox_proto_rawDesc = []byte{ + 0x0a, 0x20, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2f, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x2f, 0x76, 0x31, 0x2f, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x70, 0x72, 0x6f, + 0x74, 0x6f, 0x12, 0x1d, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, + 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, + 0x31, 0x1a, 0x19, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2f, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, + 0x75, 0x66, 0x2f, 0x61, 0x6e, 0x79, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x1a, 0x1f, 0x67, 0x6f, + 0x6f, 0x67, 0x6c, 0x65, 0x2f, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2f, 0x74, 0x69, + 0x6d, 0x65, 0x73, 0x74, 0x61, 0x6d, 0x70, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x1a, 0x13, 0x74, + 0x79, 0x70, 0x65, 0x73, 0x2f, 0x6d, 0x65, 0x74, 0x72, 0x69, 0x63, 0x73, 0x2e, 0x70, 0x72, 0x6f, + 0x74, 0x6f, 0x1a, 0x11, 0x74, 0x79, 0x70, 0x65, 0x73, 0x2f, 0x6d, 0x6f, 0x75, 0x6e, 0x74, 0x2e, + 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x1a, 0x14, 0x74, 0x79, 0x70, 0x65, 0x73, 0x2f, 0x70, 0x6c, 0x61, + 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x22, 0xfe, 0x02, 0x0a, 0x14, + 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, + 0x75, 0x65, 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, + 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x49, 0x64, 0x12, 0x1f, 0x0a, 0x0b, 0x62, 0x75, 0x6e, 0x64, 0x6c, 0x65, 0x5f, 0x70, 0x61, + 0x74, 0x68, 0x18, 0x02, 0x20, 0x01, 0x28, 0x09, 0x52, 0x0a, 0x62, 0x75, 0x6e, 0x64, 0x6c, 0x65, + 0x50, 0x61, 0x74, 0x68, 0x12, 0x2f, 0x0a, 0x06, 0x72, 0x6f, 0x6f, 0x74, 0x66, 0x73, 0x18, 0x03, + 0x20, 0x03, 0x28, 0x0b, 0x32, 0x17, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, + 0x64, 0x2e, 0x74, 0x79, 0x70, 0x65, 0x73, 0x2e, 0x4d, 0x6f, 0x75, 0x6e, 0x74, 0x52, 0x06, 0x72, + 0x6f, 0x6f, 0x74, 0x66, 0x73, 0x12, 0x2e, 0x0a, 0x07, 0x6f, 0x70, 0x74, 0x69, 0x6f, 0x6e, 0x73, + 0x18, 0x04, 0x20, 0x01, 0x28, 0x0b, 0x32, 0x14, 0x2e, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, + 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2e, 0x41, 0x6e, 0x79, 0x52, 0x07, 0x6f, 0x70, + 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x12, 0x1d, 0x0a, 0x0a, 0x6e, 0x65, 0x74, 0x6e, 0x73, 0x5f, 0x70, + 0x61, 0x74, 0x68, 0x18, 0x05, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x6e, 0x65, 0x74, 0x6e, 0x73, + 0x50, 0x61, 0x74, 0x68, 0x12, 0x66, 0x0a, 0x0b, 0x61, 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, + 0x6f, 0x6e, 0x73, 0x18, 0x06, 0x20, 0x03, 0x28, 0x0b, 0x32, 0x44, 0x2e, 0x63, 0x6f, 0x6e, 0x74, + 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, + 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, + 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x2e, 0x41, + 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x45, 0x6e, 0x74, 0x72, 0x79, 0x52, + 0x0b, 0x61, 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x1a, 0x3e, 0x0a, 0x10, + 0x41, 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x45, 0x6e, 0x74, 0x72, 0x79, + 0x12, 0x10, 0x0a, 0x03, 0x6b, 0x65, 0x79, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x03, 0x6b, + 0x65, 0x79, 0x12, 0x14, 0x0a, 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x18, 0x02, 0x20, 0x01, 0x28, + 0x09, 0x52, 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x3a, 0x02, 0x38, 0x01, 0x22, 0x17, 0x0a, 0x15, + 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, + 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x22, 0x34, 0x0a, 0x13, 0x53, 0x74, 0x61, 0x72, 0x74, 0x53, 0x61, + 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, + 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, + 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, 0x64, 0x22, 0x8d, 0x01, 0x0a, 0x14, + 0x53, 0x74, 0x61, 0x72, 0x74, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, + 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x10, 0x0a, 0x03, 0x70, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, + 0x0d, 0x52, 0x03, 0x70, 0x69, 0x64, 0x12, 0x39, 0x0a, 0x0a, 0x63, 0x72, 0x65, 0x61, 0x74, 0x65, + 0x64, 0x5f, 0x61, 0x74, 0x18, 0x02, 0x20, 0x01, 0x28, 0x0b, 0x32, 0x1a, 0x2e, 0x67, 0x6f, 0x6f, + 0x67, 0x6c, 0x65, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2e, 0x54, 0x69, 0x6d, + 0x65, 0x73, 0x74, 0x61, 0x6d, 0x70, 0x52, 0x09, 0x63, 0x72, 0x65, 0x61, 0x74, 0x65, 0x64, 0x41, + 0x74, 0x12, 0x28, 0x0a, 0x04, 0x73, 0x70, 0x65, 0x63, 0x18, 0x03, 0x20, 0x01, 0x28, 0x0b, 0x32, + 0x14, 0x2e, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, + 0x66, 0x2e, 0x41, 0x6e, 0x79, 0x52, 0x04, 0x73, 0x70, 0x65, 0x63, 0x22, 0x30, 0x0a, 0x0f, 0x50, + 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x12, 0x1d, + 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, + 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, 0x64, 0x22, 0x4a, 0x0a, + 0x10, 0x50, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, + 0x65, 0x12, 0x36, 0x0a, 0x08, 0x70, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x18, 0x01, 0x20, + 0x01, 0x28, 0x0b, 0x32, 0x1a, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, + 0x2e, 0x74, 0x79, 0x70, 0x65, 0x73, 0x2e, 0x50, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x52, + 0x08, 0x70, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x22, 0x56, 0x0a, 0x12, 0x53, 0x74, 0x6f, + 0x70, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x12, + 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, 0x18, 0x01, 0x20, + 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, 0x64, 0x12, 0x21, + 0x0a, 0x0c, 0x74, 0x69, 0x6d, 0x65, 0x6f, 0x75, 0x74, 0x5f, 0x73, 0x65, 0x63, 0x73, 0x18, 0x02, + 0x20, 0x01, 0x28, 0x0d, 0x52, 0x0b, 0x74, 0x69, 0x6d, 0x65, 0x6f, 0x75, 0x74, 0x53, 0x65, 0x63, + 0x73, 0x22, 0x15, 0x0a, 0x13, 0x53, 0x74, 0x6f, 0x70, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, + 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x22, 0x91, 0x02, 0x0a, 0x14, 0x55, 0x70, 0x64, + 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, + 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, 0x18, + 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, 0x64, + 0x12, 0x32, 0x0a, 0x09, 0x72, 0x65, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x73, 0x18, 0x02, 0x20, + 0x01, 0x28, 0x0b, 0x32, 0x14, 0x2e, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, 0x70, 0x72, 0x6f, + 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2e, 0x41, 0x6e, 0x79, 0x52, 0x09, 0x72, 0x65, 0x73, 0x6f, 0x75, + 0x72, 0x63, 0x65, 0x73, 0x12, 0x66, 0x0a, 0x0b, 0x61, 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, + 0x6f, 0x6e, 0x73, 0x18, 0x03, 0x20, 0x03, 0x28, 0x0b, 0x32, 0x44, 0x2e, 0x63, 0x6f, 0x6e, 0x74, + 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, + 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x55, 0x70, 0x64, 0x61, 0x74, 0x65, + 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x2e, 0x41, + 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x45, 0x6e, 0x74, 0x72, 0x79, 0x52, + 0x0b, 0x61, 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x1a, 0x3e, 0x0a, 0x10, + 0x41, 0x6e, 0x6e, 0x6f, 0x74, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x73, 0x45, 0x6e, 0x74, 0x72, 0x79, + 0x12, 0x10, 0x0a, 0x03, 0x6b, 0x65, 0x79, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x03, 0x6b, + 0x65, 0x79, 0x12, 0x14, 0x0a, 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x18, 0x02, 0x20, 0x01, 0x28, + 0x09, 0x52, 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x3a, 0x02, 0x38, 0x01, 0x22, 0x33, 0x0a, 0x12, + 0x57, 0x61, 0x69, 0x74, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, + 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, + 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, + 0x64, 0x22, 0x6f, 0x0a, 0x13, 0x57, 0x61, 0x69, 0x74, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, + 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x1f, 0x0a, 0x0b, 0x65, 0x78, 0x69, 0x74, + 0x5f, 0x73, 0x74, 0x61, 0x74, 0x75, 0x73, 0x18, 0x01, 0x20, 0x01, 0x28, 0x0d, 0x52, 0x0a, 0x65, + 0x78, 0x69, 0x74, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, 0x12, 0x37, 0x0a, 0x09, 0x65, 0x78, 0x69, + 0x74, 0x65, 0x64, 0x5f, 0x61, 0x74, 0x18, 0x02, 0x20, 0x01, 0x28, 0x0b, 0x32, 0x1a, 0x2e, 0x67, + 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2e, 0x54, + 0x69, 0x6d, 0x65, 0x73, 0x74, 0x61, 0x6d, 0x70, 0x52, 0x08, 0x65, 0x78, 0x69, 0x74, 0x65, 0x64, + 0x41, 0x74, 0x22, 0x17, 0x0a, 0x15, 0x55, 0x70, 0x64, 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, + 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x22, 0x4f, 0x0a, 0x14, 0x53, + 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, 0x52, 0x65, 0x71, 0x75, + 0x65, 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, + 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, + 0x49, 0x64, 0x12, 0x18, 0x0a, 0x07, 0x76, 0x65, 0x72, 0x62, 0x6f, 0x73, 0x65, 0x18, 0x02, 0x20, + 0x01, 0x28, 0x08, 0x52, 0x07, 0x76, 0x65, 0x72, 0x62, 0x6f, 0x73, 0x65, 0x22, 0x8b, 0x03, 0x0a, + 0x15, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, 0x52, 0x65, + 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x5f, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, + 0x62, 0x6f, 0x78, 0x49, 0x64, 0x12, 0x10, 0x0a, 0x03, 0x70, 0x69, 0x64, 0x18, 0x02, 0x20, 0x01, + 0x28, 0x0d, 0x52, 0x03, 0x70, 0x69, 0x64, 0x12, 0x14, 0x0a, 0x05, 0x73, 0x74, 0x61, 0x74, 0x65, + 0x18, 0x03, 0x20, 0x01, 0x28, 0x09, 0x52, 0x05, 0x73, 0x74, 0x61, 0x74, 0x65, 0x12, 0x52, 0x0a, + 0x04, 0x69, 0x6e, 0x66, 0x6f, 0x18, 0x04, 0x20, 0x03, 0x28, 0x0b, 0x32, 0x3e, 0x2e, 0x63, 0x6f, + 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, + 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x61, 0x6e, 0x64, + 0x62, 0x6f, 0x78, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, + 0x65, 0x2e, 0x49, 0x6e, 0x66, 0x6f, 0x45, 0x6e, 0x74, 0x72, 0x79, 0x52, 0x04, 0x69, 0x6e, 0x66, + 0x6f, 0x12, 0x39, 0x0a, 0x0a, 0x63, 0x72, 0x65, 0x61, 0x74, 0x65, 0x64, 0x5f, 0x61, 0x74, 0x18, + 0x05, 0x20, 0x01, 0x28, 0x0b, 0x32, 0x1a, 0x2e, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, 0x70, + 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2e, 0x54, 0x69, 0x6d, 0x65, 0x73, 0x74, 0x61, 0x6d, + 0x70, 0x52, 0x09, 0x63, 0x72, 0x65, 0x61, 0x74, 0x65, 0x64, 0x41, 0x74, 0x12, 0x37, 0x0a, 0x09, + 0x65, 0x78, 0x69, 0x74, 0x65, 0x64, 0x5f, 0x61, 0x74, 0x18, 0x06, 0x20, 0x01, 0x28, 0x0b, 0x32, + 0x1a, 0x2e, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x62, 0x75, + 0x66, 0x2e, 0x54, 0x69, 0x6d, 0x65, 0x73, 0x74, 0x61, 0x6d, 0x70, 0x52, 0x08, 0x65, 0x78, 0x69, + 0x74, 0x65, 0x64, 0x41, 0x74, 0x12, 0x2a, 0x0a, 0x05, 0x65, 0x78, 0x74, 0x72, 0x61, 0x18, 0x07, + 0x20, 0x01, 0x28, 0x0b, 0x32, 0x14, 0x2e, 0x67, 0x6f, 0x6f, 0x67, 0x6c, 0x65, 0x2e, 0x70, 0x72, + 0x6f, 0x74, 0x6f, 0x62, 0x75, 0x66, 0x2e, 0x41, 0x6e, 0x79, 0x52, 0x05, 0x65, 0x78, 0x74, 0x72, + 0x61, 0x1a, 0x37, 0x0a, 0x09, 0x49, 0x6e, 0x66, 0x6f, 0x45, 0x6e, 0x74, 0x72, 0x79, 0x12, 0x10, + 0x0a, 0x03, 0x6b, 0x65, 0x79, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x03, 0x6b, 0x65, 0x79, + 0x12, 0x14, 0x0a, 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x18, 0x02, 0x20, 0x01, 0x28, 0x09, 0x52, + 0x05, 0x76, 0x61, 0x6c, 0x75, 0x65, 0x3a, 0x02, 0x38, 0x01, 0x22, 0x2c, 0x0a, 0x0b, 0x50, 0x69, + 0x6e, 0x67, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, + 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, + 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, 0x64, 0x22, 0x0e, 0x0a, 0x0c, 0x50, 0x69, 0x6e, 0x67, + 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x22, 0x37, 0x0a, 0x16, 0x53, 0x68, 0x75, 0x74, + 0x64, 0x6f, 0x77, 0x6e, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, + 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x5f, 0x69, 0x64, + 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x49, + 0x64, 0x22, 0x19, 0x0a, 0x17, 0x53, 0x68, 0x75, 0x74, 0x64, 0x6f, 0x77, 0x6e, 0x53, 0x61, 0x6e, + 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x22, 0x36, 0x0a, 0x15, + 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x4d, 0x65, 0x74, 0x72, 0x69, 0x63, 0x73, 0x52, 0x65, + 0x71, 0x75, 0x65, 0x73, 0x74, 0x12, 0x1d, 0x0a, 0x0a, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, + 0x5f, 0x69, 0x64, 0x18, 0x01, 0x20, 0x01, 0x28, 0x09, 0x52, 0x09, 0x73, 0x61, 0x6e, 0x64, 0x62, + 0x6f, 0x78, 0x49, 0x64, 0x22, 0x4c, 0x0a, 0x16, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x4d, + 0x65, 0x74, 0x72, 0x69, 0x63, 0x73, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x32, + 0x0a, 0x07, 0x6d, 0x65, 0x74, 0x72, 0x69, 0x63, 0x73, 0x18, 0x01, 0x20, 0x01, 0x28, 0x0b, 0x32, + 0x18, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x74, 0x79, 0x70, + 0x65, 0x73, 0x2e, 0x4d, 0x65, 0x74, 0x72, 0x69, 0x63, 0x52, 0x07, 0x6d, 0x65, 0x74, 0x72, 0x69, + 0x63, 0x73, 0x32, 0xbd, 0x08, 0x0a, 0x07, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x12, 0x7a, + 0x0a, 0x0d, 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x12, + 0x33, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, + 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, + 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, + 0x75, 0x65, 0x73, 0x74, 0x1a, 0x34, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, + 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x43, 0x72, 0x65, 0x61, 0x74, 0x65, 0x53, 0x61, 0x6e, 0x64, 0x62, + 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x77, 0x0a, 0x0c, 0x53, 0x74, + 0x61, 0x72, 0x74, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x12, 0x32, 0x2e, 0x63, 0x6f, 0x6e, + 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, + 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x74, 0x61, 0x72, 0x74, + 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, 0x33, + 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, + 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, + 0x74, 0x61, 0x72, 0x74, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, 0x6f, + 0x6e, 0x73, 0x65, 0x12, 0x6b, 0x0a, 0x08, 0x50, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x12, + 0x2e, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, + 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, + 0x50, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, + 0x2f, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, + 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, + 0x50, 0x6c, 0x61, 0x74, 0x66, 0x6f, 0x72, 0x6d, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, + 0x12, 0x74, 0x0a, 0x0b, 0x53, 0x74, 0x6f, 0x70, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x12, + 0x31, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, + 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, + 0x53, 0x74, 0x6f, 0x70, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, + 0x73, 0x74, 0x1a, 0x32, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, + 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, + 0x76, 0x31, 0x2e, 0x53, 0x74, 0x6f, 0x70, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, + 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x74, 0x0a, 0x0b, 0x57, 0x61, 0x69, 0x74, 0x53, 0x61, + 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x12, 0x31, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, + 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, + 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x57, 0x61, 0x69, 0x74, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, 0x32, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, + 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, + 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x57, 0x61, 0x69, 0x74, 0x53, 0x61, 0x6e, + 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x7a, 0x0a, 0x0d, + 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, 0x12, 0x33, 0x2e, + 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, + 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x61, + 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, 0x52, 0x65, 0x71, 0x75, 0x65, + 0x73, 0x74, 0x1a, 0x34, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, + 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, + 0x76, 0x31, 0x2e, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x53, 0x74, 0x61, 0x74, 0x75, 0x73, + 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, 0x12, 0x66, 0x0a, 0x0b, 0x50, 0x69, 0x6e, 0x67, + 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x12, 0x2a, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, + 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, + 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x50, 0x69, 0x6e, 0x67, 0x52, 0x65, 0x71, 0x75, + 0x65, 0x73, 0x74, 0x1a, 0x2b, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, + 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, + 0x2e, 0x76, 0x31, 0x2e, 0x50, 0x69, 0x6e, 0x67, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, 0x73, 0x65, + 0x12, 0x80, 0x01, 0x0a, 0x0f, 0x53, 0x68, 0x75, 0x74, 0x64, 0x6f, 0x77, 0x6e, 0x53, 0x61, 0x6e, + 0x64, 0x62, 0x6f, 0x78, 0x12, 0x35, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, + 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, + 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x68, 0x75, 0x74, 0x64, 0x6f, 0x77, 0x6e, 0x53, 0x61, 0x6e, + 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, 0x36, 0x2e, 0x63, 0x6f, + 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, + 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x68, 0x75, 0x74, + 0x64, 0x6f, 0x77, 0x6e, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x52, 0x65, 0x73, 0x70, 0x6f, + 0x6e, 0x73, 0x65, 0x12, 0x7d, 0x0a, 0x0e, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x4d, 0x65, + 0x74, 0x72, 0x69, 0x63, 0x73, 0x12, 0x34, 0x2e, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, + 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, + 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x4d, 0x65, 0x74, + 0x72, 0x69, 0x63, 0x73, 0x52, 0x65, 0x71, 0x75, 0x65, 0x73, 0x74, 0x1a, 0x35, 0x2e, 0x63, 0x6f, + 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2e, 0x72, 0x75, 0x6e, 0x74, 0x69, 0x6d, 0x65, + 0x2e, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2e, 0x76, 0x31, 0x2e, 0x53, 0x61, 0x6e, 0x64, + 0x62, 0x6f, 0x78, 0x4d, 0x65, 0x74, 0x72, 0x69, 0x63, 0x73, 0x52, 0x65, 0x73, 0x70, 0x6f, 0x6e, + 0x73, 0x65, 0x42, 0x41, 0x5a, 0x3f, 0x67, 0x69, 0x74, 0x68, 0x75, 0x62, 0x2e, 0x63, 0x6f, 0x6d, + 0x2f, 0x63, 0x6f, 0x6e, 0x74, 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2f, 0x63, 0x6f, 0x6e, 0x74, + 0x61, 0x69, 0x6e, 0x65, 0x72, 0x64, 0x2f, 0x61, 0x70, 0x69, 0x2f, 0x72, 0x75, 0x6e, 0x74, 0x69, + 0x6d, 0x65, 0x2f, 0x73, 0x61, 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x2f, 0x76, 0x31, 0x3b, 0x73, 0x61, + 0x6e, 0x64, 0x62, 0x6f, 0x78, 0x62, 0x06, 0x70, 0x72, 0x6f, 0x74, 0x6f, 0x33, +} + +var ( + file_runtime_sandbox_v1_sandbox_proto_rawDescOnce sync.Once + file_runtime_sandbox_v1_sandbox_proto_rawDescData = file_runtime_sandbox_v1_sandbox_proto_rawDesc +) + +func file_runtime_sandbox_v1_sandbox_proto_rawDescGZIP() []byte { + file_runtime_sandbox_v1_sandbox_proto_rawDescOnce.Do(func() { + file_runtime_sandbox_v1_sandbox_proto_rawDescData = protoimpl.X.CompressGZIP(file_runtime_sandbox_v1_sandbox_proto_rawDescData) + }) + return file_runtime_sandbox_v1_sandbox_proto_rawDescData +} + +var file_runtime_sandbox_v1_sandbox_proto_msgTypes = make([]protoimpl.MessageInfo, 23) +var file_runtime_sandbox_v1_sandbox_proto_goTypes = []interface{}{ + (*CreateSandboxRequest)(nil), // 0: containerd.runtime.sandbox.v1.CreateSandboxRequest + (*CreateSandboxResponse)(nil), // 1: containerd.runtime.sandbox.v1.CreateSandboxResponse + (*StartSandboxRequest)(nil), // 2: containerd.runtime.sandbox.v1.StartSandboxRequest + (*StartSandboxResponse)(nil), // 3: containerd.runtime.sandbox.v1.StartSandboxResponse + (*PlatformRequest)(nil), // 4: containerd.runtime.sandbox.v1.PlatformRequest + (*PlatformResponse)(nil), // 5: containerd.runtime.sandbox.v1.PlatformResponse + (*StopSandboxRequest)(nil), // 6: containerd.runtime.sandbox.v1.StopSandboxRequest + (*StopSandboxResponse)(nil), // 7: containerd.runtime.sandbox.v1.StopSandboxResponse + (*UpdateSandboxRequest)(nil), // 8: containerd.runtime.sandbox.v1.UpdateSandboxRequest + (*WaitSandboxRequest)(nil), // 9: containerd.runtime.sandbox.v1.WaitSandboxRequest + (*WaitSandboxResponse)(nil), // 10: containerd.runtime.sandbox.v1.WaitSandboxResponse + (*UpdateSandboxResponse)(nil), // 11: containerd.runtime.sandbox.v1.UpdateSandboxResponse + (*SandboxStatusRequest)(nil), // 12: containerd.runtime.sandbox.v1.SandboxStatusRequest + (*SandboxStatusResponse)(nil), // 13: containerd.runtime.sandbox.v1.SandboxStatusResponse + (*PingRequest)(nil), // 14: containerd.runtime.sandbox.v1.PingRequest + (*PingResponse)(nil), // 15: containerd.runtime.sandbox.v1.PingResponse + (*ShutdownSandboxRequest)(nil), // 16: containerd.runtime.sandbox.v1.ShutdownSandboxRequest + (*ShutdownSandboxResponse)(nil), // 17: containerd.runtime.sandbox.v1.ShutdownSandboxResponse + (*SandboxMetricsRequest)(nil), // 18: containerd.runtime.sandbox.v1.SandboxMetricsRequest + (*SandboxMetricsResponse)(nil), // 19: containerd.runtime.sandbox.v1.SandboxMetricsResponse + nil, // 20: containerd.runtime.sandbox.v1.CreateSandboxRequest.AnnotationsEntry + nil, // 21: containerd.runtime.sandbox.v1.UpdateSandboxRequest.AnnotationsEntry + nil, // 22: containerd.runtime.sandbox.v1.SandboxStatusResponse.InfoEntry + (*types.Mount)(nil), // 23: containerd.types.Mount + (*anypb.Any)(nil), // 24: google.protobuf.Any + (*timestamppb.Timestamp)(nil), // 25: google.protobuf.Timestamp + (*types.Platform)(nil), // 26: containerd.types.Platform + (*types.Metric)(nil), // 27: containerd.types.Metric +} +var file_runtime_sandbox_v1_sandbox_proto_depIdxs = []int32{ + 23, // 0: containerd.runtime.sandbox.v1.CreateSandboxRequest.rootfs:type_name -> containerd.types.Mount + 24, // 1: containerd.runtime.sandbox.v1.CreateSandboxRequest.options:type_name -> google.protobuf.Any + 20, // 2: containerd.runtime.sandbox.v1.CreateSandboxRequest.annotations:type_name -> containerd.runtime.sandbox.v1.CreateSandboxRequest.AnnotationsEntry + 25, // 3: containerd.runtime.sandbox.v1.StartSandboxResponse.created_at:type_name -> google.protobuf.Timestamp + 24, // 4: containerd.runtime.sandbox.v1.StartSandboxResponse.spec:type_name -> google.protobuf.Any + 26, // 5: containerd.runtime.sandbox.v1.PlatformResponse.platform:type_name -> containerd.types.Platform + 24, // 6: containerd.runtime.sandbox.v1.UpdateSandboxRequest.resources:type_name -> google.protobuf.Any + 21, // 7: containerd.runtime.sandbox.v1.UpdateSandboxRequest.annotations:type_name -> containerd.runtime.sandbox.v1.UpdateSandboxRequest.AnnotationsEntry + 25, // 8: containerd.runtime.sandbox.v1.WaitSandboxResponse.exited_at:type_name -> google.protobuf.Timestamp + 22, // 9: containerd.runtime.sandbox.v1.SandboxStatusResponse.info:type_name -> containerd.runtime.sandbox.v1.SandboxStatusResponse.InfoEntry + 25, // 10: containerd.runtime.sandbox.v1.SandboxStatusResponse.created_at:type_name -> google.protobuf.Timestamp + 25, // 11: containerd.runtime.sandbox.v1.SandboxStatusResponse.exited_at:type_name -> google.protobuf.Timestamp + 24, // 12: containerd.runtime.sandbox.v1.SandboxStatusResponse.extra:type_name -> google.protobuf.Any + 27, // 13: containerd.runtime.sandbox.v1.SandboxMetricsResponse.metrics:type_name -> containerd.types.Metric + 0, // 14: containerd.runtime.sandbox.v1.Sandbox.CreateSandbox:input_type -> containerd.runtime.sandbox.v1.CreateSandboxRequest + 2, // 15: containerd.runtime.sandbox.v1.Sandbox.StartSandbox:input_type -> containerd.runtime.sandbox.v1.StartSandboxRequest + 4, // 16: containerd.runtime.sandbox.v1.Sandbox.Platform:input_type -> containerd.runtime.sandbox.v1.PlatformRequest + 6, // 17: containerd.runtime.sandbox.v1.Sandbox.StopSandbox:input_type -> containerd.runtime.sandbox.v1.StopSandboxRequest + 9, // 18: containerd.runtime.sandbox.v1.Sandbox.WaitSandbox:input_type -> containerd.runtime.sandbox.v1.WaitSandboxRequest + 12, // 19: containerd.runtime.sandbox.v1.Sandbox.SandboxStatus:input_type -> containerd.runtime.sandbox.v1.SandboxStatusRequest + 14, // 20: containerd.runtime.sandbox.v1.Sandbox.PingSandbox:input_type -> containerd.runtime.sandbox.v1.PingRequest + 16, // 21: containerd.runtime.sandbox.v1.Sandbox.ShutdownSandbox:input_type -> containerd.runtime.sandbox.v1.ShutdownSandboxRequest + 18, // 22: containerd.runtime.sandbox.v1.Sandbox.SandboxMetrics:input_type -> containerd.runtime.sandbox.v1.SandboxMetricsRequest + 1, // 23: containerd.runtime.sandbox.v1.Sandbox.CreateSandbox:output_type -> containerd.runtime.sandbox.v1.CreateSandboxResponse + 3, // 24: containerd.runtime.sandbox.v1.Sandbox.StartSandbox:output_type -> containerd.runtime.sandbox.v1.StartSandboxResponse + 5, // 25: containerd.runtime.sandbox.v1.Sandbox.Platform:output_type -> containerd.runtime.sandbox.v1.PlatformResponse + 7, // 26: containerd.runtime.sandbox.v1.Sandbox.StopSandbox:output_type -> containerd.runtime.sandbox.v1.StopSandboxResponse + 10, // 27: containerd.runtime.sandbox.v1.Sandbox.WaitSandbox:output_type -> containerd.runtime.sandbox.v1.WaitSandboxResponse + 13, // 28: containerd.runtime.sandbox.v1.Sandbox.SandboxStatus:output_type -> containerd.runtime.sandbox.v1.SandboxStatusResponse + 15, // 29: containerd.runtime.sandbox.v1.Sandbox.PingSandbox:output_type -> containerd.runtime.sandbox.v1.PingResponse + 17, // 30: containerd.runtime.sandbox.v1.Sandbox.ShutdownSandbox:output_type -> containerd.runtime.sandbox.v1.ShutdownSandboxResponse + 19, // 31: containerd.runtime.sandbox.v1.Sandbox.SandboxMetrics:output_type -> containerd.runtime.sandbox.v1.SandboxMetricsResponse + 23, // [23:32] is the sub-list for method output_type + 14, // [14:23] is the sub-list for method input_type + 14, // [14:14] is the sub-list for extension type_name + 14, // [14:14] is the sub-list for extension extendee + 0, // [0:14] is the sub-list for field type_name +} + +func init() { file_runtime_sandbox_v1_sandbox_proto_init() } +func file_runtime_sandbox_v1_sandbox_proto_init() { + if File_runtime_sandbox_v1_sandbox_proto != nil { + return + } + if !protoimpl.UnsafeEnabled { + file_runtime_sandbox_v1_sandbox_proto_msgTypes[0].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CreateSandboxRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[1].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*CreateSandboxResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[2].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*StartSandboxRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[3].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*StartSandboxResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[4].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*PlatformRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[5].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*PlatformResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[6].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*StopSandboxRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[7].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*StopSandboxResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[8].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*UpdateSandboxRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[9].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*WaitSandboxRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[10].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*WaitSandboxResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[11].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*UpdateSandboxResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[12].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*SandboxStatusRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[13].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*SandboxStatusResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[14].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*PingRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[15].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*PingResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[16].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*ShutdownSandboxRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[17].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*ShutdownSandboxResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[18].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*SandboxMetricsRequest); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + file_runtime_sandbox_v1_sandbox_proto_msgTypes[19].Exporter = func(v interface{}, i int) interface{} { + switch v := v.(*SandboxMetricsResponse); i { + case 0: + return &v.state + case 1: + return &v.sizeCache + case 2: + return &v.unknownFields + default: + return nil + } + } + } + type x struct{} + out := protoimpl.TypeBuilder{ + File: protoimpl.DescBuilder{ + GoPackagePath: reflect.TypeOf(x{}).PkgPath(), + RawDescriptor: file_runtime_sandbox_v1_sandbox_proto_rawDesc, + NumEnums: 0, + NumMessages: 23, + NumExtensions: 0, + NumServices: 1, + }, + GoTypes: file_runtime_sandbox_v1_sandbox_proto_goTypes, + DependencyIndexes: file_runtime_sandbox_v1_sandbox_proto_depIdxs, + MessageInfos: file_runtime_sandbox_v1_sandbox_proto_msgTypes, + }.Build() + File_runtime_sandbox_v1_sandbox_proto = out.File + file_runtime_sandbox_v1_sandbox_proto_rawDesc = nil + file_runtime_sandbox_v1_sandbox_proto_goTypes = nil + file_runtime_sandbox_v1_sandbox_proto_depIdxs = nil +} diff --git a/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox.proto b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox.proto new file mode 100644 index 00000000..9130c75f --- /dev/null +++ b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox.proto @@ -0,0 +1,149 @@ +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +syntax = "proto3"; + +package containerd.runtime.sandbox.v1; + +import "google/protobuf/any.proto"; +import "google/protobuf/timestamp.proto"; +import "types/metrics.proto"; +import "types/mount.proto"; +import "types/platform.proto"; + +option go_package = "github.com/containerd/containerd/api/runtime/sandbox/v1;sandbox"; + +// Sandbox is an optional interface that shim may implement to support sandboxes environments. +// A typical example of sandbox is microVM or pause container - an entity that groups containers and/or +// holds resources relevant for this group. +service Sandbox { + // CreateSandbox will be called right after sandbox shim instance launched. + // It is a good place to initialize sandbox environment. + rpc CreateSandbox(CreateSandboxRequest) returns (CreateSandboxResponse); + + // StartSandbox will start a previously created sandbox. + rpc StartSandbox(StartSandboxRequest) returns (StartSandboxResponse); + + // Platform queries the platform the sandbox is going to run containers on. + // containerd will use this to generate a proper OCI spec. + rpc Platform(PlatformRequest) returns (PlatformResponse); + + // StopSandbox will stop existing sandbox instance + rpc StopSandbox(StopSandboxRequest) returns (StopSandboxResponse); + + // WaitSandbox blocks until sandbox exits. + rpc WaitSandbox(WaitSandboxRequest) returns (WaitSandboxResponse); + + // SandboxStatus will return current status of the running sandbox instance + rpc SandboxStatus(SandboxStatusRequest) returns (SandboxStatusResponse); + + // PingSandbox is a lightweight API call to check whether sandbox alive. + rpc PingSandbox(PingRequest) returns (PingResponse); + + // ShutdownSandbox must shutdown shim instance. + rpc ShutdownSandbox(ShutdownSandboxRequest) returns (ShutdownSandboxResponse); + + // SandboxMetrics retrieves metrics about a sandbox instance. + rpc SandboxMetrics(SandboxMetricsRequest) returns (SandboxMetricsResponse); +} + +message CreateSandboxRequest { + string sandbox_id = 1; + string bundle_path = 2; + repeated containerd.types.Mount rootfs = 3; + google.protobuf.Any options = 4; + string netns_path = 5; + map annotations = 6; +} + +message CreateSandboxResponse {} + +message StartSandboxRequest { + string sandbox_id = 1; +} + +message StartSandboxResponse { + uint32 pid = 1; + google.protobuf.Timestamp created_at = 2; + google.protobuf.Any spec = 3; +} + +message PlatformRequest { + string sandbox_id = 1; +} + +message PlatformResponse { + containerd.types.Platform platform = 1; +} + +message StopSandboxRequest { + string sandbox_id = 1; + uint32 timeout_secs = 2; +} + +message StopSandboxResponse {} + +message UpdateSandboxRequest { + string sandbox_id = 1; + google.protobuf.Any resources = 2; + map annotations = 3; +} + +message WaitSandboxRequest { + string sandbox_id = 1; +} + +message WaitSandboxResponse { + uint32 exit_status = 1; + google.protobuf.Timestamp exited_at = 2; +} + +message UpdateSandboxResponse {} + +message SandboxStatusRequest { + string sandbox_id = 1; + bool verbose = 2; +} + +message SandboxStatusResponse { + string sandbox_id = 1; + uint32 pid = 2; + string state = 3; + map info = 4; + google.protobuf.Timestamp created_at = 5; + google.protobuf.Timestamp exited_at = 6; + google.protobuf.Any extra = 7; +} + +message PingRequest { + string sandbox_id = 1; +} + +message PingResponse {} + +message ShutdownSandboxRequest { + string sandbox_id = 1; +} + +message ShutdownSandboxResponse {} + +message SandboxMetricsRequest { + string sandbox_id = 1; +} + +message SandboxMetricsResponse { + containerd.types.Metric metrics = 1; +} diff --git a/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox_grpc.pb.go b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox_grpc.pb.go new file mode 100644 index 00000000..d4834638 --- /dev/null +++ b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox_grpc.pb.go @@ -0,0 +1,417 @@ +//go:build !no_grpc + +// Code generated by protoc-gen-go-grpc. DO NOT EDIT. +// versions: +// - protoc-gen-go-grpc v1.2.0 +// - protoc (unknown) +// source: runtime/sandbox/v1/sandbox.proto + +package sandbox + +import ( + context "context" + grpc "google.golang.org/grpc" + codes "google.golang.org/grpc/codes" + status "google.golang.org/grpc/status" +) + +// This is a compile-time assertion to ensure that this generated file +// is compatible with the grpc package it is being compiled against. +// Requires gRPC-Go v1.32.0 or later. +const _ = grpc.SupportPackageIsVersion7 + +// SandboxClient is the client API for Sandbox service. +// +// For semantics around ctx use and closing/ending streaming RPCs, please refer to https://pkg.go.dev/google.golang.org/grpc/?tab=doc#ClientConn.NewStream. +type SandboxClient interface { + // CreateSandbox will be called right after sandbox shim instance launched. + // It is a good place to initialize sandbox environment. + CreateSandbox(ctx context.Context, in *CreateSandboxRequest, opts ...grpc.CallOption) (*CreateSandboxResponse, error) + // StartSandbox will start a previously created sandbox. + StartSandbox(ctx context.Context, in *StartSandboxRequest, opts ...grpc.CallOption) (*StartSandboxResponse, error) + // Platform queries the platform the sandbox is going to run containers on. + // containerd will use this to generate a proper OCI spec. + Platform(ctx context.Context, in *PlatformRequest, opts ...grpc.CallOption) (*PlatformResponse, error) + // StopSandbox will stop existing sandbox instance + StopSandbox(ctx context.Context, in *StopSandboxRequest, opts ...grpc.CallOption) (*StopSandboxResponse, error) + // WaitSandbox blocks until sandbox exits. + WaitSandbox(ctx context.Context, in *WaitSandboxRequest, opts ...grpc.CallOption) (*WaitSandboxResponse, error) + // SandboxStatus will return current status of the running sandbox instance + SandboxStatus(ctx context.Context, in *SandboxStatusRequest, opts ...grpc.CallOption) (*SandboxStatusResponse, error) + // PingSandbox is a lightweight API call to check whether sandbox alive. + PingSandbox(ctx context.Context, in *PingRequest, opts ...grpc.CallOption) (*PingResponse, error) + // ShutdownSandbox must shutdown shim instance. + ShutdownSandbox(ctx context.Context, in *ShutdownSandboxRequest, opts ...grpc.CallOption) (*ShutdownSandboxResponse, error) + // SandboxMetrics retrieves metrics about a sandbox instance. + SandboxMetrics(ctx context.Context, in *SandboxMetricsRequest, opts ...grpc.CallOption) (*SandboxMetricsResponse, error) +} + +type sandboxClient struct { + cc grpc.ClientConnInterface +} + +func NewSandboxClient(cc grpc.ClientConnInterface) SandboxClient { + return &sandboxClient{cc} +} + +func (c *sandboxClient) CreateSandbox(ctx context.Context, in *CreateSandboxRequest, opts ...grpc.CallOption) (*CreateSandboxResponse, error) { + out := new(CreateSandboxResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/CreateSandbox", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) StartSandbox(ctx context.Context, in *StartSandboxRequest, opts ...grpc.CallOption) (*StartSandboxResponse, error) { + out := new(StartSandboxResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/StartSandbox", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) Platform(ctx context.Context, in *PlatformRequest, opts ...grpc.CallOption) (*PlatformResponse, error) { + out := new(PlatformResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/Platform", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) StopSandbox(ctx context.Context, in *StopSandboxRequest, opts ...grpc.CallOption) (*StopSandboxResponse, error) { + out := new(StopSandboxResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/StopSandbox", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) WaitSandbox(ctx context.Context, in *WaitSandboxRequest, opts ...grpc.CallOption) (*WaitSandboxResponse, error) { + out := new(WaitSandboxResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/WaitSandbox", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) SandboxStatus(ctx context.Context, in *SandboxStatusRequest, opts ...grpc.CallOption) (*SandboxStatusResponse, error) { + out := new(SandboxStatusResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/SandboxStatus", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) PingSandbox(ctx context.Context, in *PingRequest, opts ...grpc.CallOption) (*PingResponse, error) { + out := new(PingResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/PingSandbox", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) ShutdownSandbox(ctx context.Context, in *ShutdownSandboxRequest, opts ...grpc.CallOption) (*ShutdownSandboxResponse, error) { + out := new(ShutdownSandboxResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/ShutdownSandbox", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *sandboxClient) SandboxMetrics(ctx context.Context, in *SandboxMetricsRequest, opts ...grpc.CallOption) (*SandboxMetricsResponse, error) { + out := new(SandboxMetricsResponse) + err := c.cc.Invoke(ctx, "/containerd.runtime.sandbox.v1.Sandbox/SandboxMetrics", in, out, opts...) + if err != nil { + return nil, err + } + return out, nil +} + +// SandboxServer is the server API for Sandbox service. +// All implementations must embed UnimplementedSandboxServer +// for forward compatibility +type SandboxServer interface { + // CreateSandbox will be called right after sandbox shim instance launched. + // It is a good place to initialize sandbox environment. + CreateSandbox(context.Context, *CreateSandboxRequest) (*CreateSandboxResponse, error) + // StartSandbox will start a previously created sandbox. + StartSandbox(context.Context, *StartSandboxRequest) (*StartSandboxResponse, error) + // Platform queries the platform the sandbox is going to run containers on. + // containerd will use this to generate a proper OCI spec. + Platform(context.Context, *PlatformRequest) (*PlatformResponse, error) + // StopSandbox will stop existing sandbox instance + StopSandbox(context.Context, *StopSandboxRequest) (*StopSandboxResponse, error) + // WaitSandbox blocks until sandbox exits. + WaitSandbox(context.Context, *WaitSandboxRequest) (*WaitSandboxResponse, error) + // SandboxStatus will return current status of the running sandbox instance + SandboxStatus(context.Context, *SandboxStatusRequest) (*SandboxStatusResponse, error) + // PingSandbox is a lightweight API call to check whether sandbox alive. + PingSandbox(context.Context, *PingRequest) (*PingResponse, error) + // ShutdownSandbox must shutdown shim instance. + ShutdownSandbox(context.Context, *ShutdownSandboxRequest) (*ShutdownSandboxResponse, error) + // SandboxMetrics retrieves metrics about a sandbox instance. + SandboxMetrics(context.Context, *SandboxMetricsRequest) (*SandboxMetricsResponse, error) + mustEmbedUnimplementedSandboxServer() +} + +// UnimplementedSandboxServer must be embedded to have forward compatible implementations. +type UnimplementedSandboxServer struct { +} + +func (UnimplementedSandboxServer) CreateSandbox(context.Context, *CreateSandboxRequest) (*CreateSandboxResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method CreateSandbox not implemented") +} +func (UnimplementedSandboxServer) StartSandbox(context.Context, *StartSandboxRequest) (*StartSandboxResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method StartSandbox not implemented") +} +func (UnimplementedSandboxServer) Platform(context.Context, *PlatformRequest) (*PlatformResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method Platform not implemented") +} +func (UnimplementedSandboxServer) StopSandbox(context.Context, *StopSandboxRequest) (*StopSandboxResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method StopSandbox not implemented") +} +func (UnimplementedSandboxServer) WaitSandbox(context.Context, *WaitSandboxRequest) (*WaitSandboxResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method WaitSandbox not implemented") +} +func (UnimplementedSandboxServer) SandboxStatus(context.Context, *SandboxStatusRequest) (*SandboxStatusResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method SandboxStatus not implemented") +} +func (UnimplementedSandboxServer) PingSandbox(context.Context, *PingRequest) (*PingResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method PingSandbox not implemented") +} +func (UnimplementedSandboxServer) ShutdownSandbox(context.Context, *ShutdownSandboxRequest) (*ShutdownSandboxResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method ShutdownSandbox not implemented") +} +func (UnimplementedSandboxServer) SandboxMetrics(context.Context, *SandboxMetricsRequest) (*SandboxMetricsResponse, error) { + return nil, status.Errorf(codes.Unimplemented, "method SandboxMetrics not implemented") +} +func (UnimplementedSandboxServer) mustEmbedUnimplementedSandboxServer() {} + +// UnsafeSandboxServer may be embedded to opt out of forward compatibility for this service. +// Use of this interface is not recommended, as added methods to SandboxServer will +// result in compilation errors. +type UnsafeSandboxServer interface { + mustEmbedUnimplementedSandboxServer() +} + +func RegisterSandboxServer(s grpc.ServiceRegistrar, srv SandboxServer) { + s.RegisterService(&Sandbox_ServiceDesc, srv) +} + +func _Sandbox_CreateSandbox_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(CreateSandboxRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).CreateSandbox(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/CreateSandbox", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).CreateSandbox(ctx, req.(*CreateSandboxRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_StartSandbox_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(StartSandboxRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).StartSandbox(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/StartSandbox", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).StartSandbox(ctx, req.(*StartSandboxRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_Platform_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(PlatformRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).Platform(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/Platform", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).Platform(ctx, req.(*PlatformRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_StopSandbox_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(StopSandboxRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).StopSandbox(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/StopSandbox", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).StopSandbox(ctx, req.(*StopSandboxRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_WaitSandbox_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(WaitSandboxRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).WaitSandbox(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/WaitSandbox", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).WaitSandbox(ctx, req.(*WaitSandboxRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_SandboxStatus_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(SandboxStatusRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).SandboxStatus(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/SandboxStatus", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).SandboxStatus(ctx, req.(*SandboxStatusRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_PingSandbox_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(PingRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).PingSandbox(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/PingSandbox", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).PingSandbox(ctx, req.(*PingRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_ShutdownSandbox_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(ShutdownSandboxRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).ShutdownSandbox(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/ShutdownSandbox", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).ShutdownSandbox(ctx, req.(*ShutdownSandboxRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Sandbox_SandboxMetrics_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(SandboxMetricsRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(SandboxServer).SandboxMetrics(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: "/containerd.runtime.sandbox.v1.Sandbox/SandboxMetrics", + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(SandboxServer).SandboxMetrics(ctx, req.(*SandboxMetricsRequest)) + } + return interceptor(ctx, in, info, handler) +} + +// Sandbox_ServiceDesc is the grpc.ServiceDesc for Sandbox service. +// It's only intended for direct use with grpc.RegisterService, +// and not to be introspected or modified (even as a copy) +var Sandbox_ServiceDesc = grpc.ServiceDesc{ + ServiceName: "containerd.runtime.sandbox.v1.Sandbox", + HandlerType: (*SandboxServer)(nil), + Methods: []grpc.MethodDesc{ + { + MethodName: "CreateSandbox", + Handler: _Sandbox_CreateSandbox_Handler, + }, + { + MethodName: "StartSandbox", + Handler: _Sandbox_StartSandbox_Handler, + }, + { + MethodName: "Platform", + Handler: _Sandbox_Platform_Handler, + }, + { + MethodName: "StopSandbox", + Handler: _Sandbox_StopSandbox_Handler, + }, + { + MethodName: "WaitSandbox", + Handler: _Sandbox_WaitSandbox_Handler, + }, + { + MethodName: "SandboxStatus", + Handler: _Sandbox_SandboxStatus_Handler, + }, + { + MethodName: "PingSandbox", + Handler: _Sandbox_PingSandbox_Handler, + }, + { + MethodName: "ShutdownSandbox", + Handler: _Sandbox_ShutdownSandbox_Handler, + }, + { + MethodName: "SandboxMetrics", + Handler: _Sandbox_SandboxMetrics_Handler, + }, + }, + Streams: []grpc.StreamDesc{}, + Metadata: "runtime/sandbox/v1/sandbox.proto", +} diff --git a/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox_ttrpc.pb.go b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox_ttrpc.pb.go new file mode 100644 index 00000000..7fb6ea26 --- /dev/null +++ b/vendor/github.com/containerd/containerd/api/runtime/sandbox/v1/sandbox_ttrpc.pb.go @@ -0,0 +1,172 @@ +// Code generated by protoc-gen-go-ttrpc. DO NOT EDIT. +// source: runtime/sandbox/v1/sandbox.proto +package sandbox + +import ( + context "context" + ttrpc "github.com/containerd/ttrpc" +) + +type TTRPCSandboxService interface { + CreateSandbox(context.Context, *CreateSandboxRequest) (*CreateSandboxResponse, error) + StartSandbox(context.Context, *StartSandboxRequest) (*StartSandboxResponse, error) + Platform(context.Context, *PlatformRequest) (*PlatformResponse, error) + StopSandbox(context.Context, *StopSandboxRequest) (*StopSandboxResponse, error) + WaitSandbox(context.Context, *WaitSandboxRequest) (*WaitSandboxResponse, error) + SandboxStatus(context.Context, *SandboxStatusRequest) (*SandboxStatusResponse, error) + PingSandbox(context.Context, *PingRequest) (*PingResponse, error) + ShutdownSandbox(context.Context, *ShutdownSandboxRequest) (*ShutdownSandboxResponse, error) + SandboxMetrics(context.Context, *SandboxMetricsRequest) (*SandboxMetricsResponse, error) +} + +func RegisterTTRPCSandboxService(srv *ttrpc.Server, svc TTRPCSandboxService) { + srv.RegisterService("containerd.runtime.sandbox.v1.Sandbox", &ttrpc.ServiceDesc{ + Methods: map[string]ttrpc.Method{ + "CreateSandbox": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req CreateSandboxRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.CreateSandbox(ctx, &req) + }, + "StartSandbox": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req StartSandboxRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.StartSandbox(ctx, &req) + }, + "Platform": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req PlatformRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.Platform(ctx, &req) + }, + "StopSandbox": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req StopSandboxRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.StopSandbox(ctx, &req) + }, + "WaitSandbox": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req WaitSandboxRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.WaitSandbox(ctx, &req) + }, + "SandboxStatus": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req SandboxStatusRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.SandboxStatus(ctx, &req) + }, + "PingSandbox": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req PingRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.PingSandbox(ctx, &req) + }, + "ShutdownSandbox": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req ShutdownSandboxRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.ShutdownSandbox(ctx, &req) + }, + "SandboxMetrics": func(ctx context.Context, unmarshal func(interface{}) error) (interface{}, error) { + var req SandboxMetricsRequest + if err := unmarshal(&req); err != nil { + return nil, err + } + return svc.SandboxMetrics(ctx, &req) + }, + }, + }) +} + +type ttrpcsandboxClient struct { + client *ttrpc.Client +} + +func NewTTRPCSandboxClient(client *ttrpc.Client) TTRPCSandboxService { + return &ttrpcsandboxClient{ + client: client, + } +} + +func (c *ttrpcsandboxClient) CreateSandbox(ctx context.Context, req *CreateSandboxRequest) (*CreateSandboxResponse, error) { + var resp CreateSandboxResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "CreateSandbox", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) StartSandbox(ctx context.Context, req *StartSandboxRequest) (*StartSandboxResponse, error) { + var resp StartSandboxResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "StartSandbox", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) Platform(ctx context.Context, req *PlatformRequest) (*PlatformResponse, error) { + var resp PlatformResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "Platform", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) StopSandbox(ctx context.Context, req *StopSandboxRequest) (*StopSandboxResponse, error) { + var resp StopSandboxResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "StopSandbox", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) WaitSandbox(ctx context.Context, req *WaitSandboxRequest) (*WaitSandboxResponse, error) { + var resp WaitSandboxResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "WaitSandbox", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) SandboxStatus(ctx context.Context, req *SandboxStatusRequest) (*SandboxStatusResponse, error) { + var resp SandboxStatusResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "SandboxStatus", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) PingSandbox(ctx context.Context, req *PingRequest) (*PingResponse, error) { + var resp PingResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "PingSandbox", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) ShutdownSandbox(ctx context.Context, req *ShutdownSandboxRequest) (*ShutdownSandboxResponse, error) { + var resp ShutdownSandboxResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "ShutdownSandbox", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} + +func (c *ttrpcsandboxClient) SandboxMetrics(ctx context.Context, req *SandboxMetricsRequest) (*SandboxMetricsResponse, error) { + var resp SandboxMetricsResponse + if err := c.client.Call(ctx, "containerd.runtime.sandbox.v1.Sandbox", "SandboxMetrics", req, &resp); err != nil { + return nil, err + } + return &resp, nil +} diff --git a/vendor/github.com/containerd/shimtest/README.md b/vendor/github.com/containerd/shimtest/README.md index b563ca58..6b097273 100644 --- a/vendor/github.com/containerd/shimtest/README.md +++ b/vendor/github.com/containerd/shimtest/README.md @@ -40,7 +40,8 @@ Tests are driven by one or more JSON configuration files. See | `uid` | int | UID to run as; defaults to the current user's UID. If set to a value different from the current UID and the effective UID is 0, the harness re-execs itself as that user via `sudo` | | `gid` | int | GID to run as | | `format_mounts` | bool | Provide the rootfs as formatted erofs/ext4 images with a `format/mkdir/overlay` descriptor for the shim to mount. Default (`false`) extracts the rootfs and provides a pre-mounted overlay (or plain directory when rootless) | -| `skip` | []string | Feature names to skip (`exec`, `layers`, `net`, `oom`, `transfer`, `uds`) | +| `provide_network` | bool | Have `NetworkSuite` set up a container's network connectivity itself (a dedicated network namespace plus a `slirp4netns` process attached to it) instead of assuming the shim already gives containers a working default network path. Leave `false` for shims with real networking of their own (e.g. VM-based shims); set `true` only for shims with no network setup at all, which otherwise leave a container in an empty, unconfigured network namespace. Requires `slirp4netns` on `PATH`; skipped otherwise | +| `skip` | []string | Feature names to skip (`exec`, `layers`, `net`, `oom`, `sandbox`, `transfer`, `uds`) | | `env` | map | Additional environment variables for the test run | | `debug` | bool | Enable debug logging on the shim | @@ -114,7 +115,20 @@ config, the tree is `TestShim//`. | `OutboundTCP` | net | A container's init process opens an outbound TCP connection to a host-reachable endpoint and completes a round trip. Implementation-neutral: does not assume any particular networking mechanism, only that a container has working default outbound TCP connectivity, as "host networking" would provide. | | `OutboundUDP` | net | A container's init process exchanges a UDP datagram with a host-reachable endpoint. Same neutrality as `OutboundTCP`, for the datagram path. | | `DNSResolve` | net | A container's init process resolves a real external hostname (`example.com`, forcing an actual DNS query — not answered from `/etc/hosts`) and gets back valid IP addresses. Requires outbound internet access from the test host. | -| `Stress` | (per feature) | Long-running concurrent stress run. Composes subtests from the enabled features (currently transfer: stat/write/read). Each subtest runs as a goroutine until the test deadline approaches or any one fails (which cancels the rest). Skipped under `-test.short`. | +| `InboundTCPListen` | net | A container's init process binds a TCP listener on a concrete port chosen by the test (never one discovered from the container's own output, which some shims cannot reliably report — see pickHostPort), and the host-side test connects to it and confirms the container actually received the data. With `provide_network`, the test also registers its own port forward before the container even starts. Verifies the inbound (listen/accept) side of the default networking contract: ports bound inside the container must be reachable from the host. | +| `LoopbackWithinContainer` | net | A container's init process binds a listener on 127.0.0.1 and connects to it from within the same container over loopback. Verifies that the container's loopback interface works; a prerequisite for between-container localhost connectivity when containers share a network namespace. | +| `Lifecycle` | sandbox | Full sandbox lifecycle: `CreateSandbox` → `StartSandbox` → `SandboxStatus`(ready) → `StopSandbox` → `SandboxStatus`(stopped) → `ShutdownSandbox`. Verifies state transitions and that the bootstrap version is ≥ 3. Linux only. | +| `Platform` | sandbox | `Platform` RPC returns a non-empty OS and Architecture. Linux only. | +| `Ping` | sandbox | `PingSandbox` succeeds while the sandbox is running. Linux only. | +| `SingleContainer` | sandbox | One member container created on the shared connection produces expected output and exits cleanly. Linux only. | +| `MultipleContainers` | sandbox | Three member containers run concurrently in one sandbox VM, each with isolated rootfs and independent output. Linux only. | +| `ContainerLifecycleIndependence` | sandbox | Deleting one member container leaves the sandbox and other containers running; `SandboxStatus` remains ready. Linux only. | +| `StatusAfterStop` | sandbox | `SandboxStatus` after `StopSandbox` does not report ready; a second `StopSandbox` is idempotent. Linux only. | +| `WaitUnblocksOnStop` | sandbox | `WaitSandbox` returns within 15 s after `StopSandbox`. Linux only. | +| `CreateTwiceRejected` | sandbox | A second `CreateSandbox` on the same shim process returns `AlreadyExists`. Linux only. | +| `StartWithoutCreateRejected` | sandbox | `StartSandbox` before `CreateSandbox` returns `FailedPrecondition`. Linux only. | +| `ResourceReleaseOnShutdown` | sandbox | After `ShutdownSandbox`, no per-container mount points remain in the shim's mount namespace. Linux only. | +| `Stress` | (per feature) | Long-running concurrent stress run. Composes subtests from the enabled features (Transfer, Sandbox). Each subtest runs as a goroutine until the test deadline approaches or any one fails (which cancels the rest). Skipped under `-test.short`. The Sandbox subtest also checks host RSS growth and mount-point leaks after each run. | A separate top-level fuzz target exists alongside `TestShim`: @@ -153,6 +167,33 @@ Benchmarks live under `BenchmarkShim//`. | `StdioRoundTrip` | exec | Stdio write/read at 8B, 4KB, 4MB | | `UDSRoundTrip` | uds | UDS forwarded-socket throughput in both directions (HostToContainer, ContainerToHost) at 8B, 4KB, 4MB | | `ThirtyLayers` | layers | Bring up a container with a 30-layer erofs rootfs (same shape as `HundredLayers`, smaller). Reports per-phase metrics (`ms/shim-start`, `ms/create`, `ms/task-start`, `ms/total`) so multi-layer mount overhead can be localized. Requires `format_mounts=true` | +| `ContainerCreate` | sandbox | Per-container create/start/wait/delete cycle inside a single shared sandbox VM. The sandbox is started once before the `b.N` loop; each iteration adds one member container, runs it to completion, and removes it. Reports `ms/create`, `ms/start`, `ms/wait`, `ms/delete`, `ms/total`, and `ms/sandbox-start` (one-time amortised cost). Compare `ms/create` and `ms/total` with `RunSuite.Lifecycle` to quantify the marginal cost of a sandbox container versus a fresh-VM container. Linux only. | + +### Member-container workload contracts + +These tests (all gated on the `sandbox` feature) verify the shim API contract for member-container workloads: status fields, host-network sandboxes, exec, shared endpoints, shared namespaces, volumes, and network scoping. + +| Test | Feature | Linux only | Verifies | +|---|---|---|---| +| `StatusReportsPidAndCreatedAt` | sandbox | no | `SandboxStatus` returns a non-zero `pid` and a non-zero `created_at` after `StartSandbox`. Both let a caller reference the sandbox's namespaces (`/proc//ns/*`) and report its age (e.g. as CRI's `PodSandboxStatus.CreatedAt` does). `SandboxStatus.Info` must carry `pid` and `state` entries. | +| `HostNetworkNoNetworkSandbox` | sandbox | no | A sandbox created with an empty `netns_path` (no network sandbox provided) must succeed. Member containers must run normally. The shim must accept the no-isolation case without error. | +| `MemberContainerExec` | sandbox | no | `Task.Exec` + `Task.Start(ExecID)` must run an additional process inside a running member container with correct output and exit-status propagation. Underpins any exec-into-a-running-container use case (interactive exec, health probes, sidecar tooling). | +| `CrossContainerViaUDS` | sandbox | yes | Two exec processes running inside a sandbox member container both connect to a shared host-side UNIX domain socket forwarded in via the `uds` mount type. Proves that multiple processes in a sandbox can reach a common host-forwarded endpoint. | +| `NetworkSandboxHeldOpen` | sandbox | yes | When a non-empty `netns_path` is provided in `CreateSandboxRequest`, the shim must hold the network sandbox resource open for the sandbox lifetime. The path must remain reachable while the sandbox is ready and must be releasable after `StopSandbox`. No special privileges required. | +| `NetworkSandboxPathInStatus` | sandbox | yes | If a network sandbox path was provided in `CreateSandboxRequest`, `SandboxStatus.Info["networkSandboxPath"]` should report the same path. Informational (absence does not hard-fail); the field is optional in the base protocol. No special privileges required. | +| `ContainerOutboundTCP` | sandbox | no | A process exec'd into a member container must be able to resolve a real external hostname, proving it has a working outbound network path. Implementation-neutral: does not assume any particular networking mechanism (native netns, virtual NIC, or otherwise). DNS is used here (rather than a raw TCP round trip) because `Task.Exec` has no stdin plumbing; `NetworkSuite` separately covers TCP and UDP round trips end-to-end on the legacy path. No special privileges required. | +| `ContainerTrafficScopedToNetworkSandbox` | sandbox | yes | When a non-empty `netns_path` is provided, a member container's outbound traffic must actually originate from within that network sandbox, not merely have the path pinned open. Uses a veth pair fully contained in a real network namespace (unreachable from outside it) as a falsifiable probe: a successful round trip is only possible if the container's traffic originates inside the sandbox. Requires root (CAP_SYS_ADMIN) to create the namespace and interfaces; skipped otherwise. | +| `InboundToNetnsScopedListener` | sandbox | yes | The inbound mirror of `ContainerTrafficScopedToNetworkSandbox`: a member container binds an ephemeral TCP listener (port discovered via its stdout) and must be reachable by a caller connecting from inside the same network sandbox namespace. The same veth-pair-isolated address is used as a falsifiable probe: only a caller inside the sandbox can reach the listener. Requires root (CAP_SYS_ADMIN); skipped otherwise. Linux only. | +| `MemberContainersShareNetwork` | sandbox | yes | Member containers of the same sandbox must share a network stack: a listener started by one member container (on an ephemeral port discovered at runtime via its stdout) must be reachable from a second, independently created member container via loopback (127.0.0.1), with no explicit network configuration on either container. Implementation-neutral: does not assume any particular mechanism (a real shared network namespace, a shared virtual interface, or any other approach). No special privileges required. | +| `MemberContainerHostVolume` | sandbox | yes | A member container's OCI spec may include a "bind" mount referencing a host directory (e.g. as CRI's hostPath volumes or Kubernetes `RecursiveReadOnly=false` bind mounts produce). The shim must honor it as a live, two-way share, not a one-time copy: a file updated on the host after the container has already started must become visible inside it. Implementation-neutral: does not assume any particular sharing mechanism. No special privileges required. | +| `MemberContainersSharePID` | sandbox | yes | When a member container's OCI spec carries a host path on its PID namespace entry (e.g. as a caller uses to express Kubernetes' `shareProcessNamespace: true` or `hostPID: true`), the shim must place that container in a PID namespace shared with its sandbox peers. A process started in one member container must be visible — by PID and argv — via `/proc` in a second, independently created member container. Implementation-neutral: does not assume any particular sharing mechanism. No special privileges required. | +| `MemberContainersSharePIDKillScopedToOwnContainer` | sandbox | yes | The converse safety property to `MemberContainersSharePID`: `Task.Kill` and `Task.Pids` identify processes by container ID (and, for `Kill`, exec ID) — never by a raw PID, which the request shape doesn't even carry — so they must stay scoped to the named container even though its processes are visible to, and share a namespace with, its sandbox peers. Killing one member container with `All: true` must not affect a peer sharing the same PID namespace. No special privileges required. | +| `MemberContainersShareIPC` | sandbox | yes | When a member container's OCI spec carries a host path on its IPC namespace entry (e.g. as a caller uses to express Kubernetes' default of always sharing one IPC namespace across a pod's containers), the shim must place that container in an IPC namespace shared with its sandbox peers. A SysV shared memory segment created by one member container must be visible — by its well-known key — to a second, independently created member container. Implementation-neutral: does not assume any particular sharing mechanism. No special privileges required. | +| `MemberContainersShareDevShm` | sandbox | yes | Member containers sharing an IPC namespace (see `MemberContainersShareIPC`) must also get a shared `/dev/shm`, even though every member container's OCI spec carries the exact same, independent-looking `{Type: "tmpfs", Destination: "/dev/shm"}` mount with no separate "shared" signal. A POSIX-shared-memory-style write made through an `mmap(MAP_SHARED)` mapping by one member container must be visible through an independent mapping in a second, independently created member container. Implementation-neutral: does not assume any particular sharing mechanism. No special privileges required. | +| `MemberContainersDevShmNotSharedWithoutIPC` | sandbox | yes | The converse of `MemberContainersShareDevShm`: a member container that does not request IPC sharing must get its own private `/dev/shm`, even though its OCI spec's `/dev/shm` mount is identical in shape to a sharing container's. Guards against a shim inferring sharing from the mount's shape rather than the IPC-sharing signal. No special privileges required. | +| `MemberContainerOOMIsolation` | sandbox | yes | The OOM-specific counterpart to `ContainerLifecycleIndependence` (which only covers a peer's graceful exit): when the kernel OOM-kills one member container (a memory-limited container running a memory-hungry workload, as in the standalone `OOM` test), a sibling member container with no memory limit must be unaffected, and the sandbox itself must remain ready. No special privileges required. | +| `MemberContainersShareUTS` | sandbox | yes | When a member container's OCI spec carries a host path on its UTS namespace entry (e.g. as a caller uses to express Kubernetes' default of sharing one hostname across a pod's containers), the shim must place that container in a UTS namespace shared with its sandbox peers. A hostname change made via `sethostname(2)` (the standard `hostname ` command) by one member container must be visible — via the kernel-reported hostname, not a file — to a second, independently created member container, even after the first container has exited, proving the shared namespace is owned by the sandbox rather than tied to the setter's lifetime. Implementation-neutral: does not assume any particular sharing mechanism. The container requests `CAP_SYS_ADMIN` explicitly since the base container spec grants no capabilities; the test is skipped, not failed, if the shim does not grant it. | +| `MemberContainersUTSNotShared` | sandbox | yes | The converse of `MemberContainersShareUTS`: a member container that does not request UTS sharing must not observe a peer's hostname change, even though both belong to the same sandbox. Guards against a shim sharing UTS namespaces merely because containers are sandbox peers, rather than because sharing was requested. | ## Using shimtest in your shim's CI @@ -234,7 +275,7 @@ unbounded `Stress` run, and run active fuzzing as its own step: - **`uid`**: omit to run as the runner user. Set explicitly when you want the harness to `sudo` re-exec itself or rewrite the profile. - **`skip`**: list of feature names to disable. Currently meaningful - values are `exec`, `layers`, `net`, `oom`, `transfer`, and `uds` — + values are `exec`, `layers`, `net`, `oom`, `sandbox`, `transfer`, and `uds` — useful when your shim doesn't implement transfer/UDS forwarding, multi-layer rootfs descriptors, or when running rootless without cgroup delegation. diff --git a/vendor/github.com/containerd/shimtest/config.go b/vendor/github.com/containerd/shimtest/config.go index a7f6b7c7..621e8adb 100644 --- a/vendor/github.com/containerd/shimtest/config.go +++ b/vendor/github.com/containerd/shimtest/config.go @@ -59,4 +59,21 @@ type Config struct { // Debug enables verbose logging from the shim. Debug bool + + // ProvideNetwork tells NetworkSuite to set up a container's + // network connectivity itself (a dedicated network namespace plus + // a slirp4netns process attached to it; see attachContainerNetwork + // in helpers_netns_linux.go) rather than assuming the shim already + // provides a working default network path. + // + // Leave this false for any shim that already gives containers real + // networking on its own (e.g. a VM-based shim bridging guest + // traffic to the host) — NetworkSuite then tests that default path + // directly, exactly as it always has. Set it true only for shims + // with no network setup of their own, which otherwise leave a + // container in an empty, unconfigured network namespace: the + // suite's tests would otherwise be unsatisfiable through no fault + // of the shim, since providing container networking isn't part of + // the shim v2 API contract at all. + ProvideNetwork bool } diff --git a/vendor/github.com/containerd/shimtest/helpers.go b/vendor/github.com/containerd/shimtest/helpers.go index af29c264..f335aaa2 100644 --- a/vendor/github.com/containerd/shimtest/helpers.go +++ b/vendor/github.com/containerd/shimtest/helpers.go @@ -201,6 +201,32 @@ func withExtraMounts(mounts ...specs.Mount) func(*specs.Spec) { } } +// withHostPathNamespace returns a CreateOCISpec opt that sets (or adds) +// a namespace entry of the given type with a host path. A non-empty Path +// on an IPC/PID/network namespace entry is how an OCI spec requests that +// a container join a namespace shared with others, rather than getting a +// fresh, isolated one — for example, a host "/proc//ns/" +// path, as containerd's WithPodNamespaces sets for pod-level namespace +// sharing. The actual path value here is a placeholder — it only needs +// to be non-empty, since the shim's job is to recognize that a host path +// is present at all and substitute its own guest-side shared namespace, +// not to interpret the path itself (which is meaningless off the host +// that produced it). +func withHostPathNamespace(nsType specs.LinuxNamespaceType, path string) func(*specs.Spec) { + return func(s *specs.Spec) { + if s.Linux == nil { + s.Linux = &specs.Linux{} + } + for i, ns := range s.Linux.Namespaces { + if ns.Type == nsType { + s.Linux.Namespaces[i].Path = path + return + } + } + s.Linux.Namespaces = append(s.Linux.Namespaces, specs.LinuxNamespace{Type: nsType, Path: path}) + } +} + // withMemoryLimit returns a CreateOCISpec opt that sets the memory // limit (in bytes) on the spec, with swap clamped equal to the limit // so the container cannot grow via swap before the OOM killer fires. @@ -216,6 +242,58 @@ func withMemoryLimit(bytes int64) func(*specs.Spec) { } } +// withNewNetworkNamespace returns a CreateOCISpec opt that ensures the +// spec requests a fresh (unshared) network namespace, appending one if +// none is already present. createOCISpec already adds one for rootless +// containers as part of dropping privileges; this opt exists so a +// caller can request the same isolation for a root container too, +// where createOCISpec otherwise leaves the container in the host's own +// network namespace. Tests that provide their own container networking +// (see attachContainerNetwork) need this so the namespace they attach +// to is exclusively the container's, root or not. +func withNewNetworkNamespace() func(*specs.Spec) { + return func(s *specs.Spec) { + if s.Linux == nil { + s.Linux = &specs.Linux{} + } + for _, ns := range s.Linux.Namespaces { + if ns.Type == specs.NetworkNamespace { + return + } + } + s.Linux.Namespaces = append(s.Linux.Namespaces, specs.LinuxNamespace{Type: specs.NetworkNamespace}) + } +} + +// withCapabilities returns a CreateOCISpec opt that grants the given +// capabilities (e.g. "CAP_SYS_ADMIN") in the container's Bounding, +// Effective, and Permitted sets, in addition to whatever the spec +// already carries. +// +// The base spec createOCISpec builds has no Capabilities section at +// all, which every runtime this repo has been tested against +// (confirmed empirically with runc) treats as granting the container +// *no* capabilities whatsoever — not even the small default set a +// container engine like Docker or containerd/CRI would normally add — +// regardless of the privilege level of the process driving the test +// suite on the host. A test that needs a capability inside the +// container must therefore request it explicitly here, on the +// container's own spec; this is a property of the requested container, +// not of the host, and needs no host-level privilege to ask for. +// Whether the shim actually honors the request is exactly what a test +// using this opt is checking. +func withCapabilities(caps ...string) func(*specs.Spec) { + return func(s *specs.Spec) { + if s.Process.Capabilities == nil { + s.Process.Capabilities = &specs.LinuxCapabilities{} + } + c := s.Process.Capabilities + c.Bounding = append(c.Bounding, caps...) + c.Effective = append(c.Effective, caps...) + c.Permitted = append(c.Permitted, caps...) + } +} + // shimSetup resolves the shim binary, creates a bundle directory, and // builds rootfs mounts from the embedded testbin. Returns the shim // binary's absolute path, the bundle directory, and the rootfs diff --git a/vendor/github.com/containerd/shimtest/helpers_netns_linux.go b/vendor/github.com/containerd/shimtest/helpers_netns_linux.go new file mode 100644 index 00000000..9dc1a617 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_netns_linux.go @@ -0,0 +1,208 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "bytes" + "encoding/json" + "fmt" + "io" + "net" + "os" + "os/exec" + "path/filepath" + "strconv" + "testing" + "time" +) + +// The fixed addresses slirp4netns assigns when attached with --configure: +// slirpGuestAddr is the container-side tap address, slirpHostAddr is the +// built-in gateway that proxies to the host's own loopback (127.0.0.1), +// and slirpDNSAddr is slirp's internal DNS proxy. +const ( + slirpGuestAddr = "10.0.2.100" + slirpHostAddr = "10.0.2.2" + slirpDNSAddr = "10.0.2.3" +) + +// containerNetwork represents connectivity attached to a single +// container's network namespace via a dedicated slirp4netns process. +type containerNetwork struct { + cmd *exec.Cmd + apiSocket string +} + +// attachContainerNetwork attaches slirp4netns to the network namespace +// of the container process identified by initPID (as returned by +// Task.Create's CreateTaskResponse.Pid), giving the container outbound +// connectivity to the host (reachable at slirpHostAddr) and enabling +// inbound port forwards via AddInboundForward. Works identically for +// root and rootless shims: slirp4netns needs no special privileges of +// its own, only a target network namespace to attach to (plus +// --userns-path when that namespace's owning user namespace isn't the +// caller's own, i.e. whenever the shim under test isn't running as +// root). +// +// Must be called after Task.Create (so initPID's network namespace +// already exists) and before Task.Start (so the container's entrypoint +// sees a fully configured network from its very first syscall). This +// ordering, together with slirp4netns's --ready-fd, makes the setup +// race-free: the test blocks on a pipe that slirp4netns writes to only +// once the interface is fully configured, so no retry logic is needed +// in the container's own networking code. +// +// Skips the test if slirp4netns is not on PATH: providing container +// networking is the test harness's job here, not the shim's, so a +// missing tool is an environment limitation, not a conformance +// failure. +func attachContainerNetwork(tb testing.TB, initPID uint32) *containerNetwork { + tb.Helper() + + slirpPath, err := exec.LookPath("slirp4netns") + if err != nil { + tb.Skip("net tests require slirp4netns (not found on PATH) to provide container networking") + } + + apiSocket := filepath.Join(tb.TempDir(), "slirp-api.sock") + + readyR, readyW, err := os.Pipe() + if err != nil { + tb.Fatalf("attachContainerNetwork: pipe: %v", err) + } + defer readyW.Close() + + args := []string{ + "--configure", + "--api-socket", apiSocket, + "--ready-fd", "3", + } + if os.Getuid() != 0 { + // The container's network namespace is owned by its own user + // namespace when the shim isn't running as root; slirp4netns + // must enter that user namespace to configure the interface. + args = append(args, "--userns-path", fmt.Sprintf("/proc/%d/ns/user", initPID)) + } + args = append(args, strconv.FormatUint(uint64(initPID), 10), "tap0") + + cmd := exec.Command(slirpPath, args...) + cmd.ExtraFiles = []*os.File{readyW} + var stderr bytes.Buffer + cmd.Stderr = &stderr + if err := cmd.Start(); err != nil { + tb.Fatalf("attachContainerNetwork: start slirp4netns: %v", err) + } + // This explicit close is what actually matters: the parent must drop + // its own copy of the write end now that the child has inherited one + // via ExtraFiles, or the parent's lingering reference keeps the pipe + // open and readyR's Read below blocks forever if slirp4netns exits + // without ever writing to it. The deferred Close above still runs + // after this, but by then the fd is already closed — Close is safe + // to call twice in Go, so that's a harmless no-op, not a double + // free. Its only real job is covering the path that returns before + // reaching here: cmd.Start's own failure just above. + readyW.Close() + + readyErr := make(chan error, 1) + go func() { + buf := make([]byte, 1) + _, err := readyR.Read(buf) + readyErr <- err + }() + + select { + case err := <-readyErr: + if err != nil { + cmd.Process.Kill() //nolint:errcheck + tb.Fatalf("attachContainerNetwork: waiting for slirp4netns readiness: %v; stderr: %s", err, stderr.String()) + } + case <-time.After(10 * time.Second): + cmd.Process.Kill() //nolint:errcheck + tb.Fatalf("attachContainerNetwork: slirp4netns did not become ready within 10s; stderr: %s", stderr.String()) + } + readyR.Close() + + tb.Cleanup(func() { + cmd.Process.Kill() //nolint:errcheck + cmd.Wait() //nolint:errcheck + }) + + return &containerNetwork{cmd: cmd, apiSocket: apiSocket} +} + +// AddInboundForward registers a slirp4netns port forward so that a +// connection to the host at 127.0.0.1:hostPort is delivered to the +// container at guestPort. Because it targets a fixed, predetermined +// port rather than one discovered at runtime, this can (and should) be +// called before Task.Start: the forward is in place before the +// container's listener even exists, so there is no window in which an +// early host-side connection attempt could race the forward's +// registration. +func (n *containerNetwork) AddInboundForward(tb testing.TB, hostPort, guestPort int) { + tb.Helper() + + req := map[string]any{ + "execute": "add_hostfwd", + "arguments": map[string]any{ + "proto": "tcp", + "host_addr": "127.0.0.1", + "host_port": hostPort, + "guest_addr": slirpGuestAddr, + "guest_port": guestPort, + }, + } + body, err := json.Marshal(req) + if err != nil { + tb.Fatalf("AddInboundForward: marshal request: %v", err) + } + + conn, err := net.Dial("unix", n.apiSocket) + if err != nil { + tb.Fatalf("AddInboundForward: dial slirp4netns API socket: %v", err) + } + defer conn.Close() + + if _, err := conn.Write(body); err != nil { + tb.Fatalf("AddInboundForward: write request: %v", err) + } + // The API protocol has no keep-alive or request framing beyond the + // JSON body itself; the client signals the end of the request by + // shutting down the write side of the connection. + if cw, ok := conn.(interface{ CloseWrite() error }); ok { + if err := cw.CloseWrite(); err != nil { + tb.Fatalf("AddInboundForward: close write side: %v", err) + } + } + + resp, err := io.ReadAll(conn) + if err != nil { + tb.Fatalf("AddInboundForward: read response: %v", err) + } + + var result struct { + Return map[string]any `json:"return"` + Error any `json:"error"` + } + if err := json.Unmarshal(resp, &result); err != nil { + tb.Fatalf("AddInboundForward: unmarshal response %q: %v", resp, err) + } + if result.Error != nil { + tb.Fatalf("AddInboundForward: slirp4netns error: %v", result.Error) + } +} diff --git a/vendor/github.com/containerd/shimtest/helpers_netns_other.go b/vendor/github.com/containerd/shimtest/helpers_netns_other.go new file mode 100644 index 00000000..fe5e74ad --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_netns_other.go @@ -0,0 +1,41 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import "testing" + +// The fixed slirp4netns addresses referenced by network_suite.go; see +// helpers_netns_linux.go. Declared here too so non-Linux builds compile; +// attachContainerNetwork always skips before any of them are used. +const ( + slirpHostAddr = "10.0.2.2" + slirpDNSAddr = "10.0.2.3" +) + +type containerNetwork struct{} + +func attachContainerNetwork(tb testing.TB, _ uint32) *containerNetwork { + tb.Helper() + tb.Skip("net tests requiring in-test container networking are Linux-only") + return nil +} + +func (n *containerNetwork) AddInboundForward(tb testing.TB, _, _ int) { + tb.Helper() +} diff --git a/vendor/github.com/containerd/shimtest/helpers_networksandbox_linux.go b/vendor/github.com/containerd/shimtest/helpers_networksandbox_linux.go new file mode 100644 index 00000000..27da774b --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_networksandbox_linux.go @@ -0,0 +1,69 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "os" + "path/filepath" + "testing" + + "golang.org/x/sys/unix" +) + +// createNetworkSandbox creates a host-side network sandbox resource and +// returns its file path. Cleanup is registered on tb. +// +// The network sandbox is represented as a regular file. This is sufficient +// to test the shim's API contract — that it accepts a netns_path in +// CreateSandboxRequest, holds the resource open for the sandbox lifetime, and +// reports the path in SandboxStatus.Info — without requiring root privileges +// or kernel support for bind-mounting namespace files. +// +// In production, a caller provides a bind-mounted network namespace file +// created before calling CreateSandbox. The shim is expected to open the +// path, pin it for the sandbox lifetime, and (when running as root) enter +// the namespace so that member-container traffic originates from the +// provided netns. Entering the namespace requires CAP_SYS_ADMIN; when the +// shim lacks that capability it logs a warning and continues without +// entering. +func createNetworkSandbox(tb testing.TB) string { + tb.Helper() + + nsPath := filepath.Join(tb.TempDir(), "network-sandbox") + + f, err := os.OpenFile(nsPath, os.O_CREATE|os.O_EXCL|os.O_RDONLY, 0o444) + if err != nil { + tb.Fatalf("createNetworkSandbox: create resource file: %v", err) + } + f.Close() + + // No explicit cleanup needed: tb.TempDir() handles removal. + return nsPath +} + +// networkSandboxIsOpen returns true if the network sandbox file at path still +// exists and is stat-able. Returns false if the path is empty, does not +// exist, or cannot be accessed. +func networkSandboxIsOpen(path string) bool { + if path == "" { + return false + } + var st unix.Stat_t + return unix.Stat(path, &st) == nil +} diff --git a/vendor/github.com/containerd/shimtest/helpers_networksandbox_other.go b/vendor/github.com/containerd/shimtest/helpers_networksandbox_other.go new file mode 100644 index 00000000..2cd1d5b4 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_networksandbox_other.go @@ -0,0 +1,33 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import "testing" + +// createNetworkSandbox is not available on non-Linux platforms. +// Tests that call it are expected to guard with runtime.GOOS == "linux" +// or be in a linux-only file; this stub satisfies the compiler on other +// platforms. +func createNetworkSandbox(tb testing.TB) string { + tb.Skip("createNetworkSandbox is Linux-only") + return "" +} + +// networkSandboxIsOpen always returns false on non-Linux platforms. +func networkSandboxIsOpen(_ string) bool { return false } diff --git a/vendor/github.com/containerd/shimtest/helpers_realnetns_linux.go b/vendor/github.com/containerd/shimtest/helpers_realnetns_linux.go new file mode 100644 index 00000000..f0a9a3f9 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_realnetns_linux.go @@ -0,0 +1,276 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "fmt" + "net" + "os" + "os/exec" + "runtime" + "strings" + "sync/atomic" + "testing" + "time" + + "golang.org/x/sys/unix" +) + +// realNetworkSandbox is a genuine Linux network namespace created via the +// system "ip netns" tool — a standard unshare(CLONE_NEWNET) + bind-mount +// technique for creating an isolated, enterable network namespace. +// Unlike the fake-file sandbox used by createNetworkSandbox, a +// realNetworkSandbox can be entered with setns(2) and can carry real network +// interfaces, so it can be used to observe where a shim's network traffic +// actually originates. +// +// Creating one requires CAP_SYS_ADMIN (root) in the initial user namespace; +// callers should be prepared for createRealNetworkSandbox to skip the test. +type realNetworkSandbox struct { + name string + path string +} + +var ( + netnsCounter atomic.Int32 + vethSubnetCounter atomic.Int32 +) + +// createRealNetworkSandbox creates a genuine network namespace using the +// system "ip netns add" command. Cleanup (namespace deletion, which also +// destroys any interfaces that live only inside it) is registered on tb. +// +// Skips the test if not running as root: creating and entering a real +// network namespace requires CAP_SYS_ADMIN. +func createRealNetworkSandbox(tb testing.TB) *realNetworkSandbox { + tb.Helper() + if os.Getuid() != 0 { + tb.Skip("createRealNetworkSandbox requires root (CAP_SYS_ADMIN)") + } + if _, err := exec.LookPath("ip"); err != nil { + tb.Skip("createRealNetworkSandbox requires the \"ip\" (iproute2) command") + } + + name := fmt.Sprintf("shimtest-%d-%d", os.Getpid(), netnsCounter.Add(1)) + if out, err := exec.Command("ip", "netns", "add", name).CombinedOutput(); err != nil { + tb.Fatalf("ip netns add %s: %v: %s", name, err, out) + } + tb.Cleanup(func() { + exec.Command("ip", "netns", "delete", name).Run() //nolint:errcheck + }) + + return &realNetworkSandbox{name: name, path: "/var/run/netns/" + name} +} + +// attachVeth creates a veth pair with both ends inside the sandbox namespace +// and assigns one end a unique address on a small point-to-point subnet. +// Because neither end of the pair ever exists in the root namespace, the +// resulting subnet has no route from outside the sandbox: only a process +// actually running inside the sandbox namespace (or a caller that has +// entered it via setns) can reach the returned address. This is what makes +// the address usable as a falsifiable probe for "did this traffic originate +// inside the network sandbox". +// +// The interfaces are destroyed automatically when the sandbox namespace is +// deleted; no separate cleanup is required. +func (n *realNetworkSandbox) attachVeth(tb testing.TB) (addr string) { + tb.Helper() + + slot := vethSubnetCounter.Add(1) % 250 + addr = fmt.Sprintf("10.244.%d.1", slot) + hostIf := fmt.Sprintf("vh%d", slot) + peerIf := fmt.Sprintf("vp%d", slot) + + runIP(tb, "link", "add", hostIf, "type", "veth", "peer", "name", peerIf) + runIP(tb, "link", "set", hostIf, "netns", n.name) + runIP(tb, "link", "set", peerIf, "netns", n.name) + + runIPNetns(tb, n.name, "addr", "add", addr+"/30", "dev", hostIf) + runIPNetns(tb, n.name, "link", "set", hostIf, "up") + runIPNetns(tb, n.name, "link", "set", peerIf, "up") + runIPNetns(tb, n.name, "link", "set", "lo", "up") + + return addr +} + +// runIP runs "ip " and fails the test on error. +func runIP(tb testing.TB, args ...string) { + tb.Helper() + if out, err := exec.Command("ip", args...).CombinedOutput(); err != nil { + tb.Fatalf("ip %s: %v: %s", strings.Join(args, " "), err, out) + } +} + +// runIPNetns runs "ip netns exec ip " and fails the test on +// error. +func runIPNetns(tb testing.TB, name string, args ...string) { + tb.Helper() + full := append([]string{"netns", "exec", name, "ip"}, args...) + if out, err := exec.Command("ip", full...).CombinedOutput(); err != nil { + tb.Fatalf("ip netns exec %s ip %s: %v: %s", name, strings.Join(args, " "), err, out) + } +} + +// probeUnreachableFromCurrentNamespace asserts that addr cannot be reached +// from the calling goroutine's current network namespace. It is used as a +// sanity check that a realNetworkSandbox address is actually isolated +// before relying on it as a falsifiable probe. +func probeUnreachableFromCurrentNamespace(tb testing.TB, addr string) { + tb.Helper() + conn, err := net.DialTimeout("tcp", addr, 300*time.Millisecond) + if err == nil { + conn.Close() + tb.Fatalf("test setup error: %s is reachable from the current namespace; "+ + "the sandbox isolation this test relies on is not actually in effect", addr) + } +} + +// listenAndEchoOnceInNetns enters the network namespace at nsPath on a +// dedicated, locked OS thread, binds a TCP listener at addr, accepts a +// single connection, echoes back one line, and closes. It returns a channel +// that receives the result (nil on success) once the exchange completes or +// fails. +// +// The listener is created on the locked thread after entering the +// namespace, so binding succeeds only when addr is actually reachable from +// within that namespace. Combined with an address from attachVeth (which has +// no route from the root namespace), a successful exchange on the returned +// channel is only possible if the connecting peer's traffic actually +// originated inside the sandbox namespace. +func listenAndEchoOnceInNetns(tb testing.TB, nsPath, addr string) <-chan error { + tb.Helper() + + ready := make(chan error, 1) + result := make(chan error, 1) + + go func() { + runtime.LockOSThread() + // Intentionally never unlocked: Go retires this OS thread when the + // goroutine exits (Go 1.10+), so the namespace change does not leak + // into the shared thread pool. + + f, err := os.Open(nsPath) + if err != nil { + ready <- fmt.Errorf("open netns %q: %w", nsPath, err) + return + } + defer f.Close() + if err := unix.Setns(int(f.Fd()), unix.CLONE_NEWNET); err != nil { + ready <- fmt.Errorf("setns %q: %w", nsPath, err) + return + } + + ln, err := net.Listen("tcp", addr) + if err != nil { + ready <- fmt.Errorf("listen %s in netns: %w", addr, err) + return + } + ready <- nil + + if tcpLn, ok := ln.(*net.TCPListener); ok { + tcpLn.SetDeadline(time.Now().Add(20 * time.Second)) + } + conn, err := ln.Accept() + ln.Close() + if err != nil { + result <- fmt.Errorf("accept: %w", err) + return + } + defer conn.Close() + conn.SetDeadline(time.Now().Add(10 * time.Second)) + + buf := make([]byte, 256) + n, rerr := conn.Read(buf) + if n == 0 && rerr != nil { + result <- fmt.Errorf("read: %w", rerr) + return + } + if _, err := conn.Write(buf[:n]); err != nil { + result <- fmt.Errorf("write: %w", err) + return + } + result <- nil + }() + + if err := <-ready; err != nil { + tb.Fatalf("listenAndEchoOnceInNetns: %v", err) + } + return result +} + +// dialInNetnsRoundTrip is the complement of listenAndEchoOnceInNetns: it +// enters the network namespace at nsPath on a dedicated, locked OS thread, +// dials a TCP connection to addr, sends token, reads the echo, and +// reports the result on the returned channel. +// +// Combined with a guest container running echosrv (which binds 0.0.0.0 and +// has its traffic scoped to the same namespace by the shim), a successful +// round trip is only possible if the container's listener is genuinely +// reachable from within that namespace — proving the shim honours the +// inbound side of the network-sandbox contract, not just the outbound. +func dialInNetnsRoundTrip(tb testing.TB, nsPath, addr, token string) <-chan error { + tb.Helper() + + result := make(chan error, 1) + + go func() { + runtime.LockOSThread() + // Intentionally never unlocked: Go retires this OS thread when the + // goroutine exits (Go 1.10+), so the namespace change does not leak + // into the shared thread pool. + + f, err := os.Open(nsPath) + if err != nil { + result <- fmt.Errorf("open netns %q: %w", nsPath, err) + return + } + defer f.Close() + if err := unix.Setns(int(f.Fd()), unix.CLONE_NEWNET); err != nil { + result <- fmt.Errorf("setns %q: %w", nsPath, err) + return + } + + conn, err := net.DialTimeout("tcp", addr, 15*time.Second) + if err != nil { + result <- fmt.Errorf("dial %s in netns: %w", addr, err) + return + } + defer conn.Close() + conn.SetDeadline(time.Now().Add(10 * time.Second)) + + if _, err := conn.Write([]byte(token)); err != nil { + result <- fmt.Errorf("write: %w", err) + return + } + buf := make([]byte, 256) + n, rerr := conn.Read(buf) + if n == 0 && rerr != nil { + result <- fmt.Errorf("read: %w", rerr) + return + } + got := string(buf[:n]) + if got != token { + result <- fmt.Errorf("echo mismatch: got %q, want %q", got, token) + return + } + result <- nil + }() + + return result +} diff --git a/vendor/github.com/containerd/shimtest/helpers_sandbox.go b/vendor/github.com/containerd/shimtest/helpers_sandbox.go new file mode 100644 index 00000000..bdddfd65 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_sandbox.go @@ -0,0 +1,822 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "bufio" + "bytes" + "context" + "encoding/json" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "sync" + "testing" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + "github.com/containerd/containerd/api/types" + "github.com/containerd/containerd/v2/pkg/namespaces" + "github.com/containerd/ttrpc" + typeurl "github.com/containerd/typeurl/v2" + specs "github.com/opencontainers/runtime-spec/specs-go" + "google.golang.org/protobuf/types/known/anypb" +) + +// containerOutput holds the captured stdout for a member container. +type containerOutput struct { + buf *bytes.Buffer + mu *sync.Mutex +} + +// sandboxEnv holds the shared state for a running sandbox: the TTRPC +// client, sandbox and task service clients, and the list of member +// containers created so far. +type sandboxEnv struct { + ctx context.Context + client *ttrpc.Client + sc sandboxAPI.TTRPCSandboxService + tc taskAPI.TTRPCTaskService + sandboxID string + address string + + mu sync.Mutex + containers []string // member container IDs in creation order + stdoutBufs map[string]*containerOutput // cid -> captured stdout + stdinPaths map[string]string // cid -> stdin FIFO path (only set when withSandboxCtrStdin is used) +} + +// startSandboxShim starts the shim binary for a sandbox and drives it +// through CreateSandbox + StartSandbox. It returns a *sandboxEnv +// backed by the shared TTRPC connection. +// +// API contract enforced here: +// - Bootstrap version ≥ 3 (enables per-connection task routing). +// - CreateSandbox and StartSandbox must succeed. +// - StartSandboxResponse.Pid must be > 0. +// - StartSandboxResponse.CreatedAt must be set and non-zero. +// +// Cleanup (ShutdownSandbox + shim delete) is registered on tb. +// startSandboxShim starts a sandbox shim with no network sandbox (host-network +// or platforms where network sandboxes are not supported). +// Use startSandboxShimWithNetworkSandbox to pass a network sandbox path. +func startSandboxShim(tb testing.TB, cfg Config, sandboxID string) *sandboxEnv { + tb.Helper() + return startSandboxShimInner(tb, cfg, sandboxID, "") +} + +// writeSandboxOCISpec writes a minimal OCI config.json suitable for a +// pod-sandbox bundle. The spec carries resource annotations so that +// VM-based shims start with a small VM; shims that do not use these +// annotations ignore them. +func writeSandboxOCISpec(tb testing.TB, bundleDir string) { + tb.Helper() + spec := struct { + OciVersion string `json:"ociVersion"` + Annotations map[string]string `json:"annotations,omitempty"` + }{ + OciVersion: "1.0.2", + Annotations: map[string]string{ + "io.containerd.nerdbox.resources.cpu": "2", + "io.containerd.nerdbox.resources.memory": "2048", + }, + } + data, err := json.Marshal(spec) + if err != nil { + tb.Fatal("marshal sandbox OCI spec:", err) + } + if err := os.WriteFile(filepath.Join(bundleDir, "config.json"), data, 0o644); err != nil { + tb.Fatal("write sandbox config.json:", err) + } +} + +// shutdownSandboxShim is the cleanup function registered by +// startSandboxShim. It tears down any remaining member containers +// then calls StopSandbox and ShutdownSandbox. Errors are logged but +// not fatal so that cleanup proceeds even after a failed test. +func shutdownSandboxShim(tb testing.TB, env *sandboxEnv) { + tb.Helper() + ctx, cancel := context.WithTimeout(env.ctx, 30*time.Second) + defer cancel() + + env.mu.Lock() + ctrs := make([]string, len(env.containers)) + copy(ctrs, env.containers) + env.mu.Unlock() + + for _, cid := range ctrs { + env.tc.Kill(ctx, &taskAPI.KillRequest{ID: cid, Signal: 9, All: true}) //nolint:errcheck + env.tc.Wait(ctx, &taskAPI.WaitRequest{ID: cid}) //nolint:errcheck + env.tc.Delete(ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + } + + env.sc.StopSandbox(ctx, &sandboxAPI.StopSandboxRequest{SandboxID: env.sandboxID}) //nolint:errcheck + env.sc.ShutdownSandbox(ctx, &sandboxAPI.ShutdownSandboxRequest{SandboxID: env.sandboxID}) //nolint:errcheck +} + +// createContainerInSandbox creates (and by default starts) a member +// container inside an already-running sandbox. It uses the same +// TTRPC connection that was used for the sandbox lifecycle RPCs, which +// is the routing mechanism containerd uses in production. +// +// Stdout is captured into an internal buffer; callers read it via +// readContainerOutput(env, cid). Cleanup (kill/wait/delete) is +// registered on tb. +// +// Returns the container ID. +func createContainerInSandbox(tb testing.TB, env *sandboxEnv, args []string, specOpts ...func(*sandboxCtrSpec)) string { + tb.Helper() + + so := &sandboxCtrSpec{} + for _, opt := range specOpts { + opt(so) + } + + cid := containerID(tb) + + bundleDir := tb.TempDir() + bundleDir, err := filepath.EvalSymlinks(bundleDir) + if err != nil { + tb.Fatal("evalSymlinks member bundleDir:", err) + } + + // Member containers: build the rootfs from the embedded testbin. + // For the sandbox path, ShareRootfs on the host will assemble the + // rootfs from whatever mounts are provided. We must give the shim + // a mount spec it can execute on the host. + // + // When running as root, use FormatMounts=true (erofs images with an + // overlay descriptor) so the shim assembles the overlay properly. + // + // When running as non-root, FormatMounts=false extracts the rootfs + // directly into bundleDir/rootfs and returns nil mounts. In that + // case we provide a single bind mount of that pre-extracted dir so + // ShareRootfs can bind it into the shared dir. + cfg := Config{FormatMounts: os.Getuid() == 0} + rootfsMounts := buildEmbeddedRootfs(tb, bundleDir, cfg) + + // Non-root / pre-extracted path: nil mounts means the rootfs is + // already in bundleDir/rootfs — present it as a bind mount. + if len(rootfsMounts) == 0 { + rootfsMounts = []*types.Mount{{ + Type: "bind", + Source: filepath.Join(bundleDir, "rootfs"), + Options: []string{"ro", "rbind"}, + }} + } + + var ociOpts []func(*specs.Spec) + if len(so.extraMounts) > 0 { + ociOpts = append(ociOpts, withExtraMounts(so.extraMounts...)) + } + ociOpts = append(ociOpts, so.ociOpts...) + createOCISpec(tb, bundleDir, args, cfg, ociOpts...) + + var stdinPath, stdoutPath, stderrPath string + if so.stdin { + stdinPath, stdoutPath, stderrPath = createStdioFifos(tb, bundleDir) + } else { + stdoutPath, stderrPath = createIOFifos(tb, bundleDir) + } + // Start capturing stdout into a buffer before Task.Create so the + // shim's forwardIO can open the write end without blocking. + var stdoutBuf bytes.Buffer + var stdoutMu sync.Mutex + drainFifoInto(tb, env.ctx, stdoutPath, &stdoutBuf, &stdoutMu) + // Stderr is discarded. + drainFifo(tb, env.ctx, stderrPath) + + var req *taskAPI.CreateTaskRequest + if so.stdin { + req = newCreateTaskRequestStdin(tb, cid, bundleDir, stdinPath, stdoutPath, stderrPath, rootfsMounts) + } else { + req = newCreateTaskRequest(tb, cid, bundleDir, stdoutPath, stderrPath, rootfsMounts) + } + if _, err := env.tc.Create(env.ctx, req); err != nil { + tb.Fatalf("Task.Create member %s: %v", cid, err) + } + + if !so.noStart { + if _, err := env.tc.Start(env.ctx, &taskAPI.StartRequest{ID: cid}); err != nil { + tb.Fatalf("Task.Start member %s: %v", cid, err) + } + } + + env.mu.Lock() + env.containers = append(env.containers, cid) + env.stdoutBufs[cid] = &containerOutput{buf: &stdoutBuf, mu: &stdoutMu} + if so.stdin { + if env.stdinPaths == nil { + env.stdinPaths = make(map[string]string) + } + env.stdinPaths[cid] = stdinPath + } + env.mu.Unlock() + + tb.Cleanup(func() { + ctx, cancel := context.WithTimeout(env.ctx, 10*time.Second) + defer cancel() + env.tc.Kill(ctx, &taskAPI.KillRequest{ID: cid, Signal: 9, All: true}) //nolint:errcheck + env.tc.Wait(ctx, &taskAPI.WaitRequest{ID: cid}) //nolint:errcheck + env.tc.Delete(ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + + env.mu.Lock() + for i, id := range env.containers { + if id == cid { + env.containers = append(env.containers[:i], env.containers[i+1:]...) + break + } + } + delete(env.stdoutBufs, cid) + env.mu.Unlock() + }) + + return cid +} + +// sandboxCtrSpec carries options for createContainerInSandbox. +type sandboxCtrSpec struct { + noStart bool + stdin bool + extraMounts []specs.Mount // extra OCI mounts appended to the container spec + ociOpts []func(*specs.Spec) // extra low-level OCI spec opts (e.g. namespace sharing) +} + +// withSandboxCtrStdin requests that createContainerInSandbox wire up a +// stdin FIFO for the member container, in addition to stdout/stderr. Use +// writeContainerStdin to send data once the container is running. +func withSandboxCtrStdin() func(*sandboxCtrSpec) { + return func(o *sandboxCtrSpec) { o.stdin = true } +} + +// withSandboxCtrExtraMounts appends mounts to the container's OCI spec. +// Use this to inject shared volumes, /dev/shm bind-mounts, or UDS-mount +// entries into a sandbox member container. +func withSandboxCtrExtraMounts(mounts ...specs.Mount) func(*sandboxCtrSpec) { + return func(o *sandboxCtrSpec) { + o.extraMounts = append(o.extraMounts, mounts...) + } +} + +// withSandboxCtrNamespace requests that the given namespace type be set +// to a (placeholder) host path in the container's OCI spec, signaling +// that the container should join a namespace shared with its sandbox +// peers rather than a fresh, isolated one. See withHostPathNamespace for +// why the specific path value does not matter. +func withSandboxCtrNamespace(nsType specs.LinuxNamespaceType, path string) func(*sandboxCtrSpec) { + return func(o *sandboxCtrSpec) { + o.ociOpts = append(o.ociOpts, withHostPathNamespace(nsType, path)) + } +} + +// withSandboxCtrOCIOpts appends arbitrary low-level OCI spec opts (e.g. +// withMemoryLimit, withCapabilities) to a member container's spec. Use +// this for anything createContainerInSandbox doesn't have a +// higher-level opt for. +func withSandboxCtrOCIOpts(opts ...func(*specs.Spec)) func(*sandboxCtrSpec) { + return func(o *sandboxCtrSpec) { + o.ociOpts = append(o.ociOpts, opts...) + } +} + +// createSandboxContainerFast creates and starts a member container using +// pre-built rootfs images. It is the stress-loop counterpart of +// createContainerInSandbox: it avoids calling writeRootfsErofs / +// writeBigFileErofs on every iteration (which would exhaust tmpfs over +// thousands of iterations) by accepting images built once before the loop. +// +// preExtractedRootfs is a pre-populated directory used as the bind-mount +// source on non-root systems (where loop mounts are unavailable). Pass "" +// on root systems where FormatMounts is true (the erofs+overlay path is used +// instead). +// +// Unlike createContainerInSandbox it does NOT register a tb.Cleanup, and it +// does NOT capture stdout into a buffer. Callers must call +// releaseSandboxContainer when done with each container. Stdout/stderr are +// drained and discarded. +func createSandboxContainerFast(ctx context.Context, tb testing.TB, env *sandboxEnv, cfg Config, imgs shimImages, preExtractedRootfs string, args []string) (string, error) { + cid := containerID(tb) + + bundleDir := tb.TempDir() + bundleDir, err := filepath.EvalSymlinks(bundleDir) + if err != nil { + return "", fmt.Errorf("evalSymlinks: %w", err) + } + + rootfsDir := filepath.Join(bundleDir, "rootfs") + if err := os.MkdirAll(rootfsDir, 0755); err != nil { + return "", fmt.Errorf("mkdir rootfs: %w", err) + } + + rootfsMounts := buildSandboxMemberMountsFromImages(tb, cfg, imgs, rootfsDir, preExtractedRootfs) + + createOCISpec(tb, bundleDir, args, Config{FormatMounts: cfg.FormatMounts}) + + // Use null (empty) IO paths so the shim skips FIFO creation entirely. + // This avoids the per-container FIFO files and the goroutines that drain + // them from accumulating in the test's temp directory over thousands of + // iterations. The shim treats empty Stdout/Stderr as "discard IO". + req := newCreateTaskRequest(tb, cid, bundleDir, "", "", rootfsMounts) + if _, err := env.tc.Create(ctx, req); err != nil { + return "", fmt.Errorf("Task.Create: %w", err) + } + if _, err := env.tc.Start(ctx, &taskAPI.StartRequest{ID: cid}); err != nil { + return "", fmt.Errorf("Task.Start: %w", err) + } + + env.mu.Lock() + env.containers = append(env.containers, cid) + env.mu.Unlock() + + return cid, nil +} + +// buildSandboxMemberMountsFromImages builds rootfs mount specs for a stress +// iteration using pre-built images. It only creates the per-iteration +// writable parts (ext4 scratch or overlay upper/work), reusing the read-only +// erofs images across iterations to avoid O(N) disk consumption. +// +// When running as root with FormatMounts, the full erofs+ext4+overlay path +// is used (same as benchContainerCreate). Otherwise a bind mount of the +// given preExtractedRootfs directory is returned so ShareRootfs can copy it +// into the sandbox shared dir without needing loop devices. +// preExtractedRootfs must be pre-populated by the caller once before the +// loop; it is read-only and reused across all iterations. +func buildSandboxMemberMountsFromImages(tb testing.TB, cfg Config, imgs shimImages, rootfsDir, preExtractedRootfs string) []*types.Mount { + tb.Helper() + if cfg.FormatMounts && os.Getuid() == 0 { + // Root + format mounts: use the erofs+ext4+overlay path. + // rootfsDir gets a fresh ext4 scratch on each iteration. + return buildRootfsMountsFromImages(tb, cfg, imgs, rootfsDir) + } + // Non-root or no format mounts: point at the pre-extracted directory. + // ShareRootfs will copy it into the sandbox shared dir per container. + if preExtractedRootfs == "" { + // Fallback if caller did not pre-extract (shouldn't happen). + extractErofsIntoDir(tb, imgs.erofsImg, rootfsDir) + preExtractedRootfs = rootfsDir + } + return []*types.Mount{{ + Type: "bind", + Source: preExtractedRootfs, + Options: []string{"ro", "rbind"}, + }} +} + +// releaseSandboxContainer immediately releases a container that was created +// with createContainerInSandbox. It issues Task.Delete on the shim (which +// triggers host-side rootfs cleanup via Unshare) and removes the container +// from the env tracking maps so the memory is reclaimed during the run. +// +// This is the per-iteration counterpart to the tb.Cleanup registered by +// createContainerInSandbox. Call it in stress loops where containers are +// short-lived: it prevents env.stdoutBufs from growing unboundedly across +// thousands of iterations and avoids stacking O(N) redundant tb.Cleanup +// registrations that would fire at test teardown. +// +// After releaseSandboxContainer returns the tb.Cleanup registered at +// creation time will still fire, but it becomes a no-op: the container is +// gone from env.containers and env.stdoutBufs, so the Kill/Wait/Delete RPCs +// will return NotFound and the map deletes are idempotent. +func releaseSandboxContainer(ctx context.Context, env *sandboxEnv, cid string) error { + _, err := env.tc.Delete(ctx, &taskAPI.DeleteRequest{ID: cid}) + + env.mu.Lock() + for i, id := range env.containers { + if id == cid { + env.containers = append(env.containers[:i], env.containers[i+1:]...) + break + } + } + delete(env.stdoutBufs, cid) + env.mu.Unlock() + + return err +} + +// withSandboxCtrNoStart creates the task without issuing Task.Start. +func withSandboxCtrNoStart() func(*sandboxCtrSpec) { + return func(o *sandboxCtrSpec) { o.noStart = true } +} + +// readContainerOutput waits up to timeout for want to appear in the +// captured stdout for the container with the given ID. The container +// must have been created via createContainerInSandbox on env. +func readContainerOutput(tb testing.TB, env *sandboxEnv, cid, want string, timeout time.Duration) string { + tb.Helper() + env.mu.Lock() + co := env.stdoutBufs[cid] + env.mu.Unlock() + if co == nil { + tb.Fatalf("no captured stdout for container %s", cid) + } + deadline := time.After(timeout) + for { + co.mu.Lock() + got := co.buf.String() + co.mu.Unlock() + if strings.Contains(got, want) { + return got + } + select { + case <-deadline: + co.mu.Lock() + final := co.buf.String() + co.mu.Unlock() + tb.Fatalf("timed out waiting for %q in stdout of %s, got: %q", want, cid, final) + case <-time.After(20 * time.Millisecond): + } + } +} + +// containerOutputSnapshot returns whatever stdout has been captured so +// far for cid, with no waiting for particular content — unlike +// readContainerOutput, which blocks for a specific substring. Use this +// when the expected output isn't known in advance (e.g. it's the value +// under test, not a fixed marker) and the container's completion is +// already known some other way (typically a prior Task.Wait). +// +// Callers should allow a brief moment after Task.Wait returns before +// calling this: the container's process has exited, but the FIFO +// drain goroutine may not have flushed the last of its output quite +// yet (see the same allowance in execInSandboxContainer). +func containerOutputSnapshot(tb testing.TB, env *sandboxEnv, cid string) string { + tb.Helper() + env.mu.Lock() + co := env.stdoutBufs[cid] + env.mu.Unlock() + if co == nil { + tb.Fatalf("no captured stdout for container %s", cid) + return "" + } + co.mu.Lock() + defer co.mu.Unlock() + return co.buf.String() +} + +// writeContainerStdin writes data to the stdin FIFO of a member container +// created with withSandboxCtrStdin, then closes the write end and signals +// EOF via the CloseIO RPC. The container must have been created via +// createContainerInSandbox with the withSandboxCtrStdin option. +// +// Closing the local FIFO write end alone is not sufficient: the shim may +// hold its own write-end reference on the stdin FIFO and only release it +// upon CloseIO — exactly as documented for exec stdin in exec_suite.go and +// enforced for the non-sandboxed path in network_suite.go's testOutboundTCP. +// Without the CloseIO call here, a shim implementing that (correct) contract +// would never deliver stdin EOF to the container's process. +func writeContainerStdin(tb testing.TB, env *sandboxEnv, cid, data string) { + tb.Helper() + env.mu.Lock() + stdinPath := env.stdinPaths[cid] + env.mu.Unlock() + if stdinPath == "" { + tb.Fatalf("no stdin FIFO for container %s (was it created with withSandboxCtrStdin?)", cid) + } + w, err := openPipeWriter(env.ctx, stdinPath) + if err != nil { + tb.Fatalf("open stdin fifo for %s: %v", cid, err) + } + if _, err := w.Write([]byte(data)); err != nil { + tb.Fatalf("write stdin for %s: %v", cid, err) + } + if err := w.Close(); err != nil { + tb.Fatalf("close stdin fifo for %s: %v", cid, err) + } + if _, err := env.tc.CloseIO(env.ctx, &taskAPI.CloseIORequest{ID: cid, Stdin: true}); err != nil { + tb.Fatalf("close stdin for %s: %v", cid, err) + } +} + +// readSandboxOutput waits up to timeout for want to appear in the FIFO +// at stdoutPath, returning the full accumulated output. Prefer +// readContainerOutput when the container was created with +// createContainerInSandbox. +func readSandboxOutput(tb testing.TB, ctx context.Context, stdoutPath, want string, timeout time.Duration) string { + tb.Helper() + var buf bytes.Buffer + var mu sync.Mutex + drainFifoInto(tb, ctx, stdoutPath, &buf, &mu) + deadline := time.After(timeout) + for { + mu.Lock() + got := buf.String() + mu.Unlock() + if strings.Contains(got, want) { + return got + } + select { + case <-deadline: + mu.Lock() + final := buf.String() + mu.Unlock() + tb.Fatalf("timed out waiting for %q in stdout, got: %q", want, final) + case <-time.After(20 * time.Millisecond): + } + } +} + +// waitForContainerPort waits for a container that prints its bound port as the +// first line of stdout (e.g. echosrv or nc -l with port 0) to become ready, +// and returns the port string. It polls the captured stdout buffer until a +// newline-terminated first line appears or timeout elapses (fatal). +// +// This allows tests to synchronise with a container listener without a fixed +// sleep: once the port line has been emitted the container is ready to accept +// connections. +func waitForContainerPort(tb testing.TB, env *sandboxEnv, cid string, timeout time.Duration) string { + tb.Helper() + env.mu.Lock() + co := env.stdoutBufs[cid] + env.mu.Unlock() + if co == nil { + tb.Fatalf("waitForContainerPort: no captured stdout for container %s", cid) + } + deadline := time.After(timeout) + for { + co.mu.Lock() + out := co.buf.String() + co.mu.Unlock() + if idx := strings.Index(out, "\n"); idx >= 0 { + return strings.TrimSpace(out[:idx]) + } + select { + case <-deadline: + co.mu.Lock() + final := co.buf.String() + co.mu.Unlock() + tb.Fatalf("timed out after %v waiting for port line from container %s; got: %q", timeout, cid, final) + case <-time.After(20 * time.Millisecond): + } + } +} + +// sandboxShimPID resolves the shim OS PID via the Task.Connect RPC +// after the first member container exists. Returns 0 if unavailable. +func sandboxShimPID(env *sandboxEnv, memberCID string) int { + // The sandbox API requires bootstrap version >= 3 (enforced in + // startSandboxShimInner), so the shim is always dialed as a v3 task + // service here. + pid, err := shimPidViaConnect(env.address, memberCID, 3, 2*time.Second) + if err != nil { + return 0 + } + return pid +} + +// sandboxMountTargets returns all mount targets visible in the shim +// process's mount namespace by parsing /proc//mountinfo. +// Returns nil if pid == 0 or the file is unreadable. +func sandboxMountTargets(pid int) []string { + if pid == 0 { + return nil + } + f, err := os.Open(fmt.Sprintf("/proc/%d/mountinfo", pid)) + if err != nil { + return nil + } + defer f.Close() + + var targets []string + scanner := bufio.NewScanner(f) + for scanner.Scan() { + // mountinfo: id parent major:minor root mountpoint options ... + fields := strings.Fields(scanner.Text()) + if len(fields) >= 5 { + targets = append(targets, fields[4]) + } + } + return targets +} + +// sandboxContainersMounts returns mount targets that fall under the +// sandbox shared containers directory (i.e. paths containing +// "/containers/"). Used by the mount-leak detector. +func sandboxContainersMounts(pid int) []string { + all := sandboxMountTargets(pid) + var matched []string + for _, t := range all { + if strings.Contains(t, "/containers/") { + matched = append(matched, t) + } + } + return matched +} + +// waitForSandboxStatus polls SandboxStatus until the state matches +// want or the deadline is exceeded. +func waitForSandboxStatus(ctx context.Context, sc sandboxAPI.TTRPCSandboxService, sandboxID, want string, timeout time.Duration) error { + deadline := time.Now().Add(timeout) + for { + resp, err := sc.SandboxStatus(ctx, &sandboxAPI.SandboxStatusRequest{SandboxID: sandboxID}) + if err == nil && resp.GetState() == want { + return nil + } + if time.Now().After(deadline) { + state := "unknown" + if err == nil { + state = resp.GetState() + } + return fmt.Errorf("timed out waiting for state %q, last state %q (err: %v)", want, state, err) + } + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(50 * time.Millisecond): + } + } +} + +// Unused import guard: types is used only to satisfy the compiler when +// buildRootfsMountsForSandbox is called from sandbox_suite.go. The +// function body references the types package via buildEmbeddedRootfs. +var _ *types.Mount + +// startSandboxShimWithNetworkSandbox starts a sandbox shim and passes +// networkSandboxPath in the CreateSandboxRequest. It is otherwise identical +// to startSandboxShim. Pass an empty string for the host-network case +// (no network sandbox). +// +// The API contract: the shim must hold the host-side network sandbox open +// for the lifetime of the sandbox so that CNI and other host tooling can +// operate on it while the sandbox is running. +func startSandboxShimWithNetworkSandbox(tb testing.TB, cfg Config, sandboxID, networkSandboxPath string) *sandboxEnv { + tb.Helper() + return startSandboxShimInner(tb, cfg, sandboxID, networkSandboxPath) +} + +// startSandboxShimInner is the shared implementation; exposed via +// startSandboxShim (empty path) and startSandboxShimWithNetworkSandbox. +func startSandboxShimInner(tb testing.TB, cfg Config, sandboxID, networkSandboxPath string) *sandboxEnv { + tb.Helper() + + shimBin, err := exec.LookPath(cfg.ShimBinary) + if err != nil { + tb.Fatalf("shim binary %q not found in PATH: %v", cfg.ShimBinary, err) + } + shimDir := filepath.Dir(shimBin) + if !strings.Contains(os.Getenv("PATH"), shimDir) { + os.Setenv("PATH", shimDir+string(os.PathListSeparator)+os.Getenv("PATH")) + } + + bundleDir := tb.TempDir() + bundleDir, err = filepath.EvalSymlinks(bundleDir) + if err != nil { + tb.Fatal("evalSymlinks bundleDir:", err) + } + + writeSandboxOCISpec(tb, bundleDir) + startEventsRecorder(tb, bundleDir) + + ns := uniqueTestNamespace(tb, "sandbox") + ctx := namespaces.WithNamespace(tb.Context(), ns) + + params := startShim(tb, shimBin, bundleDir, sandboxID, ns, cfg) + + if params.Version < 3 { + tb.Fatalf("sandbox API requires bootstrap version ≥ 3, shim returned %d", params.Version) + } + + conn := connectShim(tb, params.Address) + client := ttrpc.NewClient(conn) + tb.Cleanup(func() { client.Close() }) + + sc := sandboxAPI.NewTTRPCSandboxClient(client) + tc := taskAPI.NewTTRPCTaskClient(client) + + env := &sandboxEnv{ + ctx: ctx, + client: client, + sc: sc, + tc: tc, + sandboxID: sandboxID, + address: params.Address, + stdoutBufs: make(map[string]*containerOutput), + } + + if _, err := sc.CreateSandbox(ctx, &sandboxAPI.CreateSandboxRequest{ + SandboxID: sandboxID, + BundlePath: bundleDir, + NetnsPath: networkSandboxPath, + }); err != nil { + tb.Fatalf("CreateSandbox: %v", err) + } + + startResp, err := sc.StartSandbox(ctx, &sandboxAPI.StartSandboxRequest{ + SandboxID: sandboxID, + }) + if err != nil { + tb.Fatalf("StartSandbox: %v", err) + } + if startResp.GetPid() == 0 { + tb.Error("StartSandbox returned pid=0; shim must report a non-zero pid") + } + if ts := startResp.GetCreatedAt(); ts == nil || ts.AsTime().IsZero() { + tb.Error("StartSandbox returned zero createdAt") + } + + tb.Cleanup(func() { + shutdownSandboxShim(tb, env) + }) + + return env +} + +// sandboxStatusInfo calls SandboxStatus with verbose=true and returns the +// state string and the Info map. If the RPC fails the test is failed. +func sandboxStatusInfo(tb testing.TB, env *sandboxEnv) (state string, info map[string]string) { + tb.Helper() + resp, err := env.sc.SandboxStatus(env.ctx, &sandboxAPI.SandboxStatusRequest{ + SandboxID: env.sandboxID, + Verbose: true, + }) + if err != nil { + tb.Fatalf("SandboxStatus: %v", err) + } + return resp.GetState(), resp.GetInfo() +} + +// execInSandboxContainer execs a process in a running member container and +// returns its stdout output and exit status. It blocks until the exec +// completes or timeout elapses. stderr is captured and included in the +// returned output (interleaved) so that callers can inspect error messages. +// +// The API contract: Task.Exec followed by Task.Start(ExecID) must run the +// command inside the container; Task.Wait must return the exit status after +// the process terminates. +func execInSandboxContainer(tb testing.TB, env *sandboxEnv, cid string, args []string, timeout time.Duration) (output string, exitStatus uint32) { + tb.Helper() + + execID := containerID(tb) // unique exec ID derived from test name + + execStdout, execStderr := createIOFifos(tb, tb.TempDir()) + var outBuf bytes.Buffer + var outMu sync.Mutex + drainFifoInto(tb, env.ctx, execStdout, &outBuf, &outMu) + // Also capture stderr so callers can see error messages from the exec'd process. + drainFifoInto(tb, env.ctx, execStderr, &outBuf, &outMu) + + procSpec, err := typeurl.MarshalAnyToProto(&specs.Process{ + Args: args, + Cwd: "/", + Env: []string{"PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"}, + }) + if err != nil { + tb.Fatalf("marshal exec process spec: %v", err) + } + specAny := &anypb.Any{TypeUrl: procSpec.TypeUrl, Value: procSpec.Value} + + if _, err := env.tc.Exec(env.ctx, &taskAPI.ExecProcessRequest{ + ID: cid, + ExecID: execID, + Stdout: execStdout, + Stderr: execStderr, + Spec: specAny, + }); err != nil { + tb.Fatalf("Task.Exec in %s: %v", cid, err) + } + if _, err := env.tc.Start(env.ctx, &taskAPI.StartRequest{ + ID: cid, + ExecID: execID, + }); err != nil { + tb.Fatalf("Task.Start exec %s/%s: %v", cid, execID, err) + } + + ctx, cancel := context.WithTimeout(env.ctx, timeout) + defer cancel() + + waitResp, err := env.tc.Wait(ctx, &taskAPI.WaitRequest{ + ID: cid, + ExecID: execID, + }) + if err != nil { + tb.Fatalf("Task.Wait exec %s/%s: %v", cid, execID, err) + } + + // Allow a moment for the FIFO data to drain. + time.Sleep(50 * time.Millisecond) + outMu.Lock() + output = outBuf.String() + outMu.Unlock() + + return output, waitResp.GetExitStatus() +} diff --git a/vendor/github.com/containerd/shimtest/helpers_sandbox_other.go b/vendor/github.com/containerd/shimtest/helpers_sandbox_other.go new file mode 100644 index 00000000..808637ba --- /dev/null +++ b/vendor/github.com/containerd/shimtest/helpers_sandbox_other.go @@ -0,0 +1,63 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +// Package shimtest provides sandbox helpers. On non-Linux platforms the +// sandbox suite is not supported (virtiofs-backed shared container +// filesystems require Linux). The helpers here are stubs that satisfy +// the compiler; the SandboxSuite.Run method skips the entire suite at +// runtime. + +package shimtest + +import ( + "context" + "testing" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" +) + +type sandboxEnv struct{} + +func startSandboxShim(_ testing.TB, _ Config, _ string) *sandboxEnv { + return &sandboxEnv{} +} + +func createContainerInSandbox(_ testing.TB, _ *sandboxEnv, _ []string, _ ...func(*sandboxCtrSpec)) (string, string) { + return "", "" +} + +type sandboxCtrSpec struct{} + +func withSandboxCtrNoStart() func(*sandboxCtrSpec) { return func(*sandboxCtrSpec) {} } + +func readSandboxOutput(_ testing.TB, _ context.Context, _, _ string, _ time.Duration) string { + return "" +} + +func sandboxShimPID(_ *sandboxEnv, _ string) int { return 0 } + +func sandboxContainersMounts(_ int) []string { return nil } + +func sandboxMountTargets(_ int) []string { return nil } + +func waitForSandboxStatus(_ context.Context, _ sandboxAPI.TTRPCSandboxService, _, _ string, _ time.Duration) error { + return nil +} + +func shutdownSandboxShim(_ testing.TB, _ *sandboxEnv) {} diff --git a/vendor/github.com/containerd/shimtest/network_suite.go b/vendor/github.com/containerd/shimtest/network_suite.go index 7c5b06f3..4b60df53 100644 --- a/vendor/github.com/containerd/shimtest/network_suite.go +++ b/vendor/github.com/containerd/shimtest/network_suite.go @@ -19,7 +19,11 @@ package shimtest import ( "bytes" "fmt" + "math/rand/v2" "net" + "os" + "path/filepath" + "strconv" "strings" "sync" "testing" @@ -28,8 +32,66 @@ import ( taskAPI "github.com/containerd/containerd/api/runtime/task/v3" "github.com/containerd/containerd/v2/pkg/namespaces" "github.com/containerd/ttrpc" + specs "github.com/opencontainers/runtime-spec/specs-go" ) +// networkSuitePortRangeStart and networkSuitePortRangeEnd bound the +// candidates pickHostPort chooses testInboundTCPListen's port from. +// The range is deliberately outside Linux's default ephemeral port +// range (typically 32768-60999), to reduce the chance of a candidate +// colliding with an unrelated connection's OS-assigned source port. +const ( + networkSuitePortRangeStart = 20000 + networkSuitePortRangeEnd = 29999 + networkSuitePortAttempts = 20 +) + +// pickHostPort returns a TCP port on 127.0.0.1 verified free at the +// moment of the call, by binding and immediately releasing a listener. +// testInboundTCPListen uses this instead of requesting port 0 (letting +// the container's own kernel assign an ephemeral port), for two +// reasons: +// +// - With Config.ProvideNetwork, the host-side inbound forward must be +// registered before Task.Start -- before the container's listener +// even exists -- so the port cannot be discovered from the +// container at runtime; it has to be chosen ahead of time. +// - More fundamentally, some shims proxy each socket call to the host +// independently rather than truly sharing the host's network stack +// (see containerHostAddr's doc and testLoopbackWithinContainer). +// For such a shim, binding port 0 lets the guest and the host each +// resolve "an ephemeral port" on their own, and the two are not +// guaranteed to agree: the application only ever observes the +// guest's number, which need not be the port the host is actually +// listening on. Requesting the same concrete port on both sides +// removes that ambiguity entirely, regardless of the shim's +// networking mechanism. +// +// Binding a random port from a fixed range, with a retry on conflict, +// rather than a single hardcoded constant, avoids the test failing +// outright should that one port already be in use -- e.g. a concurrent +// run of this suite on the same host. This is inherently racy (another +// process could take the port between release and the container's own +// bind), but is a reasonable, standard mitigation for a test suite; +// this suite's tests run sequentially (never in parallel with each +// other), so at least a prior iteration of this same suite cannot be +// the collision. +func pickHostPort(t *testing.T) int { + t.Helper() + for range networkSuitePortAttempts { + port := networkSuitePortRangeStart + rand.IntN(networkSuitePortRangeEnd-networkSuitePortRangeStart+1) + ln, err := net.Listen("tcp", net.JoinHostPort("127.0.0.1", strconv.Itoa(port))) + if err != nil { + continue + } + ln.Close() + return port + } + t.Fatalf("pickHostPort: no free port in %d-%d after %d attempts", + networkSuitePortRangeStart, networkSuitePortRangeEnd, networkSuitePortAttempts) + return 0 +} + // NetworkSuite contains conformance tests for a container's default // network connectivity, gated on the "net" feature. // @@ -58,6 +120,45 @@ func (s *NetworkSuite) Run(t *testing.T) { t.Run("OutboundTCP", s.testOutboundTCP) t.Run("OutboundUDP", s.testOutboundUDP) t.Run("DNSResolve", s.testDNSResolve) + t.Run("InboundTCPListen", s.testInboundTCPListen) + t.Run("LoopbackWithinContainer", s.testLoopbackWithinContainer) +} + +// containerHostAddr returns the address a member of this suite's +// containers should use to reach a host-bound listener: slirpHostAddr +// when the test itself is providing the container's networking (see +// attachContainerNetwork), or the host's own loopback when the shim is +// assumed to already give the container a working default network path +// (e.g. by sharing the host's network namespace). +func (s *NetworkSuite) containerHostAddr() string { + if s.cfg.ProvideNetwork { + return slirpHostAddr + } + return "127.0.0.1" +} + +// networkSpecOpts returns the CreateOCISpec opts needed for a +// container to get its own network namespace, when this suite is +// responsible for providing that container's networking itself +// (Config.ProvideNetwork). Returns nil otherwise: the shim is assumed +// to already provide a default network path with no special spec +// configuration required. +func (s *NetworkSuite) networkSpecOpts() []func(*specs.Spec) { + if !s.cfg.ProvideNetwork { + return nil + } + return []func(*specs.Spec){withNewNetworkNamespace()} +} + +// attachIfProvided attaches this suite's own container networking (see +// attachContainerNetwork) after Task.Create when Config.ProvideNetwork +// is set, and is a no-op otherwise. +func (s *NetworkSuite) attachIfProvided(t *testing.T, pid uint32) *containerNetwork { + t.Helper() + if !s.cfg.ProvideNetwork { + return nil + } + return attachContainerNetwork(t, pid) } // testOutboundTCP verifies that a container's init process can open an @@ -73,6 +174,13 @@ func (s *NetworkSuite) Run(t *testing.T) { // stdin to the connection and the connection to stdout. The test writes a // token to the container's stdin FIFO; the host echoes it back; the test // asserts that the container's stdout contains the echoed token. +// +// When Config.ProvideNetwork is set, connectivity is provided by the test +// itself (see attachContainerNetwork): the container reaches the host at +// slirpHostAddr, which slirp4netns proxies to the host's own 127.0.0.1, +// exactly where the listener below is bound. Otherwise the shim is assumed +// to already give the container a default network path reaching the host +// directly at 127.0.0.1. func (s *NetworkSuite) testOutboundTCP(t *testing.T) { shimBin, bundleDir, rootfsMounts := shimSetup(t, s.cfg) cid := containerID(t) @@ -83,7 +191,7 @@ func (s *NetworkSuite) testOutboundTCP(t *testing.T) { } t.Cleanup(func() { ln.Close() }) - host, port, err := net.SplitHostPort(ln.Addr().String()) + _, port, err := net.SplitHostPort(ln.Addr().String()) if err != nil { t.Fatalf("split addr: %v", err) } @@ -119,7 +227,7 @@ func (s *NetworkSuite) testOutboundTCP(t *testing.T) { }() // nc : TCP mode, bidirectional pipe between socket and stdio. - createOCISpec(t, bundleDir, []string{"/bin/nc", host, port}, s.cfg) + createOCISpec(t, bundleDir, []string{"/bin/nc", s.containerHostAddr(), port}, s.cfg, s.networkSpecOpts()...) stdinPath, stdoutPath, stderrPath := createStdioFifos(t, bundleDir) ns := uniqueTestNamespace(t, "net") @@ -148,9 +256,15 @@ func (s *NetworkSuite) testOutboundTCP(t *testing.T) { t.Fatalf("open stdin fifo: %v", err) } - if _, err := tc.Create(ctx, newCreateTaskRequestStdin(t, cid, bundleDir, stdinPath, stdoutPath, stderrPath, rootfsMounts)); err != nil { + createResp, err := tc.Create(ctx, newCreateTaskRequestStdin(t, cid, bundleDir, stdinPath, stdoutPath, stderrPath, rootfsMounts)) + if err != nil { t.Fatal("create failed:", err) } + // Attach networking (when we're responsible for providing it) after + // Create (so the container's network namespace already exists) and + // before Start (so nc's very first connect() sees a fully configured + // network) — see attachContainerNetwork. + s.attachIfProvided(t, createResp.GetPid()) if _, err := tc.Start(ctx, &taskAPI.StartRequest{ID: cid}); err != nil { t.Fatal("start failed:", err) } @@ -221,7 +335,7 @@ func (s *NetworkSuite) testOutboundUDP(t *testing.T) { } t.Cleanup(func() { pc.Close() }) - host, port, err := net.SplitHostPort(pc.LocalAddr().String()) + _, port, err := net.SplitHostPort(pc.LocalAddr().String()) if err != nil { t.Fatalf("split addr: %v", err) } @@ -248,8 +362,9 @@ func (s *NetworkSuite) testOutboundUDP(t *testing.T) { recvDone <- nil }() - // nc -u : UDP mode, one sendto (stdin) then one recvfrom (stdout). - createOCISpec(t, bundleDir, []string{"/bin/nc", "-u", host, port}, s.cfg) + // nc -u : UDP mode, one sendto (stdin) then one recvfrom + // (stdout). See testOutboundTCP for containerHostAddr/networkSpecOpts. + createOCISpec(t, bundleDir, []string{"/bin/nc", "-u", s.containerHostAddr(), port}, s.cfg, s.networkSpecOpts()...) stdinPath, stdoutPath, stderrPath := createStdioFifos(t, bundleDir) ns := uniqueTestNamespace(t, "net") @@ -275,9 +390,11 @@ func (s *NetworkSuite) testOutboundUDP(t *testing.T) { t.Fatalf("open stdin fifo: %v", err) } - if _, err := tc.Create(ctx, newCreateTaskRequestStdin(t, cid, bundleDir, stdinPath, stdoutPath, stderrPath, rootfsMounts)); err != nil { + createResp, err := tc.Create(ctx, newCreateTaskRequestStdin(t, cid, bundleDir, stdinPath, stdoutPath, stderrPath, rootfsMounts)) + if err != nil { t.Fatal("create failed:", err) } + s.attachIfProvided(t, createResp.GetPid()) if _, err := tc.Start(ctx, &taskAPI.StartRequest{ID: cid}); err != nil { t.Fatal("start failed:", err) } @@ -344,12 +461,37 @@ const dnsTestHostname = "example.com" // The container runs host(1) (host ), which prints one line per address // in the form " has address ". The test parses those lines and // validates that each address field is a valid IP address. +// +// When Config.ProvideNetwork is set, connectivity is provided by the test +// itself (see attachContainerNetwork), whose resolver is slirpDNSAddr; the +// container's /etc/resolv.conf is bind-mounted to point at it, since the +// embedded rootfs carries none of its own. Otherwise the shim is assumed to +// already give the container a default network path with working DNS. func (s *NetworkSuite) testDNSResolve(t *testing.T) { shimBin, bundleDir, rootfsMounts := shimSetup(t, s.cfg) cid := containerID(t) + specOpts := s.networkSpecOpts() + if s.cfg.ProvideNetwork { + resolvConfPath := filepath.Join(t.TempDir(), "resolv.conf") + if err := os.WriteFile(resolvConfPath, []byte("nameserver "+slirpDNSAddr+"\n"), 0o644); err != nil { + t.Fatalf("write resolv.conf: %v", err) + } + specOpts = append(specOpts, withExtraMounts(specs.Mount{ + Type: "bind", + Source: resolvConfPath, + Destination: "/etc/resolv.conf", + // Not "ro": a read-only bind mount requires a follow-up + // MS_REMOUNT|MS_RDONLY, which fails EPERM in a rootless + // user namespace. Read-write is an acceptable tradeoff here + // -- this file is test-owned scratch data, not a security + // boundary the container has any reason to tamper with. + Options: []string{"rbind"}, + })) + } + // host : prints " has address " for each resolved address. - createOCISpec(t, bundleDir, []string{"/bin/host", dnsTestHostname}, s.cfg) + createOCISpec(t, bundleDir, []string{"/bin/host", dnsTestHostname}, s.cfg, specOpts...) stdoutPath, stderrPath := createIOFifos(t, bundleDir) ns := uniqueTestNamespace(t, "net") @@ -369,9 +511,11 @@ func (s *NetworkSuite) testDNSResolve(t *testing.T) { var stderrMu sync.Mutex drainFifoInto(t, ctx, stderrPath, &stderrBuf, &stderrMu) - if _, err := tc.Create(ctx, newCreateTaskRequest(t, cid, bundleDir, stdoutPath, stderrPath, rootfsMounts)); err != nil { + createResp, err := tc.Create(ctx, newCreateTaskRequest(t, cid, bundleDir, stdoutPath, stderrPath, rootfsMounts)) + if err != nil { t.Fatal("create failed:", err) } + s.attachIfProvided(t, createResp.GetPid()) if _, err := tc.Start(ctx, &taskAPI.StartRequest{ID: cid}); err != nil { t.Fatal("start failed:", err) } @@ -421,3 +565,296 @@ func (s *NetworkSuite) testDNSResolve(t *testing.T) { shutdownTask(ctx, tc, cid) t.Log("container DNS resolution: ok, addresses:", addrs) } + +// testInboundTCPListen verifies that a container's init process can bind a +// TCP listener and accept a connection from outside the container — i.e. that +// the shim's default networking path exposes inbound TCP as well as outbound. +// +// The API contract: a container started without any network configuration must +// be able to bind a TCP listener and receive connections from the host, the +// same way a process using the host's network stack would. The shim +// documentation claims that "ports bound inside the VM are transparently +// mapped on the host"; this test verifies that claim is true in practice. +// +// The container binds a concrete port chosen by pickHostPort — never port +// 0 — regardless of Config.ProvideNetwork. See pickHostPort's doc for why +// a discovered ephemeral port is not just unnecessary but actively unsafe +// to rely on for some shims: the guest and the host can each resolve "an +// ephemeral port" independently and disagree, so the application's own +// report of its bound port is not guaranteed to be where the host is +// actually listening. With Config.ProvideNetwork the fixed port is also +// what lets the test register its own inbound port forward (see +// attachContainerNetwork.AddInboundForward) before Task.Start, before the +// container's listener even exists. +// +// nc -v reports the socket it bound on stderr before accepting; the test +// waits for that line purely as a readiness signal (the port itself is +// already known) before dialing host-loopback (127.0.0.1:) and +// sending a token. nc copies its connection to stdout and its stdin to the +// connection, but has no path that echoes received data back over the same +// connection — so the test confirms the token actually reached the +// container, which is this test's whole point, by waiting for it to appear +// in the container's own captured stdout rather than by reading a reply off +// the socket. The test hard-fails if the host cannot connect: inbound +// reachability is a required contract of the default networking path. +func (s *NetworkSuite) testInboundTCPListen(t *testing.T) { + shimBin, bundleDir, rootfsMounts := shimSetup(t, s.cfg) + cid := containerID(t) + + port := pickHostPort(t) + requestedPort := strconv.Itoa(port) + + // nc -v -l : bind, report the bound socket on stderr, accept one + // connection, then bidirectionally pipe stdio↔socket. stdout is left + // carrying connection payload only, so the token assertion below cannot + // be confused by control output. + createOCISpec(t, bundleDir, []string{"/bin/nc", "-v", "-l", requestedPort}, s.cfg, s.networkSpecOpts()...) + + stdinPath, stdoutPath, stderrPath := createStdioFifos(t, bundleDir) + ns := uniqueTestNamespace(t, "net") + ctx := namespaces.WithNamespace(t.Context(), ns) + + params := startShim(t, shimBin, bundleDir, cid, ns, s.cfg) + shimConn := connectShim(t, params.Address) + client := ttrpc.NewClient(shimConn) + defer client.Close() + + tc := newTaskClient(client, params.Version) + + var stdoutBuf bytes.Buffer + var stdoutMu sync.Mutex + stdoutDone := drainFifoIntoDone(t, ctx, stdoutPath, &stdoutBuf, &stdoutMu) + var stderrBuf bytes.Buffer + var stderrMu sync.Mutex + drainFifoInto(t, ctx, stderrPath, &stderrBuf, &stderrMu) + + createResp, err := tc.Create(ctx, newCreateTaskRequestStdin(t, cid, bundleDir, stdinPath, stdoutPath, stderrPath, rootfsMounts)) + if err != nil { + t.Fatal("create failed:", err) + } + netw := s.attachIfProvided(t, createResp.GetPid()) + if s.cfg.ProvideNetwork { + // Register the forward before Start: the port is fixed and known + // ahead of time, so there is no need to wait for the container to + // report it first, and no window in which an early host + // connection attempt could race the forward's registration. + netw.AddInboundForward(t, port, port) + } + if _, err := tc.Start(ctx, &taskAPI.StartRequest{ID: cid}); err != nil { + t.Fatal("start failed:", err) + } + + // Read the bound port from the container's stderr. nc -v reports the + // listening socket there before calling Accept, so this completes as + // soon as the listener is ready. A 15 s timeout guards against hangs. + // The port is already known (see pickHostPort); this is a readiness + // signal, and doubles as a check that the container actually bound + // the exact port it was asked to. + boundPort := waitForListeningPort(t, &stderrBuf, &stderrMu, 15*time.Second) + if boundPort != requestedPort { + t.Fatalf("container reported port %q, want %q", boundPort, requestedPort) + } + t.Logf("container listener bound on port %s", boundPort) + + // Dial the container's mapped/forwarded port on the host loopback. + hostAddr := net.JoinHostPort("127.0.0.1", boundPort) + const token = "network-suite-inbound-ok" + + hostConn, err := net.DialTimeout("tcp", hostAddr, 15*time.Second) + if err != nil { + t.Fatalf("host could not connect to container listener at %s: %v\n"+ + "(inbound port mapping is a required contract of the default networking path)", hostAddr, err) + } + hostConn.SetDeadline(time.Now().Add(10 * time.Second)) + + // Send the token to the container. + if _, err := fmt.Fprintf(hostConn, "%s\n", token); err != nil { + t.Fatalf("host write to container: %v", err) + } + + // Confirm the token actually reached the container by waiting for it + // to show up in the container's own stdout, rather than reading a + // reply back over the socket (nc -l has no echo direction — see + // above). + tokenDeadline := time.After(10 * time.Second) + var out string + for { + stdoutMu.Lock() + out = stdoutBuf.String() + stdoutMu.Unlock() + if strings.Contains(out, token) { + break + } + select { + case <-tokenDeadline: + t.Fatalf("container did not report receiving %q within 10s; stdout so far: %q", token, out) + case <-time.After(20 * time.Millisecond): + } + } + + // Close the host connection so nc's io.Copy from the connection returns. + hostConn.Close() + + // Write nothing to stdin — just close it so nc's stdin→socket copy also + // finishes; combined with hostConn.Close() nc will exit cleanly. + stdinFifo, err := openPipeWriter(ctx, stdinPath) + if err != nil { + t.Fatalf("open stdin fifo: %v", err) + } + stdinFifo.Close() + + // Signal EOF to nc via the CloseIO RPC; see testOutboundTCP for why + // closing the test's own FIFO write end alone is not sufficient. + if _, err := tc.CloseIO(ctx, &taskAPI.CloseIORequest{ID: cid, Stdin: true}); err != nil { + t.Fatal("close stdin failed:", err) + } + + waitResp, err := tc.Wait(ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatal("wait failed:", err) + } + <-stdoutDone + stdoutMu.Lock() + out = stdoutBuf.String() + stdoutMu.Unlock() + + if waitResp.ExitStatus != 0 { + t.Fatalf("container exit status: got %d, want 0; stdout: %q", waitResp.ExitStatus, out) + } + + tc.Delete(ctx, &taskAPI.DeleteRequest{ID: cid}) + shutdownTask(ctx, tc, cid) + t.Logf("container inbound TCP listen: ok (port %s)", boundPort) +} + +// testLoopbackWithinContainer verifies that a container's init process can +// connect to itself over loopback (127.0.0.1). This guards the in-container +// localhost contract: a process that binds 127.0.0.1 inside a container must +// be reachable from another process within the same container via loopback. +// +// The API contract: a container started without any network configuration must +// have a working loopback interface, exactly as any process running on the +// host's network stack would. This is a prerequisite for between-container +// localhost connectivity: if the loopback inside a single container is broken, +// containers sharing a network namespace will also fail. +// +// The container runs a small helper binary ("looptest"; see cmdLooptest in +// testbin) that binds a listener, then connects back to it over +// 127.0.0.1: from within the same process. Because both ends run in +// the same init process, only in-container loopback carries the traffic. +// looptest binds a concrete port from a fixed range with a short retry +// loop, rather than port 0, for the same reason pickHostPort exists on the +// host side: some shims proxy each socket call to the host independently, +// so a port 0 bind's two ends could each resolve "an ephemeral port" to a +// different number and never actually rendezvous. +// +// This test doesn't need attachContainerNetwork's connectivity itself (it +// never leaves the container), but when Config.ProvideNetwork is set it +// still requests a network namespace and attaches to it like every other +// test in this suite, to confirm that in-container loopback keeps working +// under exactly the same namespace setup the tests that do need outside +// connectivity depend on. +func (s *NetworkSuite) testLoopbackWithinContainer(t *testing.T) { + shimBin, bundleDir, rootfsMounts := shimSetup(t, s.cfg) + cid := containerID(t) + + const token = "network-suite-loopback-ok" + + // looptest : binds a listener on a concrete port (see + // looptestPortRangeStart in testbin), then connects back to it from + // within the same process, sends the token, and prints the echoed + // response to stdout. + createOCISpec(t, bundleDir, []string{"/bin/looptest", token}, s.cfg, s.networkSpecOpts()...) + + stdoutPath, stderrPath := createIOFifos(t, bundleDir) + ns := uniqueTestNamespace(t, "net") + ctx := namespaces.WithNamespace(t.Context(), ns) + + params := startShim(t, shimBin, bundleDir, cid, ns, s.cfg) + shimConn := connectShim(t, params.Address) + client := ttrpc.NewClient(shimConn) + defer client.Close() + + tc := newTaskClient(client, params.Version) + + var stdoutBuf bytes.Buffer + var stdoutMu sync.Mutex + stdoutDone := drainFifoIntoDone(t, ctx, stdoutPath, &stdoutBuf, &stdoutMu) + var stderrBuf bytes.Buffer + var stderrMu sync.Mutex + drainFifoInto(t, ctx, stderrPath, &stderrBuf, &stderrMu) + + createResp, err := tc.Create(ctx, newCreateTaskRequest(t, cid, bundleDir, stdoutPath, stderrPath, rootfsMounts)) + if err != nil { + t.Fatal("create failed:", err) + } + s.attachIfProvided(t, createResp.GetPid()) + if _, err := tc.Start(ctx, &taskAPI.StartRequest{ID: cid}); err != nil { + t.Fatal("start failed:", err) + } + + waitResp, err := tc.Wait(ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatal("wait failed:", err) + } + <-stdoutDone + stdoutMu.Lock() + out := stdoutBuf.String() + stdoutMu.Unlock() + stderrMu.Lock() + errOut := stderrBuf.String() + stderrMu.Unlock() + + if waitResp.ExitStatus != 0 { + t.Fatalf("container exit status: got %d, want 0; stdout: %q, stderr: %q", waitResp.ExitStatus, out, errOut) + } + if !strings.Contains(out, token) { + t.Fatalf("loopback echo not found in stdout: want %q, got %q; stderr: %q", token, out, errOut) + } + + tc.Delete(ctx, &taskAPI.DeleteRequest{ID: cid}) + shutdownTask(ctx, tc, cid) + t.Log("container in-container loopback: ok") +} + +// listeningNotice is the prefix nc(1) uses to report the socket it bound +// when run with -v. The full line has the form "Listening on ". +const listeningNotice = "Listening on " + +// waitForListeningPort polls buf (protected by mu) until nc's -v listen +// notice appears in the accumulated output, then returns the port it +// reports. It lets a test synchronously learn the port a container bound +// without opening a competing reader on the FIFO — the caller's drainFifo +// goroutine already owns the read end. +// +// Only newline-terminated lines are considered, so a port number still being +// written cannot be parsed as if it were complete. +func waitForListeningPort(t *testing.T, buf *bytes.Buffer, mu *sync.Mutex, timeout time.Duration) string { + t.Helper() + deadline := time.After(timeout) + for { + mu.Lock() + out := buf.String() + mu.Unlock() + if end := strings.LastIndex(out, "\n"); end >= 0 { + for _, line := range strings.Split(out[:end], "\n") { + line = strings.TrimSpace(line) + if !strings.HasPrefix(line, listeningNotice) { + continue + } + if fields := strings.Fields(line); len(fields) >= 2 { + return fields[len(fields)-1] + } + } + } + select { + case <-deadline: + mu.Lock() + final := buf.String() + mu.Unlock() + t.Fatalf("timed out after %v waiting for nc's %q notice; stderr so far: %q", + timeout, strings.TrimSpace(listeningNotice), final) + case <-time.After(20 * time.Millisecond): + } + } +} diff --git a/vendor/github.com/containerd/shimtest/rootfs.go b/vendor/github.com/containerd/shimtest/rootfs.go index 06244e92..97ef7b7f 100644 --- a/vendor/github.com/containerd/shimtest/rootfs.go +++ b/vendor/github.com/containerd/shimtest/rootfs.go @@ -125,7 +125,7 @@ func testbinAssetName(goarch string) string { // testbinCommands lists the commands provided by the testbin binary. // Symlinks are created in /bin for each command in the embedded // rootfs. -var testbinCommands = []string{"forever", "burstexit", "cat", "date", "echo", "exit", "hashverify", "host", "layercheck", "ls", "memhog", "nc", "tickexit"} +var testbinCommands = []string{"forever", "burstexit", "cat", "date", "echo", "echosrv", "exit", "hashverify", "host", "hostname", "layercheck", "looptest", "ls", "memhog", "nc", "pidscan", "shmread", "shmwrite", "shmmapread", "shmmapwrite", "tickexit"} // bigFileSize is the size of the IO benchmark fixture file. Large // enough to swamp small per-call overheads while still building / diff --git a/vendor/github.com/containerd/shimtest/sandbox_bench_linux.go b/vendor/github.com/containerd/shimtest/sandbox_bench_linux.go new file mode 100644 index 00000000..ae8eda55 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_bench_linux.go @@ -0,0 +1,200 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "context" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + "github.com/containerd/containerd/api/types" +) + +// Bench runs every benchmark in the SandboxSuite as a sub-benchmark of b. +func (s *SandboxSuite) Bench(b *testing.B) { + b.Helper() + b.Run("ContainerCreate", s.benchContainerCreate) +} + +// benchContainerCreate measures the per-container create/start/wait/delete +// cycle inside a single running sandbox VM. The sandbox is started once +// before the iteration loop so the VM boot cost is paid only once; each +// b.N iteration adds one member container, runs it to completion, and +// removes it. +// +// This benchmark is the sandbox-API counterpart to RunSuite.benchLifecycle. +// Because the VM is shared across all iterations, per-iteration cost reflects +// only the marginal work needed to create and run a new container: rootfs +// assembly on the host, guest bundle/mount/task RPCs, and cleanup. The +// sandbox start time is reported separately as ms/sandbox-start so the +// amortised overhead is visible. +// +// Reported metrics (all in milliseconds, averaged over b.N): +// +// - ms/create — Task.Create RPC (rootfs assembly + guest bundle/mount/task) +// - ms/start — Task.Start RPC +// - ms/wait — Task.Wait until exit +// - ms/delete — Task.Delete RPC (rootfs unshare + cleanup) +// - ms/total — sum of the four phases above +// +// Reported once (not per-iteration): +// +// - ms/sandbox-start — time from sandbox shim launch to StartSandbox response +func (s *SandboxSuite) benchContainerCreate(b *testing.B) { + shimBin, err := exec.LookPath(s.cfg.ShimBinary) + if err != nil { + b.Fatalf("shim binary %q not found in PATH: %v", s.cfg.ShimBinary, err) + } + if shimDir := filepath.Dir(shimBin); !strings.Contains(os.Getenv("PATH"), shimDir) { + os.Setenv("PATH", shimDir+string(os.PathListSeparator)+os.Getenv("PATH")) + } + + // Pre-build the read-only rootfs images once so per-iteration setup + // only needs to construct the writable layer. For the sandbox path, + // buildSandboxMemberMounts uses these to produce the mounts passed to + // ShareRootfs on each iteration. + imgs := buildShimImages(b, s.cfg) + + sandboxID := containerID(b) + base := containerID(b) + + // ── Start the sandbox (timed separately, not part of b.N loop) ─────── + tSandboxStart := time.Now() + env := startSandboxShim(b, s.cfg, sandboxID) + sandboxStartMs := float64(time.Since(tSandboxStart).Microseconds()) / 1000.0 + + b.ReportMetric(sandboxStartMs, "ms/sandbox-start") + + // ── Per-iteration state ─────────────────────────────────────────────── + var sumCreate, sumStart, sumWait, sumDelete time.Duration + + b.ResetTimer() + for i := 0; i < b.N; i++ { + b.StopTimer() + + cid := fmt.Sprintf("%s-%d", base, i) + + // Build a fresh member-container bundle and rootfs mounts. + bundleDir := b.TempDir() + bundleDir, err = filepath.EvalSymlinks(bundleDir) + if err != nil { + b.Fatal("resolve member bundle dir:", err) + } + rootfsDir := filepath.Join(bundleDir, "rootfs") + if err := os.MkdirAll(rootfsDir, 0755); err != nil { + b.Fatal("mkdir rootfs:", err) + } + rootfsMounts := buildSandboxMemberMounts(b, s.cfg, imgs, rootfsDir, bundleDir) + cfg := Config{FormatMounts: s.cfg.FormatMounts} + createOCISpec(b, bundleDir, []string{"/bin/exit", "0"}, cfg) + + stdoutPath, stderrPath := createIOFifos(b, bundleDir) + drainFifo(b, env.ctx, stdoutPath) + drainFifo(b, env.ctx, stderrPath) + + req := newCreateTaskRequest(b, cid, bundleDir, stdoutPath, stderrPath, rootfsMounts) + + b.StartTimer() + + // Create + t := time.Now() + if _, err := env.tc.Create(env.ctx, req); err != nil { + b.Fatalf("Create %s: %v", cid, err) + } + sumCreate += time.Since(t) + + // Start + t = time.Now() + if _, err := env.tc.Start(env.ctx, &taskAPI.StartRequest{ID: cid}); err != nil { + b.Fatalf("Start %s: %v", cid, err) + } + sumStart += time.Since(t) + + // Wait + t = time.Now() + if _, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}); err != nil { + b.Fatalf("Wait %s: %v", cid, err) + } + sumWait += time.Since(t) + + // Delete (triggers host-side rootfs cleanup via SharedFS.Unshare) + t = time.Now() + if _, err := env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}); err != nil { + b.Fatalf("Delete %s: %v", cid, err) + } + sumDelete += time.Since(t) + + b.StopTimer() + + // Remove from env tracking so memory does not accumulate. + env.mu.Lock() + for i, id := range env.containers { + if id == cid { + env.containers = append(env.containers[:i], env.containers[i+1:]...) + break + } + } + delete(env.stdoutBufs, cid) + env.mu.Unlock() + } + + n := float64(b.N) + reportMs := func(d time.Duration, name string) { + b.ReportMetric(float64(d.Microseconds())/n/1000.0, name) + } + reportMs(sumCreate, "ms/create") + reportMs(sumStart, "ms/start") + reportMs(sumWait, "ms/wait") + reportMs(sumDelete, "ms/delete") + reportMs(sumCreate+sumStart+sumWait+sumDelete, "ms/total") +} + +// buildSandboxMemberMounts builds the rootfs mount specs for a sandbox member +// container benchmark iteration. It mirrors the logic in +// createContainerInSandbox but is optimised for benchmarks: when FormatMounts +// is true the pre-built erofs images are reused; otherwise a bind mount of the +// pre-extracted rootfs dir is returned (same fallback that ShareRootfs handles +// by copying into the shared dir). +func buildSandboxMemberMounts(tb testing.TB, cfg Config, imgs shimImages, rootfsDir, bundleDir string) []*types.Mount { + tb.Helper() + if cfg.FormatMounts && os.Getuid() == 0 { + return buildRootfsMountsFromImages(tb, cfg, imgs, rootfsDir) + } + // Non-root or non-format: extract once into rootfsDir, then wrap as + // a bind mount so ShareRootfs can copy it into the shared directory. + if os.Getuid() != 0 { + extractErofsIntoDir(tb, imgs.erofsImg, rootfsDir) + return []*types.Mount{{ + Type: "bind", + Source: rootfsDir, + Options: []string{"ro", "rbind"}, + }} + } + _ = bundleDir + return buildRootfsMountsFromImages(tb, cfg, imgs, rootfsDir) +} + +// Ensure the context package is used (env.ctx references it implicitly). +var _ context.Context diff --git a/vendor/github.com/containerd/shimtest/sandbox_bench_other.go b/vendor/github.com/containerd/shimtest/sandbox_bench_other.go new file mode 100644 index 00000000..c7543d57 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_bench_other.go @@ -0,0 +1,26 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import "testing" + +// Bench skips sandbox benchmarks on non-Linux platforms. +func (s *SandboxSuite) Bench(b *testing.B) { + b.Skip("SandboxSuite benchmarks are Linux-only") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite.go b/vendor/github.com/containerd/shimtest/sandbox_suite.go new file mode 100644 index 00000000..34782b2f --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite.go @@ -0,0 +1,516 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "fmt" + "strings" + "sync" + "syscall" + "testing" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + tasktypes "github.com/containerd/containerd/api/types/task" + "github.com/containerd/containerd/v2/pkg/namespaces" + "github.com/containerd/ttrpc" +) + +// sandboxStateReady and sandboxStateNotReady are the two values a shim's +// SandboxStatusResponse.State must use. +// +// The runtime/sandbox/v1 proto itself only documents State as a plain +// string, with no enumerated values on the wire. But an unconstrained +// free-form string is not a usable API contract: any caller that needs to +// branch on sandbox readiness has to match against *some* fixed vocabulary, +// and a shim that invents its own spelling (a plausible one like "ready" +// included) breaks every caller that expects the specified names. State is +// therefore specified here as exactly one of these two strings, not an +// arbitrary human-readable state name. This is exactly the kind of +// externally-observable contract shimtest exists to check: it is invisible +// in the base shim-v2 sandbox protocol's type signature, but load-bearing +// for any caller that inspects sandbox readiness — for example, +// containerd's CRI layer maps these exact names onto the CRI v1 +// PodSandboxState enum when the shim sandboxer is configured. +const ( + sandboxStateReady = "SANDBOX_READY" + sandboxStateNotReady = "SANDBOX_NOTREADY" +) + +// SandboxSuite verifies the containerd sandbox shim API contract +// (runtime/sandbox/v1). Tests in this suite cover: +// +// - Sandbox lifecycle (create → start → status → stop → shutdown) +// - Platform and Ping RPCs +// - Member container routing: tasks created on the shared connection +// after StartSandbox must run correctly inside the sandbox +// - Multiple concurrent containers sharing one sandbox +// - Per-container Delete independence (does not tear down the sandbox) +// - WaitSandbox unblocks on stop +// - Protocol error cases (duplicate Create, Start before Create) +// - Resource release: no mount leaks after shutdown +// +// The suite is gated on the "sandbox" feature key; it is never skipped +// once enabled — every failure is a conformance failure. +type SandboxSuite struct { + cfg Config +} + +// NewSandboxSuite constructs a SandboxSuite from cfg. +func NewSandboxSuite(cfg Config) *SandboxSuite { + return &SandboxSuite{cfg: cfg} +} + +// Run runs every test in the suite as a subtest of t. +func (s *SandboxSuite) Run(t *testing.T) { + t.Helper() + registerShimLeakCheck(t, s.cfg.ShimBinary) + + t.Run("Lifecycle", s.testLifecycle) + t.Run("Platform", s.testPlatform) + t.Run("Ping", s.testPing) + t.Run("SingleContainer", s.testSingleContainer) + t.Run("MultipleContainers", s.testMultipleContainers) + t.Run("ContainerLifecycleIndependence", s.testContainerLifecycleIndependence) + t.Run("StatusAfterStop", s.testStatusAfterStop) + t.Run("WaitUnblocksOnStop", s.testWaitUnblocksOnStop) + t.Run("CreateTwiceRejected", s.testCreateTwiceRejected) + t.Run("StartWithoutCreateRejected", s.testStartWithoutCreateRejected) + t.Run("ResourceReleaseOnShutdown", s.testResourceReleaseOnShutdown) + + // Member-container workload contracts (exec, shared namespaces, + // volumes, networking). + t.Run("StatusReportsPidAndCreatedAt", s.testStatusReportsPidAndCreatedAt) + t.Run("HostNetworkNoNetworkSandbox", s.testHostNetworkNoNetworkSandbox) + t.Run("MemberContainerExec", s.testMemberContainerExec) + t.Run("CrossContainerViaUDS", s.testCrossContainerViaUDS) + t.Run("NetworkSandboxHeldOpen", s.testNetworkSandboxHeldOpen) + t.Run("NetworkSandboxPathInStatus", s.testNetworkSandboxPathInStatus) + t.Run("ContainerOutboundTCP", s.testContainerOutboundTCP) + t.Run("ContainerTrafficScopedToNetworkSandbox", s.testContainerTrafficScopedToNetworkSandbox) + t.Run("InboundToNetnsScopedListener", s.testInboundToNetnsScopedListener) + t.Run("MemberContainersShareNetwork", s.testMemberContainersShareNetwork) + t.Run("MemberContainerHostVolume", s.testMemberContainerHostVolume) + t.Run("MemberContainersSharePID", s.testMemberContainersSharePID) + t.Run("MemberContainersSharePIDKillScopedToOwnContainer", s.testMemberContainersSharePIDKillScopedToOwnContainer) + t.Run("MemberContainersShareIPC", s.testMemberContainersShareIPC) + t.Run("MemberContainersShareDevShm", s.testMemberContainersShareDevShm) + t.Run("MemberContainersDevShmNotSharedWithoutIPC", s.testMemberContainersDevShmNotSharedWithoutIPC) + t.Run("MemberContainerOOMIsolation", s.testMemberContainerOOMIsolation) + t.Run("MemberContainersShareUTS", s.testMemberContainersShareUTS) + t.Run("MemberContainersUTSNotShared", s.testMemberContainersUTSNotShared) +} + +// testLifecycle drives the sandbox through the full lifecycle: +// +// CreateSandbox → StartSandbox → SandboxStatus(ready) → +// StopSandbox → SandboxStatus(stopped) → ShutdownSandbox +// +// The shim must transition through the expected states and the +// ShutdownSandbox RPC must succeed. +func (s *SandboxSuite) testLifecycle(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // After StartSandbox the status must reflect a running sandbox. + status, err := env.sc.SandboxStatus(env.ctx, &sandboxAPI.SandboxStatusRequest{ + SandboxID: sandboxID, + }) + if err != nil { + t.Fatalf("SandboxStatus: %v", err) + } + if status.GetState() != sandboxStateReady { + t.Errorf("SandboxStatus state after start: got %q, want %q", status.GetState(), sandboxStateReady) + } + if status.GetPid() == 0 { + t.Error("SandboxStatus Pid must be > 0 after start") + } + t.Logf("sandbox running: state=%s pid=%d", status.GetState(), status.GetPid()) + + // StopSandbox must succeed and transition state to stopped. + if _, err := env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{ + SandboxID: sandboxID, + }); err != nil { + t.Fatalf("StopSandbox: %v", err) + } + + // Poll status: the shim must report SANDBOX_NOTREADY after stop. + if err := waitForSandboxStatus(env.ctx, env.sc, sandboxID, sandboxStateNotReady, 10*time.Second); err != nil { + t.Errorf("SandboxStatus state after stop: %v", err) + } + + // ShutdownSandbox must succeed even though the sandbox is already stopped. + if _, err := env.sc.ShutdownSandbox(env.ctx, &sandboxAPI.ShutdownSandboxRequest{ + SandboxID: sandboxID, + }); err != nil { + t.Fatalf("ShutdownSandbox: %v", err) + } + + t.Log("sandbox lifecycle complete") +} + +// testPlatform verifies that Platform returns a valid OS/architecture. +// The shim must always honour this RPC; containerd uses it to generate +// a correct OCI spec for member containers. +func (s *SandboxSuite) testPlatform(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + resp, err := env.sc.Platform(env.ctx, &sandboxAPI.PlatformRequest{SandboxID: sandboxID}) + if err != nil { + t.Fatalf("Platform: %v", err) + } + p := resp.GetPlatform() + if p == nil { + t.Fatal("Platform response missing platform field") + } + if p.GetOS() == "" { + t.Error("Platform response: OS must not be empty") + } + if p.GetArchitecture() == "" { + t.Error("Platform response: Architecture must not be empty") + } + t.Logf("platform: os=%s arch=%s variant=%s", p.GetOS(), p.GetArchitecture(), p.GetVariant()) +} + +// testPing verifies that PingSandbox succeeds while the sandbox is +// running. PingSandbox is a lightweight liveness check; it must +// return without error from a live sandbox. +func (s *SandboxSuite) testPing(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + if _, err := env.sc.PingSandbox(env.ctx, &sandboxAPI.PingRequest{SandboxID: sandboxID}); err != nil { + t.Fatalf("PingSandbox: %v", err) + } + t.Log("PingSandbox succeeded") +} + +// testSingleContainer verifies that a single member container can be +// created, started, and produce output inside the sandbox. +// +// The API contract: after StartSandbox, Task.Create on the shared +// connection must create a container that runs inside the sandbox. +// Task.Start must make the container's init process runnable, and +// Task.Wait must return the init process exit status. +func (s *SandboxSuite) testSingleContainer(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + cid := createContainerInSandbox(t, env, []string{"/bin/echo", "hello-sandbox"}) + + // Read output — the container must produce the expected string. + readContainerOutput(t, env, cid, "hello-sandbox", 30*time.Second) + + // Wait for the container's natural exit and verify exit status. + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatalf("Task.Wait: %v", err) + } + if waitResp.GetExitStatus() != 0 { + t.Errorf("expected exit status 0, got %d", waitResp.GetExitStatus()) + } + + // Delete the container task. + if _, err := env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}); err != nil { + t.Fatalf("Task.Delete: %v", err) + } + t.Log("single container complete") +} + +// testMultipleContainers verifies that three member containers can run +// concurrently inside one sandbox, each receiving its own isolated +// rootfs and producing the expected output. +// +// The API contract: the sandbox must support N concurrent member +// containers. Each container's output must be independent; the sandbox +// VM must not be torn down when one container exits. +func (s *SandboxSuite) testMultipleContainers(t *testing.T) { + const n = 3 + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + type ctrResult struct { + cid string + token string + } + + // Create all containers first, then collect output concurrently. + ctrs := make([]ctrResult, n) + for i := range ctrs { + token := fmt.Sprintf("ctr%d-token-%s", i, randomSuffix()) + cid := createContainerInSandbox(t, env, []string{"/bin/echo", token}) + ctrs[i] = ctrResult{cid: cid, token: token} + } + + // Verify each container's output in parallel. + var wg sync.WaitGroup + for _, cr := range ctrs { + wg.Add(1) + go func(cr ctrResult) { + defer wg.Done() + readContainerOutput(t, env, cr.cid, cr.token, 30*time.Second) + }(cr) + } + wg.Wait() + + // All containers should have exited cleanly by now. + for _, cr := range ctrs { + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cr.cid}) + if err != nil { + t.Errorf("Task.Wait %s: %v", cr.cid, err) + continue + } + if waitResp.GetExitStatus() != 0 { + t.Errorf("container %s exit status: got %d, want 0", cr.cid, waitResp.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cr.cid}) //nolint:errcheck + } + + t.Logf("all %d containers completed", n) +} + +// testContainerLifecycleIndependence verifies that deleting one member +// container leaves the sandbox and other containers running. +// +// The API contract: Task.Delete for a member container must clean up +// that container's resources without affecting the sandbox VM or any +// other member containers. +func (s *SandboxSuite) testContainerLifecycleIndependence(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Start a long-lived "observer" container. + observerCID := createContainerInSandbox(t, env, []string{"/bin/forever", "observer"}) + + // Start a short-lived container that exits naturally. + shortCID := createContainerInSandbox(t, env, []string{"/bin/echo", "short-lived"}) + readContainerOutput(t, env, shortCID, "short-lived", 30*time.Second) + + // Wait for the short container to exit and delete it. + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: shortCID}) //nolint:errcheck + if _, err := env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: shortCID}); err != nil { + t.Fatalf("Delete short container: %v", err) + } + + // The observer container must still be running. + stateResp, err := env.tc.State(env.ctx, &taskAPI.StateRequest{ID: observerCID}) + if err != nil { + t.Fatalf("State for observer after short-container delete: %v", err) + } + if stateResp.GetStatus() != tasktypes.Status_RUNNING { + t.Errorf("observer container status after peer delete: got %v, want RUNNING", stateResp.GetStatus()) + } + + // The sandbox itself must also still be running. + status, err := env.sc.SandboxStatus(env.ctx, &sandboxAPI.SandboxStatusRequest{SandboxID: sandboxID}) + if err != nil { + t.Fatalf("SandboxStatus after peer delete: %v", err) + } + if status.GetState() != sandboxStateReady { + t.Errorf("sandbox state after peer-container delete: got %q, want %q", status.GetState(), sandboxStateReady) + } + + t.Log("observer still running after peer delete; sandbox intact") + + // Clean up the observer. + env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: observerCID, Signal: uint32(syscall.SIGKILL), All: true}) //nolint:errcheck + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: observerCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: observerCID}) //nolint:errcheck +} + +// testStatusAfterStop verifies that SandboxStatus after StopSandbox +// reports a non-ready state, and that a second StopSandbox is +// idempotent (must not error). +func (s *SandboxSuite) testStatusAfterStop(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // First stop. + if _, err := env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{SandboxID: sandboxID}); err != nil { + t.Fatalf("StopSandbox (first): %v", err) + } + + // Status must not be sandboxStateReady after stop. + status, err := env.sc.SandboxStatus(env.ctx, &sandboxAPI.SandboxStatusRequest{SandboxID: sandboxID}) + if err != nil { + t.Logf("SandboxStatus after stop returned error (may be acceptable): %v", err) + } else if status.GetState() == sandboxStateReady { + t.Errorf("SandboxStatus after stop: state is still %q; shim must not report ready after stop", status.GetState()) + } + + // Second stop must be idempotent — must not return an error. + if _, err := env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{SandboxID: sandboxID}); err != nil { + t.Errorf("StopSandbox (second, idempotency check): %v", err) + } + + t.Log("status-after-stop and idempotency checks passed") +} + +// testWaitUnblocksOnStop verifies that WaitSandbox returns after the +// sandbox is stopped. +// +// The API contract: WaitSandbox must unblock when the sandbox exits +// (via StopSandbox or ShutdownSandbox). Callers rely on this to +// detect sandbox death. +func (s *SandboxSuite) testWaitUnblocksOnStop(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + waitDone := make(chan error, 1) + go func() { + _, err := env.sc.WaitSandbox(env.ctx, &sandboxAPI.WaitSandboxRequest{SandboxID: sandboxID}) + waitDone <- err + }() + + // Give WaitSandbox a moment to start blocking. + time.Sleep(200 * time.Millisecond) + + if _, err := env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{SandboxID: sandboxID}); err != nil { + t.Fatalf("StopSandbox: %v", err) + } + + select { + case err := <-waitDone: + if err != nil { + t.Logf("WaitSandbox returned error after stop (may be acceptable for ttrpc shutdown): %v", err) + } else { + t.Log("WaitSandbox returned cleanly after stop") + } + case <-time.After(15 * time.Second): + t.Fatal("WaitSandbox did not return within 15s after StopSandbox") + } +} + +// testCreateTwiceRejected verifies that calling CreateSandbox a second +// time on the same shim returns an AlreadyExists error. +// +// The API contract: a sandbox shim process hosts exactly one sandbox. +// A second CreateSandbox must be rejected with AlreadyExists. +func (s *SandboxSuite) testCreateTwiceRejected(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + _, err := env.sc.CreateSandbox(env.ctx, &sandboxAPI.CreateSandboxRequest{ + SandboxID: sandboxID + "-dup", + BundlePath: ".", + }) + if err == nil { + t.Fatal("second CreateSandbox must fail; got nil error") + } + errStr := strings.ToLower(err.Error()) + if !strings.Contains(errStr, "already exists") && !strings.Contains(errStr, "alreadyexists") { + t.Errorf("second CreateSandbox: expected AlreadyExists error, got: %v", err) + } + t.Logf("second CreateSandbox correctly rejected: %v", err) +} + +// testStartWithoutCreateRejected verifies that StartSandbox before +// CreateSandbox fails with FailedPrecondition. +// +// The API contract: CreateSandbox must precede StartSandbox. The shim +// must reject StartSandbox if CreateSandbox has not been called first. +func (s *SandboxSuite) testStartWithoutCreateRejected(t *testing.T) { + shimBin, bundleDir, _ := shimSetup(t, s.cfg) + sandboxID := containerID(t) + ns := uniqueTestNamespace(t, "sandbox") + ctx := namespaces.WithNamespace(t.Context(), ns) + + // The shim reads config.json from its working directory for the + // grouping label. Write a minimal sandbox spec so the shim can start. + writeSandboxOCISpec(t, bundleDir) + + // Start a fresh shim without calling CreateSandbox. + params := startShim(t, shimBin, bundleDir, sandboxID, ns, s.cfg) + conn := connectShim(t, params.Address) + client := ttrpc.NewClient(conn) + defer client.Close() + + sc := sandboxAPI.NewTTRPCSandboxClient(client) + tc := taskAPI.NewTTRPCTaskClient(client) + + _, err := sc.StartSandbox(ctx, &sandboxAPI.StartSandboxRequest{SandboxID: sandboxID}) + if err == nil { + t.Fatal("StartSandbox without CreateSandbox must fail; got nil error") + } + errStr := strings.ToLower(err.Error()) + if !strings.Contains(errStr, "precondition") && !strings.Contains(errStr, "failed_precondition") { + t.Errorf("StartSandbox without create: expected FailedPrecondition, got: %v", err) + } + t.Logf("StartSandbox before CreateSandbox correctly rejected: %v", err) + + // Shut the shim down cleanly. + shutdownTask(ctx, tc, sandboxID) +} + +// testResourceReleaseOnShutdown verifies that ShutdownSandbox releases +// per-container host resources. Specifically, if the shim uses a +// shared virtiofs directory, paths under that directory must not appear +// as mount points in the shim's mount namespace after shutdown. +// +// The API contract: ShutdownSandbox must release all resources +// allocated for the sandbox and its member containers. Leaked mount +// points can prevent bundle-directory cleanup and exhaust kernel mount +// table entries. +func (s *SandboxSuite) testResourceReleaseOnShutdown(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Run two member containers to create per-container rootfs mounts. + cid1 := createContainerInSandbox(t, env, []string{"/bin/echo", "ctr1"}) + cid2 := createContainerInSandbox(t, env, []string{"/bin/echo", "ctr2"}) + + readContainerOutput(t, env, cid1, "ctr1", 30*time.Second) + readContainerOutput(t, env, cid2, "ctr2", 30*time.Second) + + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid1}) //nolint:errcheck + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid2}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid1}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid2}) //nolint:errcheck + + // Capture the shim PID before shutdown so we can inspect its + // namespace after. Use a probe container-ID; Connect returns the + // shim PID regardless of which ID is used on some shims. + shimPID := sandboxShimPID(env, cid1) + mountsBefore := sandboxContainersMounts(shimPID) + t.Logf("shim PID: %d, container mounts before shutdown: %v", shimPID, mountsBefore) + + // Stop and shut down the sandbox. + env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{SandboxID: sandboxID}) //nolint:errcheck + env.sc.ShutdownSandbox(env.ctx, &sandboxAPI.ShutdownSandboxRequest{SandboxID: sandboxID}) //nolint:errcheck + + // Give the shim time to clean up. + time.Sleep(500 * time.Millisecond) + + // After shutdown, no per-container mounts should remain in the + // shim's namespace. We check the shim's /proc//mountinfo + // if the shim runs in a private mount namespace; if the shim + // exited (pid gone) that is also a clean result. + mountsAfter := sandboxContainersMounts(shimPID) + if len(mountsAfter) > 0 { + t.Errorf("shim left %d per-container mount(s) after ShutdownSandbox: %v", + len(mountsAfter), mountsAfter) + } else { + t.Log("no per-container mounts remain after shutdown") + } +} + +// Ensure ttrpc import is used (consumed in testStartWithoutCreateRejected). +var _ *ttrpc.Client diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_member.go b/vendor/github.com/containerd/shimtest/sandbox_suite_member.go new file mode 100644 index 00000000..4f7df794 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_member.go @@ -0,0 +1,301 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +// This file contains SandboxSuite tests that verify the shim API contract +// for member-container workloads: status fields, host-network sandboxes, +// exec, shared endpoints, and outbound networking. Every test is framed +// in terms of the shim API specification, not any particular +// implementation. + +package shimtest + +import ( + "net" + "os" + "strings" + "syscall" + "testing" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + specs "github.com/opencontainers/runtime-spec/specs-go" +) + +// testStatusReportsPidAndCreatedAt verifies that SandboxStatus always returns +// a non-zero pid and a non-zero created_at timestamp. +// +// The API contract: a caller must be able to reference the sandbox's +// namespaces (e.g. /proc//ns/*) and report the sandbox's age, so a +// shim must populate a non-zero pid and a non-zero created_at after a +// successful StartSandbox. +func (s *SandboxSuite) testStatusReportsPidAndCreatedAt(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + _, info := sandboxStatusInfo(t, env) + + resp, err := env.sc.SandboxStatus(env.ctx, &sandboxAPI.SandboxStatusRequest{ + SandboxID: sandboxID, + }) + if err != nil { + t.Fatalf("SandboxStatus: %v", err) + } + + if resp.GetPid() == 0 { + t.Error("SandboxStatus.Pid must be non-zero after StartSandbox") + } + ts := resp.GetCreatedAt() + if ts == nil || ts.AsTime().IsZero() { + t.Error("SandboxStatus.CreatedAt must be non-zero after StartSandbox") + } + if info["pid"] == "" || info["pid"] == "0" { + t.Errorf("SandboxStatus.Info[pid] must be a non-zero pid string, got %q", info["pid"]) + } + if info["state"] == "" { + t.Errorf("SandboxStatus.Info[state] must not be empty, got %q", info["state"]) + } + t.Logf("status ok: pid=%d created_at=%s info=%v", resp.GetPid(), resp.GetCreatedAt().AsTime(), info) +} + +// testHostNetworkNoNetworkSandbox verifies that a sandbox created with an +// empty NetnsPath (i.e. no network sandbox provided) succeeds and that +// member containers run normally. +// +// The API contract: an empty netns_path in CreateSandboxRequest means the +// sandbox uses the host's network stack (no isolation). The shim must accept +// this case without error and member containers must run successfully. +func (s *SandboxSuite) testHostNetworkNoNetworkSandbox(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShimWithNetworkSandbox(t, s.cfg, sandboxID, "") + + cid := createContainerInSandbox(t, env, []string{"/bin/echo", "host-network-ok"}) + readContainerOutput(t, env, cid, "host-network-ok", 30*time.Second) + + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatalf("Task.Wait: %v", err) + } + if waitResp.GetExitStatus() != 0 { + t.Errorf("container exit status: got %d, want 0", waitResp.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + t.Log("host-network sandbox: member container ran successfully") +} + +// testMemberContainerExec verifies that a process can be exec'd into a running +// member container and that its output and exit status are correctly propagated. +// +// The API contract: after Task.Create + Task.Start, a shim must accept +// Task.Exec to run an additional process inside the container. The exec +// process must run inside the container's namespace and filesystem, its +// output must arrive on the configured stdio path, and Task.Wait on the +// ExecID must return the correct exit status. +// +// This contract underpins any exec-into-a-running-container use case +// (e.g. interactive exec, health/liveness probes, sidecar tooling). +func (s *SandboxSuite) testMemberContainerExec(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Start a long-lived container to exec into. + observerCID := createContainerInSandbox(t, env, []string{"/bin/forever", "exec-target"}) + + const token = "exec-probe-ok" + out, exitStatus := execInSandboxContainer(t, env, observerCID, []string{"/bin/echo", token}, 30*time.Second) + if !strings.Contains(out, token) { + t.Errorf("exec output: want %q in output, got %q", token, out) + } + if exitStatus != 0 { + t.Errorf("exec exit status: got %d, want 0", exitStatus) + } + + // Verify non-zero exit status propagation. + _, nonZeroStatus := execInSandboxContainer(t, env, observerCID, []string{"/bin/exit", "42"}, 30*time.Second) + if nonZeroStatus != 42 { + t.Errorf("exec exit-code propagation: got %d, want 42", nonZeroStatus) + } + + env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: observerCID, Signal: uint32(syscall.SIGKILL), All: true}) //nolint:errcheck + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: observerCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: observerCID}) //nolint:errcheck + t.Log("exec in sandbox member container: ok") +} + +// testCrossContainerViaUDS verifies that two member containers in one sandbox +// can both reach a shared host-side UNIX domain socket endpoint by exec'ing +// into a container that has the socket forwarded into its filesystem. +// +// The API contract: a member container that has a "uds" mount type in its +// OCI spec must receive the corresponding host-side UNIX socket forwarded into +// its filesystem. A process exec'd into the container must be able to connect +// to that socket. This is the general contract behind any shared, pre-forwarded +// host endpoint: multiple processes in a sandbox must be able to reach the +// same endpoint through it. +// +// Test topology: +// +// host UNIX socket listener (the shared endpoint) +// └── forwarded into the shared container at /run/shared.sock +// ├── exec A: nc -U /run/shared.sock (connects, verified by host accept) +// └── exec B: nc -U /run/shared.sock (connects, verified by host accept) +// +// Using exec into a single container avoids the per-container socket-forward +// routing ambiguity that arises when multiple containers each have their own +// accept stream. +func (s *SandboxSuite) testCrossContainerViaUDS(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Create a host-side UNIX socket listener (the shared endpoint). + hostSockPath, err := makeUnixSockPath(t) + if err != nil { + t.Fatalf("create host sock path: %v", err) + } + ln, err := net.Listen("unix", hostSockPath) + if err != nil { + t.Fatalf("host unix listen: %v", err) + } + t.Cleanup(func() { ln.Close() }) + + const containerSockPath = "/run/shared.sock" + + // Accept connections from the container and immediately close them. + // Closing causes nc to see EOF on the socket and exit cleanly. + hostDone := make(chan error, 2) + acceptOne := func() { + conn, err := ln.Accept() + if err != nil { + hostDone <- err + return + } + conn.Close() + hostDone <- nil + } + go acceptOne() + go acceptOne() + + // Start a long-lived container with the host socket forwarded into it. + sharedCID := createContainerInSandbox(t, env, + []string{"/bin/forever", "uds-shared-container"}, + withSandboxCtrExtraMounts(specs.Mount{ + Type: "uds", + Source: hostSockPath, + Destination: containerSockPath, + }), + ) + + // Exec nc twice into the shared container; each connection proves that + // a process running in the container can reach the shared host endpoint. + // These model two different processes reaching a common host-forwarded + // endpoint, as multiple containers in a sandbox would. + for i := 0; i < 2; i++ { + out, exitCode := execInSandboxContainer( + t, env, sharedCID, + []string{"/bin/nc", "-U", containerSockPath}, + 30*time.Second, + ) + if exitCode != 0 { + t.Errorf("nc exec %d: exit code %d, output: %q", i+1, exitCode, out) + } + // The host must have seen a connection for this exec. + select { + case err := <-hostDone: + if err != nil { + t.Fatalf("host accept connection %d: %v", i+1, err) + } + t.Logf("cross-container UDS connection %d: ok", i+1) + case <-time.After(5 * time.Second): + t.Fatalf("host did not see connection %d within 5s", i+1) + } + } + + // Kill the shared container. + env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: sharedCID, Signal: uint32(syscall.SIGKILL), All: true}) //nolint:errcheck + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: sharedCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: sharedCID}) //nolint:errcheck + t.Log("cross-container UDS: both execs reached the shared host endpoint") +} + +// testContainerOutboundTCP verifies that a process running inside a member +// container has a working outbound network path by resolving a real +// external hostname. +// +// The API contract: a shim must give member containers a working network +// stack, regardless of the mechanism used to provide it (native network +// namespace membership, a virtual NIC, or any other in-guest networking +// approach). This test does not care how connectivity is achieved — only +// that a container can reach a resolver and get back a valid answer, +// exactly as any container workload that depends on DNS would. +// +// DNS resolution (rather than a raw TCP round trip) is used here because +// Task.Exec — the mechanism this suite uses to run a process inside an +// already-running member container — has no stdin plumbing, and a TCP +// round trip needs a way to send data. /bin/host takes its input purely +// from argv and writes its result to stdout, so it fits Task.Exec's +// existing capabilities. NetworkSuite (legacy path) separately covers TCP +// and UDP round trips end-to-end. +func (s *SandboxSuite) testContainerOutboundTCP(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + cid := createContainerInSandbox(t, env, []string{"/bin/forever", "outbound-container"}) + + // host : prints " has address " for each resolved address. + out, exitStatus := execInSandboxContainer(t, env, cid, []string{"/bin/host", dnsTestHostname}, 30*time.Second) + if exitStatus != 0 { + t.Fatalf("host exec exit status: got %d, want 0; output: %q", exitStatus, out) + } + + var addrs []string + for _, line := range strings.Split(strings.TrimSpace(out), "\n") { + line = strings.TrimSpace(line) + if line == "" { + continue + } + const marker = " has address " + idx := strings.Index(line, marker) + if idx < 0 { + t.Errorf("host %s: unexpected output line %q", dnsTestHostname, line) + continue + } + ip := line[idx+len(marker):] + if net.ParseIP(ip) == nil { + t.Errorf("host %s: %q is not a valid IP address", dnsTestHostname, ip) + continue + } + addrs = append(addrs, ip) + } + if len(addrs) == 0 { + t.Fatalf("host %s produced no addresses; output: %q", dnsTestHostname, out) + } + + t.Log("container outbound DNS resolution: ok, addresses:", addrs) +} + +// makeUnixSockPath returns a UNIX socket path under a temp directory that +// satisfies the 104-byte AF_UNIX path limit on macOS. +func makeUnixSockPath(tb testing.TB) (string, error) { + tb.Helper() + dir, err := os.MkdirTemp(unixSafeDir(), "nb-uds-") + if err != nil { + return "", err + } + tb.Cleanup(func() { os.RemoveAll(dir) }) + return dir + "/shared.sock", nil +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_member_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_member_linux.go new file mode 100644 index 00000000..e2a3accd --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_member_linux.go @@ -0,0 +1,103 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +// This file contains SandboxSuite member-container workload tests that +// require Linux kernel features (network namespaces). + +package shimtest + +import ( + "testing" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" +) + +// testNetworkSandboxHeldOpen verifies that the shim holds the host-side +// network sandbox open for the lifetime of the sandbox and releases it after +// the sandbox stops. +// +// The API contract: a shim that receives a non-empty netns_path in +// CreateSandboxRequest must pin the network sandbox resource (e.g. a Linux +// network namespace bind-mount) for the duration of the sandbox. This allows +// CNI and other host-side tools to inspect or manipulate the network sandbox +// while the sandbox is running. After StopSandbox the shim must release its +// pin so that the caller can unmount the bind-mount and reclaim the resource. +// +// The test creates a real host-side network sandbox (bind-mounted netns), +// passes its path to CreateSandbox, asserts the path is reachable while the +// sandbox is ready, then stops the sandbox and verifies the state transition. +func (s *SandboxSuite) testNetworkSandboxHeldOpen(t *testing.T) { + nsPath := createNetworkSandbox(t) + + sandboxID := containerID(t) + env := startSandboxShimWithNetworkSandbox(t, s.cfg, sandboxID, nsPath) + + // While the sandbox is running the bind-mount must still be reachable. + // A missing path means the shim (or something else) unmounted it + // prematurely — violating the hold-open contract. + if !networkSandboxIsOpen(nsPath) { + t.Fatal("network sandbox disappeared while sandbox is running; shim must hold it open") + } + t.Logf("network sandbox %q is reachable while sandbox is running", nsPath) + + // Run a member container to confirm the sandbox is fully operational. + cid := createContainerInSandbox(t, env, []string{"/bin/echo", "netns-ok"}) + readContainerOutput(t, env, cid, "netns-ok", 30*time.Second) + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + + // Stop the sandbox explicitly so we can inspect the final state. + if _, err := env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{ + SandboxID: sandboxID, + }); err != nil { + t.Fatalf("StopSandbox: %v", err) + } + + if err := waitForSandboxStatus(env.ctx, env.sc, sandboxID, sandboxStateNotReady, 10*time.Second); err != nil { + t.Errorf("SandboxStatus state after stop: %v", err) + } + t.Log("network sandbox held open while running; sandbox stopped cleanly") +} + +// testNetworkSandboxPathInStatus verifies that SandboxStatus may report the +// network sandbox path in its Info map under "networkSandboxPath". +// +// The API contract: the base sandbox TTRPC protocol does not mandate specific +// Info keys. A shim that accepts a network sandbox path is encouraged to +// expose it in Info so callers can confirm which network resource is pinned +// without side-channel lookups. The test treats absence of the key as an +// informational result rather than a hard failure. +func (s *SandboxSuite) testNetworkSandboxPathInStatus(t *testing.T) { + nsPath := createNetworkSandbox(t) + + sandboxID := containerID(t) + env := startSandboxShimWithNetworkSandbox(t, s.cfg, sandboxID, nsPath) + + _, info := sandboxStatusInfo(t, env) + reported := info["networkSandboxPath"] + if reported == "" { + t.Logf("SandboxStatus.Info does not include 'networkSandboxPath' (optional field); info=%v", info) + return + } + if reported != nsPath { + t.Errorf("SandboxStatus.Info[networkSandboxPath]: got %q, want %q", reported, nsPath) + } + t.Logf("SandboxStatus.Info[networkSandboxPath]=%q (matches provided path)", reported) +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_netns_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_netns_linux.go new file mode 100644 index 00000000..6e4988f9 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_netns_linux.go @@ -0,0 +1,179 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +// This file contains a SandboxSuite test that verifies container network +// traffic is actually scoped to the network sandbox the shim was given, as +// opposed to merely holding the resource's path open. It requires root to +// create a real network namespace and network interfaces, and is skipped +// otherwise. + +package shimtest + +import ( + "net" + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" +) + +// testContainerTrafficScopedToNetworkSandbox verifies that when a sandbox is +// given a network sandbox (a non-empty netns_path in CreateSandboxRequest), +// a member container's outbound network traffic actually originates from +// within that network sandbox, rather than from whatever network context +// the shim process happens to run in. +// +// The API contract: passing a non-empty netns_path in CreateSandboxRequest +// must result in member container traffic being scoped to that network +// sandbox — regardless of the mechanism the shim uses internally (native +// namespace membership, a virtual NIC bridged into the namespace, or any +// other approach). This test only observes the externally visible result: +// a container must be able to reach an endpoint that exists only inside the +// provided network sandbox, exactly as any container workload reaching a +// service scoped to that network sandbox would. +// +// Test topology: a veth pair is created with both ends inside a real network +// namespace, giving one end an address on a subnet that has no route from +// outside that namespace (see realNetworkSandbox.attachVeth). A listener is +// bound to that address from inside the namespace. The sandbox is started +// with the namespace's path, and a member container runs nc(1) in TCP mode +// (nc ) to reach the listener's address, sending a token via +// its stdin FIFO and printing the echoed response to stdout. Since the +// address is unreachable from any context other than the namespace itself, +// a successful round trip is only possible if the container's traffic +// actually originates there. +// +// Requires root (CAP_SYS_ADMIN) to create a network namespace and attach +// network interfaces; skipped otherwise. +func (s *SandboxSuite) testContainerTrafficScopedToNetworkSandbox(t *testing.T) { + netns := createRealNetworkSandbox(t) + addr := netns.attachVeth(t) + const port = "9191" + endpoint := addr + ":" + port + + // Sanity check: confirm the address really is unreachable from the + // namespace this test process runs in, so that a later successful + // connection can only be explained by the container's traffic + // originating inside the sandbox namespace. + probeUnreachableFromCurrentNamespace(t, endpoint) + + done := listenAndEchoOnceInNetns(t, netns.path, endpoint) + + sandboxID := containerID(t) + env := startSandboxShimWithNetworkSandbox(t, s.cfg, sandboxID, netns.path) + + const token = "netns-scoped-ok" + cid := createContainerInSandbox(t, env, []string{"/bin/nc", addr, port}, withSandboxCtrStdin()) + writeContainerStdin(t, env, cid, token+"\n") + readContainerOutput(t, env, cid, token, 30*time.Second) + + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatalf("Task.Wait: %v", err) + } + if waitResp.GetExitStatus() != 0 { + t.Fatalf("container exit status: got %d, want 0 "+ + "(container could not reach the network-sandbox-scoped endpoint; "+ + "its traffic may not be originating inside the provided network sandbox)", + waitResp.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + + select { + case err := <-done: + if err != nil { + t.Fatalf("network-sandbox-scoped endpoint did not observe a successful exchange: %v", err) + } + case <-time.After(5 * time.Second): + t.Fatal("network-sandbox-scoped endpoint did not observe a connection from the container within 5s") + } + + t.Log("container traffic is scoped to the provided network sandbox") +} + +// testInboundToNetnsScopedListener verifies the inbound (listen) side of the +// network-sandbox contract: a member container that binds a TCP listener must +// be reachable by a caller connecting from within the same network sandbox. +// +// The API contract: when a sandbox is given a non-empty netns_path in +// CreateSandboxRequest, a member container's inbound TCP listener must be +// accessible from within that network sandbox — not only from the root +// namespace. This is the mirror of testContainerTrafficScopedToNetworkSandbox: +// where that test verifies outbound traffic stays inside the sandbox, this +// test verifies that inbound connections from inside the sandbox reach the +// container listener. +// +// Test topology: the same veth-pair-inside-a-real-netns setup from +// testContainerTrafficScopedToNetworkSandbox, except the roles are reversed. +// A member container runs echosrv (which prints its bound port to stdout then +// accepts one connection and echoes). Once the port is known, the test enters +// the sandbox namespace on a locked OS thread and dials the container from +// inside it. Since the veth addresses are unreachable from outside the +// namespace, a successful round trip proves the listener is inside the sandbox. +// +// Requires root (CAP_SYS_ADMIN) to create a network namespace and attach +// network interfaces; skipped otherwise. +func (s *SandboxSuite) testInboundToNetnsScopedListener(t *testing.T) { + netns := createRealNetworkSandbox(t) + vethAddr := netns.attachVeth(t) + + // Sanity check: the veth address is unreachable from our namespace. + probeUnreachableFromCurrentNamespace(t, net.JoinHostPort(vethAddr, "1")) + + sandboxID := containerID(t) + env := startSandboxShimWithNetworkSandbox(t, s.cfg, sandboxID, netns.path) + + // Start echosrv 0 (ephemeral port). echosrv prints the bound port to + // stdout before accepting, so readContainerOutput will capture it. + listenerCID := createContainerInSandbox(t, env, []string{"/bin/echosrv", "0"}) + + // Wait for echosrv to print the port to stdout. The port appears as the + // first (and only pre-accept) line; give it 15 s which is generous for + // any VM start + listen latency. + boundPort := waitForContainerPort(t, env, listenerCID, 15*time.Second) + t.Logf("container echosrv bound on port %s", boundPort) + + // The container is inside the sandbox netns; its listener is reachable + // at the veth address from inside that namespace. Dial from inside it. + endpoint := net.JoinHostPort(vethAddr, boundPort) + const token = "netns-inbound-ok" + done := dialInNetnsRoundTrip(t, netns.path, endpoint, token) + + select { + case err := <-done: + if err != nil { + t.Fatalf("dial from inside sandbox namespace to container listener failed: %v\n"+ + "(the shim must make the container's listener reachable from within "+ + "the provided network sandbox)", err) + } + case <-time.After(15 * time.Second): + t.Fatal("timed out waiting for in-netns dial to container listener to complete") + } + + // echosrv exits after one exchange; wait for it. + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: listenerCID}) + if err != nil { + t.Fatalf("Task.Wait listener: %v", err) + } + if waitResp.GetExitStatus() != 0 { + t.Fatalf("listener container exit status: got %d, want 0", waitResp.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: listenerCID}) //nolint:errcheck + + t.Log("inbound connection from inside network sandbox to container listener: ok") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_oom_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_oom_linux.go new file mode 100644 index 00000000..e4eff676 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_oom_linux.go @@ -0,0 +1,89 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "syscall" + "testing" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + tasktypes "github.com/containerd/containerd/api/types/task" +) + +// testMemberContainerOOMIsolation verifies that the kernel OOM-killing +// one member container is scoped to that container: a peer member +// container and the sandbox itself must be unaffected. +// +// The API contract: a per-container memory limit (set via the +// container's OCI spec, exactly as OOMSuite's standalone test does) is +// a property of that one container's cgroup. The shim must not let an +// OOM kill inside one container's cgroup propagate to, or otherwise +// disturb, a sibling member container or the sandbox VM/process group +// hosting them both. This is the OOM-specific counterpart to +// testContainerLifecycleIndependence, which covers the same "one +// container's fate must not affect its peers or the sandbox" contract +// for a graceful exit rather than a kernel-forced kill. +func (s *SandboxSuite) testMemberContainerOOMIsolation(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Start a long-lived peer with no memory limit. + peerCID := createContainerInSandbox(t, env, []string{"/bin/forever", "oom-peer"}) + + // Start the victim with a tight memory limit and a memory-hungry + // workload, matching OOMSuite's standalone test. + victimCID := createContainerInSandbox(t, env, []string{"/bin/memhog"}, + withSandboxCtrOCIOpts(withMemoryLimit(128*1024*1024))) + + victimWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: victimCID}) + if err != nil { + t.Fatalf("Task.Wait victim: %v", err) + } + const sigkillExit = 128 + uint32(syscall.SIGKILL) + if victimWait.GetExitStatus() != sigkillExit { + t.Fatalf("victim container exit status: got %d, want %d (SIGKILL from OOM killer)", + victimWait.GetExitStatus(), sigkillExit) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: victimCID}) //nolint:errcheck + + // The peer, in its own unrestricted cgroup, must still be running. + stateResp, err := env.tc.State(env.ctx, &taskAPI.StateRequest{ID: peerCID}) + if err != nil { + t.Fatalf("State for peer after victim OOM kill: %v", err) + } + if stateResp.GetStatus() != tasktypes.Status_RUNNING { + t.Errorf("peer container status after victim's OOM kill: got %v, want RUNNING", stateResp.GetStatus()) + } + + // The sandbox itself must also still be running. + status, err := env.sc.SandboxStatus(env.ctx, &sandboxAPI.SandboxStatusRequest{SandboxID: sandboxID}) + if err != nil { + t.Fatalf("SandboxStatus after victim's OOM kill: %v", err) + } + if status.GetState() != sandboxStateReady { + t.Errorf("sandbox state after victim's OOM kill: got %q, want %q", status.GetState(), sandboxStateReady) + } + + t.Log("peer still running and sandbox intact after a sibling container's OOM kill") + + env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: peerCID, Signal: uint32(syscall.SIGKILL), All: true}) //nolint:errcheck + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: peerCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: peerCID}) //nolint:errcheck +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_other.go b/vendor/github.com/containerd/shimtest/sandbox_suite_other.go new file mode 100644 index 00000000..2f73c693 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_other.go @@ -0,0 +1,37 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import "testing" + +// SandboxSuite is the sandbox conformance suite. On non-Linux +// platforms the suite is not supported and every test is skipped. +type SandboxSuite struct { + cfg Config +} + +// NewSandboxSuite constructs a SandboxSuite. +func NewSandboxSuite(cfg Config) *SandboxSuite { + return &SandboxSuite{cfg: cfg} +} + +// Run skips the entire suite on non-Linux platforms. +func (s *SandboxSuite) Run(t *testing.T) { + t.Skip("SandboxSuite is Linux-only (virtiofs-backed shared filesystem)") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_shared_devshm_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_devshm_linux.go new file mode 100644 index 00000000..f4d8f77f --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_devshm_linux.go @@ -0,0 +1,167 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "strings" + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + specs "github.com/opencontainers/runtime-spec/specs-go" +) + +// devShmMount is the /dev/shm mount exactly as sent by a real CRI client +// for a pod's member containers (confirmed against a live cluster via +// `crictl inspect`): every member container gets this same, independent +// tmpfs mount in its own spec — there is no shared-mount indication in the +// request at all. Deliberately not shimtest's default: the default OCI +// spec createOCISpec builds has no /dev/shm entry, so tests that don't care +// about it aren't affected, and this test is explicit about the exact +// shape it depends on. +var devShmMount = specs.Mount{ + Destination: "/dev/shm", + Type: "tmpfs", + Source: "shm", + Options: []string{"nosuid", "noexec", "nodev", "mode=1777", "size=65536k"}, +} + +// testMemberContainersShareDevShm verifies that member containers of the +// same sandbox that share an IPC namespace also get a working, shared +// /dev/shm: a POSIX-shared-memory-style write — through an +// mmap(MAP_SHARED) mapping, not a plain write(2) — made by one member +// container is visible, through an independent mmap(MAP_SHARED) mapping, +// to a second, independently created member container. +// +// The API contract: Kubernetes pods share both an IPC namespace and +// /dev/shm across their containers by default (the same +// shareProcessNamespace / hostIPC-independent default this suite's IPC +// namespace test exercises). A container's own OCI spec carries no +// separate signal for "share /dev/shm too" — CRI clients send every +// member container the same, independent-looking +// {Type: "tmpfs", Destination: "/dev/shm"} mount — so the shim must infer +// /dev/shm sharing from the same "this container is joining a shared IPC +// namespace" signal used for IPC namespace sharing itself, and must not +// rely on the container's own /dev/shm mount already looking shared. +// +// The write and read paths deliberately go through mmap rather than +// read(2)/write(2): what's under test is whether two independent +// mmap(MAP_SHARED) mappings of what should be the same underlying file — +// made by processes in different containers, each reaching the file +// through whatever mechanism the shim uses to make it appear at +// /dev/shm — actually share memory, which is a stronger and more direct +// test of POSIX shared-memory semantics than confirming that file +// contents eventually match. +func (s *SandboxSuite) testMemberContainersShareDevShm(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + const ( + shmPath = "/dev/shm/testfile" + marker = "shared-devshm-ok" + ) + + writerCID := createContainerInSandbox(t, env, []string{"/bin/shmmapwrite", shmPath, marker}, + withSandboxCtrNamespace(specs.IPCNamespace, "/proc/1/ns/ipc"), + withSandboxCtrExtraMounts(devShmMount)) + + writerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: writerCID}) + if err != nil { + t.Fatalf("Task.Wait writer: %v", err) + } + if writerWait.GetExitStatus() != 0 { + t.Fatalf("writer container exit status: got %d, want 0", writerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: writerCID}) //nolint:errcheck + + readerCID := createContainerInSandbox(t, env, []string{"/bin/shmmapread", shmPath}, + withSandboxCtrNamespace(specs.IPCNamespace, "/proc/1/ns/ipc"), + withSandboxCtrExtraMounts(devShmMount)) + + out := readContainerOutput(t, env, readerCID, marker, 30*time.Second) + if !strings.Contains(out, marker) { + t.Fatalf("shmmapread output did not contain marker %q: %q", marker, out) + } + + readerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: readerCID}) + if err != nil { + t.Fatalf("Task.Wait reader: %v", err) + } + if readerWait.GetExitStatus() != 0 { + t.Fatalf("reader container exit status: got %d, want 0", readerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: readerCID}) //nolint:errcheck + + t.Log("member containers share /dev/shm: mmap(MAP_SHARED) write visible across containers") +} + +// testMemberContainersDevShmNotSharedWithoutIPC verifies the negative case: +// a member container that does *not* share an IPC namespace (the default — +// NamespaceMode_CONTAINER) gets its own private /dev/shm, even though its +// spec carries the exact same-looking {Type: "tmpfs", Destination: +// "/dev/shm"} mount as a sharing container's does. This guards against a +// shim inferring sharing from the mount's shape (which is identical either +// way) rather than from the IPC-sharing signal. +func (s *SandboxSuite) testMemberContainersDevShmNotSharedWithoutIPC(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + const ( + shmPath = "/dev/shm/testfile" + marker = "should-not-be-visible" + ) + + writerCID := createContainerInSandbox(t, env, []string{"/bin/shmmapwrite", shmPath, marker}, + withSandboxCtrExtraMounts(devShmMount)) + // No withSandboxCtrNamespace(IPCNamespace, ...): this container does not + // request IPC sharing. + + writerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: writerCID}) + if err != nil { + t.Fatalf("Task.Wait writer: %v", err) + } + if writerWait.GetExitStatus() != 0 { + t.Fatalf("writer container exit status: got %d, want 0", writerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: writerCID}) //nolint:errcheck + + readerCID := createContainerInSandbox(t, env, []string{"/bin/shmmapread", shmPath}, + withSandboxCtrExtraMounts(devShmMount)) + + // Wait for the reader's actual, specific "not found" output (see + // cmdShmMapRead) rather than an empty want string: strings.Contains + // treats "" as a match against anything, including a buffer that + // hasn't received any output yet, which would let this pass without + // having waited for the container to actually run at all. + out := readContainerOutput(t, env, readerCID, "NOTFOUND", 30*time.Second) + if strings.Contains(out, marker) { + t.Fatalf("reader saw writer's marker %q despite neither container sharing IPC: %q", marker, out) + } + + readerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: readerCID}) + if err != nil { + t.Fatalf("Task.Wait reader: %v", err) + } + if readerWait.GetExitStatus() == 0 { + t.Fatalf("reader container exit status: got 0, want non-zero (file should not exist in its private /dev/shm)") + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: readerCID}) //nolint:errcheck + + t.Log("member containers not sharing IPC correctly get independent, private /dev/shm instances") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_shared_ipcns_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_ipcns_linux.go new file mode 100644 index 00000000..ca04cd94 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_ipcns_linux.go @@ -0,0 +1,94 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "strings" + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + specs "github.com/opencontainers/runtime-spec/specs-go" +) + +// testMemberContainersShareIPC verifies that member containers of the +// same sandbox can share an IPC namespace: a SysV shared memory segment +// created by one member container is visible — by its well-known key — +// to a second, independently created member container. +// +// SysV IPC objects are chosen (rather than, say, a shared file) because +// their visibility is governed entirely by the process's IPC namespace, +// independent of mount namespace or any bind-mounted directory. A +// successful cross-container round trip through the same key is +// conclusive proof of a shared IPC namespace specifically, not an +// artifact of some other sharing mechanism. +// +// The API contract: when a member container's OCI spec carries a host +// path on its IPC namespace entry, the shim must place that container +// in an IPC namespace shared with its sandbox peers (e.g. this is how a +// caller expresses Kubernetes' default of always sharing one IPC +// namespace across a pod's containers). This test only observes the +// externally visible result and does not assume any particular +// mechanism a shim uses to provide it. It intentionally uses a +// placeholder host path (see withSandboxCtrNamespace) since only a +// live host has an actual sandbox PID to put there. +func (s *SandboxSuite) testMemberContainersShareIPC(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + const ( + shmKey = "424242" + marker = "shared-ipcns-ok" + ) + + writerCID := createContainerInSandbox(t, env, []string{"/bin/shmwrite", shmKey, marker}, + withSandboxCtrNamespace(specs.IPCNamespace, "/proc/1/ns/ipc")) + + writerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: writerCID}) + if err != nil { + t.Fatalf("Task.Wait writer: %v", err) + } + if writerWait.GetExitStatus() != 0 { + t.Fatalf("writer container exit status: got %d, want 0", writerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: writerCID}) //nolint:errcheck + + // The shm segment created by the writer persists in the IPC + // namespace after the writer container exits (nothing calls + // IPC_RMID on it), so there is no ordering requirement beyond the + // writer having already exited. + readerCID := createContainerInSandbox(t, env, []string{"/bin/shmread", shmKey}, + withSandboxCtrNamespace(specs.IPCNamespace, "/proc/1/ns/ipc")) + + out := readContainerOutput(t, env, readerCID, marker, 30*time.Second) + if !strings.Contains(out, marker) { + t.Fatalf("shmread output did not contain marker %q: %q", marker, out) + } + + readerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: readerCID}) + if err != nil { + t.Fatalf("Task.Wait reader: %v", err) + } + if readerWait.GetExitStatus() != 0 { + t.Fatalf("reader container exit status: got %d, want 0", readerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: readerCID}) //nolint:errcheck + + t.Log("member containers share an IPC namespace: shared memory segment visible across containers") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_shared_netns_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_netns_linux.go new file mode 100644 index 00000000..fc072508 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_netns_linux.go @@ -0,0 +1,79 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" +) + +// testMemberContainersShareNetwork verifies that member containers of the +// same sandbox share a network stack: a listener started by one member +// container is reachable from a second, independently created member +// container via loopback, with no explicit network configuration on either +// container. +// +// The API contract: a shim's sandbox model requires all member containers +// of one sandbox to share a single network identity (e.g. this is what +// backs a Kubernetes pod's shared network namespace). This test only +// observes the externally visible result of that contract — +// cross-container loopback connectivity — and does not assume any +// particular mechanism a shim uses to provide it (a real shared network +// namespace, a shared virtual interface, or any other approach). +func (s *SandboxSuite) testMemberContainersShareNetwork(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Use port 0 so the OS picks an ephemeral port, avoiding collisions + // when tests run concurrently. echosrv prints the actual bound port to + // stdout before accepting; waitForContainerPort reads it synchronously, + // replacing the old time.Sleep readiness guard with a real signal. + listenerCID := createContainerInSandbox(t, env, []string{"/bin/echosrv", "0"}) + + // Block until echosrv has bound and printed its port — no sleep needed. + port := waitForContainerPort(t, env, listenerCID, 30*time.Second) + t.Logf("listener container bound on port %s", port) + + const token = "shared-netns-ok" + clientCID := createContainerInSandbox(t, env, []string{"/bin/nc", "127.0.0.1", port}, withSandboxCtrStdin()) + writeContainerStdin(t, env, clientCID, token) + readContainerOutput(t, env, clientCID, token, 30*time.Second) + + clientWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: clientCID}) + if err != nil { + t.Fatalf("Task.Wait client: %v", err) + } + if clientWait.GetExitStatus() != 0 { + t.Fatalf("client container exit status: got %d, want 0", clientWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: clientCID}) //nolint:errcheck + + listenerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: listenerCID}) + if err != nil { + t.Fatalf("Task.Wait listener: %v", err) + } + if listenerWait.GetExitStatus() != 0 { + t.Fatalf("listener container exit status: got %d, want 0", listenerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: listenerCID}) //nolint:errcheck + + t.Log("member containers share a network namespace: loopback connectivity confirmed") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_shared_pidns_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_pidns_linux.go new file mode 100644 index 00000000..06b07798 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_pidns_linux.go @@ -0,0 +1,158 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "strings" + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + specs "github.com/opencontainers/runtime-spec/specs-go" +) + +// testMemberContainersSharePID verifies that member containers of the +// same sandbox can share a PID namespace: a process started by one +// member container is visible — by PID and argv — to a second, +// independently created member container. +// +// The API contract: when a member container's OCI spec carries a host +// path on its PID namespace entry, the shim must place that container +// in a PID namespace shared with its sandbox peers rather than a fresh, +// isolated one (e.g. this is how a caller expresses Kubernetes' +// shareProcessNamespace: true or hostPID: true for a pod). This test +// only observes the externally visible result — cross-container process +// visibility via /proc — and does not assume any particular mechanism a +// shim uses to provide it (a real shared PID namespace, or any other +// approach). It intentionally uses a placeholder host path (see +// withSandboxCtrNamespace) since only a live host has an actual sandbox +// PID to put there; the shim's job is to recognize that a host path was +// requested at all and substitute its own equivalent, not to interpret +// the specific value. +func (s *SandboxSuite) testMemberContainersSharePID(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + const marker = "pid-share-marker-forever" + + sentinelCID := createContainerInSandbox(t, env, []string{"/bin/forever", marker}, + withSandboxCtrNamespace(specs.PIDNamespace, "/proc/1/ns/pid")) + + // Give the sentinel process a moment to actually start before the + // scanner container looks for it; there is no synchronous "ready" + // signal available across two containers created independently via + // the Task API. + time.Sleep(200 * time.Millisecond) + + scannerCID := createContainerInSandbox(t, env, []string{"/bin/pidscan"}, + withSandboxCtrNamespace(specs.PIDNamespace, "/proc/1/ns/pid")) + + out := readContainerOutput(t, env, scannerCID, marker, 30*time.Second) + if !strings.Contains(out, marker) { + t.Fatalf("pidscan output did not contain sentinel marker %q: %q", marker, out) + } + + scannerWait, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: scannerCID}) + if err != nil { + t.Fatalf("Task.Wait scanner: %v", err) + } + if scannerWait.GetExitStatus() != 0 { + t.Fatalf("scanner container exit status: got %d, want 0", scannerWait.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: scannerCID}) //nolint:errcheck + + if _, err := env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: sentinelCID, Signal: 9, All: true}); err != nil { + t.Fatalf("Task.Kill sentinel: %v", err) + } + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: sentinelCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: sentinelCID}) //nolint:errcheck + + t.Log("member containers share a PID namespace: scanner observed the sentinel process's argv") +} + +// testMemberContainersSharePIDKillScopedToOwnContainer verifies the +// converse safety property to testMemberContainersSharePID: sharing a PID +// namespace makes a peer's processes *visible*, but Task.Kill and +// Task.Pids remain scoped to the container named in the request and never +// affect or report a peer's processes, even though both containers' PIDs +// live in the same namespace and are visible to each other via /proc. +// +// The API contract: Task.Kill and Task.Pids identify processes by +// container ID (and, for Kill, ExecID) — never by a raw PID — so a shim +// must resolve a request against that specific container's own tracked +// process(es), not by number within whatever PID namespace the container +// happens to be in. A caller has no way to name a PID at all through this +// API, so this test's job is to confirm a shim doesn't reintroduce that +// possibility through some other means (e.g. resolving Kill by scanning a +// shared namespace for a matching argv or similar). +func (s *SandboxSuite) testMemberContainersSharePIDKillScopedToOwnContainer(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + const ( + victimMarker = "pid-kill-scope-victim" + peerMarker = "pid-kill-scope-peer" + ) + + victimCID := createContainerInSandbox(t, env, []string{"/bin/forever", victimMarker}, + withSandboxCtrNamespace(specs.PIDNamespace, "/proc/1/ns/pid")) + peerCID := createContainerInSandbox(t, env, []string{"/bin/forever", peerMarker}, + withSandboxCtrNamespace(specs.PIDNamespace, "/proc/1/ns/pid")) + + // Give both a moment to actually start (see testMemberContainersSharePID). + time.Sleep(200 * time.Millisecond) + + // Confirm sharing actually took effect before testing scoping: a + // scanner that can't see the peer at all would make the rest of this + // test meaningless (scoping would trivially "work" for the wrong + // reason). + scannerCID := createContainerInSandbox(t, env, []string{"/bin/pidscan"}, + withSandboxCtrNamespace(specs.PIDNamespace, "/proc/1/ns/pid")) + out := readContainerOutput(t, env, scannerCID, peerMarker, 30*time.Second) + if !strings.Contains(out, peerMarker) { + t.Fatalf("pidscan output did not contain peer marker %q: %q (sharing did not take effect)", peerMarker, out) + } + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: scannerCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: scannerCID}) //nolint:errcheck + + // Kill the victim with All: true -- the broadest kill request the API + // allows -- and confirm the peer, sharing the same PID namespace, is + // unaffected. + if _, err := env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: victimCID, Signal: 9, All: true}); err != nil { + t.Fatalf("Task.Kill victim: %v", err) + } + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: victimCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: victimCID}) //nolint:errcheck + + peerPids, err := env.tc.Pids(env.ctx, &taskAPI.PidsRequest{ID: peerCID}) + if err != nil { + t.Fatalf("Task.Pids peer: %v", err) + } + if len(peerPids.GetProcesses()) == 0 { + t.Fatal("peer container reports no processes after killing the victim; it should be unaffected") + } + + if _, err := env.tc.Kill(env.ctx, &taskAPI.KillRequest{ID: peerCID, Signal: 9, All: true}); err != nil { + t.Fatalf("Task.Kill peer: %v", err) + } + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: peerCID}) //nolint:errcheck + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: peerCID}) //nolint:errcheck + + t.Log("Task.Kill and Task.Pids stayed scoped to their own container despite a shared PID namespace") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_shared_utsns_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_utsns_linux.go new file mode 100644 index 00000000..2c23ae89 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_shared_utsns_linux.go @@ -0,0 +1,171 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "strings" + "testing" + "time" + + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" + specs "github.com/opencontainers/runtime-spec/specs-go" +) + +// requireSandboxHostnameCapability probes, independently of any +// namespace-sharing behavior, whether the shim honors a member +// container's request for CAP_SYS_ADMIN by running "hostname " +// once in an otherwise-unshared container. If the set fails, the +// calling test is skipped rather than failed: granting a container's +// requested Linux capabilities is a separate contract from namespace +// sharing (which is what UTS tests in this file actually check), and a +// shim that doesn't support capability requests at all shouldn't be +// penalized on a contract it was never asked to implement here. +func requireSandboxHostnameCapability(t *testing.T, env *sandboxEnv) { + t.Helper() + + cid := createContainerInSandbox(t, env, []string{"/bin/hostname", "cap-probe-" + randomSuffix()}, + withSandboxCtrOCIOpts(withCapabilities("CAP_SYS_ADMIN"))) + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatalf("Task.Wait capability probe: %v", err) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + if waitResp.GetExitStatus() != 0 { + t.Skipf("shim did not grant the requested CAP_SYS_ADMIN to the container (hostname set failed); cannot verify UTS namespace behavior") + } +} + +// setSandboxHostname creates a member container that sets the UTS +// namespace hostname via the standard "hostname " command, +// applying opts (e.g. withSandboxCtrNamespace(UTSNamespace, ...) to +// join a shared UTS namespace) to its spec, and waits for it to exit. +// +// Callers must call requireSandboxHostnameCapability first: with the +// capability precondition already verified separately, a non-zero exit +// here is treated as a hard test failure rather than a skip. +// +// The standard "hostname " exits immediately and silently on a +// successful set rather than holding the namespace open itself (see +// cmdHostname), so a caller that later observes the change is also +// proving the namespace — and its hostname — outlives the process that +// set it, not merely that a still-running setter's own namespace is +// visible. +func setSandboxHostname(t *testing.T, env *sandboxEnv, hostname string, opts ...func(*sandboxCtrSpec)) { + t.Helper() + + opts = append(opts, withSandboxCtrOCIOpts(withCapabilities("CAP_SYS_ADMIN"))) + cid := createContainerInSandbox(t, env, []string{"/bin/hostname", hostname}, opts...) + + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatalf("Task.Wait hostname setter: %v", err) + } + if waitResp.GetExitStatus() != 0 { + t.Fatalf("hostname setter exit status: got %d, want 0 (CAP_SYS_ADMIN already verified available)", waitResp.GetExitStatus()) + } + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck +} + +// readSandboxHostname creates a member container that prints its UTS +// namespace hostname via the standard, argument-less "hostname" +// command, applying opts to its spec, waits for it to exit, and +// returns the trimmed hostname it reported. +func readSandboxHostname(t *testing.T, env *sandboxEnv, opts ...func(*sandboxCtrSpec)) string { + t.Helper() + + cid := createContainerInSandbox(t, env, []string{"/bin/hostname"}, opts...) + waitResp, err := env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: cid}) + if err != nil { + t.Fatalf("Task.Wait hostname reader: %v", err) + } + if waitResp.GetExitStatus() != 0 { + t.Fatalf("hostname reader exit status: got %d, want 0", waitResp.GetExitStatus()) + } + // Allow a moment for the last of stdout to drain after exit (see + // containerOutputSnapshot). + time.Sleep(50 * time.Millisecond) + out := strings.TrimSpace(containerOutputSnapshot(t, env, cid)) + env.tc.Delete(env.ctx, &taskAPI.DeleteRequest{ID: cid}) //nolint:errcheck + return out +} + +// testMemberContainersShareUTS verifies that member containers of the +// same sandbox can share a UTS namespace: a hostname change made by one +// member container via the standard "hostname " command is +// visible — via the kernel's reported hostname, not a file or +// environment variable — to a second, independently created member +// container, even after the container that made the change has exited. +// +// The API contract: when a member container's OCI spec carries a host +// path on its UTS namespace entry (e.g. this is how a caller expresses +// Kubernetes' default of sharing one hostname across a pod's +// containers), the shim must place that container in a UTS namespace +// shared with its sandbox peers rather than a fresh, isolated one, and +// that shared namespace must be owned by the sandbox rather than tied +// to the lifetime of whichever container last changed its hostname. +// This test only observes the externally visible result and does not +// assume any particular mechanism a shim uses to provide it. It +// intentionally uses a placeholder host path (see +// withSandboxCtrNamespace) since only a live host has an actual +// sandbox PID to put there. +func (s *SandboxSuite) testMemberContainersShareUTS(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + requireSandboxHostnameCapability(t, env) + + hostname := "shared-uts-" + randomSuffix() + setSandboxHostname(t, env, hostname, withSandboxCtrNamespace(specs.UTSNamespace, "/proc/1/ns/uts")) + + got := readSandboxHostname(t, env, withSandboxCtrNamespace(specs.UTSNamespace, "/proc/1/ns/uts")) + if got != hostname { + t.Fatalf("reader hostname: got %q, want %q", got, hostname) + } + + t.Log("member containers share a UTS namespace: hostname change outlived the container that set it and was visible to a peer") +} + +// testMemberContainersUTSNotShared verifies the converse of +// testMemberContainersShareUTS: a member container that does not +// request UTS sharing must not observe a peer's hostname change, even +// though both containers belong to the same sandbox. +// +// The API contract mirrors testMemberContainersSharePID's converse and +// testMemberContainersDevShmNotSharedWithoutIPC: the shim must key UTS +// namespace sharing off the container's own UTS-namespace-sharing +// signal, not off simply being a member of the same sandbox. +func (s *SandboxSuite) testMemberContainersUTSNotShared(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + requireSandboxHostnameCapability(t, env) + + hostname := "should-not-be-visible-" + randomSuffix() + setSandboxHostname(t, env, hostname, withSandboxCtrNamespace(specs.UTSNamespace, "/proc/1/ns/uts")) + + // No withSandboxCtrNamespace(UTSNamespace, ...): this container does + // not request UTS sharing, so it must get its own, unaffected UTS + // namespace. + got := readSandboxHostname(t, env) + if got == hostname { + t.Fatalf("reader saw writer's hostname %q despite neither container sharing UTS", hostname) + } + + t.Log("member container not sharing UTS correctly did not observe a peer's hostname change") +} diff --git a/vendor/github.com/containerd/shimtest/sandbox_suite_volumes_linux.go b/vendor/github.com/containerd/shimtest/sandbox_suite_volumes_linux.go new file mode 100644 index 00000000..3618fe59 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/sandbox_suite_volumes_linux.go @@ -0,0 +1,104 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "os" + "path/filepath" + "strings" + "testing" + "time" + + specs "github.com/opencontainers/runtime-spec/specs-go" +) + +// testMemberContainerHostVolume verifies that a member container can be +// given a host directory as a bind-mount volume (an OCI "bind" mount type +// in the container spec), and that the mount is a real, live share — not +// a one-time copy: a file updated on the host after the container has +// already started must become visible inside it. +// +// The API contract: a container's OCI spec may include "bind" mounts +// referencing host paths (e.g. this is how a caller expresses Kubernetes +// hostPath volumes or RecursiveReadOnly=false bind mounts) and the shim +// must honor them for member containers, exactly as it does for the +// top-level bundle rootfs. This test only observes the externally +// visible behavior: it does not assume any particular implementation +// mechanism (a hot-added virtual filesystem, a pre-existing shared tree, +// or anything else a shim might use to satisfy the mount). +func (s *SandboxSuite) testMemberContainerHostVolume(t *testing.T) { + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + hostDir := t.TempDir() + const ( + fileName = "data.txt" + hostToken = "AAA-initial" + containerMP = "/data" + ) + hostFile := filepath.Join(hostDir, fileName) + if err := os.WriteFile(hostFile, []byte(hostToken+"\n"), 0o644); err != nil { + t.Fatalf("write host file: %v", err) + } + + cid := createContainerInSandbox(t, env, []string{"/bin/forever", "volume-container"}, + withSandboxCtrExtraMounts(specs.Mount{ + Type: "bind", + Source: hostDir, + Destination: containerMP, + Options: []string{"rbind", "rw"}, + }), + ) + + // Host-to-container direction: the file written before the container + // started must be readable inside it. Tokens deliberately use short, + // mutually-exclusive prefixes ("AAA"/"BBB") rather than distinguishing + // suffixes: execInSandboxContainer's output capture allows only a + // fixed, short grace period for FIFO data to drain after the exec'd + // process exits, so a real but short read (e.g. just "AAA-in") must + // still unambiguously identify which file version was seen. + out, exitCode := execInSandboxContainer(t, env, cid, + []string{"/bin/cat", containerMP + "/" + fileName}, 30*time.Second) + if exitCode != 0 { + t.Fatalf("cat host file from container: exit code %d, output %q", exitCode, out) + } + if !strings.HasPrefix(out, "AAA") { + t.Errorf("container read of host file: got %q, want a prefix of %q", out, hostToken) + } + + // Container-to-host direction is exercised the other way around: update + // the file on the host after the container has already started and + // confirm the container sees the change live, proving this is a real + // shared mount and not a one-shot copy taken when the container + // started. + const updatedToken = "BBB-updated" + if err := os.WriteFile(hostFile, []byte(updatedToken+"\n"), 0o644); err != nil { + t.Fatalf("update host file: %v", err) + } + out, exitCode = execInSandboxContainer(t, env, cid, + []string{"/bin/cat", containerMP + "/" + fileName}, 30*time.Second) + if exitCode != 0 { + t.Fatalf("cat updated host file from container: exit code %d, output %q", exitCode, out) + } + if !strings.HasPrefix(out, "BBB") { + t.Errorf("container read of updated host file: got %q, want a prefix of %q", out, updatedToken) + } + + t.Log("member container host volume: live bind mount confirmed both at creation and after a host-side update") +} diff --git a/vendor/github.com/containerd/shimtest/stress_sandbox_linux.go b/vendor/github.com/containerd/shimtest/stress_sandbox_linux.go new file mode 100644 index 00000000..cfabdf9b --- /dev/null +++ b/vendor/github.com/containerd/shimtest/stress_sandbox_linux.go @@ -0,0 +1,229 @@ +//go:build linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import ( + "context" + "fmt" + "os" + "sync" + "sync/atomic" + "testing" + "time" + + sandboxAPI "github.com/containerd/containerd/api/runtime/sandbox/v1" + taskAPI "github.com/containerd/containerd/api/runtime/task/v3" +) + +// stressSandboxConcurrency is the number of member containers created +// per iteration of the sandbox stress test. +const stressSandboxConcurrency = 3 + +// stressSandboxMaxRSSGrowth is the upper bound on shim RSS growth (in +// bytes) for the sandbox stress run. VM-based shims exhibit a large +// one-time RSS step on first boot (guest RAM, VMM structures) that +// saturates quickly; growth beyond that is the signal of a per-container +// leak. +// +// 512 MiB accommodates the observed one-time VM boot step (typically +// ~200–300 MiB on Linux) with headroom. Per-container growth at +// steady state should be < 1 KiB/container. +const stressSandboxMaxRSSGrowth = 512 * 1024 * 1024 + +// testSandbox exercises the sandbox shim API under sustained container +// churn: one long-lived sandbox VM hosts repeated bursts of concurrent +// member containers. Each burst creates, starts, waits, and deletes +// stressSandboxConcurrency containers before the next burst begins. +// +// Leak-detection components: +// +// 1. Process leak: no shim processes remain after the run (enforced +// by the top-level registerShimLeakCheck in StressSuite.Run). +// 2. Host RSS growth: the shim RSS must not exceed +// stressSandboxMaxRSSGrowth bytes over the run duration. +// 3. Mount leak: no per-container mount points remain in the shim's +// mount namespace after ShutdownSandbox. +func (s *StressSuite) testSandbox(t *testing.T) { + if testing.Short() { + t.Skip("skipping sandbox stress in short mode") + } + + // Pre-build read-only rootfs images once. Reusing them across all + // iterations keeps disk consumption O(1) instead of O(iterations). + // Per-iteration setup only creates the small writable parts (ext4 + // scratch image or overlay upper/work dirs). + imgs := buildShimImages(t, s.cfg) + + // On non-root systems (where loop mounts are unavailable), pre-extract + // the rootfs erofs into a single directory that every iteration reuses + // as the bind-mount source. ShareRootfs copies it into the sandbox + // shared dir per container and Unshare removes it promptly, so disk + // consumption stays O(1) across the run. + var preExtractedRootfs string + if !s.cfg.FormatMounts || os.Getuid() != 0 { + preExtractedRootfs = t.TempDir() + extractErofsIntoDir(t, imgs.erofsImg, preExtractedRootfs) + } + + sandboxID := containerID(t) + env := startSandboxShim(t, s.cfg, sandboxID) + + // Bootstrap: create a probe container to seed the shimPID lookup + // and establish that the sandbox is functional before the loop. + probeCID := createContainerInSandbox(t, env, []string{"/bin/echo", "probe"}) + readContainerOutput(t, env, probeCID, "probe", 30*time.Second) + env.tc.Wait(env.ctx, &taskAPI.WaitRequest{ID: probeCID}) //nolint:errcheck + + shimPID := sandboxShimPID(env, probeCID) + releaseSandboxContainer(env.ctx, env, probeCID) //nolint:errcheck + t.Logf("sandbox stress: shim PID=%d", shimPID) + + // Sample RSS before the churn loop. + var rssBefore int64 + var rssOK bool + if shimPID != 0 { + var err error + rssBefore, err = readRSS(shimPID) + if err != nil { + t.Logf("cannot read pre-stress shim RSS (PID %d): %v — disabling RSS check", shimPID, err) + } else { + rssOK = true + } + } + + var iterIdx atomic.Int64 + ctx, cancel := stressCtx(t, env.ctx) + defer cancel() + + iters, elapsed, stressErr := runStress(ctx, func(iterCtx context.Context) error { + i := iterIdx.Add(1) + name := fmt.Sprintf("sbiter%05d", i) + + // Each iteration creates stressSandboxConcurrency containers using + // the pre-built images. Commands cycle: + // j%3 == 0: /bin/exit 0 (exit-code propagation) + // j%3 == 1: /bin/exit 0 + // j%3 == 2: /bin/exit 0 + // All containers exit cleanly; we just verify the exit status. + // Output is discarded (stdout/stderr FIFOs are drained silently) + // so the test exercises the create/start/wait/delete path under + // sustained load without accumulating output buffers. + type ctrInfo struct { + cid string + } + ctrs := make([]ctrInfo, stressSandboxConcurrency) + for j := range ctrs { + args := []string{"/bin/echo", fmt.Sprintf("%s-j%d", name, j)} + cid, err := createSandboxContainerFast(iterCtx, t, env, s.cfg, imgs, preExtractedRootfs, args) + if err != nil { + return fmt.Errorf("create %d: %w", j, err) + } + ctrs[j] = ctrInfo{cid: cid} + } + + // Wait for all containers concurrently. + var wg sync.WaitGroup + errs := make([]error, stressSandboxConcurrency) + for j, ci := range ctrs { + wg.Add(1) + go func(j int, ci ctrInfo) { + defer wg.Done() + subCtx, subCancel := context.WithTimeout(iterCtx, stressIterationTimeout) + defer subCancel() + + waitResp, err := env.tc.Wait(subCtx, &taskAPI.WaitRequest{ID: ci.cid}) + if err != nil { + errs[j] = fmt.Errorf("wait %s: %w", ci.cid, err) + return + } + if waitResp.GetExitStatus() != 0 { + errs[j] = fmt.Errorf("container %s exited with status %d", + ci.cid, waitResp.GetExitStatus()) + } + }(j, ci) + } + wg.Wait() + + for _, e := range errs { + if e != nil { + return e + } + } + + // Delete all containers and release tracking state immediately. + for _, ci := range ctrs { + if err := releaseSandboxContainer(iterCtx, env, ci.cid); err != nil { + return fmt.Errorf("delete %s: %w", ci.cid, err) + } + } + + return nil + }) + + rate := float64(iters) / elapsed.Seconds() + t.Logf("sandbox stress: %d iterations × %d containers = %d total containers in %s (%.1f iter/s)", + iters, stressSandboxConcurrency, iters*stressSandboxConcurrency, + elapsed.Round(time.Millisecond), rate) + + if stressErr != nil { + t.Fatalf("sandbox stress: %v", stressErr) + } + + // ── RSS growth check ────────────────────────────────────────────── + if rssOK && shimPID != 0 { + rssAfter, err := readRSS(shimPID) + if err != nil { + t.Logf("cannot read post-stress shim RSS: %v", err) + } else { + growth := rssAfter - rssBefore + threshold := int64(stressSandboxMaxRSSGrowth) + if s.options.SandboxRSSGrowthOverride > 0 { + threshold = s.options.SandboxRSSGrowthOverride + } + t.Logf("shim RSS: before=%d MiB after=%d MiB growth=%d MiB (threshold %d MiB)", + rssBefore>>20, rssAfter>>20, growth>>20, threshold>>20) + if growth > threshold { + t.Errorf("shim RSS grew %d bytes during sandbox stress (threshold %d bytes); "+ + "possible per-container memory leak", + growth, threshold) + } + } + } + + // ── Mount-leak check ───────────────────────────────────────────── + // Snapshot mounts before shutdown, then verify they are gone after. + mountsBefore := sandboxContainersMounts(shimPID) + t.Logf("per-container mounts before shutdown: %d", len(mountsBefore)) + + // Trigger shutdown (cleanup is also registered by startSandboxShim, + // but we drive it explicitly here so we can inspect state after). + env.sc.StopSandbox(env.ctx, &sandboxAPI.StopSandboxRequest{SandboxID: sandboxID}) //nolint:errcheck + env.sc.ShutdownSandbox(env.ctx, &sandboxAPI.ShutdownSandboxRequest{SandboxID: sandboxID}) //nolint:errcheck + + // Allow shim to finish cleanup. + time.Sleep(500 * time.Millisecond) + + mountsAfter := sandboxContainersMounts(shimPID) + if len(mountsAfter) > 0 { + t.Errorf("sandbox stress: %d per-container mount(s) leaked after ShutdownSandbox: %v", + len(mountsAfter), mountsAfter) + } else { + t.Log("no per-container mounts remain after shutdown (mount-leak check passed)") + } +} diff --git a/vendor/github.com/containerd/shimtest/stress_sandbox_other.go b/vendor/github.com/containerd/shimtest/stress_sandbox_other.go new file mode 100644 index 00000000..372d22e6 --- /dev/null +++ b/vendor/github.com/containerd/shimtest/stress_sandbox_other.go @@ -0,0 +1,26 @@ +//go:build !linux + +/* + Copyright The containerd Authors. + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. +*/ + +package shimtest + +import "testing" + +// testSandbox skips on non-Linux platforms. +func (s *StressSuite) testSandbox(t *testing.T) { + t.Skip("sandbox stress is Linux-only") +} diff --git a/vendor/github.com/containerd/shimtest/stress_suite.go b/vendor/github.com/containerd/shimtest/stress_suite.go index 44dcb6d5..b1286b70 100644 --- a/vendor/github.com/containerd/shimtest/stress_suite.go +++ b/vendor/github.com/containerd/shimtest/stress_suite.go @@ -64,12 +64,24 @@ type StressOptions struct { // test. The shim under test must implement the transfer service. Transfer bool + // Sandbox enables the sandbox container-churn stress test. The + // shim under test must implement the containerd sandbox shim API + // (runtime/sandbox/v1). + Sandbox bool + // ExecRSSGrowthOverride, when non-zero, replaces the platform default // RSS growth threshold for the exec stress test. Use this for shims // that host a VM or other large runtime in-process and therefore have // a higher expected one-time RSS step than a thin supervisor shim. // The value is in bytes. ExecRSSGrowthOverride int64 + + // SandboxRSSGrowthOverride, when non-zero, replaces the platform default + // RSS growth threshold for the sandbox stress test. VM-based shims + // have a large one-time RSS step from the VM boot that saturates + // early; set this to accommodate the expected baseline. + // The value is in bytes. + SandboxRSSGrowthOverride int64 } // NewStressSuite constructs a StressSuite from cfg and options. @@ -92,6 +104,9 @@ func (s *StressSuite) Run(t *testing.T) { if s.options.Transfer { t.Run("Transfer", s.testTransfer) } + if s.options.Sandbox { + t.Run("Sandbox", s.testSandbox) + } } // testLifecycle exercises the full create/start/run/kill/wait/delete diff --git a/vendor/github.com/containerd/shimtest/testbin/testbin.go b/vendor/github.com/containerd/shimtest/testbin/testbin.go index fcc1baee..42a797ff 100644 --- a/vendor/github.com/containerd/shimtest/testbin/testbin.go +++ b/vendor/github.com/containerd/shimtest/testbin/testbin.go @@ -28,9 +28,11 @@ package testbin import ( + "bytes" "fmt" "hash/crc32" "io" + "math/rand/v2" "net" "os" "os/signal" @@ -38,7 +40,9 @@ import ( "strconv" "strings" "sync" + "syscall" "time" + "unsafe" ) // Main is the entry point for the testbin multicall binary. It dispatches @@ -86,8 +90,24 @@ func Main() { cmdNC(args) case "host": cmdHost(args) + case "looptest": + cmdLooptest(args) + case "echosrv": + cmdEchoServer(args) case "tickexit": cmdTickexit(args) + case "pidscan": + cmdPidscan(args) + case "shmwrite": + cmdShmWrite(args) + case "shmread": + cmdShmRead(args) + case "shmmapwrite": + cmdShmMapWrite(args) + case "shmmapread": + cmdShmMapRead(args) + case "hostname": + cmdHostname(args) default: fmt.Fprintf(os.Stderr, "testbin: unknown command: %s\n", cmd) os.Exit(127) @@ -446,26 +466,68 @@ func cmdBurstexit(args []string) { os.Exit(exitCode) } -// cmdNC is a minimal netcat-compatible tool supporting three modes: +// cmdNC is a minimal netcat-compatible tool supporting four modes: // // nc -U connect to a unix domain socket +// nc [-v] -l listen on TCP (all IPv4 interfaces), accept one +// connection, then pipe stdio // nc connect via TCP // nc -u exchange a single UDP datagram // -// In all modes data flows verbatim between the network endpoint and stdio, -// matching the behaviour of the standard nc(1) utility: -// - TCP / unix: bidirectional io.Copy (stdin→socket, socket→stdout). -// - UDP: one unconnected sendto (stdin→remote) then one recvfrom -// (remote→stdout). The socket is unconnected (ListenPacket / WriteTo / -// ReadFrom, i.e. sendto/recvfrom) so that shim networking layers cannot -// short-circuit routing based on the local connect(2) call, which for UDP -// always succeeds regardless of whether any peer is listening. +// In stream modes (TCP / unix) data flows verbatim between the network +// endpoint and stdio, matching the behaviour of the standard nc(1) utility: +// stdout carries connection payload and nothing else. +// +// Listen mode (nc -l): binds tcp4 0.0.0.0: (port 0 = ephemeral), then +// accepts one connection and enters the same bidirectional stdio↔socket pipe +// as connect mode. Listening is restricted to IPv4 (unlike connect mode, +// which is dual-stack) to keep the address a peer must dial unambiguous. +// +// As with standard nc, the listening socket is reported only when -v is +// given, and then on stderr -- never on stdout, which would corrupt the +// payload stream. The line format matches nc -v exactly: +// +// Listening on 0.0.0.0 39117 +// +// It is written before the accept(2) call, so a caller that waits for it can +// use it both to learn an ephemeral port and as a readiness signal. Note +// that -v is honoured for listen mode only; standard nc is also verbose +// about outgoing connections, which is not implemented here. +// +// nc's stream modes are a symmetric bidirectional pipe that only terminates +// once *both* copy directions are done: the peer has closed the connection +// *and* stdin has reached EOF. This matches standard nc, which likewise +// keeps running after a peer close so that a half-closed connection can +// still be written to. A caller driving nc through a shim must therefore +// issue Task.CloseIO to signal stdin EOF; closing its own write end of the +// stdin FIFO is not sufficient. For "one container listens, another +// connects" scenarios where neither end is a host-side Go program able to +// explicitly Close() once done, see echosrv, a purpose-built one-shot +// responder that always terminates on its own. +// +// UDP mode: one unconnected sendto (stdin→remote) then one recvfrom +// (remote→stdout). The socket is unconnected (ListenPacket / WriteTo / +// ReadFrom, i.e. sendto/recvfrom) so that shim networking layers cannot +// short-circuit routing based on the local connect(2) call, which for UDP +// always succeeds regardless of whether any peer is listening. func cmdNC(args []string) { if len(args) < 2 { - fmt.Fprintln(os.Stderr, "usage: nc [-u] | nc -U ") + fmt.Fprintln(os.Stderr, "usage: nc [-u] | nc [-v] -l | nc -U ") os.Exit(1) } + // Standard nc is silent unless -v is given. Strip it here so the mode + // dispatch below sees the same argument shape either way. + verbose := false + if args[1] == "-v" { + verbose = true + args = append([]string{args[0]}, args[2:]...) + if len(args) < 2 { + fmt.Fprintln(os.Stderr, "usage: nc [-v] -l ") + os.Exit(1) + } + } + switch args[1] { case "-U": // Unix domain socket mode. @@ -481,6 +543,41 @@ func cmdNC(args []string) { defer conn.Close() ncStream(conn) + case "-l": + // Listen mode: nc [-v] -l + // Binds tcp4 0.0.0.0: (0 = ephemeral), accepts one connection, + // then bidirectionally pipes stdio↔socket. See the function comment + // for the -v listen notice and for the termination conditions. + if len(args) < 3 { + fmt.Fprintln(os.Stderr, "usage: nc [-v] -l ") + os.Exit(1) + } + ln, err := net.Listen("tcp4", "0.0.0.0:"+args[2]) + if err != nil { + fmt.Fprintf(os.Stderr, "nc: listen 0.0.0.0:%s: %v\n", args[2], err) + os.Exit(1) + } + if verbose { + // Report the socket actually bound, so a caller need not assume + // the kernel honoured the requested port. os.Stderr is + // unbuffered, so this reaches the caller before the accept(2) + // below and is therefore usable as a readiness signal. + host, boundPort, err := net.SplitHostPort(ln.Addr().String()) + if err != nil { + fmt.Fprintf(os.Stderr, "nc: splithost: %v\n", err) + os.Exit(1) + } + fmt.Fprintf(os.Stderr, "Listening on %s %s\n", host, boundPort) + } + conn, err := ln.Accept() + ln.Close() + if err != nil { + fmt.Fprintf(os.Stderr, "nc: accept: %v\n", err) + os.Exit(1) + } + defer conn.Close() + ncStream(conn) + case "-u": // UDP datagram mode. if len(args) < 4 { @@ -508,6 +605,14 @@ func cmdNC(args []string) { fmt.Fprintf(os.Stderr, "nc: udp send: %v\n", err) os.Exit(1) } + // A deadline here is deliberately not a retry: the caller (see + // attachContainerNetwork) is responsible for the container's + // network being fully ready before this process ever runs, so a + // reply that doesn't show up within the deadline is a real + // failure, not something to wait out. Its only job is to turn a + // missing reply into a fast, legible error instead of blocking + // forever. + pc.SetReadDeadline(time.Now().Add(10 * time.Second)) buf := make([]byte, 65536) n, _, err := pc.ReadFrom(buf) if err != nil { @@ -578,3 +683,455 @@ func cmdHost(args []string) { fmt.Printf("%s has address %s\n", name, a) } } + +// looptestPortRangeStart and looptestPortRangeEnd bound the candidates +// cmdLooptest picks its listener's port from, instead of asking the guest +// kernel for an ephemeral one (port 0). Some shims proxy each socket call +// to the host independently of the others rather than truly sharing a +// single network stack; under such a shim, a port-0 bind lets each side +// independently pick "an ephemeral port," with no guarantee the two agree +// -- the listener and the connector could each resolve to a different +// number and never actually rendezvous. Binding a concrete port removes +// that ambiguity: both ends of the same in-process test necessarily agree +// on the number, because there is only one port to have picked. The range +// is deliberately outside Linux's default ephemeral range (typically +// 32768-60999), to reduce the chance of colliding with an unrelated +// connection's OS-assigned source port. A short retry loop, rather than a +// single fixed port, absorbs the (small, and this being an in-container +// listener, more theoretical than practical) chance of a collision with +// another process already using a given candidate. +const ( + looptestPortRangeStart = 20000 + looptestPortRangeEnd = 29999 + looptestBindAttempts = 20 +) + +// cmdLooptest verifies in-container loopback connectivity by starting an +// echo listener inside the same process, connecting to it over +// 127.0.0.1, sending a token, and printing the echo to stdout. +// +// It is a self-contained in-process test that does not fork subprocesses: it +// implements a minimal echo server directly, running it on a goroutine. This +// avoids exec dependencies on the container's filesystem while keeping the +// test agnostic to PID-namespace configuration. +// +// Usage: looptest +// +// Exits 0 and prints the echoed token on success. Exits 1 with a diagnostic +// on stderr if the listener, connection, or echo fails. +func cmdLooptest(args []string) { + if len(args) < 2 { + fmt.Fprintln(os.Stderr, "usage: looptest ") + os.Exit(1) + } + token := args[1] + + // See looptestPortRangeStart's doc for why this binds a concrete port + // from a fixed range, with a short retry loop, rather than port 0. + var ln net.Listener + var lastErr error + for range looptestBindAttempts { + port := looptestPortRangeStart + rand.IntN(looptestPortRangeEnd-looptestPortRangeStart+1) + l, err := net.Listen("tcp4", net.JoinHostPort("127.0.0.1", strconv.Itoa(port))) + if err != nil { + lastErr = err + continue + } + ln = l + break + } + if ln == nil { + fmt.Fprintf(os.Stderr, "looptest: listen: %v\n", lastErr) + os.Exit(1) + } + _, boundPort, err := net.SplitHostPort(ln.Addr().String()) + if err != nil { + fmt.Fprintf(os.Stderr, "looptest: splithost: %v\n", err) + os.Exit(1) + } + + // Run the echo server on a goroutine. + srvDone := make(chan error, 1) + go func() { + conn, err := ln.Accept() + ln.Close() + if err != nil { + srvDone <- fmt.Errorf("accept: %w", err) + return + } + defer conn.Close() + buf := make([]byte, 4096) + n, rerr := conn.Read(buf) + if n == 0 && rerr != nil { + srvDone <- fmt.Errorf("read: %w", rerr) + return + } + if _, err := conn.Write(buf[:n]); err != nil { + srvDone <- fmt.Errorf("write: %w", err) + return + } + srvDone <- nil + }() + + // Dial back over loopback. + conn, err := net.DialTimeout("tcp", net.JoinHostPort("127.0.0.1", boundPort), 10*time.Second) + if err != nil { + fmt.Fprintf(os.Stderr, "looptest: dial 127.0.0.1:%s: %v\n", boundPort, err) + os.Exit(1) + } + conn.SetDeadline(time.Now().Add(10 * time.Second)) + + if _, err := conn.Write([]byte(token)); err != nil { + fmt.Fprintf(os.Stderr, "looptest: write: %v\n", err) + os.Exit(1) + } + + buf := make([]byte, 4096) + n, rerr := conn.Read(buf) + conn.Close() + if n == 0 && rerr != nil { + fmt.Fprintf(os.Stderr, "looptest: read echo: %v\n", rerr) + os.Exit(1) + } + got := string(buf[:n]) + if got != token { + fmt.Fprintf(os.Stderr, "looptest: echo mismatch: got %q, want %q\n", got, token) + os.Exit(1) + } + fmt.Println(got) + + if err := <-srvDone; err != nil { + fmt.Fprintf(os.Stderr, "looptest: server: %v\n", err) + os.Exit(1) + } +} + +// cmdEchoServer listens on TCP port (all interfaces), accepts exactly +// one connection, reads exactly one chunk of data (up to 4096 bytes), writes +// the same bytes back verbatim, closes the connection, and exits 0. +// +// Unlike "nc -l", which is a general-purpose bidirectional stream pipe (and +// so never closes the connection on its own — the peer must close it), +// echosrv is purpose-built as a one-shot round-trip responder: it always +// terminates on its own once one exchange completes, which is what makes it +// usable as a container's main process in a test that needs the container to +// exit cleanly after proving connectivity (e.g. two containers exchanging +// data over a shared network namespace, where neither side is a host-side Go +// program that can explicitly Close() to signal completion). +// +// Usage: echosrv +func cmdEchoServer(args []string) { + if len(args) < 2 { + fmt.Fprintln(os.Stderr, "usage: echosrv ") + os.Exit(1) + } + // tcp4/0.0.0.0 explicitly, not "tcp"/":" (which defaults to a + // dual-stack IPv6 socket on Linux): shimtest does not assume a shim's + // default container networking path supports IPv6, only IPv4. + ln, err := net.Listen("tcp4", "0.0.0.0:"+args[1]) + if err != nil { + fmt.Fprintf(os.Stderr, "echosrv: listen 0.0.0.0:%s: %v\n", args[1], err) + os.Exit(1) + } + // Always print the actual bound port before accepting, so the test can + // discover it even when port 0 (ephemeral) was requested. + _, boundPort, err := net.SplitHostPort(ln.Addr().String()) + if err != nil { + fmt.Fprintf(os.Stderr, "echosrv: splithost: %v\n", err) + os.Exit(1) + } + fmt.Println(boundPort) + conn, err := ln.Accept() + ln.Close() + if err != nil { + fmt.Fprintf(os.Stderr, "echosrv: accept: %v\n", err) + os.Exit(1) + } + defer conn.Close() + + buf := make([]byte, 4096) + n, err := conn.Read(buf) + if n == 0 && err != nil { + fmt.Fprintf(os.Stderr, "echosrv: read: %v\n", err) + os.Exit(1) + } + if _, err := conn.Write(buf[:n]); err != nil { + fmt.Fprintf(os.Stderr, "echosrv: write: %v\n", err) + os.Exit(1) + } +} + +// cmdPidscan lists every PID visible in this process's PID namespace +// along with its cmdline, by scanning /proc. Used by shimtest to verify +// PID namespace sharing across member containers: the test does not +// know the PID number of the sentinel process it is looking for ahead +// of time (only a unique marker string baked into that process's +// argv), so it scans every visible PID's cmdline rather than checking +// one specific PID. +func cmdPidscan(_ []string) { + entries, err := os.ReadDir("/proc") + if err != nil { + fmt.Fprintf(os.Stderr, "pidscan: readdir /proc: %v\n", err) + os.Exit(1) + } + for _, e := range entries { + name := e.Name() + if _, err := strconv.Atoi(name); err != nil { + continue // not a PID directory + } + data, err := os.ReadFile(filepath.Join("/proc", name, "cmdline")) + if err != nil { + // The process may have exited between the readdir and this + // read; that race is expected and not an error. + continue + } + cmdline := strings.ReplaceAll(strings.TrimRight(string(data), "\x00"), "\x00", " ") + fmt.Printf("%s %s\n", name, cmdline) + } +} + +const ( + shmSize = 4096 + // ipcCreat is IPC_CREAT, from linux/ipc.h. The stdlib syscall + // package exposes SysV shm's syscall numbers (SYS_SHMGET etc.) but + // not its flag constants, so this is hardcoded. + ipcCreat = 0o1000 +) + +// cmdShmWrite creates (or reuses) a SysV shared memory segment +// identified by a fixed numeric key and writes a marker string into it, +// then detaches — but does not remove — the segment, leaving it behind +// for a later shmread call to find. +// +// Used by shimtest to verify IPC namespace sharing: SysV IPC objects +// are keyed and visible only within the creating process's IPC +// namespace, independent of mount namespace or any bind-mounted +// /dev/shm, so a successful cross-container shmwrite/shmread round +// trip through the same key is conclusive proof of a shared IPC +// namespace (and not, for instance, an artifact of a shared /dev/shm +// bind mount). +// +// Usage: shmwrite +func cmdShmWrite(args []string) { + if len(args) < 3 { + fmt.Fprintln(os.Stderr, "usage: shmwrite ") + os.Exit(1) + } + key, err := strconv.ParseInt(args[1], 10, 64) + if err != nil { + fmt.Fprintf(os.Stderr, "shmwrite: invalid key %q: %v\n", args[1], err) + os.Exit(1) + } + marker := args[2] + if len(marker) >= shmSize { + fmt.Fprintln(os.Stderr, "shmwrite: marker too large") + os.Exit(1) + } + + shmid, _, errno := syscall.Syscall(syscall.SYS_SHMGET, uintptr(key), shmSize, ipcCreat|0600) + if errno != 0 { + fmt.Fprintf(os.Stderr, "shmwrite: shmget: %v\n", errno) + os.Exit(1) + } + addr, _, errno := syscall.Syscall(syscall.SYS_SHMAT, shmid, 0, 0) + if errno != 0 { + fmt.Fprintf(os.Stderr, "shmwrite: shmat: %v\n", errno) + os.Exit(1) + } + // addr is a raw address returned by the shmat(2) syscall, not derived + // from a Go pointer, so it doesn't fit vet's recognized safe-conversion + // patterns even though the conversion itself is valid here. + buf := (*[shmSize]byte)(unsafe.Pointer(addr)) //nolint:govet + n := copy(buf[:], marker) + buf[n] = 0 + syscall.Syscall(syscall.SYS_SHMDT, addr, 0, 0) //nolint:errcheck + + fmt.Println("shmwrite: ok") +} + +// cmdShmRead attaches to an existing SysV shared memory segment +// identified by a fixed numeric key (created by a prior shmwrite call, +// possibly in a different container) and prints the marker string +// found in it. +// +// It deliberately omits IPC_CREAT: if the segment does not already +// exist in this process's IPC namespace, that is exactly the "not +// shared" case and must be reported as a failure, rather than silently +// creating a fresh, empty segment that would make a broken test look +// like it passed. +// +// Usage: shmread +func cmdShmRead(args []string) { + if len(args) < 2 { + fmt.Fprintln(os.Stderr, "usage: shmread ") + os.Exit(1) + } + key, err := strconv.ParseInt(args[1], 10, 64) + if err != nil { + fmt.Fprintf(os.Stderr, "shmread: invalid key %q: %v\n", args[1], err) + os.Exit(1) + } + + shmid, _, errno := syscall.Syscall(syscall.SYS_SHMGET, uintptr(key), shmSize, 0600) + if errno != 0 { + fmt.Println("shmread: NOTFOUND") + os.Exit(1) + } + addr, _, errno := syscall.Syscall(syscall.SYS_SHMAT, shmid, 0, 0) + if errno != 0 { + fmt.Fprintf(os.Stderr, "shmread: shmat: %v\n", errno) + os.Exit(1) + } + // See the matching comment in cmdShmWrite: addr comes from shmat(2), + // not from a Go pointer, so vet can't recognize this as a safe + // conversion even though it is one. + buf := (*[shmSize]byte)(unsafe.Pointer(addr)) //nolint:govet + end := bytes.IndexByte(buf[:], 0) + if end < 0 { + end = shmSize + } + fmt.Println(string(buf[:end])) + syscall.Syscall(syscall.SYS_SHMDT, addr, 0, 0) //nolint:errcheck +} + +// shmMapSize is the file/mapping size used by cmdShmMapWrite and +// cmdShmMapRead. Independent of shmSize (the SysV segment size above): +// these test a different sharing mechanism and there's no reason to couple +// their sizes. +const shmMapSize = 4096 + +// cmdShmMapWrite creates (or truncates) the file at path to shmMapSize and +// writes marker into it through an mmap(MAP_SHARED) mapping — not via a +// write(2) call — then unmaps and exits, leaving the file (and, since the +// mapping is MAP_SHARED, its written contents) behind for a later +// shmmapread call to find. +// +// This, together with shmmapread, is a POSIX-shared-memory-style access +// pattern layered on an ordinary file (as, for example, glibc's +// shm_open()+mmap() is): the file's path is what a caller controls to +// target /dev/shm specifically or any other shared location, but the +// read/write path deliberately goes through the mapping, not the file +// descriptor, since what's under test is whether two independent +// mmap(MAP_SHARED) calls on the same underlying file — potentially made by +// processes in different containers, each reaching the file through its +// own bind mount of a shared directory — actually share memory, rather +// than each seeing an independent, disconnected copy. +// +// Usage: shmmapwrite +func cmdShmMapWrite(args []string) { + if len(args) < 3 { + fmt.Fprintln(os.Stderr, "usage: shmmapwrite ") + os.Exit(1) + } + path := args[1] + marker := args[2] + if len(marker) >= shmMapSize { + fmt.Println("shmmapwrite: marker too large") + os.Exit(1) + } + + f, err := os.OpenFile(path, os.O_RDWR|os.O_CREATE, 0o644) + if err != nil { + fmt.Printf("shmmapwrite: open: %v\n", err) + os.Exit(1) + } + defer f.Close() + if err := f.Truncate(shmMapSize); err != nil { + fmt.Printf("shmmapwrite: truncate: %v\n", err) + os.Exit(1) + } + + data, err := syscall.Mmap(int(f.Fd()), 0, shmMapSize, syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_SHARED) + if err != nil { + fmt.Printf("shmmapwrite: mmap: %v\n", err) + os.Exit(1) + } + + n := copy(data, marker) + data[n] = 0 + + if err := syscall.Munmap(data); err != nil { + fmt.Printf("shmmapwrite: munmap: %v\n", err) + os.Exit(1) + } + + fmt.Println("shmmapwrite: ok") +} + +// cmdShmMapRead opens the file at path (created by a prior shmmapwrite +// call, possibly in a different container) and reads the marker string +// back through an mmap(MAP_SHARED) mapping — not via a read(2) call. See +// cmdShmMapWrite for why the access pattern matters. +// +// It deliberately does not create path if missing: a missing file is +// exactly the "not shared" case and must be reported as a failure, rather +// than silently creating a fresh, empty one that would make a broken test +// look like it passed. +// +// Usage: shmmapread +func cmdShmMapRead(args []string) { + if len(args) < 2 { + fmt.Fprintln(os.Stderr, "usage: shmmapread ") + os.Exit(1) + } + path := args[1] + + f, err := os.OpenFile(path, os.O_RDONLY, 0) + if err != nil { + fmt.Println("shmmapread: NOTFOUND") + os.Exit(1) + } + defer f.Close() + + data, err := syscall.Mmap(int(f.Fd()), 0, shmMapSize, syscall.PROT_READ, syscall.MAP_SHARED) + if err != nil { + fmt.Fprintf(os.Stderr, "shmmapread: mmap: %v\n", err) + os.Exit(1) + } + defer syscall.Munmap(data) //nolint:errcheck + + end := bytes.IndexByte(data, 0) + if end < 0 { + end = shmMapSize + } + fmt.Println(string(data[:end])) +} + +// cmdHostname mirrors the standard "hostname" utility's CLI: with no +// argument it prints the calling process's UTS namespace hostname as +// reported by the kernel (via gethostname(2), not /etc/hostname or an +// env var); with one argument it sets the hostname (via sethostname(2)) +// and, matching the standard utility, exits immediately and silently on +// success rather than staying running or printing anything. +// +// sethostname(2) requires CAP_SYS_ADMIN in the user namespace that owns +// the target UTS namespace; the container's OCI spec must request that +// capability explicitly (shimtest's base spec grants none) for a set to +// succeed at all. On failure (of either form) a "hostname: ..." message +// is printed to stderr and the process exits non-zero, matching the +// standard utility's error convention. +// +// Because this command exits immediately after a successful set rather +// than holding the UTS namespace open itself, a caller that verifies a +// hostname change is later visible to a different process is also +// proving that the namespace — and its hostname — outlives the process +// that set it, not merely that a still-running setter's own namespace +// is visible. +// +// Usage: hostname [name] +func cmdHostname(args []string) { + if len(args) < 2 { + name, err := os.Hostname() + if err != nil { + fmt.Fprintf(os.Stderr, "hostname: %v\n", err) + os.Exit(1) + } + fmt.Println(name) + return + } + + if err := syscall.Sethostname([]byte(args[1])); err != nil { + fmt.Fprintf(os.Stderr, "hostname: %v\n", err) + os.Exit(1) + } +} diff --git a/vendor/modules.txt b/vendor/modules.txt index 5d13523b..a5f39316 100644 --- a/vendor/modules.txt +++ b/vendor/modules.txt @@ -67,6 +67,7 @@ github.com/containerd/console ## explicit; go 1.24.0 github.com/containerd/containerd/api/events github.com/containerd/containerd/api/runtime/bootstrap/v1 +github.com/containerd/containerd/api/runtime/sandbox/v1 github.com/containerd/containerd/api/runtime/task/v2 github.com/containerd/containerd/api/runtime/task/v3 github.com/containerd/containerd/api/services/streaming/v1 @@ -146,7 +147,7 @@ github.com/containerd/platforms ## explicit; go 1.22 github.com/containerd/plugin github.com/containerd/plugin/registry -# github.com/containerd/shimtest v0.3.3 +# github.com/containerd/shimtest v0.3.4-0.20260820001033-a0143efafccb ## explicit; go 1.26.3 github.com/containerd/shimtest github.com/containerd/shimtest/internal/transfer