mirror of
https://github.com/izzy2lost/xemu.git
synced 2026-07-06 00:20:22 -07:00
Merge remote-tracking branch 'remotes/mst/tags/for_upstream' into staging
pci, pc, virtio: features, fixes, cleanups intel-iommu scalable option pcie acs emulation beginning for vhost-user-blk reconnect and of vhost-user backend work misc fixes and cleanups Signed-off-by: Michael S. Tsirkin <mst@redhat.com> # gpg: Signature made Wed 13 Mar 2019 02:52:02 GMT # gpg: using RSA key 281F0DB8D28D5469 # gpg: Good signature from "Michael S. Tsirkin <mst@kernel.org>" [full] # gpg: aka "Michael S. Tsirkin <mst@redhat.com>" [full] # Primary key fingerprint: 0270 606B 6F3C DF3D 0B17 0970 C350 3912 AFBE 8E67 # Subkey fingerprint: 5D09 FD08 71C8 F85B 94CA 8A0D 281F 0DB8 D28D 5469 * remotes/mst/tags/for_upstream: (26 commits) i386, acpi: check acpi_memory_hotplug capacity in pre_plug gen_pcie_root_port: Add ACS (Access Control Services) capability pcie: Add a simple PCIe ACS (Access Control Services) helper function vhost-user-blk: Add support to get/set inflight buffer libvhost-user: Support tracking inflight I/O in shared memory libvhost-user: Introduce vu_queue_map_desc() libvhost-user: Remove unnecessary FD flag check for event file descriptors vhost-user: Support transferring inflight buffer between qemu and backend nvdimm: use NVDIMM_ACPI_IO_LEN for the proper IO size nvdimm: use *function* directly instead of allocating it again nvdimm: fix typo in nvdimm_build_nvdimm_devices argument intel_iommu: add scalable-mode option to make scalable mode work intel_iommu: add 256 bits qi_desc support intel_iommu: scalable mode emulation libvhost-user: add vu_queue_unpop() libvhost-user-glib: export vug_source_new() vhost-user: split vhost_user_read() vhost-user: wrap some read/write with retry handling libvhost-user: exit by default on VHOST_USER_NONE vhost-user: simplify vhost_user_init/vhost_user_cleanup ... Signed-off-by: Peter Maydell <peter.maydell@linaro.org>
This commit is contained in:
@@ -1455,6 +1455,7 @@ vhost
|
||||
M: Michael S. Tsirkin <mst@redhat.com>
|
||||
S: Supported
|
||||
F: hw/*/*vhost*
|
||||
F: docs/interop/vhost-user.json
|
||||
F: docs/interop/vhost-user.txt
|
||||
F: contrib/vhost-user-*/
|
||||
|
||||
|
||||
@@ -497,7 +497,7 @@ Makefile: $(version-obj-y)
|
||||
# Build libraries
|
||||
|
||||
libqemuutil.a: $(util-obj-y) $(trace-obj-y) $(stub-obj-y)
|
||||
libvhost-user.a: $(libvhost-user-obj-y)
|
||||
libvhost-user.a: $(libvhost-user-obj-y) $(util-obj-y) $(stub-obj-y)
|
||||
|
||||
######################################################################
|
||||
|
||||
|
||||
@@ -47,7 +47,7 @@
|
||||
typedef struct CryptoDevBackendVhostUser {
|
||||
CryptoDevBackend parent_obj;
|
||||
|
||||
VhostUserState *vhost_user;
|
||||
VhostUserState vhost_user;
|
||||
CharBackend chr;
|
||||
char *chr_name;
|
||||
bool opened;
|
||||
@@ -104,7 +104,7 @@ cryptodev_vhost_user_start(int queues,
|
||||
continue;
|
||||
}
|
||||
|
||||
options.opaque = s->vhost_user;
|
||||
options.opaque = &s->vhost_user;
|
||||
options.backend_type = VHOST_BACKEND_TYPE_USER;
|
||||
options.cc = b->conf.peers.ccs[i];
|
||||
s->vhost_crypto[i] = cryptodev_vhost_init(&options);
|
||||
@@ -182,7 +182,6 @@ static void cryptodev_vhost_user_init(
|
||||
size_t i;
|
||||
Error *local_err = NULL;
|
||||
Chardev *chr;
|
||||
VhostUserState *user;
|
||||
CryptoDevBackendClient *cc;
|
||||
CryptoDevBackendVhostUser *s =
|
||||
CRYPTODEV_BACKEND_VHOST_USER(backend);
|
||||
@@ -213,15 +212,10 @@ static void cryptodev_vhost_user_init(
|
||||
}
|
||||
}
|
||||
|
||||
user = vhost_user_init();
|
||||
if (!user) {
|
||||
error_setg(errp, "Failed to init vhost_user");
|
||||
if (!vhost_user_init(&s->vhost_user, &s->chr, errp)) {
|
||||
return;
|
||||
}
|
||||
|
||||
user->chr = &s->chr;
|
||||
s->vhost_user = user;
|
||||
|
||||
qemu_chr_fe_set_handlers(&s->chr, NULL, NULL,
|
||||
cryptodev_vhost_user_event, NULL, s, NULL, true);
|
||||
|
||||
@@ -307,11 +301,7 @@ static void cryptodev_vhost_user_cleanup(
|
||||
}
|
||||
}
|
||||
|
||||
if (s->vhost_user) {
|
||||
vhost_user_cleanup(s->vhost_user);
|
||||
g_free(s->vhost_user);
|
||||
s->vhost_user = NULL;
|
||||
}
|
||||
vhost_user_cleanup(&s->vhost_user);
|
||||
}
|
||||
|
||||
static void cryptodev_vhost_user_set_chardev(Object *obj,
|
||||
|
||||
@@ -68,15 +68,16 @@ static GSourceFuncs vug_src_funcs = {
|
||||
NULL
|
||||
};
|
||||
|
||||
static GSource *
|
||||
vug_source_new(VuDev *dev, int fd, GIOCondition cond,
|
||||
GSource *
|
||||
vug_source_new(VugDev *gdev, int fd, GIOCondition cond,
|
||||
vu_watch_cb vu_cb, gpointer data)
|
||||
{
|
||||
VuDev *dev = &gdev->parent;
|
||||
GSource *gsrc;
|
||||
VugSrc *src;
|
||||
guint id;
|
||||
|
||||
g_assert(dev);
|
||||
g_assert(gdev);
|
||||
g_assert(fd >= 0);
|
||||
g_assert(vu_cb);
|
||||
|
||||
@@ -106,7 +107,7 @@ set_watch(VuDev *vu_dev, int fd, int vu_evt, vu_watch_cb cb, void *pvt)
|
||||
g_assert(cb);
|
||||
|
||||
dev = container_of(vu_dev, VugDev, parent);
|
||||
src = vug_source_new(vu_dev, fd, vu_evt, cb, pvt);
|
||||
src = vug_source_new(dev, fd, vu_evt, cb, pvt);
|
||||
g_hash_table_replace(dev->fdmap, GINT_TO_POINTER(fd), src);
|
||||
}
|
||||
|
||||
@@ -141,7 +142,7 @@ vug_init(VugDev *dev, int socket,
|
||||
dev->fdmap = g_hash_table_new_full(NULL, NULL, NULL,
|
||||
(GDestroyNotify) g_source_destroy);
|
||||
|
||||
dev->src = vug_source_new(&dev->parent, socket, G_IO_IN, vug_watch, NULL);
|
||||
dev->src = vug_source_new(dev, socket, G_IO_IN, vug_watch, NULL);
|
||||
}
|
||||
|
||||
void
|
||||
|
||||
@@ -29,4 +29,7 @@ void vug_init(VugDev *dev, int socket,
|
||||
vu_panic_cb panic, const VuDevIface *iface);
|
||||
void vug_deinit(VugDev *dev);
|
||||
|
||||
GSource *vug_source_new(VugDev *dev, int fd, GIOCondition cond,
|
||||
vu_watch_cb vu_cb, gpointer data);
|
||||
|
||||
#endif /* LIBVHOST_USER_GLIB_H */
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -53,6 +53,7 @@ enum VhostUserProtocolFeature {
|
||||
VHOST_USER_PROTOCOL_F_CONFIG = 9,
|
||||
VHOST_USER_PROTOCOL_F_SLAVE_SEND_FD = 10,
|
||||
VHOST_USER_PROTOCOL_F_HOST_NOTIFIER = 11,
|
||||
VHOST_USER_PROTOCOL_F_INFLIGHT_SHMFD = 12,
|
||||
|
||||
VHOST_USER_PROTOCOL_F_MAX
|
||||
};
|
||||
@@ -91,6 +92,8 @@ typedef enum VhostUserRequest {
|
||||
VHOST_USER_POSTCOPY_ADVISE = 28,
|
||||
VHOST_USER_POSTCOPY_LISTEN = 29,
|
||||
VHOST_USER_POSTCOPY_END = 30,
|
||||
VHOST_USER_GET_INFLIGHT_FD = 31,
|
||||
VHOST_USER_SET_INFLIGHT_FD = 32,
|
||||
VHOST_USER_MAX
|
||||
} VhostUserRequest;
|
||||
|
||||
@@ -138,6 +141,13 @@ typedef struct VhostUserVringArea {
|
||||
uint64_t offset;
|
||||
} VhostUserVringArea;
|
||||
|
||||
typedef struct VhostUserInflight {
|
||||
uint64_t mmap_size;
|
||||
uint64_t mmap_offset;
|
||||
uint16_t num_queues;
|
||||
uint16_t queue_size;
|
||||
} VhostUserInflight;
|
||||
|
||||
#if defined(_WIN32)
|
||||
# define VU_PACKED __attribute__((gcc_struct, packed))
|
||||
#else
|
||||
@@ -145,7 +155,7 @@ typedef struct VhostUserVringArea {
|
||||
#endif
|
||||
|
||||
typedef struct VhostUserMsg {
|
||||
VhostUserRequest request;
|
||||
int request;
|
||||
|
||||
#define VHOST_USER_VERSION_MASK (0x3)
|
||||
#define VHOST_USER_REPLY_MASK (0x1 << 2)
|
||||
@@ -163,6 +173,7 @@ typedef struct VhostUserMsg {
|
||||
VhostUserLog log;
|
||||
VhostUserConfig config;
|
||||
VhostUserVringArea area;
|
||||
VhostUserInflight inflight;
|
||||
} payload;
|
||||
|
||||
int fds[VHOST_MEMORY_MAX_NREGIONS];
|
||||
@@ -234,9 +245,61 @@ typedef struct VuRing {
|
||||
uint32_t flags;
|
||||
} VuRing;
|
||||
|
||||
typedef struct VuDescStateSplit {
|
||||
/* Indicate whether this descriptor is inflight or not.
|
||||
* Only available for head-descriptor. */
|
||||
uint8_t inflight;
|
||||
|
||||
/* Padding */
|
||||
uint8_t padding[5];
|
||||
|
||||
/* Maintain a list for the last batch of used descriptors.
|
||||
* Only available when batching is used for submitting */
|
||||
uint16_t next;
|
||||
|
||||
/* Used to preserve the order of fetching available descriptors.
|
||||
* Only available for head-descriptor. */
|
||||
uint64_t counter;
|
||||
} VuDescStateSplit;
|
||||
|
||||
typedef struct VuVirtqInflight {
|
||||
/* The feature flags of this region. Now it's initialized to 0. */
|
||||
uint64_t features;
|
||||
|
||||
/* The version of this region. It's 1 currently.
|
||||
* Zero value indicates a vm reset happened. */
|
||||
uint16_t version;
|
||||
|
||||
/* The size of VuDescStateSplit array. It's equal to the virtqueue
|
||||
* size. Slave could get it from queue size field of VhostUserInflight. */
|
||||
uint16_t desc_num;
|
||||
|
||||
/* The head of list that track the last batch of used descriptors. */
|
||||
uint16_t last_batch_head;
|
||||
|
||||
/* Storing the idx value of used ring */
|
||||
uint16_t used_idx;
|
||||
|
||||
/* Used to track the state of each descriptor in descriptor table */
|
||||
VuDescStateSplit desc[0];
|
||||
} VuVirtqInflight;
|
||||
|
||||
typedef struct VuVirtqInflightDesc {
|
||||
uint16_t index;
|
||||
uint64_t counter;
|
||||
} VuVirtqInflightDesc;
|
||||
|
||||
typedef struct VuVirtq {
|
||||
VuRing vring;
|
||||
|
||||
VuVirtqInflight *inflight;
|
||||
|
||||
VuVirtqInflightDesc *resubmit_list;
|
||||
|
||||
uint16_t resubmit_num;
|
||||
|
||||
uint64_t counter;
|
||||
|
||||
/* Next head to pop */
|
||||
uint16_t last_avail_idx;
|
||||
|
||||
@@ -279,11 +342,18 @@ typedef void (*vu_set_watch_cb) (VuDev *dev, int fd, int condition,
|
||||
vu_watch_cb cb, void *data);
|
||||
typedef void (*vu_remove_watch_cb) (VuDev *dev, int fd);
|
||||
|
||||
typedef struct VuDevInflightInfo {
|
||||
int fd;
|
||||
void *addr;
|
||||
uint64_t size;
|
||||
} VuDevInflightInfo;
|
||||
|
||||
struct VuDev {
|
||||
int sock;
|
||||
uint32_t nregions;
|
||||
VuDevRegion regions[VHOST_MEMORY_MAX_NREGIONS];
|
||||
VuVirtq vq[VHOST_MAX_NR_VIRTQUEUE];
|
||||
VuDevInflightInfo inflight_info;
|
||||
int log_call_fd;
|
||||
int slave_fd;
|
||||
uint64_t log_size;
|
||||
@@ -458,6 +528,20 @@ void vu_queue_notify(VuDev *dev, VuVirtq *vq);
|
||||
*/
|
||||
void *vu_queue_pop(VuDev *dev, VuVirtq *vq, size_t sz);
|
||||
|
||||
|
||||
/**
|
||||
* vu_queue_unpop:
|
||||
* @dev: a VuDev context
|
||||
* @vq: a VuVirtq queue
|
||||
* @elem: The #VuVirtqElement
|
||||
* @len: number of bytes written
|
||||
*
|
||||
* Pretend the most recent element wasn't popped from the virtqueue. The next
|
||||
* call to vu_queue_pop() will refetch the element.
|
||||
*/
|
||||
void vu_queue_unpop(VuDev *dev, VuVirtq *vq, VuVirtqElement *elem,
|
||||
size_t len);
|
||||
|
||||
/**
|
||||
* vu_queue_rewind:
|
||||
* @dev: a VuDev context
|
||||
|
||||
@@ -0,0 +1,232 @@
|
||||
# -*- Mode: Python -*-
|
||||
#
|
||||
# Copyright (C) 2018 Red Hat, Inc.
|
||||
#
|
||||
# Authors:
|
||||
# Marc-André Lureau <marcandre.lureau@redhat.com>
|
||||
#
|
||||
# This work is licensed under the terms of the GNU GPL, version 2 or
|
||||
# later. See the COPYING file in the top-level directory.
|
||||
|
||||
##
|
||||
# = vhost user backend discovery & capabilities
|
||||
##
|
||||
|
||||
##
|
||||
# @VHostUserBackendType:
|
||||
#
|
||||
# List the various vhost user backend types.
|
||||
#
|
||||
# @9p: 9p virtio console
|
||||
# @balloon: virtio balloon
|
||||
# @block: virtio block
|
||||
# @caif: virtio caif
|
||||
# @console: virtio console
|
||||
# @crypto: virtio crypto
|
||||
# @gpu: virtio gpu
|
||||
# @input: virtio input
|
||||
# @net: virtio net
|
||||
# @rng: virtio rng
|
||||
# @rpmsg: virtio remote processor messaging
|
||||
# @rproc-serial: virtio remoteproc serial link
|
||||
# @scsi: virtio scsi
|
||||
# @vsock: virtio vsock transport
|
||||
#
|
||||
# Since: 4.0
|
||||
##
|
||||
{
|
||||
'enum': 'VHostUserBackendType',
|
||||
'data': [
|
||||
'9p',
|
||||
'balloon',
|
||||
'block',
|
||||
'caif',
|
||||
'console',
|
||||
'crypto',
|
||||
'gpu',
|
||||
'input',
|
||||
'net',
|
||||
'rng',
|
||||
'rpmsg',
|
||||
'rproc-serial',
|
||||
'scsi',
|
||||
'vsock'
|
||||
]
|
||||
}
|
||||
|
||||
##
|
||||
# @VHostUserBackendInputFeature:
|
||||
#
|
||||
# List of vhost user "input" features.
|
||||
#
|
||||
# @evdev-path: The --evdev-path command line option is supported.
|
||||
# @no-grab: The --no-grab command line option is supported.
|
||||
#
|
||||
# Since: 4.0
|
||||
##
|
||||
{
|
||||
'enum': 'VHostUserBackendInputFeature',
|
||||
'data': [ 'evdev-path', 'no-grab' ]
|
||||
}
|
||||
|
||||
##
|
||||
# @VHostUserBackendCapabilitiesInput:
|
||||
#
|
||||
# Capabilities reported by vhost user "input" backends
|
||||
#
|
||||
# @features: list of supported features.
|
||||
#
|
||||
# Since: 4.0
|
||||
##
|
||||
{
|
||||
'struct': 'VHostUserBackendCapabilitiesInput',
|
||||
'data': {
|
||||
'features': [ 'VHostUserBackendInputFeature' ]
|
||||
}
|
||||
}
|
||||
|
||||
##
|
||||
# @VHostUserBackendGPUFeature:
|
||||
#
|
||||
# List of vhost user "gpu" features.
|
||||
#
|
||||
# @render-node: The --render-node command line option is supported.
|
||||
# @virgl: The --virgl command line option is supported.
|
||||
#
|
||||
# Since: 4.0
|
||||
##
|
||||
{
|
||||
'enum': 'VHostUserBackendGPUFeature',
|
||||
'data': [ 'render-node', 'virgl' ]
|
||||
}
|
||||
|
||||
##
|
||||
# @VHostUserBackendCapabilitiesGPU:
|
||||
#
|
||||
# Capabilities reported by vhost user "gpu" backends.
|
||||
#
|
||||
# @features: list of supported features.
|
||||
#
|
||||
# Since: 4.0
|
||||
##
|
||||
{
|
||||
'struct': 'VHostUserBackendCapabilitiesGPU',
|
||||
'data': {
|
||||
'features': [ 'VHostUserBackendGPUFeature' ]
|
||||
}
|
||||
}
|
||||
|
||||
##
|
||||
# @VHostUserBackendCapabilities:
|
||||
#
|
||||
# Capabilities reported by vhost user backends.
|
||||
#
|
||||
# @type: The vhost user backend type.
|
||||
#
|
||||
# Since: 4.0
|
||||
##
|
||||
{
|
||||
'union': 'VHostUserBackendCapabilities',
|
||||
'base': { 'type': 'VHostUserBackendType' },
|
||||
'discriminator': 'type',
|
||||
'data': {
|
||||
'input': 'VHostUserBackendCapabilitiesInput',
|
||||
'gpu': 'VHostUserBackendCapabilitiesGPU'
|
||||
}
|
||||
}
|
||||
|
||||
##
|
||||
# @VhostUserBackend:
|
||||
#
|
||||
# Describes a vhost user backend to management software.
|
||||
#
|
||||
# It is possible for multiple @VhostUserBackend elements to match the
|
||||
# search criteria of management software. Applications thus need rules
|
||||
# to pick one of the many matches, and users need the ability to
|
||||
# override distro defaults.
|
||||
#
|
||||
# It is recommended to create vhost user backend JSON files (each
|
||||
# containing a single @VhostUserBackend root element) with a
|
||||
# double-digit prefix, for example "50-qemu-gpu.json",
|
||||
# "50-crosvm-gpu.json", etc, so they can be sorted in predictable
|
||||
# order. The backend JSON files should be searched for in three
|
||||
# directories:
|
||||
#
|
||||
# - /usr/share/qemu/vhost-user -- populated by distro-provided
|
||||
# packages (XDG_DATA_DIRS covers
|
||||
# /usr/share by default),
|
||||
#
|
||||
# - /etc/qemu/vhost-user -- exclusively for sysadmins' local additions,
|
||||
#
|
||||
# - $XDG_CONFIG_HOME/qemu/vhost-user -- exclusively for per-user local
|
||||
# additions (XDG_CONFIG_HOME
|
||||
# defaults to $HOME/.config).
|
||||
#
|
||||
# Top-down, the list of directories goes from general to specific.
|
||||
#
|
||||
# Management software should build a list of files from all three
|
||||
# locations, then sort the list by filename (i.e., basename
|
||||
# component). Management software should choose the first JSON file on
|
||||
# the sorted list that matches the search criteria. If a more specific
|
||||
# directory has a file with same name as a less specific directory,
|
||||
# then the file in the more specific directory takes effect. If the
|
||||
# more specific file is zero length, it hides the less specific one.
|
||||
#
|
||||
# For example, if a distro ships
|
||||
#
|
||||
# - /usr/share/qemu/vhost-user/50-qemu-gpu.json
|
||||
#
|
||||
# - /usr/share/qemu/vhost-user/50-crosvm-gpu.json
|
||||
#
|
||||
# then the sysadmin can prevent the default QEMU being used at all with
|
||||
#
|
||||
# $ touch /etc/qemu/vhost-user/50-qemu-gpu.json
|
||||
#
|
||||
# The sysadmin can replace/alter the distro default OVMF with
|
||||
#
|
||||
# $ vim /etc/qemu/vhost-user/50-qemu-gpu.json
|
||||
#
|
||||
# or they can provide a parallel QEMU GPU with higher priority
|
||||
#
|
||||
# $ vim /etc/qemu/vhost-user/10-qemu-gpu.json
|
||||
#
|
||||
# or they can provide a parallel OVMF with lower priority
|
||||
#
|
||||
# $ vim /etc/qemu/vhost-user/99-qemu-gpu.json
|
||||
#
|
||||
# @type: The vhost user backend type.
|
||||
#
|
||||
# @description: Provides a human-readable description of the backend.
|
||||
# Management software may or may not display @description.
|
||||
#
|
||||
# @binary: Absolute path to the backend binary.
|
||||
#
|
||||
# @tags: An optional list of auxiliary strings associated with the
|
||||
# backend for which @description is not appropriate, due to the
|
||||
# latter's possible exposure to the end-user. @tags serves
|
||||
# development and debugging purposes only, and management
|
||||
# software shall explicitly ignore it.
|
||||
#
|
||||
# Since: 4.0
|
||||
#
|
||||
# Example:
|
||||
#
|
||||
# {
|
||||
# "description": "QEMU vhost-user-gpu",
|
||||
# "type": "gpu",
|
||||
# "binary": "/usr/libexec/qemu/vhost-user-gpu",
|
||||
# "tags": [
|
||||
# "CONFIG_OPENGL_DMABUF=y"
|
||||
# ]
|
||||
# }
|
||||
#
|
||||
##
|
||||
{
|
||||
'struct' : 'VhostUserBackend',
|
||||
'data' : {
|
||||
'description': 'str',
|
||||
'type': 'VHostUserBackendType',
|
||||
'binary': 'str',
|
||||
'*tags': [ 'str' ]
|
||||
}
|
||||
}
|
||||
+384
-2
@@ -17,8 +17,13 @@ The protocol defines 2 sides of the communication, master and slave. Master is
|
||||
the application that shares its virtqueues, in our case QEMU. Slave is the
|
||||
consumer of the virtqueues.
|
||||
|
||||
In the current implementation QEMU is the Master, and the Slave is intended to
|
||||
be a software Ethernet switch running in user space, such as Snabbswitch.
|
||||
In the current implementation QEMU is the Master, and the Slave is the
|
||||
external process consuming the virtio queues, for example a software
|
||||
Ethernet switch running in user space, such as Snabbswitch, or a block
|
||||
device backend processing read & write to a virtual disk. In order to
|
||||
facilitate interoperability between various backend implementations,
|
||||
it is recommended to follow the "Backend program conventions"
|
||||
described in this document.
|
||||
|
||||
Master and slave can be either a client (i.e. connecting) or server (listening)
|
||||
in the socket communication.
|
||||
@@ -142,6 +147,17 @@ Depending on the request type, payload can be:
|
||||
Offset: a 64-bit offset of this area from the start of the
|
||||
supplied file descriptor
|
||||
|
||||
* Inflight description
|
||||
-----------------------------------------------------
|
||||
| mmap size | mmap offset | num queues | queue size |
|
||||
-----------------------------------------------------
|
||||
|
||||
mmap size: a 64-bit size of area to track inflight I/O
|
||||
mmap offset: a 64-bit offset of this area from the start
|
||||
of the supplied file descriptor
|
||||
num queues: a 16-bit number of virtqueues
|
||||
queue size: a 16-bit size of virtqueues
|
||||
|
||||
In QEMU the vhost-user message is implemented with the following struct:
|
||||
|
||||
typedef struct VhostUserMsg {
|
||||
@@ -157,6 +173,7 @@ typedef struct VhostUserMsg {
|
||||
struct vhost_iotlb_msg iotlb;
|
||||
VhostUserConfig config;
|
||||
VhostUserVringArea area;
|
||||
VhostUserInflight inflight;
|
||||
};
|
||||
} QEMU_PACKED VhostUserMsg;
|
||||
|
||||
@@ -175,6 +192,7 @@ the ones that do:
|
||||
* VHOST_USER_GET_PROTOCOL_FEATURES
|
||||
* VHOST_USER_GET_VRING_BASE
|
||||
* VHOST_USER_SET_LOG_BASE (if VHOST_USER_PROTOCOL_F_LOG_SHMFD)
|
||||
* VHOST_USER_GET_INFLIGHT_FD (if VHOST_USER_PROTOCOL_F_INFLIGHT_SHMFD)
|
||||
|
||||
[ Also see the section on REPLY_ACK protocol extension. ]
|
||||
|
||||
@@ -188,6 +206,7 @@ in the ancillary data:
|
||||
* VHOST_USER_SET_VRING_CALL
|
||||
* VHOST_USER_SET_VRING_ERR
|
||||
* VHOST_USER_SET_SLAVE_REQ_FD
|
||||
* VHOST_USER_SET_INFLIGHT_FD (if VHOST_USER_PROTOCOL_F_INFLIGHT_SHMFD)
|
||||
|
||||
If Master is unable to send the full message or receives a wrong reply it will
|
||||
close the connection. An optional reconnection mechanism can be implemented.
|
||||
@@ -382,6 +401,256 @@ If VHOST_USER_PROTOCOL_F_SLAVE_SEND_FD protocol feature is negotiated,
|
||||
slave can send file descriptors (at most 8 descriptors in each message)
|
||||
to master via ancillary data using this fd communication channel.
|
||||
|
||||
Inflight I/O tracking
|
||||
---------------------
|
||||
|
||||
To support reconnecting after restart or crash, slave may need to resubmit
|
||||
inflight I/Os. If virtqueue is processed in order, we can easily achieve
|
||||
that by getting the inflight descriptors from descriptor table (split virtqueue)
|
||||
or descriptor ring (packed virtqueue). However, it can't work when we process
|
||||
descriptors out-of-order because some entries which store the information of
|
||||
inflight descriptors in available ring (split virtqueue) or descriptor
|
||||
ring (packed virtqueue) might be overrided by new entries. To solve this
|
||||
problem, slave need to allocate an extra buffer to store this information of inflight
|
||||
descriptors and share it with master for persistent. VHOST_USER_GET_INFLIGHT_FD and
|
||||
VHOST_USER_SET_INFLIGHT_FD are used to transfer this buffer between master
|
||||
and slave. And the format of this buffer is described below:
|
||||
|
||||
-------------------------------------------------------
|
||||
| queue0 region | queue1 region | ... | queueN region |
|
||||
-------------------------------------------------------
|
||||
|
||||
N is the number of available virtqueues. Slave could get it from num queues
|
||||
field of VhostUserInflight.
|
||||
|
||||
For split virtqueue, queue region can be implemented as:
|
||||
|
||||
typedef struct DescStateSplit {
|
||||
/* Indicate whether this descriptor is inflight or not.
|
||||
* Only available for head-descriptor. */
|
||||
uint8_t inflight;
|
||||
|
||||
/* Padding */
|
||||
uint8_t padding[5];
|
||||
|
||||
/* Maintain a list for the last batch of used descriptors.
|
||||
* Only available when batching is used for submitting */
|
||||
uint16_t next;
|
||||
|
||||
/* Used to preserve the order of fetching available descriptors.
|
||||
* Only available for head-descriptor. */
|
||||
uint64_t counter;
|
||||
} DescStateSplit;
|
||||
|
||||
typedef struct QueueRegionSplit {
|
||||
/* The feature flags of this region. Now it's initialized to 0. */
|
||||
uint64_t features;
|
||||
|
||||
/* The version of this region. It's 1 currently.
|
||||
* Zero value indicates an uninitialized buffer */
|
||||
uint16_t version;
|
||||
|
||||
/* The size of DescStateSplit array. It's equal to the virtqueue
|
||||
* size. Slave could get it from queue size field of VhostUserInflight. */
|
||||
uint16_t desc_num;
|
||||
|
||||
/* The head of list that track the last batch of used descriptors. */
|
||||
uint16_t last_batch_head;
|
||||
|
||||
/* Store the idx value of used ring */
|
||||
uint16_t used_idx;
|
||||
|
||||
/* Used to track the state of each descriptor in descriptor table */
|
||||
DescStateSplit desc[0];
|
||||
} QueueRegionSplit;
|
||||
|
||||
To track inflight I/O, the queue region should be processed as follows:
|
||||
|
||||
When receiving available buffers from the driver:
|
||||
|
||||
1. Get the next available head-descriptor index from available ring, i
|
||||
|
||||
2. Set desc[i].counter to the value of global counter
|
||||
|
||||
3. Increase global counter by 1
|
||||
|
||||
4. Set desc[i].inflight to 1
|
||||
|
||||
When supplying used buffers to the driver:
|
||||
|
||||
1. Get corresponding used head-descriptor index, i
|
||||
|
||||
2. Set desc[i].next to last_batch_head
|
||||
|
||||
3. Set last_batch_head to i
|
||||
|
||||
4. Steps 1,2,3 may be performed repeatedly if batching is possible
|
||||
|
||||
5. Increase the idx value of used ring by the size of the batch
|
||||
|
||||
6. Set the inflight field of each DescStateSplit entry in the batch to 0
|
||||
|
||||
7. Set used_idx to the idx value of used ring
|
||||
|
||||
When reconnecting:
|
||||
|
||||
1. If the value of used_idx does not match the idx value of used ring (means
|
||||
the inflight field of DescStateSplit entries in last batch may be incorrect),
|
||||
|
||||
(a) Subtract the value of used_idx from the idx value of used ring to get
|
||||
last batch size of DescStateSplit entries
|
||||
|
||||
(b) Set the inflight field of each DescStateSplit entry to 0 in last batch
|
||||
list which starts from last_batch_head
|
||||
|
||||
(c) Set used_idx to the idx value of used ring
|
||||
|
||||
2. Resubmit inflight DescStateSplit entries in order of their counter value
|
||||
|
||||
For packed virtqueue, queue region can be implemented as:
|
||||
|
||||
typedef struct DescStatePacked {
|
||||
/* Indicate whether this descriptor is inflight or not.
|
||||
* Only available for head-descriptor. */
|
||||
uint8_t inflight;
|
||||
|
||||
/* Padding */
|
||||
uint8_t padding;
|
||||
|
||||
/* Link to the next free entry */
|
||||
uint16_t next;
|
||||
|
||||
/* Link to the last entry of descriptor list.
|
||||
* Only available for head-descriptor. */
|
||||
uint16_t last;
|
||||
|
||||
/* The length of descriptor list.
|
||||
* Only available for head-descriptor. */
|
||||
uint16_t num;
|
||||
|
||||
/* Used to preserve the order of fetching available descriptors.
|
||||
* Only available for head-descriptor. */
|
||||
uint64_t counter;
|
||||
|
||||
/* The buffer id */
|
||||
uint16_t id;
|
||||
|
||||
/* The descriptor flags */
|
||||
uint16_t flags;
|
||||
|
||||
/* The buffer length */
|
||||
uint32_t len;
|
||||
|
||||
/* The buffer address */
|
||||
uint64_t addr;
|
||||
} DescStatePacked;
|
||||
|
||||
typedef struct QueueRegionPacked {
|
||||
/* The feature flags of this region. Now it's initialized to 0. */
|
||||
uint64_t features;
|
||||
|
||||
/* The version of this region. It's 1 currently.
|
||||
* Zero value indicates an uninitialized buffer */
|
||||
uint16_t version;
|
||||
|
||||
/* The size of DescStatePacked array. It's equal to the virtqueue
|
||||
* size. Slave could get it from queue size field of VhostUserInflight. */
|
||||
uint16_t desc_num;
|
||||
|
||||
/* The head of free DescStatePacked entry list */
|
||||
uint16_t free_head;
|
||||
|
||||
/* The old head of free DescStatePacked entry list */
|
||||
uint16_t old_free_head;
|
||||
|
||||
/* The used index of descriptor ring */
|
||||
uint16_t used_idx;
|
||||
|
||||
/* The old used index of descriptor ring */
|
||||
uint16_t old_used_idx;
|
||||
|
||||
/* Device ring wrap counter */
|
||||
uint8_t used_wrap_counter;
|
||||
|
||||
/* The old device ring wrap counter */
|
||||
uint8_t old_used_wrap_counter;
|
||||
|
||||
/* Padding */
|
||||
uint8_t padding[7];
|
||||
|
||||
/* Used to track the state of each descriptor fetched from descriptor ring */
|
||||
DescStatePacked desc[0];
|
||||
} QueueRegionPacked;
|
||||
|
||||
To track inflight I/O, the queue region should be processed as follows:
|
||||
|
||||
When receiving available buffers from the driver:
|
||||
|
||||
1. Get the next available descriptor entry from descriptor ring, d
|
||||
|
||||
2. If d is head descriptor,
|
||||
|
||||
(a) Set desc[old_free_head].num to 0
|
||||
|
||||
(b) Set desc[old_free_head].counter to the value of global counter
|
||||
|
||||
(c) Increase global counter by 1
|
||||
|
||||
(d) Set desc[old_free_head].inflight to 1
|
||||
|
||||
3. If d is last descriptor, set desc[old_free_head].last to free_head
|
||||
|
||||
4. Increase desc[old_free_head].num by 1
|
||||
|
||||
5. Set desc[free_head].addr, desc[free_head].len, desc[free_head].flags,
|
||||
desc[free_head].id to d.addr, d.len, d.flags, d.id
|
||||
|
||||
6. Set free_head to desc[free_head].next
|
||||
|
||||
7. If d is last descriptor, set old_free_head to free_head
|
||||
|
||||
When supplying used buffers to the driver:
|
||||
|
||||
1. Get corresponding used head-descriptor entry from descriptor ring, d
|
||||
|
||||
2. Get corresponding DescStatePacked entry, e
|
||||
|
||||
3. Set desc[e.last].next to free_head
|
||||
|
||||
4. Set free_head to the index of e
|
||||
|
||||
5. Steps 1,2,3,4 may be performed repeatedly if batching is possible
|
||||
|
||||
6. Increase used_idx by the size of the batch and update used_wrap_counter if needed
|
||||
|
||||
7. Update d.flags
|
||||
|
||||
8. Set the inflight field of each head DescStatePacked entry in the batch to 0
|
||||
|
||||
9. Set old_free_head, old_used_idx, old_used_wrap_counter to free_head, used_idx,
|
||||
used_wrap_counter
|
||||
|
||||
When reconnecting:
|
||||
|
||||
1. If used_idx does not match old_used_idx (means the inflight field of DescStatePacked
|
||||
entries in last batch may be incorrect),
|
||||
|
||||
(a) Get the next descriptor ring entry through old_used_idx, d
|
||||
|
||||
(b) Use old_used_wrap_counter to calculate the available flags
|
||||
|
||||
(c) If d.flags is not equal to the calculated flags value (means slave has
|
||||
submitted the buffer to guest driver before crash, so it has to commit the
|
||||
in-progres update), set old_free_head, old_used_idx, old_used_wrap_counter
|
||||
to free_head, used_idx, used_wrap_counter
|
||||
|
||||
2. Set free_head, used_idx, used_wrap_counter to old_free_head, old_used_idx,
|
||||
old_used_wrap_counter (roll back any in-progress update)
|
||||
|
||||
3. Set the inflight field of each DescStatePacked entry in free list to 0
|
||||
|
||||
4. Resubmit inflight DescStatePacked entries in order of their counter value
|
||||
|
||||
Protocol features
|
||||
-----------------
|
||||
|
||||
@@ -397,6 +666,7 @@ Protocol features
|
||||
#define VHOST_USER_PROTOCOL_F_CONFIG 9
|
||||
#define VHOST_USER_PROTOCOL_F_SLAVE_SEND_FD 10
|
||||
#define VHOST_USER_PROTOCOL_F_HOST_NOTIFIER 11
|
||||
#define VHOST_USER_PROTOCOL_F_INFLIGHT_SHMFD 12
|
||||
|
||||
Master message types
|
||||
--------------------
|
||||
@@ -761,6 +1031,26 @@ Master message types
|
||||
was previously sent.
|
||||
The value returned is an error indication; 0 is success.
|
||||
|
||||
* VHOST_USER_GET_INFLIGHT_FD
|
||||
Id: 31
|
||||
Equivalent ioctl: N/A
|
||||
Master payload: inflight description
|
||||
|
||||
When VHOST_USER_PROTOCOL_F_INFLIGHT_SHMFD protocol feature has been
|
||||
successfully negotiated, this message is submitted by master to get
|
||||
a shared buffer from slave. The shared buffer will be used to track
|
||||
inflight I/O by slave. QEMU should retrieve a new one when vm reset.
|
||||
|
||||
* VHOST_USER_SET_INFLIGHT_FD
|
||||
Id: 32
|
||||
Equivalent ioctl: N/A
|
||||
Master payload: inflight description
|
||||
|
||||
When VHOST_USER_PROTOCOL_F_INFLIGHT_SHMFD protocol feature has been
|
||||
successfully negotiated, this message is submitted by master to send
|
||||
the shared inflight buffer back to slave so that slave could get
|
||||
inflight I/O after a crash or restart.
|
||||
|
||||
Slave message types
|
||||
-------------------
|
||||
|
||||
@@ -835,3 +1125,95 @@ resilient for selective requests.
|
||||
For the message types that already solicit a reply from the client, the
|
||||
presence of VHOST_USER_PROTOCOL_F_REPLY_ACK or need_reply bit being set brings
|
||||
no behavioural change. (See the 'Communication' section for details.)
|
||||
|
||||
Backend program conventions
|
||||
---------------------------
|
||||
|
||||
vhost-user backends can provide various devices & services and may
|
||||
need to be configured manually depending on the use case. However, it
|
||||
is a good idea to follow the conventions listed here when
|
||||
possible. Users, QEMU or libvirt, can then rely on some common
|
||||
behaviour to avoid heterogenous configuration and management of the
|
||||
backend programs and facilitate interoperability.
|
||||
|
||||
Each backend installed on a host system should come with at least one
|
||||
JSON file that conforms to the vhost-user.json schema. Each file
|
||||
informs the management applications about the backend type, and binary
|
||||
location. In addition, it defines rules for management apps for
|
||||
picking the highest priority backend when multiple match the search
|
||||
criteria (see @VhostUserBackend documentation in the schema file).
|
||||
|
||||
If the backend is not capable of enabling a requested feature on the
|
||||
host (such as 3D acceleration with virgl), or the initialization
|
||||
failed, the backend should fail to start early and exit with a status
|
||||
!= 0. It may also print a message to stderr for further details.
|
||||
|
||||
The backend program must not daemonize itself, but it may be
|
||||
daemonized by the management layer. It may also have a restricted
|
||||
access to the system.
|
||||
|
||||
File descriptors 0, 1 and 2 will exist, and have regular
|
||||
stdin/stdout/stderr usage (they may have been redirected to /dev/null
|
||||
by the management layer, or to a log handler).
|
||||
|
||||
The backend program must end (as quickly and cleanly as possible) when
|
||||
the SIGTERM signal is received. Eventually, it may receive SIGKILL by
|
||||
the management layer after a few seconds.
|
||||
|
||||
The following command line options have an expected behaviour. They
|
||||
are mandatory, unless explicitly said differently:
|
||||
|
||||
* --socket-path=PATH
|
||||
|
||||
This option specify the location of the vhost-user Unix domain socket.
|
||||
It is incompatible with --fd.
|
||||
|
||||
* --fd=FDNUM
|
||||
|
||||
When this argument is given, the backend program is started with the
|
||||
vhost-user socket as file descriptor FDNUM. It is incompatible with
|
||||
--socket-path.
|
||||
|
||||
* --print-capabilities
|
||||
|
||||
Output to stdout the backend capabilities in JSON format, and then
|
||||
exit successfully. Other options and arguments should be ignored, and
|
||||
the backend program should not perform its normal function. The
|
||||
capabilities can be reported dynamically depending on the host
|
||||
capabilities.
|
||||
|
||||
The JSON output is described in the vhost-user.json schema, by
|
||||
@VHostUserBackendCapabilities. Example:
|
||||
{
|
||||
"type": "foo",
|
||||
"features": [
|
||||
"feature-a",
|
||||
"feature-b"
|
||||
]
|
||||
}
|
||||
|
||||
vhost-user-input
|
||||
----------------
|
||||
|
||||
Command line options:
|
||||
|
||||
* --evdev-path=PATH (optional)
|
||||
|
||||
Specify the linux input device.
|
||||
|
||||
* --no-grab (optional)
|
||||
|
||||
Do no request exclusive access to the input device.
|
||||
|
||||
vhost-user-gpu
|
||||
--------------
|
||||
|
||||
Command line options:
|
||||
|
||||
* --render-node=PATH (optional)
|
||||
|
||||
Specify the GPU DRM render node.
|
||||
|
||||
* --virgl (optional)
|
||||
|
||||
Enable virgl rendering support.
|
||||
|
||||
+13
-2
@@ -483,13 +483,24 @@ void ich9_pm_add_properties(Object *obj, ICH9LPCPMRegs *pm, Error **errp)
|
||||
NULL);
|
||||
}
|
||||
|
||||
void ich9_pm_device_pre_plug_cb(HotplugHandler *hotplug_dev, DeviceState *dev,
|
||||
Error **errp)
|
||||
{
|
||||
ICH9LPCState *lpc = ICH9_LPC_DEVICE(hotplug_dev);
|
||||
|
||||
if (object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM) &&
|
||||
!lpc->pm.acpi_memory_hotplug.is_enabled)
|
||||
error_setg(errp,
|
||||
"memory hotplug is not enabled: %s.memory-hotplug-support "
|
||||
"is not set", object_get_typename(OBJECT(lpc)));
|
||||
}
|
||||
|
||||
void ich9_pm_device_plug_cb(HotplugHandler *hotplug_dev, DeviceState *dev,
|
||||
Error **errp)
|
||||
{
|
||||
ICH9LPCState *lpc = ICH9_LPC_DEVICE(hotplug_dev);
|
||||
|
||||
if (lpc->pm.acpi_memory_hotplug.is_enabled &&
|
||||
object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM)) {
|
||||
if (object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM)) {
|
||||
if (object_dynamic_cast(OBJECT(dev), TYPE_NVDIMM)) {
|
||||
nvdimm_acpi_plug_cb(hotplug_dev, dev);
|
||||
} else {
|
||||
|
||||
+4
-4
@@ -992,7 +992,7 @@ static void nvdimm_build_common_dsm(Aml *dev)
|
||||
field = aml_field(NVDIMM_DSM_IOPORT, AML_DWORD_ACC, AML_NOLOCK,
|
||||
AML_PRESERVE);
|
||||
aml_append(field, aml_named_field(NVDIMM_DSM_NOTIFY,
|
||||
sizeof(uint32_t) * BITS_PER_BYTE));
|
||||
NVDIMM_ACPI_IO_LEN * BITS_PER_BYTE));
|
||||
aml_append(method, field);
|
||||
|
||||
/*
|
||||
@@ -1086,7 +1086,7 @@ static void nvdimm_build_common_dsm(Aml *dev)
|
||||
*/
|
||||
aml_append(method, aml_store(handle, aml_name(NVDIMM_DSM_HANDLE)));
|
||||
aml_append(method, aml_store(aml_arg(1), aml_name(NVDIMM_DSM_REVISION)));
|
||||
aml_append(method, aml_store(aml_arg(2), aml_name(NVDIMM_DSM_FUNCTION)));
|
||||
aml_append(method, aml_store(function, aml_name(NVDIMM_DSM_FUNCTION)));
|
||||
|
||||
/*
|
||||
* The fourth parameter (Arg3) of _DSM is a package which contains
|
||||
@@ -1260,7 +1260,7 @@ static void nvdimm_build_nvdimm_devices(Aml *root_dev, uint32_t ram_slots)
|
||||
}
|
||||
|
||||
static void nvdimm_build_ssdt(GArray *table_offsets, GArray *table_data,
|
||||
BIOSLinker *linker, GArray *dsm_dma_arrea,
|
||||
BIOSLinker *linker, GArray *dsm_dma_area,
|
||||
uint32_t ram_slots)
|
||||
{
|
||||
Aml *ssdt, *sb_scope, *dev;
|
||||
@@ -1307,7 +1307,7 @@ static void nvdimm_build_ssdt(GArray *table_offsets, GArray *table_data,
|
||||
NVDIMM_ACPI_MEM_ADDR);
|
||||
|
||||
bios_linker_loader_alloc(linker,
|
||||
NVDIMM_DSM_MEM_FILE, dsm_dma_arrea,
|
||||
NVDIMM_DSM_MEM_FILE, dsm_dma_area,
|
||||
sizeof(NvdimmDsmIn), false /* high memory */);
|
||||
bios_linker_loader_add_pointer(linker,
|
||||
ACPI_BUILD_TABLE_FILE, mem_addr_offset, sizeof(uint32_t),
|
||||
|
||||
+10
-3
@@ -380,9 +380,17 @@ static void piix4_pm_powerdown_req(Notifier *n, void *opaque)
|
||||
static void piix4_device_pre_plug_cb(HotplugHandler *hotplug_dev,
|
||||
DeviceState *dev, Error **errp)
|
||||
{
|
||||
PIIX4PMState *s = PIIX4_PM(hotplug_dev);
|
||||
|
||||
if (object_dynamic_cast(OBJECT(dev), TYPE_PCI_DEVICE)) {
|
||||
acpi_pcihp_device_pre_plug_cb(hotplug_dev, dev, errp);
|
||||
} else if (!object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM) &&
|
||||
} else if (object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM)) {
|
||||
if (!s->acpi_memory_hotplug.is_enabled) {
|
||||
error_setg(errp,
|
||||
"memory hotplug is not enabled: %s.memory-hotplug-support "
|
||||
"is not set", object_get_typename(OBJECT(s)));
|
||||
}
|
||||
} else if (
|
||||
!object_dynamic_cast(OBJECT(dev), TYPE_CPU)) {
|
||||
error_setg(errp, "acpi: device pre plug request for not supported"
|
||||
" device type: %s", object_get_typename(OBJECT(dev)));
|
||||
@@ -394,8 +402,7 @@ static void piix4_device_plug_cb(HotplugHandler *hotplug_dev,
|
||||
{
|
||||
PIIX4PMState *s = PIIX4_PM(hotplug_dev);
|
||||
|
||||
if (s->acpi_memory_hotplug.is_enabled &&
|
||||
object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM)) {
|
||||
if (object_dynamic_cast(OBJECT(dev), TYPE_PC_DIMM)) {
|
||||
if (object_dynamic_cast(OBJECT(dev), TYPE_NVDIMM)) {
|
||||
nvdimm_acpi_plug_cb(hotplug_dev, dev);
|
||||
} else {
|
||||
|
||||
+32
-18
@@ -128,6 +128,21 @@ static void vhost_user_blk_start(VirtIODevice *vdev)
|
||||
}
|
||||
|
||||
s->dev.acked_features = vdev->guest_features;
|
||||
|
||||
if (!s->inflight->addr) {
|
||||
ret = vhost_dev_get_inflight(&s->dev, s->queue_size, s->inflight);
|
||||
if (ret < 0) {
|
||||
error_report("Error get inflight: %d", -ret);
|
||||
goto err_guest_notifiers;
|
||||
}
|
||||
}
|
||||
|
||||
ret = vhost_dev_set_inflight(&s->dev, s->inflight);
|
||||
if (ret < 0) {
|
||||
error_report("Error set inflight: %d", -ret);
|
||||
goto err_guest_notifiers;
|
||||
}
|
||||
|
||||
ret = vhost_dev_start(&s->dev, vdev);
|
||||
if (ret < 0) {
|
||||
error_report("Error starting vhost: %d", -ret);
|
||||
@@ -249,11 +264,17 @@ static void vhost_user_blk_handle_output(VirtIODevice *vdev, VirtQueue *vq)
|
||||
}
|
||||
}
|
||||
|
||||
static void vhost_user_blk_reset(VirtIODevice *vdev)
|
||||
{
|
||||
VHostUserBlk *s = VHOST_USER_BLK(vdev);
|
||||
|
||||
vhost_dev_free_inflight(s->inflight);
|
||||
}
|
||||
|
||||
static void vhost_user_blk_device_realize(DeviceState *dev, Error **errp)
|
||||
{
|
||||
VirtIODevice *vdev = VIRTIO_DEVICE(dev);
|
||||
VHostUserBlk *s = VHOST_USER_BLK(vdev);
|
||||
VhostUserState *user;
|
||||
struct vhost_virtqueue *vqs = NULL;
|
||||
int i, ret;
|
||||
|
||||
@@ -272,15 +293,10 @@ static void vhost_user_blk_device_realize(DeviceState *dev, Error **errp)
|
||||
return;
|
||||
}
|
||||
|
||||
user = vhost_user_init();
|
||||
if (!user) {
|
||||
error_setg(errp, "vhost-user-blk: failed to init vhost_user");
|
||||
if (!vhost_user_init(&s->vhost_user, &s->chardev, errp)) {
|
||||
return;
|
||||
}
|
||||
|
||||
user->chr = &s->chardev;
|
||||
s->vhost_user = user;
|
||||
|
||||
virtio_init(vdev, "virtio-blk", VIRTIO_ID_BLOCK,
|
||||
sizeof(struct virtio_blk_config));
|
||||
|
||||
@@ -289,6 +305,8 @@ static void vhost_user_blk_device_realize(DeviceState *dev, Error **errp)
|
||||
vhost_user_blk_handle_output);
|
||||
}
|
||||
|
||||
s->inflight = g_new0(struct vhost_inflight, 1);
|
||||
|
||||
s->dev.nvqs = s->num_queues;
|
||||
s->dev.vqs = g_new(struct vhost_virtqueue, s->dev.nvqs);
|
||||
s->dev.vq_index = 0;
|
||||
@@ -297,7 +315,7 @@ static void vhost_user_blk_device_realize(DeviceState *dev, Error **errp)
|
||||
|
||||
vhost_dev_set_config_notifier(&s->dev, &blk_ops);
|
||||
|
||||
ret = vhost_dev_init(&s->dev, s->vhost_user, VHOST_BACKEND_TYPE_USER, 0);
|
||||
ret = vhost_dev_init(&s->dev, &s->vhost_user, VHOST_BACKEND_TYPE_USER, 0);
|
||||
if (ret < 0) {
|
||||
error_setg(errp, "vhost-user-blk: vhost initialization failed: %s",
|
||||
strerror(-ret));
|
||||
@@ -321,11 +339,9 @@ vhost_err:
|
||||
vhost_dev_cleanup(&s->dev);
|
||||
virtio_err:
|
||||
g_free(vqs);
|
||||
g_free(s->inflight);
|
||||
virtio_cleanup(vdev);
|
||||
|
||||
vhost_user_cleanup(user);
|
||||
g_free(user);
|
||||
s->vhost_user = NULL;
|
||||
vhost_user_cleanup(&s->vhost_user);
|
||||
}
|
||||
|
||||
static void vhost_user_blk_device_unrealize(DeviceState *dev, Error **errp)
|
||||
@@ -336,14 +352,11 @@ static void vhost_user_blk_device_unrealize(DeviceState *dev, Error **errp)
|
||||
|
||||
vhost_user_blk_set_status(vdev, 0);
|
||||
vhost_dev_cleanup(&s->dev);
|
||||
vhost_dev_free_inflight(s->inflight);
|
||||
g_free(vqs);
|
||||
g_free(s->inflight);
|
||||
virtio_cleanup(vdev);
|
||||
|
||||
if (s->vhost_user) {
|
||||
vhost_user_cleanup(s->vhost_user);
|
||||
g_free(s->vhost_user);
|
||||
s->vhost_user = NULL;
|
||||
}
|
||||
vhost_user_cleanup(&s->vhost_user);
|
||||
}
|
||||
|
||||
static void vhost_user_blk_instance_init(Object *obj)
|
||||
@@ -386,6 +399,7 @@ static void vhost_user_blk_class_init(ObjectClass *klass, void *data)
|
||||
vdc->set_config = vhost_user_blk_set_config;
|
||||
vdc->get_features = vhost_user_blk_get_features;
|
||||
vdc->set_status = vhost_user_blk_set_status;
|
||||
vdc->reset = vhost_user_blk_reset;
|
||||
}
|
||||
|
||||
static const TypeInfo vhost_user_blk_info = {
|
||||
|
||||
+458
-105
File diff suppressed because it is too large
Load Diff
@@ -172,6 +172,7 @@
|
||||
|
||||
/* RTADDR_REG */
|
||||
#define VTD_RTADDR_RTT (1ULL << 11)
|
||||
#define VTD_RTADDR_SMT (1ULL << 10)
|
||||
#define VTD_RTADDR_ADDR_MASK(aw) (VTD_HAW_MASK(aw) ^ 0xfffULL)
|
||||
|
||||
/* IRTA_REG */
|
||||
@@ -189,6 +190,9 @@
|
||||
#define VTD_ECAP_EIM (1ULL << 4)
|
||||
#define VTD_ECAP_PT (1ULL << 6)
|
||||
#define VTD_ECAP_MHMV (15ULL << 20)
|
||||
#define VTD_ECAP_SRS (1ULL << 31)
|
||||
#define VTD_ECAP_SMTS (1ULL << 43)
|
||||
#define VTD_ECAP_SLTS (1ULL << 46)
|
||||
|
||||
/* CAP_REG */
|
||||
/* (offset >> 4) << 24 */
|
||||
@@ -217,11 +221,14 @@
|
||||
#define VTD_CAP_SAGAW_48bit (0x4ULL << VTD_CAP_SAGAW_SHIFT)
|
||||
|
||||
/* IQT_REG */
|
||||
#define VTD_IQT_QT(val) (((val) >> 4) & 0x7fffULL)
|
||||
#define VTD_IQT_QT(dw_bit, val) (dw_bit ? (((val) >> 5) & 0x3fffULL) : \
|
||||
(((val) >> 4) & 0x7fffULL))
|
||||
#define VTD_IQT_QT_256_RSV_BIT 0x10
|
||||
|
||||
/* IQA_REG */
|
||||
#define VTD_IQA_IQA_MASK(aw) (VTD_HAW_MASK(aw) ^ 0xfffULL)
|
||||
#define VTD_IQA_QS 0x7ULL
|
||||
#define VTD_IQA_DW_MASK 0x800
|
||||
|
||||
/* IQH_REG */
|
||||
#define VTD_IQH_QH_SHIFT 4
|
||||
@@ -294,6 +301,8 @@ typedef enum VTDFaultReason {
|
||||
* request while disabled */
|
||||
VTD_FR_IR_SID_ERR = 0x26, /* Invalid Source-ID */
|
||||
|
||||
VTD_FR_PASID_TABLE_INV = 0x58, /*Invalid PASID table entry */
|
||||
|
||||
/* This is not a normal fault reason. We use this to indicate some faults
|
||||
* that are not referenced by the VT-d specification.
|
||||
* Fault event with such reason should not be recorded.
|
||||
@@ -321,6 +330,9 @@ union VTDInvDesc {
|
||||
uint64_t lo;
|
||||
uint64_t hi;
|
||||
};
|
||||
struct {
|
||||
uint64_t val[4];
|
||||
};
|
||||
union {
|
||||
VTDInvDescIEC iec;
|
||||
};
|
||||
@@ -335,6 +347,8 @@ typedef union VTDInvDesc VTDInvDesc;
|
||||
#define VTD_INV_DESC_IEC 0x4 /* Interrupt Entry Cache
|
||||
Invalidate Descriptor */
|
||||
#define VTD_INV_DESC_WAIT 0x5 /* Invalidation Wait Descriptor */
|
||||
#define VTD_INV_DESC_PIOTLB 0x6 /* PASID-IOTLB Invalidate Desc */
|
||||
#define VTD_INV_DESC_PC 0x7 /* PASID-cache Invalidate Desc */
|
||||
#define VTD_INV_DESC_NONE 0 /* Not an Invalidate Descriptor */
|
||||
|
||||
/* Masks for Invalidation Wait Descriptor*/
|
||||
@@ -411,8 +425,8 @@ typedef struct VTDIOTLBPageInvInfo VTDIOTLBPageInvInfo;
|
||||
#define VTD_PAGE_MASK_1G (~((1ULL << VTD_PAGE_SHIFT_1G) - 1))
|
||||
|
||||
struct VTDRootEntry {
|
||||
uint64_t val;
|
||||
uint64_t rsvd;
|
||||
uint64_t lo;
|
||||
uint64_t hi;
|
||||
};
|
||||
typedef struct VTDRootEntry VTDRootEntry;
|
||||
|
||||
@@ -423,6 +437,8 @@ typedef struct VTDRootEntry VTDRootEntry;
|
||||
#define VTD_ROOT_ENTRY_NR (VTD_PAGE_SIZE / sizeof(VTDRootEntry))
|
||||
#define VTD_ROOT_ENTRY_RSVD(aw) (0xffeULL | ~VTD_HAW_MASK(aw))
|
||||
|
||||
#define VTD_DEVFN_CHECK_MASK 0x80
|
||||
|
||||
/* Masks for struct VTDContextEntry */
|
||||
/* lo */
|
||||
#define VTD_CONTEXT_ENTRY_P (1ULL << 0)
|
||||
@@ -441,6 +457,38 @@ typedef struct VTDRootEntry VTDRootEntry;
|
||||
|
||||
#define VTD_CONTEXT_ENTRY_NR (VTD_PAGE_SIZE / sizeof(VTDContextEntry))
|
||||
|
||||
#define VTD_CTX_ENTRY_LEGACY_SIZE 16
|
||||
#define VTD_CTX_ENTRY_SCALABLE_SIZE 32
|
||||
|
||||
#define VTD_SM_CONTEXT_ENTRY_RID2PASID_MASK 0xfffff
|
||||
#define VTD_SM_CONTEXT_ENTRY_RSVD_VAL0(aw) (0x1e0ULL | ~VTD_HAW_MASK(aw))
|
||||
#define VTD_SM_CONTEXT_ENTRY_RSVD_VAL1 0xffffffffffe00000ULL
|
||||
|
||||
/* PASID Table Related Definitions */
|
||||
#define VTD_PASID_DIR_BASE_ADDR_MASK (~0xfffULL)
|
||||
#define VTD_PASID_TABLE_BASE_ADDR_MASK (~0xfffULL)
|
||||
#define VTD_PASID_DIR_ENTRY_SIZE 8
|
||||
#define VTD_PASID_ENTRY_SIZE 64
|
||||
#define VTD_PASID_DIR_BITS_MASK (0x3fffULL)
|
||||
#define VTD_PASID_DIR_INDEX(pasid) (((pasid) >> 6) & VTD_PASID_DIR_BITS_MASK)
|
||||
#define VTD_PASID_DIR_FPD (1ULL << 1) /* Fault Processing Disable */
|
||||
#define VTD_PASID_TABLE_BITS_MASK (0x3fULL)
|
||||
#define VTD_PASID_TABLE_INDEX(pasid) ((pasid) & VTD_PASID_TABLE_BITS_MASK)
|
||||
#define VTD_PASID_ENTRY_FPD (1ULL << 1) /* Fault Processing Disable */
|
||||
|
||||
/* PASID Granular Translation Type Mask */
|
||||
#define VTD_SM_PASID_ENTRY_PGTT (7ULL << 6)
|
||||
#define VTD_SM_PASID_ENTRY_FLT (1ULL << 6)
|
||||
#define VTD_SM_PASID_ENTRY_SLT (2ULL << 6)
|
||||
#define VTD_SM_PASID_ENTRY_NESTED (3ULL << 6)
|
||||
#define VTD_SM_PASID_ENTRY_PT (4ULL << 6)
|
||||
|
||||
#define VTD_SM_PASID_ENTRY_AW 7ULL /* Adjusted guest-address-width */
|
||||
#define VTD_SM_PASID_ENTRY_DID(val) ((val) & VTD_DOMAIN_ID_MASK)
|
||||
|
||||
/* Second Level Page Translation Pointer*/
|
||||
#define VTD_SM_PASID_ENTRY_SLPTPTR (~0xfffULL)
|
||||
|
||||
/* Paging Structure common */
|
||||
#define VTD_SL_PT_PAGE_SIZE_MASK (1ULL << 7)
|
||||
/* Bits to decide the offset for each level */
|
||||
|
||||
@@ -2090,6 +2090,8 @@ static void pc_memory_pre_plug(HotplugHandler *hotplug_dev, DeviceState *dev,
|
||||
return;
|
||||
}
|
||||
|
||||
hotplug_handler_pre_plug(pcms->acpi_dev, dev, errp);
|
||||
|
||||
if (is_nvdimm && !ms->nvdimms_state->is_enabled) {
|
||||
error_setg(errp, "nvdimm is not enabled: missing 'nvdimm' in '-M'");
|
||||
return;
|
||||
|
||||
@@ -30,7 +30,7 @@ vtd_iotlb_cc_hit(uint8_t bus, uint8_t devfn, uint64_t high, uint64_t low, uint32
|
||||
vtd_iotlb_cc_update(uint8_t bus, uint8_t devfn, uint64_t high, uint64_t low, uint32_t gen1, uint32_t gen2) "IOTLB context update bus 0x%"PRIx8" devfn 0x%"PRIx8" high 0x%"PRIx64" low 0x%"PRIx64" gen %"PRIu32" -> gen %"PRIu32
|
||||
vtd_iotlb_reset(const char *reason) "IOTLB reset (reason: %s)"
|
||||
vtd_fault_disabled(void) "Fault processing disabled for context entry"
|
||||
vtd_replay_ce_valid(uint8_t bus, uint8_t dev, uint8_t fn, uint16_t domain, uint64_t hi, uint64_t lo) "replay valid context device %02"PRIx8":%02"PRIx8".%02"PRIx8" domain 0x%"PRIx16" hi 0x%"PRIx64" lo 0x%"PRIx64
|
||||
vtd_replay_ce_valid(const char *mode, uint8_t bus, uint8_t dev, uint8_t fn, uint16_t domain, uint64_t hi, uint64_t lo) "%s: replay valid context device %02"PRIx8":%02"PRIx8".%02"PRIx8" domain 0x%"PRIx16" hi 0x%"PRIx64" lo 0x%"PRIx64
|
||||
vtd_replay_ce_invalid(uint8_t bus, uint8_t dev, uint8_t fn) "replay invalid context device %02"PRIx8":%02"PRIx8".%02"PRIx8
|
||||
vtd_page_walk_level(uint64_t addr, uint32_t level, uint64_t start, uint64_t end) "walk (base=0x%"PRIx64", level=%"PRIu32") iova range 0x%"PRIx64" - 0x%"PRIx64
|
||||
vtd_page_walk_one(uint16_t domain, uint64_t iova, uint64_t gpa, uint64_t mask, int perm) "domain 0x%"PRIu16" iova 0x%"PRIx64" -> gpa 0x%"PRIx64" mask 0x%"PRIx64" perm %d"
|
||||
|
||||
@@ -805,6 +805,7 @@ static void ich9_lpc_class_init(ObjectClass *klass, void *data)
|
||||
* pc_q35_init()
|
||||
*/
|
||||
dc->user_creatable = false;
|
||||
hc->pre_plug = ich9_pm_device_pre_plug_cb;
|
||||
hc->plug = ich9_pm_device_plug_cb;
|
||||
hc->unplug_request = ich9_pm_device_unplug_request_cb;
|
||||
hc->unplug = ich9_pm_device_unplug_cb;
|
||||
|
||||
@@ -20,6 +20,9 @@
|
||||
OBJECT_CHECK(GenPCIERootPort, (obj), TYPE_GEN_PCIE_ROOT_PORT)
|
||||
|
||||
#define GEN_PCIE_ROOT_PORT_AER_OFFSET 0x100
|
||||
#define GEN_PCIE_ROOT_PORT_ACS_OFFSET \
|
||||
(GEN_PCIE_ROOT_PORT_AER_OFFSET + PCI_ERR_SIZEOF)
|
||||
|
||||
#define GEN_PCIE_ROOT_PORT_MSIX_NR_VECTOR 1
|
||||
|
||||
typedef struct GenPCIERootPort {
|
||||
@@ -149,6 +152,7 @@ static void gen_rp_dev_class_init(ObjectClass *klass, void *data)
|
||||
rpc->interrupts_init = gen_rp_interrupts_init;
|
||||
rpc->interrupts_uninit = gen_rp_interrupts_uninit;
|
||||
rpc->aer_offset = GEN_PCIE_ROOT_PORT_AER_OFFSET;
|
||||
rpc->acs_offset = GEN_PCIE_ROOT_PORT_ACS_OFFSET;
|
||||
}
|
||||
|
||||
static const TypeInfo gen_rp_dev_info = {
|
||||
|
||||
@@ -47,6 +47,7 @@ static void rp_reset(DeviceState *qdev)
|
||||
pcie_cap_deverr_reset(d);
|
||||
pcie_cap_slot_reset(d);
|
||||
pcie_cap_arifwd_reset(d);
|
||||
pcie_acs_reset(d);
|
||||
pcie_aer_root_reset(d);
|
||||
pci_bridge_reset(qdev);
|
||||
pci_bridge_disable_base_limit(d);
|
||||
@@ -106,6 +107,9 @@ static void rp_realize(PCIDevice *d, Error **errp)
|
||||
pcie_aer_root_init(d);
|
||||
rp_aer_vector_update(d);
|
||||
|
||||
if (rpc->acs_offset) {
|
||||
pcie_acs_init(d, rpc->acs_offset);
|
||||
}
|
||||
return;
|
||||
|
||||
err:
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user