:p
atchew
Login
Hi all! Here is a new migration parameter "local", which allows to enable local migration of TAP virtio-net backend (and maybe other devices and backends in future), including its properties and open fds. With this new option, management software doesn't need to initialize new TAP and do a switch to it. Nothing should be done around virtio-net in local migration: it just migrates and continues to use same TAP device. So we avoid extra logic in management software, extra allocations in kernel (for new TAP), and corresponding extra delay in migration downtime. v15: 03: add a-b by Peter 05: rebase on master, fix compat property to be part of 11.0 v15 pushed as branch up-tap-fd-migration-with-bk-opt at https://gitlab.com/vsementsov/qemu.git Based-on: <20260318113144.15697-1-vsementsov@yandex-team.ru> "[PATCH v4 00/13] net: refactoring and fixes" Vladimir Sementsov-Ogievskiy (8): net/tap: move vhost-net open() calls to tap_parse_vhost_fds() net/tap: move vhost initialization to tap_setup_vhost() qapi: add local migration parameter net: introduce vmstate_net_peer_backend virtio-net: support local migration of backend net/tap: support local migration with virtio-net tests/functional: add skipWithoutSudo() decorator tests/functional: add test_tap_migration hw/core/machine.c | 5 + hw/i386/pc_q35.c | 1 + hw/net/virtio-net.c | 137 +++++- include/hw/core/boards.h | 3 + include/hw/virtio/virtio-net.h | 2 + include/migration/misc.h | 2 + include/net/net.h | 6 + migration/options.c | 18 +- net/net.c | 47 ++ net/tap.c | 246 ++++++++-- qapi/migration.json | 12 +- qapi/net.json | 10 +- tests/functional/qemu_test/decorators.py | 16 + tests/functional/x86_64/meson.build | 1 + tests/functional/x86_64/test_tap_migration.py | 456 ++++++++++++++++++ 15 files changed, 911 insertions(+), 51 deletions(-) create mode 100755 tests/functional/x86_64/test_tap_migration.py -- 2.52.0
1. Simplify code path: get vhostfds for all cases in one function. 2. Prepare for further tap-fd-migraton feature, when we'll need to postpone vhost initialization up to post-load stage. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 39 ++++++++++++++++++++++----------------- 1 file changed, 22 insertions(+), 17 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, } } - if (tap->has_vhost ? tap->vhost : - (vhostfd != -1) || (tap->has_vhostforce && tap->vhostforce)) { + if (vhostfd != -1) { VhostNetOptions options; options.backend_type = VHOST_BACKEND_TYPE_KERNEL; @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, } else { options.busyloop_timeout = 0; } - - if (vhostfd == -1) { - vhostfd = open("/dev/vhost-net", O_RDWR); - if (vhostfd < 0) { - error_setg_file_open(errp, errno, "/dev/vhost-net"); - goto failed; - } - if (!qemu_set_blocking(vhostfd, false, errp)) { - goto failed; - } - } options.opaque = (void *)(uintptr_t)vhostfd; options.nvqs = 2; options.feature_bits = kernel_feature_bits; @@ -XXX,XX +XXX,XX @@ static int tap_parse_fds_and_queues(const NetdevTapOptions *tap, int **fds, static bool tap_parse_vhost_fds(const NetdevTapOptions *tap, int **vhost_fds, int queues, Error **errp) { - if (!(tap->vhostfd || tap->vhostfds)) { + bool need_vhost = tap->has_vhost ? tap->vhost : + ((tap->vhostfd || tap->vhostfds) || + (tap->has_vhostforce && tap->vhostforce)); + + if (!need_vhost) { *vhost_fds = NULL; return true; } - if (net_parse_fds(tap->vhostfd ?: tap->vhostfds, - vhost_fds, queues, errp) < 0) { - return false; + if (tap->vhostfd || tap->vhostfds) { + if (net_parse_fds(tap->vhostfd ?: tap->vhostfds, + vhost_fds, queues, errp) < 0) { + return false; + } + } else { + *vhost_fds = g_new(int, queues); + for (int i = 0; i < queues; i++) { + int vhostfd = open("/dev/vhost-net", O_RDWR); + if (vhostfd < 0) { + error_setg_file_open(errp, errno, "/dev/vhost-net"); + net_free_fds(*vhost_fds, i); + return false; + } + (*vhost_fds)[i] = vhostfd; + } } if (!unblock_fds(*vhost_fds, queues, errp)) { -- 2.52.0
Make a new helper function in a way it can be reused later for TAP fd-migration feature: we'll need to initialize vhost in a later point when we doesn't have access to QAPI parameters. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 62 ++++++++++++++++++++++++++++++++++--------------------- 1 file changed, 38 insertions(+), 24 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static const int kernel_feature_bits[] = { typedef struct TAPState { NetClientState nc; int fd; + int vhostfd; + uint32_t vhost_busyloop_timeout; char down_script[1024]; char down_script_arg[128]; uint8_t buf[NET_BUFSIZE]; @@ -XXX,XX +XXX,XX @@ static int net_tap_init(const NetdevTapOptions *tap, int *vnet_hdr, return fd; } +static bool tap_setup_vhost(TAPState *s, Error **errp) +{ + VhostNetOptions options; + + if (s->vhostfd == -1) { + return true; + } + + options.backend_type = VHOST_BACKEND_TYPE_KERNEL; + options.net_backend = &s->nc; + options.busyloop_timeout = s->vhost_busyloop_timeout; + options.opaque = (void *)(uintptr_t)s->vhostfd; + options.nvqs = 2; + options.feature_bits = kernel_feature_bits; + options.get_acked_features = NULL; + options.save_acked_features = NULL; + options.max_tx_queue_size = 0; + options.is_vhost_user = false; + + s->vhost_net = vhost_net_init(&options); + if (!s->vhost_net) { + error_setg(errp, + "vhost-net requested but could not be initialized"); + return false; + } + + /* vhostfd ownership is passed to s->vhost_net */ + s->vhostfd = -1; + + return true; +} + static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, const char *name, const char *ifname, const char *script, @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, } } - if (vhostfd != -1) { - VhostNetOptions options; - - options.backend_type = VHOST_BACKEND_TYPE_KERNEL; - options.net_backend = &s->nc; - if (tap->has_poll_us) { - options.busyloop_timeout = tap->poll_us; - } else { - options.busyloop_timeout = 0; - } - options.opaque = (void *)(uintptr_t)vhostfd; - options.nvqs = 2; - options.feature_bits = kernel_feature_bits; - options.get_acked_features = NULL; - options.save_acked_features = NULL; - options.max_tx_queue_size = 0; - options.is_vhost_user = false; - - s->vhost_net = vhost_net_init(&options); - if (!s->vhost_net) { - error_setg(errp, - "vhost-net requested but could not be initialized"); - goto failed; - } + s->vhostfd = vhostfd; + s->vhost_busyloop_timeout = tap->has_poll_us ? tap->poll_us : 0; + if (!tap_setup_vhost(s, errp)) { + return false; } return true; -- 2.52.0
We are going to implement local-migration feature: some devices will be able to transfer open file descriptors through migration stream (which must UNIX domain socket for that purpose). This allows to transfer the whole backend state without reconnecting and restarting the backend service. For example, virtio-net will migrate its attached TAP netdev, together with its connected file descriptors. In this commit we introduce a migration parameter, which enables the feature for devices that support it (none at the moment). Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Acked-by: Markus Armbruster <armbru@redhat.com> Acked-by: Peter Xu <peterx@redhat.com> --- include/migration/misc.h | 2 ++ migration/options.c | 18 +++++++++++++++++- qapi/migration.json | 12 ++++++++++-- 3 files changed, 29 insertions(+), 3 deletions(-) diff --git a/include/migration/misc.h b/include/migration/misc.h index XXXXXXX..XXXXXXX 100644 --- a/include/migration/misc.h +++ b/include/migration/misc.h @@ -XXX,XX +XXX,XX @@ bool multifd_device_state_save_thread_should_exit(void); void multifd_abort_device_state_save_threads(void); bool multifd_join_device_state_save_threads(void); +bool migrate_local(void); + #endif diff --git a/migration/options.c b/migration/options.c index XXXXXXX..XXXXXXX 100644 --- a/migration/options.c +++ b/migration/options.c @@ -XXX,XX +XXX,XX @@ #include "qemu/osdep.h" #include "qemu/error-report.h" +#include "qapi/util.h" #include "exec/target_page.h" #include "qapi/clone-visitor.h" #include "qapi/error.h" @@ -XXX,XX +XXX,XX @@ #include "migration/colo.h" #include "migration/cpr.h" #include "migration/misc.h" +#include "migration/options.h" #include "migration.h" #include "migration-stats.h" #include "qemu-file.h" @@ -XXX,XX +XXX,XX @@ bool migrate_mapped_ram(void) return s->capabilities[MIGRATION_CAPABILITY_MAPPED_RAM]; } +bool migrate_local(void) +{ + MigrationState *s = migrate_get_current(); + return s->parameters.local; +} + bool migrate_ignore_shared(void) { MigrationState *s = migrate_get_current(); @@ -XXX,XX +XXX,XX @@ static void migrate_mark_all_params_present(MigrationParameters *p) &p->has_announce_step, &p->has_block_bitmap_mapping, &p->has_x_vcpu_dirty_limit_period, &p->has_vcpu_dirty_limit, &p->has_mode, &p->has_zero_page_detection, &p->has_direct_io, - &p->has_cpr_exec_command, + &p->has_cpr_exec_command, &p->has_local, }; len = ARRAY_SIZE(has_fields); @@ -XXX,XX +XXX,XX @@ static void migrate_params_test_apply(MigrationParameters *params, if (params->has_cpr_exec_command) { dest->cpr_exec_command = params->cpr_exec_command; } + + if (params->has_local) { + dest->local = params->local; + } } static void migrate_params_apply(MigrationParameters *params) @@ -XXX,XX +XXX,XX @@ static void migrate_params_apply(MigrationParameters *params) s->parameters.cpr_exec_command = QAPI_CLONE(strList, params->cpr_exec_command); } + + if (params->has_local) { + s->parameters.local = params->local; + } } void qmp_migrate_set_parameters(MigrationParameters *params, Error **errp) diff --git a/qapi/migration.json b/qapi/migration.json index XXXXXXX..XXXXXXX 100644 --- a/qapi/migration.json +++ b/qapi/migration.json @@ -XXX,XX +XXX,XX @@ 'mode', 'zero-page-detection', 'direct-io', - 'cpr-exec-command'] } + 'cpr-exec-command', + 'local'] } ## # @migrate-set-parameters: @@ -XXX,XX +XXX,XX @@ # is @cpr-exec. The first list element is the program's filename, # the remainder its arguments. (Since 10.2) # +# @local: Enable local migration for devices that support it. Backend +# state and its file descriptors can then be passed to the +# destination in the migration channel. The migration channel +# must be a Unix domain socket. Usually needs to be enabled per +# device. (Since 11.1) +# # Features: # # @unstable: Members @x-checkpoint-delay and @@ -XXX,XX +XXX,XX @@ '*mode': 'MigMode', '*zero-page-detection': 'ZeroPageDetection', '*direct-io': 'bool', - '*cpr-exec-command': [ 'str' ]} } + '*cpr-exec-command': [ 'str' ], + '*local': 'bool' } } ## # @query-migrate-parameters: -- 2.52.0
To implement backend migration in virtio-net in the next commit, we need a generic API to migrate net backend. Here is it. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- include/net/net.h | 4 ++++ net/net.c | 47 +++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 51 insertions(+) diff --git a/include/net/net.h b/include/net/net.h index XXXXXXX..XXXXXXX 100644 --- a/include/net/net.h +++ b/include/net/net.h @@ -XXX,XX +XXX,XX @@ #include "qapi/qapi-types-net.h" #include "net/queue.h" #include "hw/core/qdev-properties-system.h" +#include "migration/vmstate.h" #define MAC_FMT "%02X:%02X:%02X:%02X:%02X:%02X" #define MAC_ARG(x) ((uint8_t *)(x))[0], ((uint8_t *)(x))[1], \ @@ -XXX,XX +XXX,XX @@ typedef struct NetClientInfo { SetSteeringEBPF *set_steering_ebpf; NetCheckPeerType *check_peer_type; GetVHostNet *get_vhost_net; + const VMStateDescription *backend_vmsd; } NetClientInfo; struct NetClientState { @@ -XXX,XX +XXX,XX @@ static inline bool net_peer_needs_padding(NetClientState *nc) return nc->peer && !nc->peer->do_not_pad; } +extern const VMStateInfo vmstate_net_peer_backend; + #endif diff --git a/net/net.c b/net/net.c index XXXXXXX..XXXXXXX 100644 --- a/net/net.c +++ b/net/net.c @@ -XXX,XX +XXX,XX @@ #include "qapi/string-output-visitor.h" #include "qapi/qobject-input-visitor.h" #include "standard-headers/linux/virtio_net.h" +#include "migration/vmstate.h" /* Net bridge is currently not supported for W32. */ #if !defined(_WIN32) @@ -XXX,XX +XXX,XX @@ int net_fill_rstate(SocketReadState *rs, const uint8_t *buf, int size) assert(size == 0); return 0; } + +static int get_peer_backend(QEMUFile *f, void *pv, size_t size, + const VMStateField *field) +{ + NetClientState *nc = pv; + Error *local_err = NULL; + int ret; + + if (!nc->peer) { + return -EINVAL; + } + nc = nc->peer; + + ret = vmstate_load_state(f, nc->info->backend_vmsd, nc, 0, &local_err); + if (ret < 0) { + error_report_err(local_err); + } + + return ret; +} + +static int put_peer_backend(QEMUFile *f, void *pv, size_t size, + const VMStateField *field, JSONWriter *vmdesc) +{ + NetClientState *nc = pv; + Error *local_err = NULL; + int ret; + + if (!nc->peer) { + return -EINVAL; + } + nc = nc->peer; + + ret = vmstate_save_state(f, nc->info->backend_vmsd, nc, 0, &local_err); + if (ret < 0) { + error_report_err(local_err); + } + + return ret; +} + +const VMStateInfo vmstate_net_peer_backend = { + .name = "virtio-net-nic-nc-backend", + .get = get_peer_backend, + .put = put_peer_backend, +}; -- 2.52.0
Add virtio-net option local-migration, which is true by default, but false for older machine types, which doesn't support the feature. When both global migration parameter "local" and new virtio-net parameter "local-migration" are true, virtio-net transfer the whole net backend to the destination, including open file descriptors. Of-course, its only for local migration and the channel must be UNIX domain socket. This way management tool should not care about creating new TAP, and should not handle switching to it. Migration downtime become shorter. Support for TAP will come in the next commit. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- hw/core/machine.c | 5 ++ hw/i386/pc_q35.c | 1 + hw/net/virtio-net.c | 137 ++++++++++++++++++++++++++++++++- include/hw/core/boards.h | 3 + include/hw/virtio/virtio-net.h | 2 + include/net/net.h | 2 + 6 files changed, 149 insertions(+), 1 deletion(-) diff --git a/hw/core/machine.c b/hw/core/machine.c index XXXXXXX..XXXXXXX 100644 --- a/hw/core/machine.c +++ b/hw/core/machine.c @@ -XXX,XX +XXX,XX @@ #include "hw/acpi/generic_event_device.h" #include "qemu/audio.h" +GlobalProperty hw_compat_11_0[] = { + { TYPE_VIRTIO_NET, "local-migration", "false" }, +}; +const size_t hw_compat_11_0_len = G_N_ELEMENTS(hw_compat_11_0); + GlobalProperty hw_compat_10_2[] = { { "scsi-block", "migrate-pr", "off" }, { "isa-cirrus-vga", "global-vmstate", "true" }, diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c index XXXXXXX..XXXXXXX 100644 --- a/hw/i386/pc_q35.c +++ b/hw/i386/pc_q35.c @@ -XXX,XX +XXX,XX @@ static void pc_q35_machine_options(MachineClass *m) static void pc_q35_machine_11_0_options(MachineClass *m) { pc_q35_machine_options(m); + compat_props_add(m->compat_props, hw_compat_11_0, hw_compat_11_0_len); } DEFINE_Q35_MACHINE_AS_LATEST(11, 0); diff --git a/hw/net/virtio-net.c b/hw/net/virtio-net.c index XXXXXXX..XXXXXXX 100644 --- a/hw/net/virtio-net.c +++ b/hw/net/virtio-net.c @@ -XXX,XX +XXX,XX @@ #include "qapi/qapi-events-migration.h" #include "hw/virtio/virtio-access.h" #include "migration/misc.h" +#include "migration/options.h" #include "standard-headers/linux/ethtool.h" #include "system/system.h" +#include "system/runstate.h" #include "system/replay.h" #include "trace.h" #include "monitor/qdev.h" @@ -XXX,XX +XXX,XX @@ static void virtio_net_set_multiqueue(VirtIONet *n, int multiqueue) n->multiqueue = multiqueue; virtio_net_change_num_queues(n, max * 2 + 1); - virtio_net_set_queue_pairs(n); + /* + * virtio_net_set_multiqueue() called from set_features(0) on early + * reset, when peer may wait for incoming (and is not initialized + * yet). + * Don't worry about it: virtio_net_set_queue_pairs() will be called + * later form virtio_net_post_load_device(), and anyway will be + * noop for local incoming migration with live backend passing. + */ + if (!n->peers_wait_incoming) { + virtio_net_set_queue_pairs(n); + } } static int virtio_net_pre_load_queues(VirtIODevice *vdev, uint32_t n) @@ -XXX,XX +XXX,XX @@ static void virtio_net_get_features(VirtIODevice *vdev, uint64_t *features, virtio_add_feature_ex(features, VIRTIO_NET_F_MAC); + if (n->peers_wait_incoming) { + /* + * Excessive feature set is OK for early initialization when + * we wait for local incoming migration: actual guest-negotiated + * features will come with migration stream anyway. And we are sure + * that we support same host-features as source, because the backend + * is the same (the same TAP device, for example). + */ + return; + } + if (!peer_has_vnet_hdr(n)) { virtio_clear_feature_ex(features, VIRTIO_NET_F_CSUM); virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_TSO4); @@ -XXX,XX +XXX,XX @@ static void virtio_net_get_features(VirtIODevice *vdev, uint64_t *features, } } +static bool virtio_net_update_host_features(VirtIONet *n, Error **errp) +{ + ERRP_GUARD(); + VirtIODevice *vdev = VIRTIO_DEVICE(n); + + peer_test_vnet_hdr(n); + + virtio_net_get_features(vdev, &vdev->host_features, errp); + + return !*errp; +} + static int virtio_net_post_load_device(void *opaque, int version_id) { VirtIONet *n = opaque; @@ -XXX,XX +XXX,XX @@ struct VirtIONetMigTmp { uint16_t curr_queue_pairs_1; uint8_t has_ufo; uint32_t has_vnet_hdr; + + NetClientState *ncs; + uint32_t max_queue_pairs; }; /* The 2nd and subsequent tx_waiting flags are loaded later than @@ -XXX,XX +XXX,XX @@ static const VMStateDescription vhost_user_net_backend_state = { } }; +static bool virtio_net_migrate_local(void *opaque, int version_id) +{ + VirtIONet *n = opaque; + + return migrate_local() && n->local_migration; +} + +static int virtio_net_nic_pre_save(void *opaque) +{ + struct VirtIONetMigTmp *tmp = opaque; + + tmp->ncs = tmp->parent->nic->ncs; + tmp->max_queue_pairs = tmp->parent->max_queue_pairs; + + return 0; +} + +static int virtio_net_nic_pre_load(void *opaque) +{ + /* Reuse the pointer setup from save */ + virtio_net_nic_pre_save(opaque); + + return 0; +} + +static int virtio_net_nic_post_load(void *opaque, int version_id) +{ + struct VirtIONetMigTmp *tmp = opaque; + Error *local_err = NULL; + + if (!virtio_net_update_host_features(tmp->parent, &local_err)) { + error_report_err(local_err); + return -EINVAL; + } + + return 0; +} + +static const VMStateDescription vmstate_virtio_net_nic = { + .name = "virtio-net-nic", + .pre_load = virtio_net_nic_pre_load, + .pre_save = virtio_net_nic_pre_save, + .post_load = virtio_net_nic_post_load, + .fields = (const VMStateField[]) { + VMSTATE_VARRAY_UINT32(ncs, struct VirtIONetMigTmp, + max_queue_pairs, 0, vmstate_net_peer_backend, + NetClientState), + VMSTATE_END_OF_LIST() + }, +}; + static const VMStateDescription vmstate_virtio_net_device = { .name = "virtio-net-device", .version_id = VIRTIO_NET_VM_VERSION, @@ -XXX,XX +XXX,XX @@ static const VMStateDescription vmstate_virtio_net_device = { * but based on the uint. */ VMSTATE_BUFFER_POINTER_UNSAFE(vlans, VirtIONet, 0, MAX_VLAN >> 3), + VMSTATE_WITH_TMP_TEST(VirtIONet, virtio_net_migrate_local, + struct VirtIONetMigTmp, + vmstate_virtio_net_nic), VMSTATE_WITH_TMP(VirtIONet, struct VirtIONetMigTmp, vmstate_virtio_net_has_vnet), VMSTATE_UINT8(mac_table.multi_overflow, VirtIONet), @@ -XXX,XX +XXX,XX @@ static bool failover_hide_primary_device(DeviceListener *listener, return qatomic_read(&n->failover_primary_hidden); } +static bool virtio_net_check_peers_wait_incoming(VirtIONet *n, bool *waiting, + Error **errp) +{ + bool has_waiting = false; + bool has_not_waiting = false; + + for (int i = 0; i < n->max_queue_pairs; i++) { + NetClientState *peer = n->nic->ncs[i].peer; + if (!peer) { + continue; + } + + if (peer->info->is_wait_incoming && + peer->info->is_wait_incoming(peer)) { + has_waiting = true; + } else { + has_not_waiting = true; + } + + if (has_waiting && has_not_waiting) { + error_setg(errp, "Mixed peer states: some peers wait for incoming " + "migration while others don't"); + return false; + } + } + + if (has_waiting && !runstate_check(RUN_STATE_INMIGRATE)) { + error_setg(errp, "Peers wait for incoming, but it's not an incoming " + "migration."); + return false; + } + + *waiting = has_waiting; + return true; +} + static void virtio_net_device_realize(DeviceState *dev, Error **errp) { VirtIODevice *vdev = VIRTIO_DEVICE(dev); @@ -XXX,XX +XXX,XX @@ static void virtio_net_device_realize(DeviceState *dev, Error **errp) n->nic->ncs[i].do_not_pad = true; } + if (!virtio_net_check_peers_wait_incoming(n, &n->peers_wait_incoming, + errp)) { + virtio_cleanup(vdev); + return; + } + peer_test_vnet_hdr(n); if (peer_has_vnet_hdr(n)) { n->host_hdr_len = sizeof(struct virtio_net_hdr); @@ -XXX,XX +XXX,XX @@ static const Property virtio_net_properties[] = { host_features_ex, VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM, true), + DEFINE_PROP_BOOL("local-migration", VirtIONet, local_migration, true), }; static void virtio_net_class_init(ObjectClass *klass, const void *data) diff --git a/include/hw/core/boards.h b/include/hw/core/boards.h index XXXXXXX..XXXXXXX 100644 --- a/include/hw/core/boards.h +++ b/include/hw/core/boards.h @@ -XXX,XX +XXX,XX @@ struct MachineState { } \ } while (0) +extern GlobalProperty hw_compat_11_0[]; +extern const size_t hw_compat_11_0_len; + extern GlobalProperty hw_compat_10_2[]; extern const size_t hw_compat_10_2_len; diff --git a/include/hw/virtio/virtio-net.h b/include/hw/virtio/virtio-net.h index XXXXXXX..XXXXXXX 100644 --- a/include/hw/virtio/virtio-net.h +++ b/include/hw/virtio/virtio-net.h @@ -XXX,XX +XXX,XX @@ struct VirtIONet { struct EBPFRSSContext ebpf_rss; uint32_t nr_ebpf_rss_fds; char **ebpf_rss_fds; + bool peers_wait_incoming; + bool local_migration; }; size_t virtio_net_handle_ctrl_iov(VirtIODevice *vdev, diff --git a/include/net/net.h b/include/net/net.h index XXXXXXX..XXXXXXX 100644 --- a/include/net/net.h +++ b/include/net/net.h @@ -XXX,XX +XXX,XX @@ typedef void (SocketReadStateFinalize)(SocketReadState *rs); typedef void (NetAnnounce)(NetClientState *); typedef bool (SetSteeringEBPF)(NetClientState *, int); typedef bool (NetCheckPeerType)(NetClientState *, ObjectClass *, Error **); +typedef bool (IsWaitIncoming)(NetClientState *); typedef struct vhost_net *(GetVHostNet)(NetClientState *nc); typedef struct NetClientInfo { @@ -XXX,XX +XXX,XX @@ typedef struct NetClientInfo { NetAnnounce *announce; SetSteeringEBPF *set_steering_ebpf; NetCheckPeerType *check_peer_type; + IsWaitIncoming *is_wait_incoming; GetVHostNet *get_vhost_net; const VMStateDescription *backend_vmsd; } NetClientInfo; -- 2.52.0
Support transferring of TAP state (including open fd) through migration stream as part of viritio-net "local-migration". Add new option, incoming-fds, which should be set to true to trigger new logic. For new option require explicitly unset script and downscript, to keep possibility of implementing support for them in future. Note disabling read polling on source stop for TAP migration: otherwise, source process may steal packages from TAP fd even after source vm STOP. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 147 +++++++++++++++++++++++++++++++++++++++++++++++--- qapi/net.json | 10 +++- 2 files changed, 150 insertions(+), 7 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ #include "net/net.h" #include "clients.h" #include "monitor/monitor.h" +#include "system/runstate.h" #include "system/system.h" #include "qapi/error.h" #include "qemu/cutils.h" @@ -XXX,XX +XXX,XX @@ typedef struct TAPState { VHostNetState *vhost_net; unsigned host_vnet_hdr_len; Notifier exit; + + bool read_poll_detached; + VMChangeStateEntry *vmstate; } TAPState; static void launch_script(const char *setup_script, const char *ifname, @@ -XXX,XX +XXX,XX @@ static void launch_script(const char *setup_script, const char *ifname, static void tap_send(void *opaque); static void tap_writable(void *opaque); +static bool tap_is_explicit_no_script(const char *script_arg) +{ + return script_arg && + (script_arg[0] == '\0' || strcmp(script_arg, "no") == 0); +} + static char *tap_parse_script(const char *script_arg, const char *default_path) { g_autofree char *res = g_strdup(script_arg); - if (!res) { - res = get_relocated_path(default_path); + if (tap_is_explicit_no_script(script_arg)) { + return NULL; } - if (res[0] == '\0' || strcmp(res, "no") == 0) { - return NULL; + if (!script_arg) { + return get_relocated_path(default_path); } - return g_steal_pointer(&res); + return g_strdup(script_arg); } static void tap_update_fd_handler(TAPState *s) @@ -XXX,XX +XXX,XX @@ static void tap_read_poll(TAPState *s, bool enable) tap_update_fd_handler(s); } +static void tap_vm_state_change(void *opaque, bool running, RunState state) +{ + TAPState *s = opaque; + + if (running) { + if (s->read_poll_detached) { + tap_read_poll(s, true); + s->read_poll_detached = false; + } + } else if (state == RUN_STATE_FINISH_MIGRATE) { + if (s->read_poll) { + s->read_poll_detached = true; + tap_read_poll(s, false); + } + } +} + static void tap_write_poll(TAPState *s, bool enable) { s->write_poll = enable; @@ -XXX,XX +XXX,XX @@ static void tap_cleanup(NetClientState *nc) s->exit.notify = NULL; } + if (s->vmstate) { + qemu_del_vm_change_state_handler(s->vmstate); + s->vmstate = NULL; + } + tap_read_poll(s, false); tap_write_poll(s, false); close(s->fd); @@ -XXX,XX +XXX,XX @@ static VHostNetState *tap_get_vhost_net(NetClientState *nc) return s->vhost_net; } +static bool tap_is_wait_incoming(NetClientState *nc) +{ + TAPState *s = DO_UPCAST(TAPState, nc, nc); + assert(nc->info->type == NET_CLIENT_DRIVER_TAP); + return s->fd == -1; +} + +static int tap_pre_load(void *opaque) +{ + TAPState *s = opaque; + + if (s->fd != -1) { + error_report( + "TAP is already initialized and cannot receive incoming fd"); + return -EINVAL; + } + + return 0; +} + +static bool tap_setup_vhost(TAPState *s, Error **errp); + +static int tap_post_load(void *opaque, int version_id) +{ + TAPState *s = opaque; + Error *local_err = NULL; + + tap_read_poll(s, true); + + if (s->fd < 0) { + return -1; + } + + if (!tap_setup_vhost(s, &local_err)) { + error_prepend(&local_err, + "Failed to setup vhost during TAP post-load: "); + error_report_err(local_err); + return -1; + } + + return 0; +} + +static const VMStateDescription vmstate_tap = { + .name = "net-tap", + .pre_load = tap_pre_load, + .post_load = tap_post_load, + .fields = (const VMStateField[]) { + VMSTATE_FD(fd, TAPState), + VMSTATE_BOOL(using_vnet_hdr, TAPState), + VMSTATE_BOOL(has_ufo, TAPState), + VMSTATE_BOOL(has_uso, TAPState), + VMSTATE_BOOL(has_tunnel, TAPState), + VMSTATE_BOOL(enabled, TAPState), + VMSTATE_UINT32(host_vnet_hdr_len, TAPState), + VMSTATE_END_OF_LIST() + } +}; + /* fd support */ static NetClientInfo net_tap_info = { @@ -XXX,XX +XXX,XX @@ static NetClientInfo net_tap_info = { .set_vnet_le = tap_set_vnet_le, .set_vnet_be = tap_set_vnet_be, .set_steering_ebpf = tap_set_steering_ebpf, + .is_wait_incoming = tap_is_wait_incoming, .get_vhost_net = tap_get_vhost_net, + .backend_vmsd = &vmstate_tap, }; static TAPState *net_tap_fd_init(NetClientState *peer, @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, int sndbuf = (tap->has_sndbuf && tap->sndbuf) ? MIN(tap->sndbuf, INT_MAX) : INT_MAX; + s->read_poll_detached = false; + s->vmstate = qemu_add_vm_change_state_handler(tap_vm_state_change, s); + if (!tap_set_sndbuf(fd, sndbuf, sndbuf_required ? errp : NULL) && sndbuf_required) { goto failed; @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, return true; failed: + qemu_del_vm_change_state_handler(s->vmstate); + s->vmstate = NULL; qemu_del_net_client(&s->nc); return false; } @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, return -1; } + if (tap->incoming_fds && + (tap->fd || tap->fds || tap->helper || tap->br || tap->ifname || + tap->has_sndbuf || tap->has_vnet_hdr)) { + error_setg(errp, "incoming-fds is incompatible with " + "fd=, fds=, helper=, br=, ifname=, sndbuf= and vnet_hdr="); + return -1; + } + + if (tap->incoming_fds && + !(tap_is_explicit_no_script(tap->script) && + tap_is_explicit_no_script(tap->downscript))) { + /* + * script="" and downscript="" are silently supported to be consistent + * with cases without incoming_fds, but do not care to put this into + * error message. + */ + error_setg(errp, "incoming-fds requires script=no and downscript=no"); + return -1; + } + queues = tap_parse_fds_and_queues(tap, &fds, errp); if (queues < 0) { return -1; @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, goto fail; } - if (fds) { + if (tap->incoming_fds) { + for (i = 0; i < queues; i++) { + NetClientState *nc; + TAPState *s; + + nc = qemu_new_net_client(&net_tap_info, peer, "tap", name); + qemu_set_info_str(nc, "incoming"); + + s = DO_UPCAST(TAPState, nc, nc); + s->fd = -1; + if (vhost_fds) { + s->vhostfd = vhost_fds[i]; + s->vhost_busyloop_timeout = tap->has_poll_us ? tap->poll_us : 0; + } else { + s->vhostfd = -1; + } + } + } else if (fds) { for (i = 0; i < queues; i++) { if (i == 0) { vnet_hdr = tap_probe_vnet_hdr(fds[i], errp); diff --git a/qapi/net.json b/qapi/net.json index XXXXXXX..XXXXXXX 100644 --- a/qapi/net.json +++ b/qapi/net.json @@ -XXX,XX +XXX,XX @@ # @poll-us: maximum number of microseconds that could be spent on busy # polling for tap (since 2.7) # +# @incoming-fds: do not open or create any TAP devices. Prepare for +# getting TAP file descriptors from incoming migration stream. +# The option is incompatible with any of @fd, @fds, @helper, @br, +# @ifname, @sndbuf and @vnet_hdr options, and requires @script and +# @downscript be explicitly set to nothing (empty string or "no") +# (Since 11.1) +# # Since: 1.2 ## { 'struct': 'NetdevTapOptions', @@ -XXX,XX +XXX,XX @@ '*vhostfds': 'str', '*vhostforce': 'bool', '*queues': 'uint32', - '*poll-us': 'uint32'} } + '*poll-us': 'uint32', + '*incoming-fds': 'bool' } } ## # @NetdevSocketOptions: -- 2.52.0
To be used in the next commit: that would be a test for TAP networking, and it will need to setup TAP device. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Reviewed-by: Daniel P. Berrangé <berrange@redhat.com> Reviewed-by: Thomas Huth <thuth@redhat.com> Tested-by: Lei Yang <leiyang@redhat.com> Reviewed-by: Maksim Davydov <davydov-max@yandex-team.ru> --- tests/functional/qemu_test/decorators.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tests/functional/qemu_test/decorators.py b/tests/functional/qemu_test/decorators.py index XXXXXXX..XXXXXXX 100644 --- a/tests/functional/qemu_test/decorators.py +++ b/tests/functional/qemu_test/decorators.py @@ -XXX,XX +XXX,XX @@ import os import platform import resource +import subprocess from unittest import skipIf, skipUnless from .cmd import which @@ -XXX,XX +XXX,XX @@ def skipLockedMemoryTest(locked_memory): ulimit_memory == resource.RLIM_INFINITY or ulimit_memory >= locked_memory * 1024, f'Test required {locked_memory} kB of available locked memory', ) + +''' +Decorator to skip execution of a test if passwordless +sudo command is not available. +''' +def skipWithoutSudo(): + proc = subprocess.run(["sudo", "-n", "/bin/true"], + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + universal_newlines=True, + check=False) + + return skipUnless(proc.returncode == 0, + f'requires password-less sudo access: {proc.stdout}') -- 2.52.0
Add test for a new local-migration migration of virtio-net/tap, with fd passing through UNIX socket. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- tests/functional/x86_64/meson.build | 1 + tests/functional/x86_64/test_tap_migration.py | 456 ++++++++++++++++++ 2 files changed, 457 insertions(+) create mode 100755 tests/functional/x86_64/test_tap_migration.py diff --git a/tests/functional/x86_64/meson.build b/tests/functional/x86_64/meson.build index XXXXXXX..XXXXXXX 100644 --- a/tests/functional/x86_64/meson.build +++ b/tests/functional/x86_64/meson.build @@ -XXX,XX +XXX,XX @@ tests_x86_64_system_thorough = [ 'virtio_balloon', 'virtio_gpu', 'rebuild_vmfd', + 'tap_migration', ] diff --git a/tests/functional/x86_64/test_tap_migration.py b/tests/functional/x86_64/test_tap_migration.py new file mode 100755 index XXXXXXX..XXXXXXX --- /dev/null +++ b/tests/functional/x86_64/test_tap_migration.py @@ -XXX,XX +XXX,XX @@ +#!/usr/bin/env python3 +# +# Functional test that tests TAP local migration +# with fd passing +# +# Copyright (c) Yandex Technologies LLC, 2026 +# +# SPDX-License-Identifier: GPL-2.0-or-later + +import os +import time +import subprocess +from subprocess import run +import signal +import ctypes +import ctypes.util +import unittest +from contextlib import contextmanager, ExitStack +from typing import Tuple + +from qemu_test import ( + LinuxKernelTest, + Asset, + exec_command_and_wait_for_pattern, +) +from qemu_test.decorators import skipWithoutSudo + + +GUEST_IP = "192.168.100.2" +GUEST_IP_MASK = f"{GUEST_IP}/24" +GUEST_MAC = "d6:0d:75:f8:0f:b7" +HOST_IP = "192.168.100.1" +HOST_IP_MASK = f"{HOST_IP}/24" +TAP_ID = "tap0" +TAP_ID2 = "tap1" +TAP_MAC = "e6:1d:44:b5:03:5d" +NETNS = f"qemu_test_ns_{os.getpid()}" + + +def ip(args, check=True) -> None: + """Run ip command with sudo""" + run(["sudo", "ip"] + args, check=check) + + +@contextmanager +def switch_netns(netns_name): + libc = ctypes.CDLL(ctypes.util.find_library("c")) + netns_path = f"/var/run/netns/{netns_name}" + + def switch_to_fd(fd, check: bool = False): + """Switch to netns by file descriptor""" + SYS_setns = 308 + CLONE_NEWNET = 0x40000000 + ret = libc.syscall(SYS_setns, fd, CLONE_NEWNET) + if check and ret != 0: + raise RuntimeError("syscall SETNS failed") + + with ExitStack() as stack: + original_netns_fd = os.open("/proc/self/ns/net", os.O_RDONLY) + stack.callback(os.close, original_netns_fd) + + ip(["netns", "add", netns_name]) + stack.callback(ip, ["netns", "del", netns_name], check=False) + + new_netns_fd = os.open(netns_path, os.O_RDONLY) + stack.callback(os.close, new_netns_fd) + + switch_to_fd(new_netns_fd) + stack.callback(switch_to_fd, original_netns_fd, check=False) + + yield + + +def del_tap(tap_name: str = TAP_ID) -> None: + ip(["tuntap", "del", tap_name, "mode", "tap", "multi_queue"], check=False) + + +def init_tap(tap_name: str = TAP_ID, with_ip: bool = True) -> None: + ip(["tuntap", "add", "dev", tap_name, "mode", "tap", "multi_queue"]) + if with_ip: + ip(["link", "set", "dev", tap_name, "address", TAP_MAC]) + ip(["addr", "add", HOST_IP_MASK, "dev", tap_name]) + ip(["link", "set", tap_name, "up"]) + + +def switch_network_to_tap2() -> None: + ip(["link", "set", TAP_ID2, "down"]) + ip(["link", "set", TAP_ID, "down"]) + ip(["addr", "delete", HOST_IP_MASK, "dev", TAP_ID]) + ip(["link", "set", "dev", TAP_ID2, "address", TAP_MAC]) + ip(["addr", "add", HOST_IP_MASK, "dev", TAP_ID2]) + ip(["link", "set", TAP_ID2, "up"]) + + +def parse_ping_line(line: str) -> float: + # suspect lines like + # [1748524876.590509] 64 bytes from 94.245.155.3 \ + # (94.245.155.3): icmp_seq=1 ttl=250 time=101 ms + spl = line.split() + return float(spl[0][1:-1]) + + +def parse_ping_output(out) -> Tuple[bool, float, float]: + lines = [x for x in out.split("\n") if x.startswith("[")] + + try: + first_no_ans = next( + (ind for ind in range(len(lines)) if lines[ind][20:26] == "no ans") + ) + except StopIteration: + return False, parse_ping_line(lines[0]), parse_ping_line(lines[-1]) + + last_no_ans = next( + ind + for ind in range(len(lines) - 1, -1, -1) + if lines[ind][20:26] == "no ans" + ) + + return ( + True, + parse_ping_line(lines[first_no_ans]), + parse_ping_line(lines[last_no_ans]), + ) + + +def wait_migration_finish(source_vm, target_vm): + migr_events = ( + ("MIGRATION", {"data": {"status": "completed"}}), + ("MIGRATION", {"data": {"status": "failed"}}), + ) + + source_e = source_vm.events_wait(migr_events)["data"] + target_e = target_vm.events_wait(migr_events)["data"] + + source_s = source_vm.cmd("query-status")["status"] + target_s = target_vm.cmd("query-status")["status"] + + assert ( + source_e["status"] == "completed" + and target_e["status"] == "completed" + and source_s == "postmigrate" + and target_s == "paused" + ), f"""Migration failed: + SRC status: {source_s} + SRC event: {source_e} + TGT status: {target_s} + TGT event:{target_e}""" + + +@skipWithoutSudo() +class TAPFdMigration(LinuxKernelTest): + + ASSET_KERNEL = Asset( + ( + "https://archives.fedoraproject.org/pub/archive/fedora/linux/releases" + "/31/Server/x86_64/os/images/pxeboot/vmlinuz" + ), + "d4738d03dbbe083ca610d0821d0a8f1488bebbdccef54ce33e3adb35fda00129", + ) + + ASSET_INITRD = Asset( + ( + "https://archives.fedoraproject.org/pub/archive/fedora/linux/releases" + "/31/Server/x86_64/os/images/pxeboot/initrd.img" + ), + "277cd6c7adf77c7e63d73bbb2cded8ef9e2d3a2f100000e92ff1f8396513cd8b", + ) + + ASSET_ALPINE_ISO = Asset( + ( + "https://dl-cdn.alpinelinux.org/" + "alpine/v3.22/releases/x86_64/alpine-standard-3.22.1-x86_64.iso" + ), + "96d1b44ea1b8a5a884f193526d92edb4676054e9fa903ad2f016441a0fe13089", + ) + + @classmethod + def setUpClass(cls): + super().setUpClass() + + try: + cls.netns_context = switch_netns(NETNS) + cls.netns_context.__enter__() + except (OSError, subprocess.CalledProcessError) as e: + raise unittest.SkipTest(f"can't switch network namespace: {e}") + + @classmethod + def tearDownClass(cls): + if hasattr(cls, "netns_context"): + cls.netns_context.__exit__(None, None, None) + super().tearDownClass() + + def setUp(self): + super().setUp() + + init_tap() + + self.outer_ping_proc = None + self.shm_path = None + + def tearDown(self): + try: + del_tap(TAP_ID) + del_tap(TAP_ID2) + + if self.outer_ping_proc: + self.stop_outer_ping() + + if self.shm_path: + os.unlink(self.shm_path) + finally: + super().tearDown() + + def start_outer_ping(self) -> None: + assert self.outer_ping_proc is None + self.outer_ping_log = self.scratch_file("ping.log") + with open(self.outer_ping_log, "w") as f: + self.outer_ping_proc = subprocess.Popen( + ["ping", "-i", "0", "-O", "-D", GUEST_IP], + text=True, + stdout=f, + ) + + def stop_outer_ping(self) -> str: + assert self.outer_ping_proc + self.outer_ping_proc.send_signal(signal.SIGINT) + + self.outer_ping_proc.communicate(timeout=5) + self.outer_ping_proc = None + + with open(self.outer_ping_log) as f: + return f.read() + + def stop_ping_and_check(self, stop_time, resume_time): + ping_res = self.stop_outer_ping() + + discon, a, b = parse_ping_output(ping_res) + + if not discon: + text = ( + f"STOP: {stop_time}, RESUME: {resume_time}," f"PING: {a} - {b}" + ) + if a > stop_time or b < resume_time: + self.fail(f"PING failed: {text}") + self.log.info(f"PING: no packets lost: {text}") + return + + text = ( + f"STOP: {stop_time}, RESUME: {resume_time}," + f"PING: disconnect: {a} - {b}" + ) + self.log.info(text) + eps = 0.05 + if a < stop_time - eps or b > resume_time + eps: + self.fail(text) + + def one_ping_from_guest(self, vm) -> None: + exec_command_and_wait_for_pattern( + self, + f"ping -c 1 -W 1 {HOST_IP}", + "1 packets transmitted, 1 packets received", + "1 packets transmitted, 0 packets received", + vm=vm, + ) + self.wait_for_console_pattern("# ", vm=vm) + + def one_ping_from_host(self) -> None: + run( + ["ping", "-c", "1", "-W", "1", GUEST_IP], + stdout=subprocess.DEVNULL, + check=True, + ) + + def setup_shared_memory(self): + self.shm_path = f"/dev/shm/qemu_test_{os.getpid()}" + + try: + with open(self.shm_path, "wb") as f: + f.write(b"\0" * (1024 * 1024 * 1024)) # 1GB + except Exception as e: + self.fail(f"Failed to create shared memory file: {e}") + + def prepare_and_launch_vm( + self, shm_path, vhost, incoming=False, vm=None, local=True + ): + if not vm: + vm = self.vm + + vm.set_console() + vm.add_args("-accel", "kvm") + vm.add_args("-device", "pcie-pci-bridge,id=pci.1,bus=pcie.0") + vm.add_args("-m", "1G") + + vm.add_args( + "-object", + f"memory-backend-file,id=ram0,size=1G,mem-path={shm_path},share=on", + ) + vm.add_args("-machine", "memory-backend=ram0") + + vm.add_args( + "-drive", + f"file={self.ASSET_ALPINE_ISO.fetch()},media=cdrom,format=raw", + ) + + vm.add_args("-S") + + if incoming: + vm.add_args("-incoming", "defer") + + vm_s = "target" if incoming else "source" + self.log.info(f"Launching {vm_s} VM") + vm.launch() + + if not local: + tap_name = TAP_ID2 if incoming else TAP_ID + else: + tap_name = TAP_ID + + self.add_virtio_net(vm, vhost, tap_name, local, incoming) + + self.set_migration_capabilities(vm, local) + + def add_virtio_net( + self, vm, vhost: bool, tap_name: str, local: bool, incoming: bool + ): + incoming_fds = local and incoming + netdev_params = { + "id": "netdev.1", + "vhost": vhost, + "type": "tap", + "ifname": tap_name, + "queues": 4, + "vnet_hdr": True, + "incoming-fds": incoming_fds, + "script": "no", + "downscript": "no", + } + + vm.cmd("netdev_add", netdev_params) + + vm.cmd( + "device_add", + driver="virtio-net-pci", + romfile="", + id="vnet.1", + netdev="netdev.1", + mq=True, + vectors=18, + bus="pci.1", + mac=GUEST_MAC, + disable_legacy="off", + local_migration=local, + ) + + def set_migration_capabilities(self, vm, local=True): + vm.cmd( + "migrate-set-capabilities", + { + "capabilities": [ + {"capability": "events", "state": True}, + {"capability": "x-ignore-shared", "state": True}, + ] + }, + ) + vm.cmd("migrate-set-parameters", {"local": local}) + + def setup_guest_network(self) -> None: + exec_command_and_wait_for_pattern(self, "ip addr", "# ") + exec_command_and_wait_for_pattern( + self, + f"ip addr add {GUEST_IP_MASK} dev eth0 && " + "ip link set eth0 up && echo OK", + "OK", + ) + self.wait_for_console_pattern("# ") + + def do_test_tap_fd_migration(self, vhost, local=True): + self.require_accelerator("kvm") + self.set_machine("q35") + + socket_dir = self.socket_dir() + migration_socket = os.path.join(socket_dir.name, "migration.sock") + + self.setup_shared_memory() + + # Setup second TAP if needed + if not local: + del_tap(TAP_ID2) + init_tap(TAP_ID2, with_ip=False) + + self.prepare_and_launch_vm(self.shm_path, vhost, local=local) + self.vm.cmd("cont") + self.wait_for_console_pattern("login:") + exec_command_and_wait_for_pattern(self, "root", "# ") + + self.setup_guest_network() + + self.one_ping_from_guest(self.vm) + self.one_ping_from_host() + self.start_outer_ping() + + # Get some successful pings before migration + time.sleep(0.5) + + target_vm = self.get_vm(name="target") + self.prepare_and_launch_vm( + self.shm_path, + vhost, + incoming=True, + vm=target_vm, + local=local, + ) + + target_vm.cmd("migrate-incoming", {"uri": f"unix:{migration_socket}"}) + + self.log.info("Starting migration") + freeze_start = time.time() + self.vm.cmd("migrate", {"uri": f"unix:{migration_socket}"}) + + self.log.info("Waiting for migration completion") + wait_migration_finish(self.vm, target_vm) + + # Switch network to tap1 if not using local-migration + if not local: + switch_network_to_tap2() + + target_vm.cmd("cont") + freeze_end = time.time() + + self.vm.shutdown() + + self.log.info("Verifying PING on target VM after migration") + self.one_ping_from_guest(target_vm) + self.one_ping_from_host() + + # And a bit more pings after source shutdown + time.sleep(0.3) + self.stop_ping_and_check(freeze_start, freeze_end) + + target_vm.shutdown() + + def test_tap_fd_migration(self): + self.do_test_tap_fd_migration(False) + + def test_tap_fd_migration_vhost(self): + self.do_test_tap_fd_migration(True) + + def test_tap_new_tap_migration(self): + self.do_test_tap_fd_migration(False, local=False) + + def test_tap_new_tap_migration_vhost(self): + self.do_test_tap_fd_migration(True, local=False) + + +if __name__ == "__main__": + LinuxKernelTest.main() -- 2.52.0
Hi all! Here is a migration for TAP net backend, including its properties and open fds. With this new feature, management software doesn't need to initialize new TAP and do a switch to it. Nothing should be done around virtio-net in local migration: it just migrates and continues to use same TAP device. So we avoid extra logic in management software, extra allocations in kernel (for new TAP), and corresponding extra delay in migration downtime. v19: 02: add r-b by Markus 03: pass parameter name to improve error message; info str: "no" -> "" 09: better documentation wording, keep a-b marks 10: new 13: - better wording - keep new property false by default - better error message - move to errp-variants of pre/post _load functions v19 is pushed to https://gitlab.com/vsementsov/qemu.git tag: up-tap-fd-migration-with-bk-opt-v19 To run the test, use sudo, as test needs to configure TAP device: sudo PYTHONPATH=python:tests/functional \ QEMU_TEST_QEMU_BINARY=$PWD/build/qemu-system-x86_64 \ MESON_BUILD_ROOT=$PWD/build \ ./build/pyvenv/bin/python3 tests/functional/x86_64/test_tap_migration.py Or, to test the feature by hand, you may follow the instruction. The walkthrough uses four terminals: source-cmd -- source VM console (serial output, guest login) source-qmp -- QMP connection to the source QEMU target-cmd -- target VM console (serial output, guest login after migration) target-qmp -- QMP connection to the target QEMU 1. Prerequisites ---------------- QEMU=/path/to/your/build/qemu-system-x86_64 # download same image as in test wget -O /tmp/alpine.iso "https://dl-cdn.alpinelinux.org/alpine/v3.22/releases/x86_64/alpine-standard-3.22.1-x86_64.iso" # prepare tap device (be careful to not break your own networks) sudo ip tuntap add dev tap0 mode tap multi_queue sudo ip addr add 192.168.100.1/24 dev tap0 sudo ip link set tap0 up 2. Start source VM ------------------ In source-cmd, run: $QEMU \ -name source \ -machine q35 \ -accel kvm \ -m 1G \ -object memory-backend-file,id=ram0,size=1G,mem-path=/dev/shm/qemu_migration_test,share=on \ -machine memory-backend=ram0 \ -drive file=/tmp/alpine.iso,media=cdrom,format=raw \ -device pcie-pci-bridge,id=pci.1,bus=pcie.0 \ -serial stdio \ -nographic \ -S \ -qmp unix:/tmp/qmp-source.sock,server=on,wait=off Note: the netdev and virtio-net device are added via QMP below for symmetry with the target. On the source you could also pass them on the command line (with local-migration-supported=on). In source-qmp, connect and add the TAP netdev and virtio-net device: socat - UNIX-CONNECT:/tmp/qmp-source.sock {"execute": "qmp_capabilities"} {"execute": "netdev_add", "arguments": { "id": "netdev.1", "type": "tap", "ifname": "tap0", "queues": 4, "vnet_hdr": true, "script": "no", "downscript": "no", "local-migration-supported": true }} {"execute": "device_add", "arguments": { "driver": "virtio-net-pci", "id": "vnet.1", "netdev": "netdev.1", "bus": "pci.1", "mq": true, "vectors": 18, "romfile": "", "disable-legacy": "off" }} {"execute": "cont"} Wait for Alpine to boot in source-cmd. When you see the login prompt, log in as root (no password): localhost login: root Configure the guest network: ip addr add 192.168.100.2/24 dev eth0 ip link set eth0 up Verify connectivity from the guest: ping -c 3 192.168.100.1 And from the host (in any spare terminal): ping -c 3 192.168.100.2 3. Start target VM ------------------ The TAP netdev must be created via QMP after enabling the "local" migration parameter — the target will not open tap0 itself; instead it will receive the TAP file descriptors from the source over the migration channel. In target-cmd, run: $QEMU \ -name target \ -machine q35 \ -accel kvm \ -m 1G \ -object memory-backend-file,id=ram0,size=1G,mem-path=/dev/shm/qemu_migration_test,share=on \ -machine memory-backend=ram0 \ -drive file=/tmp/alpine.iso,media=cdrom,format=raw \ -device pcie-pci-bridge,id=pci.1,bus=pcie.0 \ -serial stdio \ -nographic \ -qmp unix:/tmp/qmp-target.sock,server=on,wait=off \ -incoming defer In target-qmp, connect, enable local migration, and add the TAP netdev and virtio-net device. The "local" parameter must be set before creating the TAP. Do not pass ifname/fd — the fd will arrive via the migration channel: socat - UNIX-CONNECT:/tmp/qmp-target.sock {"execute": "qmp_capabilities"} {"execute": "migrate-set-capabilities", "arguments": { "capabilities": [ {"capability": "events", "state": true}, {"capability": "x-ignore-shared", "state": true} ] }} {"execute": "migrate-set-parameters", "arguments": {"local": true}} {"execute": "netdev_add", "arguments": { "id": "netdev.1", "type": "tap", "queues": 4, "script": "", "downscript": "", "local-migration-supported": true }} {"execute": "device_add", "arguments": { "driver": "virtio-net-pci", "id": "vnet.1", "netdev": "netdev.1", "bus": "pci.1", "mq": true, "vectors": 18, "romfile": "", "disable-legacy": "off" }} 4. Start migration ------------------ In target-qmp, tell the target to listen for the incoming migration: {"execute": "migrate-incoming", "arguments": {"uri": "unix:/tmp/migration.sock"}} In source-qmp, configure migration capabilities and parameters: {"execute": "migrate-set-capabilities", "arguments": { "capabilities": [ {"capability": "events", "state": true}, {"capability": "x-ignore-shared", "state": true} ] }} {"execute": "migrate-set-parameters", "arguments": {"local": true}} In source-qmp, trigger the migration: {"execute": "migrate", "arguments": {"uri": "unix:/tmp/migration.sock"}} Poll migration status until it completes (source-qmp): {"execute": "query-migrate"} # repeat until "status" == "completed" Or just wait for the MIGRATION event that QEMU emits automatically: # {"event": "MIGRATION", "data": {"status": "completed"}, ...} Once the source reports "completed", resume the target VM (target-qmp): {"execute": "cont"} The target VM is now running with the migrated state and the TAP file descriptors that were passed from the source. Still, target console (in target-cmd) may still be empty, until you at least press Enter in it. Verify that the guest is still reachable from the host: ping -c 3 192.168.100.2 And from inside the guest in target-cmd: ping -c 3 192.168.100.1 5. Cleanup ---------- Shut down the target VM (target-qmp): {"execute": "quit"} Shut down the source VM (source-qmp): {"execute": "quit"} Remove the TAP device: sudo ip tuntap del tap0 mode tap multi_queue Remove the shared memory file: rm /dev/shm/qemu_migration_test Remove leftover sockets if they still exist: rm -f /tmp/migration.sock /tmp/qmp-source.sock /tmp/qmp-target.sock Vladimir Sementsov-Ogievskiy (15): net/tap: rework tap_parse_script net/tap: improve script/downscript options documentation net/tap: deprecate "no" as special value for script/downscript net/tap: move vhost-net open() calls to tap_parse_vhost_fds() net/tap: move vhost initialization to tap_setup_vhost() net/tap: use container_of instead of DO_UPCAST net/tap: QOMify tap backend net/tap: add TYPE_VMSTATE_IF interface qapi: add local migration parameter migration/channel: check that transfer is UNIX socket when "local" set virtio-net: support local migration of backend net/tap: disable read polling for stopped VM net/tap: support local migration with virtio-net tests/functional: add skipWithoutSudo() decorator tests/functional: add test_tap_migration docs/about/deprecated.rst | 18 + docs/system/i386/microvm.rst | 4 +- docs/system/i386/xenpvh.rst | 2 +- docs/system/ppc/ppce500.rst | 4 +- docs/system/riscv/microchip-icicle-kit.rst | 2 +- docs/system/riscv/sifive_u.rst | 2 +- hw/net/virtio-net.c | 89 +++- include/hw/virtio/virtio-net.h | 1 + include/migration/misc.h | 2 + include/migration/vmstate.h | 2 + include/net/net.h | 9 + include/net/tap.h | 2 + migration/channel.c | 17 + migration/options.c | 18 +- net/net.c | 14 +- net/tap.c | 441 +++++++++++++---- qapi/migration.json | 15 +- qapi/net.json | 38 +- qemu-options.hx | 12 +- tests/functional/qemu_test/decorators.py | 16 + tests/functional/x86_64/meson.build | 1 + tests/functional/x86_64/test_tap_migration.py | 455 ++++++++++++++++++ 22 files changed, 1057 insertions(+), 107 deletions(-) create mode 100755 tests/functional/x86_64/test_tap_migration.py -- 2.43.0
Factor out tap_is_explicit_no_script() helper, to simplify further changes. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 27 +++++++++++++++++++++------ 1 file changed, 21 insertions(+), 6 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static void launch_script(const char *setup_script, const char *ifname, static void tap_send(void *opaque); static void tap_writable(void *opaque); -static char *tap_parse_script(const char *script_arg, const char *default_path) +static bool tap_is_explicit_no_script(const char *script_arg) { - g_autofree char *res = g_strdup(script_arg); + if (!script_arg) { + return false; + } - if (!res) { - res = get_relocated_path(default_path); + if (script_arg[0] == '\0') { + return true; + } + + if (strcmp(script_arg, "no") == 0) { + return true; } - if (res[0] == '\0' || strcmp(res, "no") == 0) { + return false; +} + +static char *tap_parse_script(const char *script_arg, const char *default_path) +{ + if (tap_is_explicit_no_script(script_arg)) { return NULL; } - return g_steal_pointer(&res); + if (!script_arg) { + return get_relocated_path(default_path); + } + + return g_strdup(script_arg); } static void tap_update_fd_handler(TAPState *s) -- 2.43.0
Properly document defaults and special values of "" and "no". Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Reviewed-by: Markus Armbruster <armbru@redhat.com> --- qapi/net.json | 12 +++++++++--- qemu-options.hx | 9 +++++---- 2 files changed, 14 insertions(+), 7 deletions(-) diff --git a/qapi/net.json b/qapi/net.json index XXXXXXX..XXXXXXX 100644 --- a/qapi/net.json +++ b/qapi/net.json @@ -XXX,XX +XXX,XX @@ # @fds: multiple file descriptors of already opened multiqueue capable # tap # -# @script: script to initialize the interface -# -# @downscript: script to shut down the interface +# @script: script to initialize the interface. An empty string or +# "no" disables script execution. Defaults to +# ``<sysconfdir>/qemu-ifup``, where ``<sysconfdir>`` is the +# system configuration directory at build time (typically /etc). +# +# @downscript: script to shut down the interface. An empty string or +# "no" disables script execution. Defaults to +# ``<sysconfdir>/qemu-ifdown``, where ``<sysconfdir>`` is the +# system configuration directory at build time (typically /etc). # # @br: bridge name (since 2.8) # diff --git a/qemu-options.hx b/qemu-options.hx index XXXXXXX..XXXXXXX 100644 --- a/qemu-options.hx +++ b/qemu-options.hx @@ -XXX,XX +XXX,XX @@ DEF("netdev", HAS_ARG, QEMU_OPTION_netdev, " use network scripts 'file' (default=" DEFAULT_NETWORK_SCRIPT ")\n" " to configure it and 'dfile' (default=" DEFAULT_NETWORK_DOWN_SCRIPT ")\n" " to deconfigure it\n" - " use '[down]script=no' to disable script execution\n" + " use '[down]script=no' or '[down]script=' to disable script execution\n" " use network helper 'helper' (default=" DEFAULT_BRIDGE_HELPER ") to\n" " configure it\n" " use 'fd=h' to connect to an already opened TAP interface\n" @@ -XXX,XX +XXX,XX @@ SRST Use the network script file to configure it and the network script dfile to deconfigure it. If name is not provided, the OS automatically provides one. The default network configure script is - ``/etc/qemu-ifup`` and the default network deconfigure script is - ``/etc/qemu-ifdown``. Use ``script=no`` or ``downscript=no`` to - disable script execution. + ``<sysconfdir>/qemu-ifup`` and the default network deconfigure script is + ``<sysconfdir>/qemu-ifdown``, where ``<sysconfdir>`` is the system + configuration directory at build time (typically ``/etc``). + Use ``[down]script=no`` or ``[down]script=`` to disable script execution. If running QEMU as an unprivileged user, use the network helper to configure the TAP interface and attach it to the bridge. -- 2.43.0
The interface is ambiguous, as "no" is valid file name. So, using "no" as a special value to disable script is deprecated. Use an empty string ("script=" / "downscript=") instead. In a future version, "no" will be treated as a plain file name, just like any other non-empty value. Document the deprecation in docs/about/deprecated.rst, qapi/net.json, and qemu-options.hx. Update other docs to use empty string instead of "no". Add a warning. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- docs/about/deprecated.rst | 18 ++++++++++++++ docs/system/i386/microvm.rst | 4 +-- docs/system/i386/xenpvh.rst | 2 +- docs/system/ppc/ppce500.rst | 4 +-- docs/system/riscv/microchip-icicle-kit.rst | 2 +- docs/system/riscv/sifive_u.rst | 2 +- net/tap.c | 29 ++++++++++++++-------- qapi/net.json | 12 ++++++--- qemu-options.hx | 7 ++++-- 9 files changed, 56 insertions(+), 24 deletions(-) diff --git a/docs/about/deprecated.rst b/docs/about/deprecated.rst index XXXXXXX..XXXXXXX 100644 --- a/docs/about/deprecated.rst +++ b/docs/about/deprecated.rst @@ -XXX,XX +XXX,XX @@ flexible enough. The monitor objects have been converted to QOM, so ``-mon mode=control`` is replaced by ``-object monitor-qmp``. The short convenience options are not deprecated, only ``-mon``. +``script=no`` and ``downscript=no`` for ``-netdev tap`` (since 11.2) +''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''' + +The special value ``"no"`` for the ``script`` and ``downscript`` +parameters of ``-netdev tap`` disables script execution. This special +treatment of ``"no"`` is deprecated. Use an empty string (``script=`` +or ``downscript=``) to disable script execution instead. In a future +version, ``"no"`` will be treated as a plain file name. + QEMU Machine Protocol (QMP) commands ------------------------------------ @@ -XXX,XX +XXX,XX @@ Use ``job-finalize`` instead. Use ``query-accelerators`` instead. +``"no"`` as value of ``script``/``downscript`` for tap in ``netdev_add`` (since 11.2) +''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''''' + +The special value ``"no"`` for the ``script`` and ``downscript`` +parameters of ``netdev_add`` with ``type=tap`` disables script +execution. This special treatment of ``"no"`` is deprecated. Use an +empty string instead. In a future version, ``"no"`` will be treated as +a plain file name. + Human Machine Protocol (HMP) commands ------------------------------------- diff --git a/docs/system/i386/microvm.rst b/docs/system/i386/microvm.rst index XXXXXXX..XXXXXXX 100644 --- a/docs/system/i386/microvm.rst +++ b/docs/system/i386/microvm.rst @@ -XXX,XX +XXX,XX @@ legacy ``ISA serial`` device as console:: -serial stdio \ -drive id=test,file=test.img,format=raw,if=none \ -device virtio-blk-device,drive=test \ - -netdev tap,id=tap0,script=no,downscript=no \ + -netdev tap,id=tap0,script=,downscript= \ -device virtio-net-device,netdev=tap0 While the example above works, you might be interested in reducing the @@ -XXX,XX +XXX,XX @@ disabled:: -device virtconsole,chardev=virtiocon0 \ -drive id=test,file=test.img,format=raw,if=none \ -device virtio-blk-device,drive=test \ - -netdev tap,id=tap0,script=no,downscript=no \ + -netdev tap,id=tap0,script=,downscript= \ -device virtio-net-device,netdev=tap0 diff --git a/docs/system/i386/xenpvh.rst b/docs/system/i386/xenpvh.rst index XXXXXXX..XXXXXXX 100644 --- a/docs/system/i386/xenpvh.rst +++ b/docs/system/i386/xenpvh.rst @@ -XXX,XX +XXX,XX @@ case you need to construct one manually: -vnc none \ -display none \ -device virtio-net-pci,id=nic0,netdev=net0,mac=00:16:3e:5c:81:78 \ - -netdev type=tap,id=net0,ifname=vif3.0-emu,br=xenbr0,script=no,downscript=no \ + -netdev type=tap,id=net0,ifname=vif3.0-emu,br=xenbr0,script=,downscript= \ -smp 4,maxcpus=4 \ -nographic \ -machine xenpvh,ram-low-base=0,ram-low-size=2147483648,ram-high-base=4294967296,ram-high-size=2147483648,pci-ecam-base=824633720832,pci-ecam-size=268435456,pci-mmio-base=4026531840,pci-mmio-size=33554432,pci-mmio-high-base=824902156288,pci-mmio-high-size=68719476736 \ diff --git a/docs/system/ppc/ppce500.rst b/docs/system/ppc/ppce500.rst index XXXXXXX..XXXXXXX 100644 --- a/docs/system/ppc/ppce500.rst +++ b/docs/system/ppc/ppce500.rst @@ -XXX,XX +XXX,XX @@ interface at PCI address 0.1.0, but we can switch that to an e1000 NIC by: $ qemu-system-ppc64 -M ppce500 -smp 4 -m 2G \ -display none -serial stdio \ -bios u-boot \ - -nic tap,ifname=tap0,script=no,downscript=no,model=e1000 + -nic tap,ifname=tap0,script=,downscript=,model=e1000 The QEMU ``ppce500`` machine can also dynamically instantiate an eTSEC device if “-device eTSEC” is given to QEMU: .. code-block:: bash - -netdev tap,ifname=tap0,script=no,downscript=no,id=net0 -device eTSEC,netdev=net0 + -netdev tap,ifname=tap0,script=,downscript=,id=net0 -device eTSEC,netdev=net0 Root file system on flash drive ------------------------------- diff --git a/docs/system/riscv/microchip-icicle-kit.rst b/docs/system/riscv/microchip-icicle-kit.rst index XXXXXXX..XXXXXXX 100644 --- a/docs/system/riscv/microchip-icicle-kit.rst +++ b/docs/system/riscv/microchip-icicle-kit.rst @@ -XXX,XX +XXX,XX @@ Then we can boot the machine by: $ qemu-system-riscv64 -M microchip-icicle-kit -smp 5 -m 2G \ -sd path/to/sdcard.img \ -nic user,model=cadence_gem \ - -nic tap,ifname=tap,model=cadence_gem,script=no \ + -nic tap,ifname=tap,model=cadence_gem,script= \ -display none -serial stdio \ -kernel path/to/u-boot/build/dir/u-boot.bin \ -dtb path/to/u-boot/build/dir/u-boot.dtb diff --git a/docs/system/riscv/sifive_u.rst b/docs/system/riscv/sifive_u.rst index XXXXXXX..XXXXXXX 100644 --- a/docs/system/riscv/sifive_u.rst +++ b/docs/system/riscv/sifive_u.rst @@ -XXX,XX +XXX,XX @@ To boot the VxWorks kernel in QEMU with the ``sifive_u`` machine, use: $ qemu-system-riscv64 -M sifive_u -smp 5 -m 2G \ -display none -serial stdio \ - -nic tap,ifname=tap0,script=no,downscript=no \ + -nic tap,ifname=tap0,script=,downscript= \ -kernel /path/to/vxWorks \ -append "gem(0,0)host:vxWorks h=192.168.200.1 e=192.168.200.2:ffffff00 u=target pw=vxTarget f=0x01" diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static void launch_script(const char *setup_script, const char *ifname, static void tap_send(void *opaque); static void tap_writable(void *opaque); -static bool tap_is_explicit_no_script(const char *script_arg) +static bool tap_is_explicit_no_script(const char *script_arg_name, + const char *script_arg_value) { - if (!script_arg) { + if (!script_arg_value) { return false; } - if (script_arg[0] == '\0') { + if (script_arg_value[0] == '\0') { return true; } - if (strcmp(script_arg, "no") == 0) { + if (strcmp(script_arg_value, "no") == 0) { + warn_report("%s=no is deprecated; use %s= instead " + "(empty string instead of 'no')", + script_arg_name, script_arg_name); return true; } return false; } -static char *tap_parse_script(const char *script_arg, const char *default_path) +static char *tap_parse_script(const char *script_arg_name, + const char *script_arg_value, + const char *default_path) { - if (tap_is_explicit_no_script(script_arg)) { + if (tap_is_explicit_no_script(script_arg_name, script_arg_value)) { return NULL; } - if (!script_arg) { + if (!script_arg_value) { return get_relocated_path(default_path); } - return g_strdup(script_arg); + return g_strdup(script_arg_value); } static void tap_update_fd_handler(TAPState *s) @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, qemu_set_info_str(&s->nc, "helper=%s", tap->helper); } else { qemu_set_info_str(&s->nc, "ifname=%s,script=%s,downscript=%s", ifname, - script ?: "no", downscript ?: "no"); + script ?: "", downscript ?: ""); if (downscript) { snprintf(s->down_script, sizeof(s->down_script), "%s", downscript); @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, } } else { g_autofree char *script = - tap_parse_script(tap->script, DEFAULT_NETWORK_SCRIPT); + tap_parse_script("script", tap->script, DEFAULT_NETWORK_SCRIPT); g_autofree char *downscript = - tap_parse_script(tap->downscript, DEFAULT_NETWORK_DOWN_SCRIPT); + tap_parse_script("downscript", tap->downscript, + DEFAULT_NETWORK_DOWN_SCRIPT); if (tap->ifname) { pstrcpy(ifname, sizeof ifname, tap->ifname); diff --git a/qapi/net.json b/qapi/net.json index XXXXXXX..XXXXXXX 100644 --- a/qapi/net.json +++ b/qapi/net.json @@ -XXX,XX +XXX,XX @@ # @fds: multiple file descriptors of already opened multiqueue capable # tap # -# @script: script to initialize the interface. An empty string or -# "no" disables script execution. Defaults to +# @script: script to initialize the interface. An empty string +# disables script execution. Defaults to # ``<sysconfdir>/qemu-ifup``, where ``<sysconfdir>`` is the # system configuration directory at build time (typically /etc). +# Using "no" to disable script execution is deprecated (since +# 11.2); use an empty string instead. # -# @downscript: script to shut down the interface. An empty string or -# "no" disables script execution. Defaults to +# @downscript: script to shut down the interface. An empty string +# disables script execution. Defaults to # ``<sysconfdir>/qemu-ifdown``, where ``<sysconfdir>`` is the # system configuration directory at build time (typically /etc). +# Using "no" to disable script execution is deprecated (since +# 11.2); use an empty string instead. # # @br: bridge name (since 2.8) # diff --git a/qemu-options.hx b/qemu-options.hx index XXXXXXX..XXXXXXX 100644 --- a/qemu-options.hx +++ b/qemu-options.hx @@ -XXX,XX +XXX,XX @@ DEF("netdev", HAS_ARG, QEMU_OPTION_netdev, " use network scripts 'file' (default=" DEFAULT_NETWORK_SCRIPT ")\n" " to configure it and 'dfile' (default=" DEFAULT_NETWORK_DOWN_SCRIPT ")\n" " to deconfigure it\n" - " use '[down]script=no' or '[down]script=' to disable script execution\n" + " use '[down]script=' to disable script execution\n" + " ('[down]script=no' is deprecated and will be treated as a file name in future)\n" " use network helper 'helper' (default=" DEFAULT_BRIDGE_HELPER ") to\n" " configure it\n" " use 'fd=h' to connect to an already opened TAP interface\n" @@ -XXX,XX +XXX,XX @@ SRST ``<sysconfdir>/qemu-ifup`` and the default network deconfigure script is ``<sysconfdir>/qemu-ifdown``, where ``<sysconfdir>`` is the system configuration directory at build time (typically ``/etc``). - Use ``[down]script=no`` or ``[down]script=`` to disable script execution. + Use ``[down]script=`` to disable script execution. + Using ``[down]script=no`` is deprecated; in a future version it will + be treated as a plain file name. If running QEMU as an unprivileged user, use the network helper to configure the TAP interface and attach it to the bridge. -- 2.43.0
1. Simplify code path: get vhostfds for all cases in one function. 2. Prepare for further tap-fd-migraton feature, when we'll need to postpone vhost initialization up to post-load stage. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Reviewed-by: Ben Chaney <bchaney@akamai.com> --- net/tap.c | 39 ++++++++++++++++++++++----------------- 1 file changed, 22 insertions(+), 17 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, } } - if (tap->has_vhost ? tap->vhost : - (vhostfd != -1) || (tap->has_vhostforce && tap->vhostforce)) { + if (vhostfd != -1) { VhostNetOptions options; options.backend_type = VHOST_BACKEND_TYPE_KERNEL; @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, } else { options.busyloop_timeout = 0; } - - if (vhostfd == -1) { - vhostfd = open("/dev/vhost-net", O_RDWR); - if (vhostfd < 0) { - error_setg_file_open(errp, errno, "/dev/vhost-net"); - goto failed; - } - if (!qemu_set_blocking(vhostfd, false, errp)) { - goto failed; - } - } options.opaque = (void *)(uintptr_t)vhostfd; options.nvqs = 2; options.feature_bits = kernel_feature_bits; @@ -XXX,XX +XXX,XX @@ static int tap_parse_fds_and_queues(const NetdevTapOptions *tap, int **fds, static bool tap_parse_vhost_fds(const NetdevTapOptions *tap, int **vhost_fds, int queues, Error **errp) { - if (!(tap->vhostfd || tap->vhostfds)) { + bool need_vhost = tap->has_vhost ? tap->vhost : + ((tap->vhostfd || tap->vhostfds) || + (tap->has_vhostforce && tap->vhostforce)); + + if (!need_vhost) { *vhost_fds = NULL; return true; } - if (net_parse_fds(tap->vhostfd ?: tap->vhostfds, - vhost_fds, queues, errp) < 0) { - return false; + if (tap->vhostfd || tap->vhostfds) { + if (net_parse_fds(tap->vhostfd ?: tap->vhostfds, + vhost_fds, queues, errp) < 0) { + return false; + } + } else { + *vhost_fds = g_new(int, queues); + for (int i = 0; i < queues; i++) { + int vhostfd = open("/dev/vhost-net", O_RDWR); + if (vhostfd < 0) { + error_setg_file_open(errp, errno, "/dev/vhost-net"); + net_free_fds(*vhost_fds, i); + return false; + } + (*vhost_fds)[i] = vhostfd; + } } if (!unblock_fds(*vhost_fds, queues, errp)) { -- 2.43.0
Make a new helper function in a way it can be reused later for TAP fd-migration feature: we'll need to initialize vhost in a later point when we doesn't have access to QAPI parameters. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Reviewed-by: Ben Chaney <bchaney@akamai.com> --- net/tap.c | 62 ++++++++++++++++++++++++++++++++++--------------------- 1 file changed, 38 insertions(+), 24 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static const int kernel_feature_bits[] = { typedef struct TAPState { NetClientState nc; int fd; + int vhostfd; + uint32_t vhost_busyloop_timeout; char down_script[1024]; char down_script_arg[128]; uint8_t buf[NET_BUFSIZE]; @@ -XXX,XX +XXX,XX @@ static int net_tap_init(const NetdevTapOptions *tap, int *vnet_hdr, return fd; } +static bool tap_setup_vhost(TAPState *s, Error **errp) +{ + VhostNetOptions options; + + if (s->vhostfd == -1) { + return true; + } + + options.backend_type = VHOST_BACKEND_TYPE_KERNEL; + options.net_backend = &s->nc; + options.busyloop_timeout = s->vhost_busyloop_timeout; + options.opaque = (void *)(uintptr_t)s->vhostfd; + options.nvqs = 2; + options.feature_bits = kernel_feature_bits; + options.get_acked_features = NULL; + options.save_acked_features = NULL; + options.max_tx_queue_size = 0; + options.is_vhost_user = false; + + s->vhost_net = vhost_net_init(&options); + if (!s->vhost_net) { + error_setg(errp, + "vhost-net requested but could not be initialized"); + return false; + } + + /* vhostfd ownership is passed to s->vhost_net */ + s->vhostfd = -1; + + return true; +} + static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, const char *name, const char *ifname, const char *script, @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, } } - if (vhostfd != -1) { - VhostNetOptions options; - - options.backend_type = VHOST_BACKEND_TYPE_KERNEL; - options.net_backend = &s->nc; - if (tap->has_poll_us) { - options.busyloop_timeout = tap->poll_us; - } else { - options.busyloop_timeout = 0; - } - options.opaque = (void *)(uintptr_t)vhostfd; - options.nvqs = 2; - options.feature_bits = kernel_feature_bits; - options.get_acked_features = NULL; - options.save_acked_features = NULL; - options.max_tx_queue_size = 0; - options.is_vhost_user = false; - - s->vhost_net = vhost_net_init(&options); - if (!s->vhost_net) { - error_setg(errp, - "vhost-net requested but could not be initialized"); - goto failed; - } + s->vhostfd = vhostfd; + s->vhost_busyloop_timeout = tap->has_poll_us ? tap->poll_us : 0; + if (!tap_setup_vhost(s, errp)) { + return false; } return true; -- 2.43.0
We are going to QOMify tap backend, which includes deriving TAPState from Object. So "NetClientState nc" will not be a first member. Let's parepare for this change, and use container_of(), which will work regardless position of "nc" field. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 36 ++++++++++++++++++------------------ 1 file changed, 18 insertions(+), 18 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ static ssize_t tap_write_packet(TAPState *s, const struct iovec *iov, int iovcnt static ssize_t tap_receive_iov(NetClientState *nc, const struct iovec *iov, int iovcnt) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); const struct iovec *iovp = iov; g_autofree struct iovec *iov_copy = NULL; struct virtio_net_hdr hdr = { }; @@ -XXX,XX +XXX,XX @@ ssize_t tap_read_packet(int tapfd, uint8_t *buf, int maxlen) static void tap_send_completed(NetClientState *nc, ssize_t len) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); tap_read_poll(s, true); } @@ -XXX,XX +XXX,XX @@ static void tap_send(void *opaque) static bool tap_has_ufo(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); @@ -XXX,XX +XXX,XX @@ static bool tap_has_ufo(NetClientState *nc) static bool tap_has_uso(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); @@ -XXX,XX +XXX,XX @@ static bool tap_has_uso(NetClientState *nc) static bool tap_has_tunnel(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); return s->has_tunnel; @@ -XXX,XX +XXX,XX @@ static bool tap_has_tunnel(NetClientState *nc) static bool tap_has_vnet_hdr(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); @@ -XXX,XX +XXX,XX @@ static bool tap_has_vnet_hdr_len(NetClientState *nc, int len) static void tap_set_vnet_hdr_len(NetClientState *nc, int len) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); @@ -XXX,XX +XXX,XX @@ static void tap_set_vnet_hdr_len(NetClientState *nc, int len) static int tap_set_vnet_le(NetClientState *nc, bool is_le) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); return tap_fd_set_vnet_le(s->fd, is_le); } static int tap_set_vnet_be(NetClientState *nc, bool is_be) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); return tap_fd_set_vnet_be(s->fd, is_be); } static void tap_set_offload(NetClientState *nc, const NetOffloads *ol) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); if (s->fd < 0) { return; } @@ -XXX,XX +XXX,XX @@ static void tap_exit_notify(Notifier *notifier, void *data) static void tap_cleanup(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); if (s->vhost_net) { vhost_net_cleanup(s->vhost_net); @@ -XXX,XX +XXX,XX @@ static void tap_cleanup(NetClientState *nc) static void tap_poll(NetClientState *nc, bool enable) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); tap_read_poll(s, enable); tap_write_poll(s, enable); } static bool tap_set_steering_ebpf(NetClientState *nc, int prog_fd) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); return tap_fd_set_steering_ebpf(s->fd, prog_fd) == 0; @@ -XXX,XX +XXX,XX @@ static bool tap_set_steering_ebpf(NetClientState *nc, int prog_fd) int tap_get_fd(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); return s->fd; } @@ -XXX,XX +XXX,XX @@ int tap_get_fd(NetClientState *nc) */ static VHostNetState *tap_get_vhost_net(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); assert(nc->info->type == NET_CLIENT_DRIVER_TAP); return s->vhost_net; } @@ -XXX,XX +XXX,XX @@ static TAPState *net_tap_fd_init(NetClientState *peer, nc = qemu_new_net_client(&net_tap_info, peer, model, name); - s = DO_UPCAST(TAPState, nc, nc); + s = container_of(nc, TAPState, nc); s->fd = fd; s->host_vnet_hdr_len = vnet_hdr ? sizeof(struct virtio_net_hdr) : 0; @@ -XXX,XX +XXX,XX @@ fail: int tap_enable(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); int ret; if (s->enabled) { @@ -XXX,XX +XXX,XX @@ int tap_enable(NetClientState *nc) int tap_disable(NetClientState *nc) { - TAPState *s = DO_UPCAST(TAPState, nc, nc); + TAPState *s = container_of(nc, TAPState, nc); int ret; if (s->enabled == 0) { -- 2.43.0
We prepare for being able to migrate TAP backend. We'll need a user change-able property for it, which can be set from machine type. So, let's QOMify it first. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- include/net/net.h | 7 +++++++ include/net/tap.h | 2 ++ net/net.c | 14 +++++++------- net/tap.c | 48 +++++++++++++++++++++++++++++++++++++++-------- 4 files changed, 56 insertions(+), 15 deletions(-) diff --git a/include/net/net.h b/include/net/net.h index XXXXXXX..XXXXXXX 100644 --- a/include/net/net.h +++ b/include/net/net.h @@ -XXX,XX +XXX,XX @@ char *qemu_mac_strdup_printf(const uint8_t *macaddr); NetClientState *qemu_find_netdev(const char *id); int qemu_find_net_clients_except(const char *id, NetClientState **ncs, NetClientDriver type, int max); +void qemu_net_client_setup(NetClientState *nc, + NetClientInfo *info, + NetClientState *peer, + const char *model, + const char *name, + NetClientDestructor *destructor, + bool is_datapath); NetClientState *qemu_new_net_client(NetClientInfo *info, NetClientState *peer, const char *model, diff --git a/include/net/tap.h b/include/net/tap.h index XXXXXXX..XXXXXXX 100644 --- a/include/net/tap.h +++ b/include/net/tap.h @@ -XXX,XX +XXX,XX @@ #include "standard-headers/linux/virtio_net.h" +#define TYPE_TAP_NETDEV "tap-netdev" + int tap_enable(NetClientState *nc); int tap_disable(NetClientState *nc); diff --git a/net/net.c b/net/net.c index XXXXXXX..XXXXXXX 100644 --- a/net/net.c +++ b/net/net.c @@ -XXX,XX +XXX,XX @@ static ssize_t qemu_deliver_packet_iov(NetClientState *sender, int iovcnt, void *opaque); -static void qemu_net_client_setup(NetClientState *nc, - NetClientInfo *info, - NetClientState *peer, - const char *model, - const char *name, - NetClientDestructor *destructor, - bool is_datapath) +void qemu_net_client_setup(NetClientState *nc, + NetClientInfo *info, + NetClientState *peer, + const char *model, + const char *name, + NetClientDestructor *destructor, + bool is_datapath) { nc->info = info; nc->model = g_strdup(model); diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ #include "qemu/main-loop.h" #include "qemu/sockets.h" #include "hw/virtio/vhost.h" +#include "qom/object.h" #include "net/tap.h" #include "net/util.h" @@ -XXX,XX +XXX,XX @@ static const int kernel_feature_bits[] = { VHOST_INVALID_FEATURE_BIT }; -typedef struct TAPState { +OBJECT_DECLARE_SIMPLE_TYPE(TAPState, TAP_NETDEV) + +struct TAPState { + Object parent_obj; + NetClientState nc; int fd; int vhostfd; @@ -XXX,XX +XXX,XX @@ typedef struct TAPState { VHostNetState *vhost_net; unsigned host_vnet_hdr_len; Notifier exit; -} TAPState; +}; static void launch_script(const char *setup_script, const char *ifname, int fd, Error **errp); @@ -XXX,XX +XXX,XX @@ static VHostNetState *tap_get_vhost_net(NetClientState *nc) return s->vhost_net; } + +static const TypeInfo tap_netdev_info = { + .name = TYPE_TAP_NETDEV, + .parent = TYPE_OBJECT, + .instance_size = sizeof(TAPState), +}; + +static void tap_net_client_destructor(NetClientState *nc) +{ + TAPState *s = container_of(nc, TAPState, nc); + object_unref(OBJECT(s)); +} + /* fd support */ static NetClientInfo net_tap_info = { @@ -XXX,XX +XXX,XX @@ static NetClientInfo net_tap_info = { .get_vhost_net = tap_get_vhost_net, }; +static TAPState *new_tap(NetClientState *peer, + const char *model, + const char *name) +{ + TAPState *s = TAP_NETDEV(object_new(TYPE_TAP_NETDEV)); + + qemu_net_client_setup(&s->nc, &net_tap_info, peer, model, name, + tap_net_client_destructor, true); + + return s; +} + static TAPState *net_tap_fd_init(NetClientState *peer, const char *model, const char *name, @@ -XXX,XX +XXX,XX @@ static TAPState *net_tap_fd_init(NetClientState *peer, int vnet_hdr) { NetOffloads ol = {}; - NetClientState *nc; - TAPState *s; - - nc = qemu_new_net_client(&net_tap_info, peer, model, name); - - s = container_of(nc, TAPState, nc); + TAPState *s = new_tap(peer, model, name); s->fd = fd; s->host_vnet_hdr_len = vnet_hdr ? sizeof(struct virtio_net_hdr) : 0; @@ -XXX,XX +XXX,XX @@ int tap_disable(NetClientState *nc) return ret; } } + +static void tap_register_types(void) +{ + type_register_static(&tap_netdev_info); +} + +type_init(tap_register_types) -- 2.43.0
We'll need it to implement TAP backend live migration. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 43 ++++++++++++++++++++++++++++++++++--------- 1 file changed, 34 insertions(+), 9 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ #include "qemu/main-loop.h" #include "qemu/sockets.h" #include "hw/virtio/vhost.h" -#include "qom/object.h" #include "net/tap.h" #include "net/util.h" @@ -XXX,XX +XXX,XX @@ struct TAPState { VHostNetState *vhost_net; unsigned host_vnet_hdr_len; Notifier exit; + + int queue_index; }; static void launch_script(const char *setup_script, const char *ifname, @@ -XXX,XX +XXX,XX @@ static VHostNetState *tap_get_vhost_net(NetClientState *nc) } +static char *tap_vmstate_if_get_id(VMStateIf *obj) +{ + TAPState *s = TAP_NETDEV(obj); + char *res = g_strdup_printf("%s/%d", s->nc.name, s->queue_index); + return res; +} + +static void tap_class_init(ObjectClass *klass, const void *data) +{ + VMStateIfClass *vc = VMSTATE_IF_CLASS(klass); + + vc->get_id = tap_vmstate_if_get_id; +} + static const TypeInfo tap_netdev_info = { .name = TYPE_TAP_NETDEV, .parent = TYPE_OBJECT, .instance_size = sizeof(TAPState), + .class_init = tap_class_init, + .interfaces = (const InterfaceInfo[]) { + { TYPE_VMSTATE_IF }, + { } + }, }; static void tap_net_client_destructor(NetClientState *nc) @@ -XXX,XX +XXX,XX @@ static NetClientInfo net_tap_info = { static TAPState *new_tap(NetClientState *peer, const char *model, - const char *name) + const char *name, + int queue_index) { TAPState *s = TAP_NETDEV(object_new(TYPE_TAP_NETDEV)); qemu_net_client_setup(&s->nc, &net_tap_info, peer, model, name, tap_net_client_destructor, true); + s->queue_index = queue_index; + return s; } @@ -XXX,XX +XXX,XX @@ static TAPState *net_tap_fd_init(NetClientState *peer, const char *model, const char *name, int fd, - int vnet_hdr) + int vnet_hdr, + int queue_index) { NetOffloads ol = {}; - TAPState *s = new_tap(peer, model, name); + TAPState *s = new_tap(peer, model, name, queue_index); s->fd = fd; s->host_vnet_hdr_len = vnet_hdr ? sizeof(struct virtio_net_hdr) : 0; @@ -XXX,XX +XXX,XX @@ int net_init_bridge(const Netdev *netdev, const char *name, close(fd); return -1; } - s = net_tap_fd_init(peer, "bridge", name, fd, vnet_hdr); + s = net_tap_fd_init(peer, "bridge", name, fd, vnet_hdr, 0); qemu_set_info_str(&s->nc, "helper=%s,br=%s", helper, br); @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, const char *name, const char *ifname, const char *script, const char *downscript, int vhostfd, - int vnet_hdr, int fd, Error **errp) + int vnet_hdr, int fd, int queue_index, + Error **errp) { TAPState *s = net_tap_fd_init(peer, tap->helper ? "bridge" : "tap", - name, fd, vnet_hdr); + name, fd, vnet_hdr, queue_index); bool sndbuf_required = tap->has_sndbuf; int sndbuf = (tap->has_sndbuf && tap->sndbuf) ? MIN(tap->sndbuf, INT_MAX) : INT_MAX; @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, if (!net_init_tap_one(tap, peer, name, ifname, NULL, NULL, vhost_fds ? vhost_fds[i] : -1, - vnet_hdr, fds[i], errp)) { + vnet_hdr, fds[i], i, errp)) { goto fail; } } @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, i >= 1 ? NULL : script, i >= 1 ? NULL : downscript, vhost_fds ? vhost_fds[i] : -1, - vnet_hdr, fd, errp)) { + vnet_hdr, fd, i, errp)) { goto fail; } } -- 2.43.0
We are going to implement local-migration feature: some devices will be able to transfer open file descriptors through migration stream (which must UNIX domain socket for that purpose). This allows to transfer the whole backend state without reconnecting and restarting the backend service. For example, virtio-net will migrate its attached TAP netdev, together with its connected file descriptors. In this commit we introduce a migration parameter, which enables the feature for devices that support it (none at the moment). Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Acked-by: Markus Armbruster <armbru@redhat.com> Acked-by: Peter Xu <peterx@redhat.com> Reviewed-by: Ben Chaney <bchaney@akamai.com> --- include/migration/misc.h | 2 ++ migration/options.c | 18 +++++++++++++++++- qapi/migration.json | 15 +++++++++++++-- 3 files changed, 32 insertions(+), 3 deletions(-) diff --git a/include/migration/misc.h b/include/migration/misc.h index XXXXXXX..XXXXXXX 100644 --- a/include/migration/misc.h +++ b/include/migration/misc.h @@ -XXX,XX +XXX,XX @@ bool multifd_join_device_state_save_threads(void); void migration_request_switchover_ack_legacy(const char *requester); +bool migrate_local(void); + #endif diff --git a/migration/options.c b/migration/options.c index XXXXXXX..XXXXXXX 100644 --- a/migration/options.c +++ b/migration/options.c @@ -XXX,XX +XXX,XX @@ #include "qemu/osdep.h" #include "qemu/error-report.h" #include "qemu/units.h" +#include "qapi/util.h" #include "exec/target_page.h" #include "qapi/clone-visitor.h" #include "qapi/error.h" @@ -XXX,XX +XXX,XX @@ #include "migration/colo.h" #include "migration/cpr.h" #include "migration/misc.h" +#include "migration/options.h" #include "migration.h" #include "migration-stats.h" #include "qemu-file.h" @@ -XXX,XX +XXX,XX @@ bool migrate_mapped_ram(void) return s->capabilities[MIGRATION_CAPABILITY_MAPPED_RAM]; } +bool migrate_local(void) +{ + MigrationState *s = migrate_get_current(); + return s->parameters.local; +} + bool migrate_ignore_shared(void) { MigrationState *s = migrate_get_current(); @@ -XXX,XX +XXX,XX @@ static void migrate_mark_all_params_present(MigrationParameters *p) &p->has_announce_step, &p->has_block_bitmap_mapping, &p->has_x_vcpu_dirty_limit_period, &p->has_vcpu_dirty_limit, &p->has_mode, &p->has_zero_page_detection, &p->has_direct_io, - &p->has_x_rdma_chunk_size, &p->has_cpr_exec_command, + &p->has_x_rdma_chunk_size, &p->has_cpr_exec_command, &p->has_local, }; len = ARRAY_SIZE(has_fields); @@ -XXX,XX +XXX,XX @@ static void migrate_params_test_apply(MigrationParameters *params, qapi_free_strList(dest->cpr_exec_command); dest->cpr_exec_command = QAPI_CLONE(strList, params->cpr_exec_command); } + + if (params->has_local) { + dest->local = params->local; + } } static void migrate_params_apply(MigrationParameters *params) @@ -XXX,XX +XXX,XX @@ static void migrate_params_apply(MigrationParameters *params) s->parameters.cpr_exec_command = QAPI_CLONE(strList, params->cpr_exec_command); } + + if (params->has_local) { + s->parameters.local = params->local; + } } void qmp_migrate_set_parameters(MigrationParameters *params, Error **errp) diff --git a/qapi/migration.json b/qapi/migration.json index XXXXXXX..XXXXXXX 100644 --- a/qapi/migration.json +++ b/qapi/migration.json @@ -XXX,XX +XXX,XX @@ 'zero-page-detection', 'direct-io', { 'name': 'x-rdma-chunk-size', 'features': [ 'unstable' ] }, - 'cpr-exec-command'] } + 'cpr-exec-command', + 'local'] } ## # @migrate-set-parameters: @@ -XXX,XX +XXX,XX @@ # Must be set to the same value on both source and destination # before migration starts. (Since 11.1) # +# @local: Permit the use of optimizations for local migration. +# This must only be set when both the source and destination +# QEMU processes are on the same OS and directly connected +# with a UNIX domain socket as the migration channel to enable +# use of file descriptor passing. Individual device backends +# may need additional configuration flags set to enable local +# migration optimizations. This will be documented against the +# device backends where it applies. (Since 11.2) +# # Features: # # @unstable: Members @x-checkpoint-delay, @x-rdma-chunk-size, and @@ -XXX,XX +XXX,XX @@ '*direct-io': 'bool', '*x-rdma-chunk-size': { 'type': 'uint64', 'features': [ 'unstable' ] }, - '*cpr-exec-command': [ 'str' ]} } + '*cpr-exec-command': [ 'str' ], + '*local': 'bool' } } ## # @query-migrate-parameters: -- 2.43.0
As documented, for "local", the migration channel must be direct UNIX socket connection from source to target. We can't check for it being "direct", but let's at least check that we deal with UNIX socket (fd-passing supported). Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- migration/channel.c | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/migration/channel.c b/migration/channel.c index XXXXXXX..XXXXXXX 100644 --- a/migration/channel.c +++ b/migration/channel.c @@ -XXX,XX +XXX,XX @@ void migration_channel_process_incoming(QIOChannel *ioc) trace_migration_set_incoming_channel( ioc, object_get_typename(OBJECT(ioc))); + if (migrate_local() && + !qio_channel_has_feature(ioc, QIO_CHANNEL_FEATURE_FD_PASS)) { + error_setg(&local_err, + "local migration requires a UNIX domain socket channel"); + goto out; + } + if (migrate_channel_requires_tls_upgrade(ioc)) { migration_tls_channel_process_incoming(ioc, &local_err); } else { @@ -XXX,XX +XXX,XX @@ void migration_channel_connect_outgoing(MigrationState *s, QIOChannel *ioc) { trace_migration_set_outgoing_channel(ioc, object_get_typename(OBJECT(ioc))); + if (migrate_local() && + !qio_channel_has_feature(ioc, QIO_CHANNEL_FEATURE_FD_PASS)) { + Error *local_err = NULL; + + error_setg(&local_err, + "local migration requires a UNIX domain socket channel"); + migration_connect_error_propagate(s, local_err); + return; + } + if (migrate_channel_requires_tls_upgrade(ioc)) { Error *local_err = NULL; -- 2.43.0
Next commit will introduce live-migration (with fd-passing) for TAP net backend. So, now we prepare virtio-net for it Add virtio-net option local-migration, which is true by default, but false for older machine types, which doesn't support the feature. We introduce interface for live-migrating backends: 1. ->is_wait_incoming() handler, so that virtio-net knows, that backend is not fully intialized, as it waits for incoming migration stream. 2. MIG_PRI_BACKEND priority: backends should migrate with higher priority than virtio-net, so that we can do final preparations here in post-load handlers and be sure, that backends are already prepared. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- hw/net/virtio-net.c | 89 +++++++++++++++++++++++++++++++++- include/hw/virtio/virtio-net.h | 1 + include/migration/vmstate.h | 2 + include/net/net.h | 2 + 4 files changed, 93 insertions(+), 1 deletion(-) diff --git a/hw/net/virtio-net.c b/hw/net/virtio-net.c index XXXXXXX..XXXXXXX 100644 --- a/hw/net/virtio-net.c +++ b/hw/net/virtio-net.c @@ -XXX,XX +XXX,XX @@ #include "migration/misc.h" #include "standard-headers/linux/ethtool.h" #include "system/system.h" +#include "system/runstate.h" #include "system/replay.h" #include "trace.h" #include "monitor/qdev.h" @@ -XXX,XX +XXX,XX @@ static void virtio_net_set_multiqueue(VirtIONet *n, int multiqueue) n->multiqueue = multiqueue; virtio_net_change_num_queues(n, max * 2 + 1); - virtio_net_set_queue_pairs(n); + /* + * virtio_net_set_multiqueue() called from set_features(0) on early + * reset, when peer may wait for incoming (and is not initialized + * yet). + * Don't worry about it: virtio_net_set_queue_pairs() will be called + * later from virtio_net_post_load_device(), and anyway will be + * no-op for local incoming migration with live backend passing. + */ + if (!n->peers_wait_incoming) { + virtio_net_set_queue_pairs(n); + } } static int virtio_net_pre_load_queues(VirtIODevice *vdev, uint32_t n) @@ -XXX,XX +XXX,XX @@ static void virtio_net_get_features(VirtIODevice *vdev, uint64_t *features, virtio_add_feature_ex(features, VIRTIO_NET_F_MAC); + if (n->peers_wait_incoming) { + /* + * Excessive feature set is OK for early initialization when + * we wait for local incoming migration: actual guest-negotiated + * features will come with migration stream anyway. And we are sure + * that we support same host-features as source, because the backend + * is the same (the same TAP device, for example). + */ + return; + } + if (!peer_has_vnet_hdr(n)) { virtio_clear_feature_ex(features, VIRTIO_NET_F_CSUM); virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_TSO4); @@ -XXX,XX +XXX,XX @@ static int virtio_net_post_load_device(void *opaque, int version_id) VirtIODevice *vdev = VIRTIO_DEVICE(n); int i, link_down; bool has_tunnel_hdr = virtio_has_tunnel_hdr(vdev->guest_features_ex); + Error *local_err = NULL; trace_virtio_net_post_load_device(); virtio_net_set_mrg_rx_bufs(n, n->mergeable_rx_bufs, @@ -XXX,XX +XXX,XX @@ static int virtio_net_post_load_device(void *opaque, int version_id) } virtio_net_commit_rss_config(n); + + /* + * If live-migration is enabled for some backend, than backend + * has already been migrated at higher priority (MIG_PRI_BACKEND) + * and virtio_net_vnet_post_load() has already called + * peer_test_vnet_hdr(). Recompute host_features so that virtio-net + * reflects the capabilities of the restored backend. + */ + virtio_net_get_features(vdev, &vdev->host_features, &local_err); + if (local_err) { + error_report_err(local_err); + return -EINVAL; + } + return 0; } @@ -XXX,XX +XXX,XX @@ static int virtio_net_vnet_post_load(void *opaque, int version_id) { struct VirtIONetMigTmp *tmp = opaque; + /* + * If live-migration is enabled for some backend, than backend + * has already been migrated at higher priority (MIG_PRI_BACKEND), + * so n->has_vnet_hdr can be refreshed from the live backend right + * here. + */ + peer_test_vnet_hdr(tmp->parent); + if (tmp->has_vnet_hdr && !peer_has_vnet_hdr(tmp->parent)) { error_report("virtio-net: saved image requires vnet_hdr=on"); return -EINVAL; @@ -XXX,XX +XXX,XX @@ static bool failover_hide_primary_device(DeviceListener *listener, return qatomic_read(&n->failover_primary_hidden); } +static bool virtio_net_check_peers_wait_incoming(VirtIONet *n, bool *waiting, + Error **errp) +{ + bool has_waiting = false; + bool has_not_waiting = false; + + for (int i = 0; i < n->max_queue_pairs; i++) { + NetClientState *peer = n->nic->ncs[i].peer; + if (!peer) { + continue; + } + + if (peer->info->is_wait_incoming && + peer->info->is_wait_incoming(peer)) { + has_waiting = true; + } else { + has_not_waiting = true; + } + + if (has_waiting && has_not_waiting) { + error_setg(errp, "Mixed peer states: some peers wait for incoming " + "migration while others don't"); + return false; + } + } + + if (has_waiting && !runstate_check(RUN_STATE_INMIGRATE)) { + error_setg(errp, "Peers wait for incoming, but it's not an incoming " + "migration."); + return false; + } + + *waiting = has_waiting; + return true; +} + static void virtio_net_device_realize(DeviceState *dev, Error **errp) { VirtIODevice *vdev = VIRTIO_DEVICE(dev); @@ -XXX,XX +XXX,XX @@ static void virtio_net_device_realize(DeviceState *dev, Error **errp) n->nic->ncs[i].do_not_pad = true; } + if (!virtio_net_check_peers_wait_incoming(n, &n->peers_wait_incoming, + errp)) { + virtio_cleanup(vdev); + return; + } + peer_test_vnet_hdr(n); if (peer_has_vnet_hdr(n)) { n->host_hdr_len = sizeof(struct virtio_net_hdr); diff --git a/include/hw/virtio/virtio-net.h b/include/hw/virtio/virtio-net.h index XXXXXXX..XXXXXXX 100644 --- a/include/hw/virtio/virtio-net.h +++ b/include/hw/virtio/virtio-net.h @@ -XXX,XX +XXX,XX @@ struct VirtIONet { struct EBPFRSSContext ebpf_rss; uint32_t nr_ebpf_rss_fds; char **ebpf_rss_fds; + bool peers_wait_incoming; }; size_t virtio_net_handle_ctrl_iov(VirtIODevice *vdev, diff --git a/include/migration/vmstate.h b/include/migration/vmstate.h index XXXXXXX..XXXXXXX 100644 --- a/include/migration/vmstate.h +++ b/include/migration/vmstate.h @@ -XXX,XX +XXX,XX @@ typedef enum { MIG_PRI_LOW, /* Must happen after default */ MIG_PRI_DEFAULT, + MIG_PRI_BACKEND, /* Must happen before emulated devices, */ + /* e.g. virtio-net */ MIG_PRI_IOMMU, /* Must happen before PCI devices */ MIG_PRI_PCI_BUS, /* Must happen before IOMMU */ MIG_PRI_VIRTIO_MEM, /* Must happen before IOMMU */ diff --git a/include/net/net.h b/include/net/net.h index XXXXXXX..XXXXXXX 100644 --- a/include/net/net.h +++ b/include/net/net.h @@ -XXX,XX +XXX,XX @@ typedef void (SocketReadStateFinalize)(SocketReadState *rs); typedef void (NetAnnounce)(NetClientState *); typedef bool (SetSteeringEBPF)(NetClientState *, int); typedef bool (NetCheckPeerType)(NetClientState *, ObjectClass *, Error **); +typedef bool (IsWaitIncoming)(NetClientState *); typedef struct vhost_net *(GetVHostNet)(NetClientState *nc); typedef struct NetClientInfo { @@ -XXX,XX +XXX,XX @@ typedef struct NetClientInfo { NetAnnounce *announce; SetSteeringEBPF *set_steering_ebpf; NetCheckPeerType *check_peer_type; + IsWaitIncoming *is_wait_incoming; GetVHostNet *get_vhost_net; } NetClientInfo; -- 2.43.0
Polling when VM is stopped doesn't make real sense, as stopped VM can't handle incoming traffic anyway. And it's critical for introduction of local TAP migration feature in the next commit: the TAP device will be transferred to the target (open fd will be passed through migration channel), and if we continue polling on source, we may get a package, which we'll never handle on source (already stopped), it will be lost. Better is save this package for target VM to handle. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ #include "net/net.h" #include "clients.h" #include "monitor/monitor.h" +#include "system/runstate.h" #include "system/system.h" #include "qapi/error.h" #include "qemu/cutils.h" @@ -XXX,XX +XXX,XX @@ struct TAPState { Notifier exit; int queue_index; + bool read_poll_detached; + VMChangeStateEntry *vmstate; }; static void launch_script(const char *setup_script, const char *ifname, @@ -XXX,XX +XXX,XX @@ static void tap_read_poll(TAPState *s, bool enable) tap_update_fd_handler(s); } +static void tap_vm_state_change(void *opaque, bool running, RunState state) +{ + TAPState *s = opaque; + + if (running) { + if (s->read_poll_detached) { + tap_read_poll(s, true); + s->read_poll_detached = false; + } + } else if (state == RUN_STATE_FINISH_MIGRATE) { + if (s->read_poll) { + s->read_poll_detached = true; + tap_read_poll(s, false); + } + } +} + static void tap_write_poll(TAPState *s, bool enable) { s->write_poll = enable; @@ -XXX,XX +XXX,XX @@ static void tap_cleanup(NetClientState *nc) s->exit.notify = NULL; } + if (s->vmstate) { + qemu_del_vm_change_state_handler(s->vmstate); + s->vmstate = NULL; + } + tap_read_poll(s, false); tap_write_poll(s, false); close(s->fd); @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, int sndbuf = (tap->has_sndbuf && tap->sndbuf) ? MIN(tap->sndbuf, INT_MAX) : INT_MAX; + s->read_poll_detached = false; + s->vmstate = qemu_add_vm_change_state_handler(tap_vm_state_change, s); + if (!tap_set_sndbuf(fd, sndbuf, sndbuf_required ? errp : NULL) && sndbuf_required) { goto failed; -- 2.43.0
Support transferring of TAP state (including open fd). Add new property "local-migration-supported", which defines whether local-migration is actually supported for this TAP device. Starting from 11.2 QEMU Machine Types it's enabled by default. Note that local-migration (including migrating opened FDs through migration channel, which must be UNIX socket) is enabled by global "local" migration parameters. But individual devices may have additional options to enable/disable it per device. The tricky thing is that we need to know whether to call open/connect in TAP initialization code, i.e. we need to know the value of migration parameter "local" when creating the TAP device. For incoming migration, we can know only for TAP devices created with QMP after setting the migration parameter with QMP. So the full picture is: On source, to start outgoing "local" migration you need: - migration parameter "local" set to true - "local-migration-supported" TAP option set to true (the default, starting from 11.2 QEMU Machine Types) If at least one of these options is not set, TAP backend doesn't participate in migration. On target, things are more difficult: Same, you need both "local" and "local-migration-supported" be set. And same, if one of them is not set, TAP backend is initialized as usual, and doesn't accept any incoming state. Additionally, if you are going to set "local", it must be set before creating the TAP device. If TAP device created with "local" unset, it initializes as usual. If you enable "local" after it and start incoming migration, it will fail in .pre_load handler of TAP backend. Moreover, there are interface restrictions: if you create TAP device when QEMU is in INCOMING state, and both "local" and "local-migration-supported" set, most of TAP options are not allowed, and script/downscript are required to be explicitly unset (set to "" or "no"). Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> --- net/tap.c | 165 ++++++++++++++++++++++++++++++++++++++++++++++++-- qapi/net.json | 22 ++++++- 2 files changed, 180 insertions(+), 7 deletions(-) diff --git a/net/tap.c b/net/tap.c index XXXXXXX..XXXXXXX 100644 --- a/net/tap.c +++ b/net/tap.c @@ -XXX,XX +XXX,XX @@ #include "monitor/monitor.h" #include "system/runstate.h" #include "system/system.h" +#include "migration/misc.h" #include "qapi/error.h" #include "qemu/cutils.h" #include "qemu/error-report.h" #include "qemu/main-loop.h" #include "qemu/sockets.h" #include "hw/virtio/vhost.h" +#include "hw/core/vmstate-if.h" +#include "migration/vmstate.h" +#include "qom/object.h" +#include "qom/compat-properties.h" #include "net/tap.h" #include "net/util.h" @@ -XXX,XX +XXX,XX @@ static const int kernel_feature_bits[] = { OBJECT_DECLARE_SIMPLE_TYPE(TAPState, TAP_NETDEV) +static const VMStateDescription vmstate_tap; + struct TAPState { Object parent_obj; @@ -XXX,XX +XXX,XX @@ struct TAPState { int queue_index; bool read_poll_detached; VMChangeStateEntry *vmstate; + bool local_migration_supported; }; static void launch_script(const char *setup_script, const char *ifname, @@ -XXX,XX +XXX,XX @@ static void tap_cleanup(NetClientState *nc) tap_write_poll(s, false); close(s->fd); s->fd = -1; + + vmstate_unregister(VMSTATE_IF(s), &vmstate_tap, s); } static void tap_poll(NetClientState *nc, bool enable) @@ -XXX,XX +XXX,XX @@ static VHostNetState *tap_get_vhost_net(NetClientState *nc) return s->vhost_net; } +static bool tap_is_wait_incoming(NetClientState *nc) +{ + TAPState *s = container_of(nc, TAPState, nc); + assert(nc->info->type == NET_CLIENT_DRIVER_TAP); + return s->fd == -1; +} + +static bool tap_pre_load(void *opaque, Error **errp) +{ + TAPState *s = opaque; + + if (s->fd != -1) { + error_setg(errp, + "TAP is already initialized and cannot receive " + "incoming fd. For local migration, 'local' " + "migration parameter must be set _before_ " + "creating TAP device."); + return false; + } + + return true; +} + +static bool tap_setup_vhost(TAPState *s, Error **errp); + +static bool tap_post_load(void *opaque, int version_id, Error **errp) +{ + ERRP_GUARD(); + TAPState *s = opaque; + + tap_read_poll(s, true); + + if (s->fd < 0) { + error_setg(errp, "FD was not loaded during incoming migration"); + return false; + } + + if (!tap_setup_vhost(s, errp)) { + error_prepend(errp, + "Failed to setup vhost during TAP post-load: "); + return false; + } + + return true; +} + +static bool tap_needed(void *opaque) +{ + TAPState *s = opaque; + + return s->local_migration_supported && migrate_local(); +} + +static const VMStateDescription vmstate_tap = { + .name = "net-tap", + .priority = MIG_PRI_BACKEND, + .pre_load_errp = tap_pre_load, + .post_load_errp = tap_post_load, + .needed = tap_needed, + .fields = (const VMStateField[]) { + VMSTATE_FD(fd, TAPState), + VMSTATE_BOOL(using_vnet_hdr, TAPState), + VMSTATE_BOOL(has_ufo, TAPState), + VMSTATE_BOOL(has_uso, TAPState), + VMSTATE_BOOL(has_tunnel, TAPState), + VMSTATE_BOOL(enabled, TAPState), + VMSTATE_UINT32(host_vnet_hdr_len, TAPState), + VMSTATE_END_OF_LIST() + } +}; static char *tap_vmstate_if_get_id(VMStateIf *obj) { @@ -XXX,XX +XXX,XX @@ static char *tap_vmstate_if_get_id(VMStateIf *obj) return res; } +static bool tap_get_local_migration_supported_prop(Object *obj, Error **errp) +{ + TAPState *s = TAP_NETDEV(obj); + return s->local_migration_supported; +} + +static void tap_set_local_migration_supported_prop(Object *obj, bool value, + Error **errp) +{ + TAPState *s = TAP_NETDEV(obj); + s->local_migration_supported = value; +} + +static void tap_instance_init(Object *obj) +{ + TAPState *s = TAP_NETDEV(obj); + s->local_migration_supported = false; +} + static void tap_class_init(ObjectClass *klass, const void *data) { VMStateIfClass *vc = VMSTATE_IF_CLASS(klass); vc->get_id = tap_vmstate_if_get_id; + + object_class_property_add_bool(klass, "local-migration-supported", + tap_get_local_migration_supported_prop, + tap_set_local_migration_supported_prop); } static const TypeInfo tap_netdev_info = { .name = TYPE_TAP_NETDEV, .parent = TYPE_OBJECT, .instance_size = sizeof(TAPState), + .instance_init = tap_instance_init, + .instance_post_init = object_apply_compat_props, .class_init = tap_class_init, .interfaces = (const InterfaceInfo[]) { { TYPE_VMSTATE_IF }, @@ -XXX,XX +XXX,XX @@ static NetClientInfo net_tap_info = { .set_vnet_le = tap_set_vnet_le, .set_vnet_be = tap_set_vnet_be, .set_steering_ebpf = tap_set_steering_ebpf, + .is_wait_incoming = tap_is_wait_incoming, .get_vhost_net = tap_get_vhost_net, }; static TAPState *new_tap(NetClientState *peer, const char *model, const char *name, - int queue_index) + int queue_index, + bool has_local_migration_supported, + bool local_migration_supported) { TAPState *s = TAP_NETDEV(object_new(TYPE_TAP_NETDEV)); @@ -XXX,XX +XXX,XX @@ static TAPState *new_tap(NetClientState *peer, s->queue_index = queue_index; + if (has_local_migration_supported) { + s->local_migration_supported = local_migration_supported; + } + + vmstate_register(VMSTATE_IF(s), VMSTATE_INSTANCE_ID_ANY, &vmstate_tap, s); + return s; } @@ -XXX,XX +XXX,XX @@ static TAPState *net_tap_fd_init(NetClientState *peer, const char *name, int fd, int vnet_hdr, - int queue_index) + int queue_index, + bool has_local_migration_supported, + bool local_migration_supported) { NetOffloads ol = {}; - TAPState *s = new_tap(peer, model, name, queue_index); + TAPState *s = new_tap(peer, model, name, queue_index, + has_local_migration_supported, + local_migration_supported); s->fd = fd; s->host_vnet_hdr_len = vnet_hdr ? sizeof(struct virtio_net_hdr) : 0; @@ -XXX,XX +XXX,XX @@ int net_init_bridge(const Netdev *netdev, const char *name, close(fd); return -1; } - s = net_tap_fd_init(peer, "bridge", name, fd, vnet_hdr, 0); + s = net_tap_fd_init(peer, "bridge", name, fd, vnet_hdr, 0, true, false); qemu_set_info_str(&s->nc, "helper=%s,br=%s", helper, br); @@ -XXX,XX +XXX,XX @@ static bool net_init_tap_one(const NetdevTapOptions *tap, NetClientState *peer, Error **errp) { TAPState *s = net_tap_fd_init(peer, tap->helper ? "bridge" : "tap", - name, fd, vnet_hdr, queue_index); + name, fd, vnet_hdr, queue_index, + tap->has_local_migration_supported, + tap->local_migration_supported); bool sndbuf_required = tap->has_sndbuf; int sndbuf = (tap->has_sndbuf && tap->sndbuf) ? MIN(tap->sndbuf, INT_MAX) : INT_MAX; @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, /* for the no-fd, no-helper case */ char ifname[128]; int *fds = NULL, *vhost_fds = NULL; + bool incoming_fds; assert(netdev->type == NET_CLIENT_DRIVER_TAP); tap = &netdev->u.tap; @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, return -1; } + incoming_fds = tap->local_migration_supported && migrate_local() && + runstate_check(RUN_STATE_INMIGRATE); + + if (incoming_fds && + (tap->fd || tap->fds || tap->helper || tap->br || tap->ifname || + tap->has_sndbuf || tap->has_vnet_hdr || + !tap_is_explicit_no_script("script", tap->script) || + !tap_is_explicit_no_script("downscript", tap->downscript))) { + error_setg(errp, "Local incoming migration of TAP device (-incoming, " + "migration parameter @local is set, " + "TAP parameter @local-migration-supported is set) " + "is incompatible with " + "fd=, fds=, helper=, br=, ifname=, sndbuf= and vnet_hdr=, " + "and requires explicit empty script= and downscript="); + return -1; + } + queues = tap_parse_fds_and_queues(tap, &fds, errp); if (queues < 0) { return -1; @@ -XXX,XX +XXX,XX @@ int net_init_tap(const Netdev *netdev, const char *name, goto fail; } - if (fds) { + if (incoming_fds) { + for (i = 0; i < queues; i++) { + TAPState *s = new_tap(peer, "tap", name, i, + tap->has_local_migration_supported, + tap->local_migration_supported); + qemu_set_info_str(&s->nc, "incoming"); + + s->fd = -1; + if (vhost_fds) { + s->vhostfd = vhost_fds[i]; + s->vhost_busyloop_timeout = tap->has_poll_us ? tap->poll_us : 0; + } else { + s->vhostfd = -1; + } + } + } else if (fds) { for (i = 0; i < queues; i++) { if (i == 0) { vnet_hdr = tap_probe_vnet_hdr(fds[i], errp); diff --git a/qapi/net.json b/qapi/net.json index XXXXXXX..XXXXXXX 100644 --- a/qapi/net.json +++ b/qapi/net.json @@ -XXX,XX +XXX,XX @@ # @poll-us: maximum number of microseconds that could be spent on busy # polling for tap (since 2.7) # +# @local-migration-supported: enable local migration for this TAP +# backend. When set, local migration is enabled/disabled by +# migration parameter @local for this TAP backend. When unset, +# migration parameter @local is ignored for this TAP backend. +# To be able to do incoming local migration of a TAP backend, +# migration parameter @local must be set _before_ creating the +# TAP backend. Otherwise, TAP backend is initialized as usual, +# opening/creating TAP devices in kernel. In this case further +# local incoming migration (with migration parameter @local set +# after creating TAP backend with @local-migration-supporeted +# parameter set) will simply fail. +# Moreover, when QEMU is in incoming migration state, migration +# parameter @local is set and @local-migration-supported is set, +# the following options are not supported and must not be set: +# @fd, @fds, @helper, @br, @ifname, @sndbuf, @vnet_hdr. +# Additionally in this case @script and @downscipt must be +# explicitly disabled (empty strings or "no"). +# (Since 11.2) +# # Since: 1.2 ## { 'struct': 'NetdevTapOptions', @@ -XXX,XX +XXX,XX @@ '*vhostfds': 'str', '*vhostforce': 'bool', '*queues': 'uint32', - '*poll-us': 'uint32'} } + '*poll-us': 'uint32', + '*local-migration-supported': 'bool' } } ## # @NetdevSocketOptions: -- 2.43.0
To be used in the next commit: that would be a test for TAP networking, and it will need to setup TAP device. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Reviewed-by: Daniel P. Berrangé <berrange@redhat.com> Reviewed-by: Thomas Huth <thuth@redhat.com> Tested-by: Lei Yang <leiyang@redhat.com> Reviewed-by: Maksim Davydov <davydov-max@yandex-team.ru> Reviewed-by: Ben Chaney <bchaney@akamai.com> --- tests/functional/qemu_test/decorators.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tests/functional/qemu_test/decorators.py b/tests/functional/qemu_test/decorators.py index XXXXXXX..XXXXXXX 100644 --- a/tests/functional/qemu_test/decorators.py +++ b/tests/functional/qemu_test/decorators.py @@ -XXX,XX +XXX,XX @@ import os import platform import resource +import subprocess from unittest import skipIf, skipUnless from .cmd import which @@ -XXX,XX +XXX,XX @@ def skipLockedMemoryTest(locked_memory): ulimit_memory == resource.RLIM_INFINITY or ulimit_memory >= locked_memory * 1024, f'Test required {locked_memory} kB of available locked memory', ) + +''' +Decorator to skip execution of a test if passwordless +sudo command is not available. +''' +def skipWithoutSudo(): + proc = subprocess.run(["sudo", "-n", "/bin/true"], + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + universal_newlines=True, + check=False) + + return skipUnless(proc.returncode == 0, + f'requires password-less sudo access: {proc.stdout}') -- 2.43.0
Add test for a new local-migration migration of virtio-net/tap, with fd passing through UNIX socket. Signed-off-by: Vladimir Sementsov-Ogievskiy <vsementsov@yandex-team.ru> Reviewed-by: Ben Chaney <bchaney@akamai.com> --- tests/functional/x86_64/meson.build | 1 + tests/functional/x86_64/test_tap_migration.py | 455 ++++++++++++++++++ 2 files changed, 456 insertions(+) create mode 100755 tests/functional/x86_64/test_tap_migration.py diff --git a/tests/functional/x86_64/meson.build b/tests/functional/x86_64/meson.build index XXXXXXX..XXXXXXX 100644 --- a/tests/functional/x86_64/meson.build +++ b/tests/functional/x86_64/meson.build @@ -XXX,XX +XXX,XX @@ tests_x86_64_system_thorough = [ 'virtio_balloon', 'virtio_gpu', 'rebuild_vmfd', + 'tap_migration', ] diff --git a/tests/functional/x86_64/test_tap_migration.py b/tests/functional/x86_64/test_tap_migration.py new file mode 100755 index XXXXXXX..XXXXXXX --- /dev/null +++ b/tests/functional/x86_64/test_tap_migration.py @@ -XXX,XX +XXX,XX @@ +#!/usr/bin/env python3 +# +# Functional test that tests TAP local migration +# with fd passing +# +# Copyright (c) Yandex Technologies LLC, 2026 +# +# SPDX-License-Identifier: GPL-2.0-or-later + +import os +import time +import subprocess +from subprocess import run +import signal +import ctypes +import ctypes.util +import unittest +from contextlib import contextmanager, ExitStack +from typing import Tuple + +from qemu_test import ( + LinuxKernelTest, + Asset, + exec_command_and_wait_for_pattern, +) +from qemu_test.decorators import skipWithoutSudo + + +GUEST_IP = "192.168.100.2" +GUEST_IP_MASK = f"{GUEST_IP}/24" +GUEST_MAC = "d6:0d:75:f8:0f:b7" +HOST_IP = "192.168.100.1" +HOST_IP_MASK = f"{HOST_IP}/24" +TAP_ID = "tap0" +TAP_ID2 = "tap1" +TAP_MAC = "e6:1d:44:b5:03:5d" +NETNS = f"qemu_test_ns_{os.getpid()}" + + +def ip(args, check=True) -> None: + """Run ip command with sudo""" + run(["sudo", "ip"] + args, check=check) + + +@contextmanager +def switch_netns(netns_name): + libc = ctypes.CDLL(ctypes.util.find_library("c")) + netns_path = f"/var/run/netns/{netns_name}" + + def switch_to_fd(fd, check: bool = False): + """Switch to netns by file descriptor""" + SYS_setns = 308 + CLONE_NEWNET = 0x40000000 + ret = libc.syscall(SYS_setns, fd, CLONE_NEWNET) + if check and ret != 0: + raise RuntimeError("syscall SETNS failed") + + with ExitStack() as stack: + original_netns_fd = os.open("/proc/self/ns/net", os.O_RDONLY) + stack.callback(os.close, original_netns_fd) + + ip(["netns", "add", netns_name]) + stack.callback(ip, ["netns", "del", netns_name], check=False) + + new_netns_fd = os.open(netns_path, os.O_RDONLY) + stack.callback(os.close, new_netns_fd) + + switch_to_fd(new_netns_fd) + stack.callback(switch_to_fd, original_netns_fd, check=False) + + yield + + +def del_tap(tap_name: str = TAP_ID) -> None: + ip(["tuntap", "del", tap_name, "mode", "tap", "multi_queue"], check=False) + + +def init_tap(tap_name: str = TAP_ID, with_ip: bool = True) -> None: + ip(["tuntap", "add", "dev", tap_name, "mode", "tap", "multi_queue"]) + if with_ip: + ip(["link", "set", "dev", tap_name, "address", TAP_MAC]) + ip(["addr", "add", HOST_IP_MASK, "dev", tap_name]) + ip(["link", "set", tap_name, "up"]) + + +def switch_network_to_tap2() -> None: + ip(["link", "set", TAP_ID2, "down"]) + ip(["link", "set", TAP_ID, "down"]) + ip(["addr", "delete", HOST_IP_MASK, "dev", TAP_ID]) + ip(["link", "set", "dev", TAP_ID2, "address", TAP_MAC]) + ip(["addr", "add", HOST_IP_MASK, "dev", TAP_ID2]) + ip(["link", "set", TAP_ID2, "up"]) + + +def parse_ping_line(line: str) -> float: + # suspect lines like + # [1748524876.590509] 64 bytes from 94.245.155.3 \ + # (94.245.155.3): icmp_seq=1 ttl=250 time=101 ms + spl = line.split() + return float(spl[0][1:-1]) + + +def parse_ping_output(out) -> Tuple[bool, float, float]: + lines = [x for x in out.split("\n") if x.startswith("[")] + + try: + first_no_ans = next( + (ind for ind in range(len(lines)) if lines[ind][20:26] == "no ans") + ) + except StopIteration: + return False, parse_ping_line(lines[0]), parse_ping_line(lines[-1]) + + last_no_ans = next( + ind + for ind in range(len(lines) - 1, -1, -1) + if lines[ind][20:26] == "no ans" + ) + + return ( + True, + parse_ping_line(lines[first_no_ans]), + parse_ping_line(lines[last_no_ans]), + ) + + +def wait_migration_finish(source_vm, target_vm): + migr_events = ( + ("MIGRATION", {"data": {"status": "completed"}}), + ("MIGRATION", {"data": {"status": "failed"}}), + ) + + source_e = source_vm.events_wait(migr_events)["data"] + target_e = target_vm.events_wait(migr_events)["data"] + + source_s = source_vm.cmd("query-status")["status"] + target_s = target_vm.cmd("query-status")["status"] + + assert ( + source_e["status"] == "completed" + and target_e["status"] == "completed" + and source_s == "postmigrate" + and target_s == "paused" + ), f"""Migration failed: + SRC status: {source_s} + SRC event: {source_e} + TGT status: {target_s} + TGT event:{target_e}""" + + +@skipWithoutSudo() +class TAPFdMigration(LinuxKernelTest): + + ASSET_KERNEL = Asset( + ( + "https://archives.fedoraproject.org/pub/archive/fedora/linux/releases" + "/31/Server/x86_64/os/images/pxeboot/vmlinuz" + ), + "d4738d03dbbe083ca610d0821d0a8f1488bebbdccef54ce33e3adb35fda00129", + ) + + ASSET_INITRD = Asset( + ( + "https://archives.fedoraproject.org/pub/archive/fedora/linux/releases" + "/31/Server/x86_64/os/images/pxeboot/initrd.img" + ), + "277cd6c7adf77c7e63d73bbb2cded8ef9e2d3a2f100000e92ff1f8396513cd8b", + ) + + ASSET_ALPINE_ISO = Asset( + ( + "https://dl-cdn.alpinelinux.org/" + "alpine/v3.22/releases/x86_64/alpine-standard-3.22.1-x86_64.iso" + ), + "96d1b44ea1b8a5a884f193526d92edb4676054e9fa903ad2f016441a0fe13089", + ) + + @classmethod + def setUpClass(cls): + super().setUpClass() + + try: + cls.netns_context = switch_netns(NETNS) + cls.netns_context.__enter__() + except (OSError, subprocess.CalledProcessError) as e: + raise unittest.SkipTest(f"can't switch network namespace: {e}") + + @classmethod + def tearDownClass(cls): + if hasattr(cls, "netns_context"): + cls.netns_context.__exit__(None, None, None) + super().tearDownClass() + + def setUp(self): + super().setUp() + + init_tap() + + self.outer_ping_proc = None + self.shm_path = None + + def tearDown(self): + try: + del_tap(TAP_ID) + del_tap(TAP_ID2) + + if self.outer_ping_proc: + self.stop_outer_ping() + + if self.shm_path: + os.unlink(self.shm_path) + finally: + super().tearDown() + + def start_outer_ping(self) -> None: + assert self.outer_ping_proc is None + self.outer_ping_log = self.scratch_file("ping.log") + with open(self.outer_ping_log, "w") as f: + self.outer_ping_proc = subprocess.Popen( + ["ping", "-i", "0", "-O", "-D", GUEST_IP], + text=True, + stdout=f, + ) + + def stop_outer_ping(self) -> str: + assert self.outer_ping_proc + self.outer_ping_proc.send_signal(signal.SIGINT) + + self.outer_ping_proc.communicate(timeout=5) + self.outer_ping_proc = None + + with open(self.outer_ping_log) as f: + return f.read() + + def stop_ping_and_check(self, stop_time, resume_time): + ping_res = self.stop_outer_ping() + + discon, a, b = parse_ping_output(ping_res) + + if not discon: + text = ( + f"STOP: {stop_time}, RESUME: {resume_time}," f"PING: {a} - {b}" + ) + if a > stop_time or b < resume_time: + self.fail(f"PING failed: {text}") + self.log.info(f"PING: no packets lost: {text}") + return + + text = ( + f"STOP: {stop_time}, RESUME: {resume_time}," + f"PING: disconnect: {a} - {b}" + ) + self.log.info(text) + eps = 0.05 + if a < stop_time - eps or b > resume_time + eps: + self.fail(text) + + def one_ping_from_guest(self, vm) -> None: + exec_command_and_wait_for_pattern( + self, + f"ping -c 1 -W 1 {HOST_IP}", + "1 packets transmitted, 1 packets received", + "1 packets transmitted, 0 packets received", + vm=vm, + ) + self.wait_for_console_pattern("# ", vm=vm) + + def one_ping_from_host(self) -> None: + run( + ["ping", "-c", "1", "-W", "1", GUEST_IP], + stdout=subprocess.DEVNULL, + check=True, + ) + + def setup_shared_memory(self): + self.shm_path = f"/dev/shm/qemu_test_{os.getpid()}" + + try: + with open(self.shm_path, "wb") as f: + f.write(b"\0" * (1024 * 1024 * 1024)) # 1GB + except Exception as e: + self.fail(f"Failed to create shared memory file: {e}") + + def prepare_and_launch_vm( + self, shm_path, vhost, incoming=False, vm=None, local=True + ): + if not vm: + vm = self.vm + + vm.set_console() + vm.add_args("-accel", "kvm") + vm.add_args("-device", "pcie-pci-bridge,id=pci.1,bus=pcie.0") + vm.add_args("-m", "1G") + + vm.add_args( + "-object", + f"memory-backend-file,id=ram0,size=1G,mem-path={shm_path},share=on", + ) + vm.add_args("-machine", "memory-backend=ram0") + + vm.add_args( + "-drive", + f"file={self.ASSET_ALPINE_ISO.fetch()},media=cdrom,format=raw", + ) + + vm.add_args("-S") + + if incoming: + vm.add_args("-incoming", "defer") + + vm_s = "target" if incoming else "source" + self.log.info(f"Launching {vm_s} VM") + vm.launch() + + if not local: + tap_name = TAP_ID2 if incoming else TAP_ID + else: + tap_name = TAP_ID + + self.set_migration_capabilities(vm, local) + self.add_virtio_net(vm, vhost, tap_name, local, incoming) + + def add_virtio_net( + self, vm, vhost: bool, tap_name: str, local: bool, incoming: bool + ): + netdev_params = { + "id": "netdev.1", + "vhost": vhost, + "type": "tap", + "queues": 4, + "script": "no", + "downscript": "no", + "local-migration-supported": local, + } + + if not (local and incoming): + netdev_params["vnet_hdr"] = True + netdev_params["ifname"] = tap_name + + vm.cmd("netdev_add", netdev_params) + + vm.cmd( + "device_add", + driver="virtio-net-pci", + romfile="", + id="vnet.1", + netdev="netdev.1", + mq=True, + vectors=18, + bus="pci.1", + mac=GUEST_MAC, + disable_legacy="off", + ) + + def set_migration_capabilities(self, vm, local=True): + vm.cmd( + "migrate-set-capabilities", + { + "capabilities": [ + {"capability": "events", "state": True}, + {"capability": "x-ignore-shared", "state": True}, + ] + }, + ) + vm.cmd("migrate-set-parameters", {"local": local}) + + def setup_guest_network(self) -> None: + exec_command_and_wait_for_pattern(self, "ip addr", "# ") + exec_command_and_wait_for_pattern( + self, + f"ip addr add {GUEST_IP_MASK} dev eth0 && " + "ip link set eth0 up && echo OK", + "OK", + ) + self.wait_for_console_pattern("# ") + + def do_test_tap_fd_migration(self, vhost, local=True): + self.require_accelerator("kvm") + self.set_machine("q35") + + socket_dir = self.socket_dir() + migration_socket = os.path.join(socket_dir.name, "migration.sock") + + self.setup_shared_memory() + + # Setup second TAP if needed + if not local: + del_tap(TAP_ID2) + init_tap(TAP_ID2, with_ip=False) + + self.prepare_and_launch_vm(self.shm_path, vhost, local=local) + self.vm.cmd("cont") + self.wait_for_console_pattern("login:") + exec_command_and_wait_for_pattern(self, "root", "# ") + + self.setup_guest_network() + + self.one_ping_from_guest(self.vm) + self.one_ping_from_host() + self.start_outer_ping() + + # Get some successful pings before migration + time.sleep(0.5) + + target_vm = self.get_vm(name="target") + self.prepare_and_launch_vm( + self.shm_path, + vhost, + incoming=True, + vm=target_vm, + local=local, + ) + + target_vm.cmd("migrate-incoming", {"uri": f"unix:{migration_socket}"}) + + self.log.info("Starting migration") + freeze_start = time.time() + self.vm.cmd("migrate", {"uri": f"unix:{migration_socket}"}) + + self.log.info("Waiting for migration completion") + wait_migration_finish(self.vm, target_vm) + + # Switch network to tap1 if not using local-migration + if not local: + switch_network_to_tap2() + + target_vm.cmd("cont") + freeze_end = time.time() + + self.vm.shutdown() + + self.log.info("Verifying PING on target VM after migration") + self.one_ping_from_guest(target_vm) + self.one_ping_from_host() + + # And a bit more pings after source shutdown + time.sleep(0.3) + self.stop_ping_and_check(freeze_start, freeze_end) + + target_vm.shutdown() + + def test_tap_fd_migration(self): + self.do_test_tap_fd_migration(False) + + def test_tap_fd_migration_vhost(self): + self.do_test_tap_fd_migration(True) + + def test_tap_new_tap_migration(self): + self.do_test_tap_fd_migration(False, local=False) + + def test_tap_new_tap_migration_vhost(self): + self.do_test_tap_fd_migration(True, local=False) + + +if __name__ == "__main__": + LinuxKernelTest.main() -- 2.43.0