@samitouri / QOSamiQemu / commits / dd67c149e2

multifd: Add COLO support

Like in the normal ram_load() path, put the received pages into the colo cache and mark the pages in the bitmap so that they will be flushed to the guest later. Multifd with COLO is useful to reduce the VM pause time during checkpointing for latency sensitive workloads. In such workloads the worst-case latency is especially important. Also, this is already worth it for the precopy phase as it helps with converging. Moreover, multifd migration is the preferred way to do migration nowadays and this allows to use multifd compression with COLO. Benchmark: Cluster nodes - Intel Xenon E5-2630 v3 - 48Gb RAM - 10G Ethernet Guest - Windows Server 2016 - 6Gb RAM - 4 cores Workload - Upload a file to the guest with SMB to simulate moderate memory dirtying - Measure the memory transfer time portion of each checkpoint - 600ms COLO checkpoint interval Results Plain idle mean: 4.50ms 99per: 10.33ms load mean: 24.30ms 99per: 78.05ms Multifd-4 idle mean: 6.48ms 99per: 10.41ms load mean: 14.12ms 99per: 31.27ms Evaluation While multifd has slightly higher latency when the guest idles, it is 10ms faster under load and more importantly it's worst case latency is less than 1/2 of plain under load as can be seen in the 99. Percentile. Co-authored-by: Juan Quintela <quintela@redhat.com> [farosas: changed SoB to coauthored as Juan doesn't own that email address anymore] Reviewed-by: Fabiano Rosas <farosas@suse.de> Reviewed-by: Peter Xu <peterx@redhat.com> Signed-off-by: Lukas Straub <lukasstraub2@web.de> Link: https://lore.kernel.org/qemu-devel/20260302-colo_unit_test_multifd-v11-8-d653fb3b1d80@web.de [removed license boilerplate] Signed-off-by: Fabiano Rosas <farosas@suse.de>

Lukas Straub committed Mar 2, 2026 at 12:43 UTC dd67c149e2b1325138e45f619dd18d7ab3680ec0
7 files changed +87 -3
MAINTAINERS
+1
@@ -3883,6 +3883,7 @@ COLO Framework
3883 M: Lukas Straub <lukasstraub2@web.de>
3884 S: Maintained
3885 F: migration/colo*
3886 +F: migration/multifd-colo.*
3887 F: include/migration/colo.h
3888 F: include/migration/failover.h
3889 F: docs/COLO-FT.txt
migration/meson.build
+1 -1
@@ -39,7 +39,7 @@ system_ss.add(files(
39 ), gnutls, zlib)
40
41 if get_option('replication').allowed()
42 - system_ss.add(files('colo-failover.c', 'colo.c'))
42 + system_ss.add(files('colo-failover.c', 'colo.c', 'multifd-colo.c'))
43 else
44 system_ss.add(files('colo-stubs.c'))
45 endif
migration/multifd-colo.c new
+41
@@ -0,0 +1,41 @@
1 +/*
2 + * SPDX-License-Identifier: GPL-2.0-or-later
3 + *
4 + * multifd colo implementation
5 + *
6 + * Copyright (c) Lukas Straub <lukasstraub2@web.de>
7 + */
8 +
9 +#include "qemu/osdep.h"
10 +#include "multifd.h"
11 +#include "multifd-colo.h"
12 +#include "migration/colo.h"
13 +#include "system/ramblock.h"
14 +
15 +void multifd_colo_prepare_recv(MultiFDRecvParams *p)
16 +{
17 + /*
18 + * While we're still in precopy state (not yet in colo state), we copy
19 + * received pages to both guest and cache. No need to set dirty bits,
20 + * since guest and cache memory are in sync.
21 + */
22 + if (migration_incoming_in_colo_state()) {
23 + colo_record_bitmap(p->block, p->normal, p->normal_num);
24 + colo_record_bitmap(p->block, p->zero, p->zero_num);
25 + }
26 +}
27 +
28 +void multifd_colo_process_recv(MultiFDRecvParams *p)
29 +{
30 + if (!migration_incoming_in_colo_state()) {
31 + for (int i = 0; i < p->normal_num; i++) {
32 + void *guest = p->block->host + p->normal[i];
33 + void *cache = p->host + p->normal[i];
34 + memcpy(guest, cache, multifd_ram_page_size());
35 + }
36 + for (int i = 0; i < p->zero_num; i++) {
37 + void *guest = p->block->host + p->zero[i];
38 + memset(guest, 0, multifd_ram_page_size());
39 + }
40 + }
41 +}
migration/multifd-colo.h new
+23
@@ -0,0 +1,23 @@
1 +/*
2 + * SPDX-License-Identifier: GPL-2.0-or-later
3 + *
4 + * multifd colo header
5 + *
6 + * Copyright (c) Lukas Straub <lukasstraub2@web.de>
7 + */
8 +
9 +#ifndef QEMU_MIGRATION_MULTIFD_COLO_H
10 +#define QEMU_MIGRATION_MULTIFD_COLO_H
11 +
12 +#ifdef CONFIG_REPLICATION
13 +
14 +void multifd_colo_prepare_recv(MultiFDRecvParams *p);
15 +void multifd_colo_process_recv(MultiFDRecvParams *p);
16 +
17 +#else
18 +
19 +static inline void multifd_colo_prepare_recv(MultiFDRecvParams *p) {}
20 +static inline void multifd_colo_process_recv(MultiFDRecvParams *p) {}
21 +
22 +#endif
23 +#endif
migration/multifd-nocomp.c
+9 -1
@@ -16,6 +16,7 @@
16 #include "file.h"
17 #include "migration-stats.h"
18 #include "multifd.h"
19 +#include "multifd-colo.h"
20 #include "options.h"
21 #include "migration.h"
22 #include "qapi/error.h"
@@ -269,7 +270,6 @@ int multifd_ram_unfill_packet(MultiFDRecvParams *p, Error **errp)
270 return -1;
271 }
272
272 - p->host = p->block->host;
273 for (i = 0; i < p->normal_num; i++) {
274 uint64_t offset = be64_to_cpu(packet->offset[i]);
275
@@ -294,6 +294,14 @@ int multifd_ram_unfill_packet(MultiFDRecvParams *p, Error **errp)
294 p->zero[i] = offset;
295 }
296
297 + if (migrate_colo()) {
298 + multifd_colo_prepare_recv(p);
299 + assert(p->block->colo_cache);
300 + p->host = p->block->colo_cache;
301 + } else {
302 + p->host = p->block->host;
303 + }
304 +
305 return 0;
306 }
307
migration/multifd.c
+8
@@ -29,6 +29,7 @@
29 #include "qemu-file.h"
30 #include "trace.h"
31 #include "multifd.h"
32 +#include "multifd-colo.h"
33 #include "options.h"
34 #include "qemu/yank.h"
35 #include "io/channel-file.h"
@@ -1258,6 +1259,13 @@ static int multifd_ram_state_recv(MultiFDRecvParams *p, Error **errp)
1259 int ret;
1260
1261 ret = multifd_recv_state->ops->recv(p, errp);
1262 + if (ret != 0) {
1263 + return ret;
1264 + }
1265 +
1266 + if (migrate_colo()) {
1267 + multifd_colo_process_recv(p);
1268 + }
1269
1270 return ret;
1271 }
migration/multifd.h
+4 -1
@@ -279,7 +279,10 @@ typedef struct {
279 uint64_t packets_recved;
280 /* ramblock */
281 RAMBlock *block;
282 - /* ramblock host address */
282 + /*
283 + * Normally, it points to ramblock's host address. When COLO
284 + * is enabled, it points to the mirror cache for the ramblock.
285 + */
286 uint8_t *host;
287 /* buffers to recv */
288 struct iovec *iov;