accel/kvm: add changes required to support KVM VM file descriptor change
This change adds common kvm specific support to handle KVM VM file descriptor change. KVM VM file descriptor can change as a part of confidential guest reset mechanism. A new function api kvm_arch_on_vmfd_change() per architecture platform is added in order to implement architecture specific changes required to support it. A subsequent patch will add x86 specific implementation for kvm_arch_on_vmfd_change() as currently only x86 supports confidential guest reset. Signed-off-by: Ani Sinha <anisinha@redhat.com> Link: https://lore.kernel.org/r/20260225035000.385950-6-anisinha@redhat.com Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
Ani Sinha committed
Feb 25, 2026 at 09:19 UTC
98884e0cc10997a17ce9abfd6ff10be19224ca6a
7 files changed
+128
-3
MAINTAINERS
+6
@@ -152,6 +152,12 @@ F: tools/i386/
152
F: tests/functional/i386/
153
F: tests/functional/x86_64/
154
155
+X86 VM file descriptor change on reset test
156
+M: Ani Sinha <anisinha@redhat.com>
157
+M: Paolo Bonzini <pbonzini@redhat.com>
158
+S: Maintained
159
+F: stubs/kvm.c
160
+
161
Guest CPU cores (TCG)
162
---------------------
163
Overall TCG CPUs
accel/kvm/kvm-all.c
+85
-3
@@ -2415,11 +2415,9 @@ void kvm_irqchip_set_qemuirq_gsi(KVMState *s, qemu_irq irq, int gsi)
2415
g_hash_table_insert(s->gsimap, irq, GINT_TO_POINTER(gsi));
2416
}
2417
2418
-static void kvm_irqchip_create(KVMState *s)
2418
+static void do_kvm_irqchip_create(KVMState *s)
2419
{
2420
int ret;
2421
-
2422
- assert(s->kernel_irqchip_split != ON_OFF_AUTO_AUTO);
2421
if (kvm_check_extension(s, KVM_CAP_IRQCHIP)) {
2422
;
2423
} else if (kvm_check_extension(s, KVM_CAP_S390_IRQCHIP)) {
@@ -2452,7 +2450,13 @@ static void kvm_irqchip_create(KVMState *s)
2450
fprintf(stderr, "Create kernel irqchip failed: %s\n", strerror(-ret));
2451
exit(1);
2452
}
2453
+}
2454
+
2455
+static void kvm_irqchip_create(KVMState *s)
2456
+{
2457
+ assert(s->kernel_irqchip_split != ON_OFF_AUTO_AUTO);
2458
2459
+ do_kvm_irqchip_create(s);
2460
kvm_kernel_irqchip = true;
2461
/* If we have an in-kernel IRQ chip then we must have asynchronous
2462
* interrupt delivery (though the reverse is not necessarily true)
@@ -2607,6 +2611,83 @@ static int kvm_setup_dirty_ring(KVMState *s)
2611
return 0;
2612
}
2613
2614
+static int kvm_reset_vmfd(MachineState *ms)
2615
+{
2616
+ KVMState *s;
2617
+ KVMMemoryListener *kml;
2618
+ int ret = 0, type;
2619
+ Error *err = NULL;
2620
+
2621
+ /*
2622
+ * bail if the current architecture does not support VM file
2623
+ * descriptor change.
2624
+ */
2625
+ if (!kvm_arch_supports_vmfd_change()) {
2626
+ error_report("This target architecture does not support KVM VM "
2627
+ "file descriptor change.");
2628
+ return -EOPNOTSUPP;
2629
+ }
2630
+
2631
+ s = KVM_STATE(ms->accelerator);
2632
+ kml = &s->memory_listener;
2633
+
2634
+ memory_listener_unregister(&kml->listener);
2635
+ memory_listener_unregister(&kvm_io_listener);
2636
+
2637
+ if (s->vmfd >= 0) {
2638
+ close(s->vmfd);
2639
+ }
2640
+
2641
+ type = find_kvm_machine_type(ms);
2642
+ if (type < 0) {
2643
+ return -EINVAL;
2644
+ }
2645
+
2646
+ ret = do_kvm_create_vm(s, type);
2647
+ if (ret < 0) {
2648
+ return ret;
2649
+ }
2650
+
2651
+ s->vmfd = ret;
2652
+
2653
+ kvm_setup_dirty_ring(s);
2654
+
2655
+ /* rebind memory to new vm fd */
2656
+ ret = ram_block_rebind(&err);
2657
+ if (ret < 0) {
2658
+ return ret;
2659
+ }
2660
+ assert(!err);
2661
+
2662
+ ret = kvm_arch_on_vmfd_change(ms, s);
2663
+ if (ret < 0) {
2664
+ return ret;
2665
+ }
2666
+
2667
+ if (s->kernel_irqchip_allowed) {
2668
+ do_kvm_irqchip_create(s);
2669
+ }
2670
+
2671
+ /* these can be only called after ram_block_rebind() */
2672
+ memory_listener_register(&kml->listener, &address_space_memory);
2673
+ memory_listener_register(&kvm_io_listener, &address_space_io);
2674
+
2675
+ /*
2676
+ * kvm fd has changed. Commit the irq routes to KVM once more.
2677
+ */
2678
+ kvm_irqchip_commit_routes(s);
2679
+ /*
2680
+ * for confidential guest, this is the last possible place where we
2681
+ * can call synchronize_all_post_init() to sync all vcpu states to
2682
+ * kvm.
2683
+ */
2684
+ if (ms->cgs) {
2685
+ cpu_synchronize_all_post_init();
2686
+ }
2687
+ trace_kvm_reset_vmfd();
2688
+ return ret;
2689
+}
2690
+
2691
static int kvm_init(AccelState *as, MachineState *ms)
2692
{
2693
MachineClass *mc = MACHINE_GET_CLASS(ms);
@@ -4015,6 +4096,7 @@ static void kvm_accel_class_init(ObjectClass *oc, const void *data)
4096
AccelClass *ac = ACCEL_CLASS(oc);
4097
ac->name = "KVM";
4098
ac->init_machine = kvm_init;
4099
+ ac->rebuild_guest = kvm_reset_vmfd;
4100
ac->has_memory = kvm_accel_has_memory;
4101
ac->allowed = &kvm_allowed;
4102
ac->gdbstub_supported_sstep_flags = kvm_gdbstub_sstep_flags;
accel/kvm/trace-events
+1
@@ -14,6 +14,7 @@ kvm_destroy_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu"
14
kvm_park_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu"
15
kvm_unpark_vcpu(unsigned long arch_cpu_id, const char *msg) "id: %lu %s"
16
kvm_irqchip_commit_routes(void) ""
17
+kvm_reset_vmfd(void) ""
18
kvm_irqchip_add_msi_route(char *name, int vector, int virq) "dev %s vector %d virq %d"
19
kvm_irqchip_update_msi_route(int virq) "Updating MSI route virq=%d"
20
kvm_irqchip_release_virq(int virq) "virq %d"
include/system/kvm.h
+3
@@ -456,6 +456,9 @@ int kvm_physical_memory_addr_from_host(KVMState *s, void *ram_addr,
456
457
#endif /* COMPILING_PER_TARGET */
458
459
+bool kvm_arch_supports_vmfd_change(void);
460
+int kvm_arch_on_vmfd_change(MachineState *ms, KVMState *s);
461
+
462
void kvm_cpu_synchronize_state(CPUState *cpu);
463
464
void kvm_init_cpu_signals(CPUState *cpu);
stubs/kvm.c
new
+22
@@ -0,0 +1,22 @@
1
+/*
2
+ * kvm target arch specific stubs
3
+ *
4
+ * Copyright (c) 2026 Red Hat, Inc.
5
+ *
6
+ * Author:
7
+ * Ani Sinha <anisinha@redhat.com>
8
+ *
9
+ * SPDX-License-Identifier: GPL-2.0-or-later
10
+ */
11
+#include "qemu/osdep.h"
12
+#include "system/kvm.h"
13
+
14
+int kvm_arch_on_vmfd_change(MachineState *ms, KVMState *s)
15
+{
16
+ abort();
17
+}
18
+
19
+bool kvm_arch_supports_vmfd_change(void)
20
+{
21
+ return false;
22
+}
stubs/meson.build
+1
@@ -74,6 +74,7 @@ if have_system
74
if igvm.found()
75
stub_ss.add(files('igvm.c'))
76
endif
77
+ stub_ss.add(files('kvm.c'))
78
stub_ss.add(files('target-get-monitor-def.c'))
79
stub_ss.add(files('target-monitor-defs.c'))
80
stub_ss.add(files('win32-kbd-hook.c'))
target/i386/kvm/kvm.c
+10
@@ -3389,6 +3389,16 @@ static int kvm_vm_enable_energy_msrs(KVMState *s)
3389
return 0;
3390
}
3391
3392
+int kvm_arch_on_vmfd_change(MachineState *ms, KVMState *s)
3393
+{
3394
+ abort();
3395
+}
3396
+
3397
+bool kvm_arch_supports_vmfd_change(void)
3398
+{
3399
+ return false;
3400
+}
3401
+
3402
int kvm_arch_init(MachineState *ms, KVMState *s)
3403
{
3404
int ret;