@samitouri / QOSamiQemu / commits / 98884e0cc1

accel/kvm: add changes required to support KVM VM file descriptor change

This change adds common kvm specific support to handle KVM VM file descriptor change. KVM VM file descriptor can change as a part of confidential guest reset mechanism. A new function api kvm_arch_on_vmfd_change() per architecture platform is added in order to implement architecture specific changes required to support it. A subsequent patch will add x86 specific implementation for kvm_arch_on_vmfd_change() as currently only x86 supports confidential guest reset. Signed-off-by: Ani Sinha <anisinha@redhat.com> Link: https://lore.kernel.org/r/20260225035000.385950-6-anisinha@redhat.com Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>

Ani Sinha committed Feb 25, 2026 at 09:19 UTC 98884e0cc10997a17ce9abfd6ff10be19224ca6a
7 files changed +128 -3
MAINTAINERS
+6
@@ -152,6 +152,12 @@ F: tools/i386/
152 F: tests/functional/i386/
153 F: tests/functional/x86_64/
154
155 +X86 VM file descriptor change on reset test
156 +M: Ani Sinha <anisinha@redhat.com>
157 +M: Paolo Bonzini <pbonzini@redhat.com>
158 +S: Maintained
159 +F: stubs/kvm.c
160 +
161 Guest CPU cores (TCG)
162 ---------------------
163 Overall TCG CPUs
accel/kvm/kvm-all.c
+85 -3
@@ -2415,11 +2415,9 @@ void kvm_irqchip_set_qemuirq_gsi(KVMState *s, qemu_irq irq, int gsi)
2415 g_hash_table_insert(s->gsimap, irq, GINT_TO_POINTER(gsi));
2416 }
2417
2418 -static void kvm_irqchip_create(KVMState *s)
2418 +static void do_kvm_irqchip_create(KVMState *s)
2419 {
2420 int ret;
2421 -
2422 - assert(s->kernel_irqchip_split != ON_OFF_AUTO_AUTO);
2421 if (kvm_check_extension(s, KVM_CAP_IRQCHIP)) {
2422 ;
2423 } else if (kvm_check_extension(s, KVM_CAP_S390_IRQCHIP)) {
@@ -2452,7 +2450,13 @@ static void kvm_irqchip_create(KVMState *s)
2450 fprintf(stderr, "Create kernel irqchip failed: %s\n", strerror(-ret));
2451 exit(1);
2452 }
2453 +}
2454 +
2455 +static void kvm_irqchip_create(KVMState *s)
2456 +{
2457 + assert(s->kernel_irqchip_split != ON_OFF_AUTO_AUTO);
2458
2459 + do_kvm_irqchip_create(s);
2460 kvm_kernel_irqchip = true;
2461 /* If we have an in-kernel IRQ chip then we must have asynchronous
2462 * interrupt delivery (though the reverse is not necessarily true)
@@ -2607,6 +2611,83 @@ static int kvm_setup_dirty_ring(KVMState *s)
2611 return 0;
2612 }
2613
2614 +static int kvm_reset_vmfd(MachineState *ms)
2615 +{
2616 + KVMState *s;
2617 + KVMMemoryListener *kml;
2618 + int ret = 0, type;
2619 + Error *err = NULL;
2620 +
2621 + /*
2622 + * bail if the current architecture does not support VM file
2623 + * descriptor change.
2624 + */
2625 + if (!kvm_arch_supports_vmfd_change()) {
2626 + error_report("This target architecture does not support KVM VM "
2627 + "file descriptor change.");
2628 + return -EOPNOTSUPP;
2629 + }
2630 +
2631 + s = KVM_STATE(ms->accelerator);
2632 + kml = &s->memory_listener;
2633 +
2634 + memory_listener_unregister(&kml->listener);
2635 + memory_listener_unregister(&kvm_io_listener);
2636 +
2637 + if (s->vmfd >= 0) {
2638 + close(s->vmfd);
2639 + }
2640 +
2641 + type = find_kvm_machine_type(ms);
2642 + if (type < 0) {
2643 + return -EINVAL;
2644 + }
2645 +
2646 + ret = do_kvm_create_vm(s, type);
2647 + if (ret < 0) {
2648 + return ret;
2649 + }
2650 +
2651 + s->vmfd = ret;
2652 +
2653 + kvm_setup_dirty_ring(s);
2654 +
2655 + /* rebind memory to new vm fd */
2656 + ret = ram_block_rebind(&err);
2657 + if (ret < 0) {
2658 + return ret;
2659 + }
2660 + assert(!err);
2661 +
2662 + ret = kvm_arch_on_vmfd_change(ms, s);
2663 + if (ret < 0) {
2664 + return ret;
2665 + }
2666 +
2667 + if (s->kernel_irqchip_allowed) {
2668 + do_kvm_irqchip_create(s);
2669 + }
2670 +
2671 + /* these can be only called after ram_block_rebind() */
2672 + memory_listener_register(&kml->listener, &address_space_memory);
2673 + memory_listener_register(&kvm_io_listener, &address_space_io);
2674 +
2675 + /*
2676 + * kvm fd has changed. Commit the irq routes to KVM once more.
2677 + */
2678 + kvm_irqchip_commit_routes(s);
2679 + /*
2680 + * for confidential guest, this is the last possible place where we
2681 + * can call synchronize_all_post_init() to sync all vcpu states to
2682 + * kvm.
2683 + */
2684 + if (ms->cgs) {
2685 + cpu_synchronize_all_post_init();
2686 + }
2687 + trace_kvm_reset_vmfd();
2688 + return ret;
2689 +}
2690 +
2691 static int kvm_init(AccelState *as, MachineState *ms)
2692 {
2693 MachineClass *mc = MACHINE_GET_CLASS(ms);
@@ -4015,6 +4096,7 @@ static void kvm_accel_class_init(ObjectClass *oc, const void *data)
4096 AccelClass *ac = ACCEL_CLASS(oc);
4097 ac->name = "KVM";
4098 ac->init_machine = kvm_init;
4099 + ac->rebuild_guest = kvm_reset_vmfd;
4100 ac->has_memory = kvm_accel_has_memory;
4101 ac->allowed = &kvm_allowed;
4102 ac->gdbstub_supported_sstep_flags = kvm_gdbstub_sstep_flags;
accel/kvm/trace-events
+1
@@ -14,6 +14,7 @@ kvm_destroy_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu"
14 kvm_park_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu"
15 kvm_unpark_vcpu(unsigned long arch_cpu_id, const char *msg) "id: %lu %s"
16 kvm_irqchip_commit_routes(void) ""
17 +kvm_reset_vmfd(void) ""
18 kvm_irqchip_add_msi_route(char *name, int vector, int virq) "dev %s vector %d virq %d"
19 kvm_irqchip_update_msi_route(int virq) "Updating MSI route virq=%d"
20 kvm_irqchip_release_virq(int virq) "virq %d"
include/system/kvm.h
+3
@@ -456,6 +456,9 @@ int kvm_physical_memory_addr_from_host(KVMState *s, void *ram_addr,
456
457 #endif /* COMPILING_PER_TARGET */
458
459 +bool kvm_arch_supports_vmfd_change(void);
460 +int kvm_arch_on_vmfd_change(MachineState *ms, KVMState *s);
461 +
462 void kvm_cpu_synchronize_state(CPUState *cpu);
463
464 void kvm_init_cpu_signals(CPUState *cpu);
stubs/kvm.c new
+22
@@ -0,0 +1,22 @@
1 +/*
2 + * kvm target arch specific stubs
3 + *
4 + * Copyright (c) 2026 Red Hat, Inc.
5 + *
6 + * Author:
7 + * Ani Sinha <anisinha@redhat.com>
8 + *
9 + * SPDX-License-Identifier: GPL-2.0-or-later
10 + */
11 +#include "qemu/osdep.h"
12 +#include "system/kvm.h"
13 +
14 +int kvm_arch_on_vmfd_change(MachineState *ms, KVMState *s)
15 +{
16 + abort();
17 +}
18 +
19 +bool kvm_arch_supports_vmfd_change(void)
20 +{
21 + return false;
22 +}
stubs/meson.build
+1
@@ -74,6 +74,7 @@ if have_system
74 if igvm.found()
75 stub_ss.add(files('igvm.c'))
76 endif
77 + stub_ss.add(files('kvm.c'))
78 stub_ss.add(files('target-get-monitor-def.c'))
79 stub_ss.add(files('target-monitor-defs.c'))
80 stub_ss.add(files('win32-kbd-hook.c'))
target/i386/kvm/kvm.c
+10
@@ -3389,6 +3389,16 @@ static int kvm_vm_enable_energy_msrs(KVMState *s)
3389 return 0;
3390 }
3391
3392 +int kvm_arch_on_vmfd_change(MachineState *ms, KVMState *s)
3393 +{
3394 + abort();
3395 +}
3396 +
3397 +bool kvm_arch_supports_vmfd_change(void)
3398 +{
3399 + return false;
3400 +}
3401 +
3402 int kvm_arch_init(MachineState *ms, KVMState *s)
3403 {
3404 int ret;