accel/kvm: rebind current VCPUs to the new KVM VM file descriptor upon reset
Confidential guests needs to generate a new KVM file descriptor upon virtual machine reset. Existing VCPUs needs to be reattached to this new KVM VM file descriptor. As a part of this, new VCPU file descriptors against this new KVM VM file descriptor needs to be created and re-initialized. Resources allocated against the old VCPU fds needs to be released. This change makes this happen. Signed-off-by: Ani Sinha <anisinha@redhat.com> Link: https://lore.kernel.org/r/20260225035000.385950-16-anisinha@redhat.com Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
Ani Sinha committed
Feb 25, 2026 at 09:19 UTC
083ce77fc435660330d7497927340020969a29c3
2 files changed
+174
-42
accel/kvm/kvm-all.c
+173
-42
@@ -127,6 +127,10 @@ static NotifierList kvm_irqchip_change_notifiers =
127
static NotifierWithReturnList register_vmfd_changed_notifiers =
128
NOTIFIER_WITH_RETURN_LIST_INITIALIZER(register_vmfd_changed_notifiers);
129
130
+static int map_kvm_run(KVMState *s, CPUState *cpu, Error **errp);
131
+static int map_kvm_dirty_gfns(KVMState *s, CPUState *cpu, Error **errp);
132
+static int vcpu_unmap_regions(KVMState *s, CPUState *cpu);
133
+
134
struct KVMResampleFd {
135
int gsi;
136
EventNotifier *resample_event;
@@ -420,6 +424,90 @@ err:
424
return ret;
425
}
426
427
+static void kvm_create_vcpu_internal(CPUState *cpu, KVMState *s, int kvm_fd)
428
+{
429
+ cpu->kvm_fd = kvm_fd;
430
+ cpu->kvm_state = s;
431
+ if (!s->guest_state_protected) {
432
+ cpu->vcpu_dirty = true;
433
+ }
434
+ cpu->dirty_pages = 0;
435
+ cpu->throttle_us_per_full = 0;
436
+
437
+ return;
438
+}
439
+
440
+static int kvm_rebind_vcpus(Error **errp)
441
+{
442
+ CPUState *cpu;
443
+ unsigned long vcpu_id;
444
+ KVMState *s = kvm_state;
445
+ int kvm_fd, ret = 0;
446
+
447
+ CPU_FOREACH(cpu) {
448
+ vcpu_id = kvm_arch_vcpu_id(cpu);
449
+
450
+ if (cpu->kvm_fd) {
451
+ close(cpu->kvm_fd);
452
+ }
453
+
454
+ ret = kvm_arch_destroy_vcpu(cpu);
455
+ if (ret < 0) {
456
+ goto err;
457
+ }
458
+
459
+ if (s->coalesced_mmio_ring == (void *)cpu->kvm_run + PAGE_SIZE) {
460
+ s->coalesced_mmio_ring = NULL;
461
+ }
462
+
463
+ ret = vcpu_unmap_regions(s, cpu);
464
+ if (ret < 0) {
465
+ goto err;
466
+ }
467
+
468
+ ret = kvm_arch_pre_create_vcpu(cpu, errp);
469
+ if (ret < 0) {
470
+ goto err;
471
+ }
472
+
473
+ kvm_fd = kvm_vm_ioctl(s, KVM_CREATE_VCPU, vcpu_id);
474
+ if (kvm_fd < 0) {
475
+ error_report("KVM_CREATE_VCPU IOCTL failed for vCPU %lu (%s)",
476
+ vcpu_id, strerror(kvm_fd));
477
+ return kvm_fd;
478
+ }
479
+
480
+ kvm_create_vcpu_internal(cpu, s, kvm_fd);
481
+
482
+ ret = map_kvm_run(s, cpu, errp);
483
+ if (ret < 0) {
484
+ goto err;
485
+ }
486
+
487
+ if (s->kvm_dirty_ring_size) {
488
+ ret = map_kvm_dirty_gfns(s, cpu, errp);
489
+ if (ret < 0) {
490
+ goto err;
491
+ }
492
+ }
493
+
494
+ ret = kvm_arch_init_vcpu(cpu);
495
+ if (ret < 0) {
496
+ error_setg_errno(errp, -ret,
497
+ "kvm_init_vcpu: kvm_arch_init_vcpu failed (%lu)",
498
+ vcpu_id);
499
+ }
500
+
501
+ close(cpu->kvm_vcpu_stats_fd);
502
+ cpu->kvm_vcpu_stats_fd = kvm_vcpu_ioctl(cpu, KVM_GET_STATS_FD, NULL);
503
+ kvm_init_cpu_signals(cpu);
504
+ }
505
+ trace_kvm_rebind_vcpus();
506
+
507
+ err:
508
+ return ret;
509
+}
510
+
511
static void kvm_park_vcpu(CPUState *cpu)
512
{
513
struct KVMParkedVcpu *vcpu;
@@ -483,13 +571,7 @@ static int kvm_create_vcpu(CPUState *cpu)
571
}
572
}
573
486
- cpu->kvm_fd = kvm_fd;
487
- cpu->kvm_state = s;
488
- if (!s->guest_state_protected) {
489
- cpu->vcpu_dirty = true;
490
- }
491
- cpu->dirty_pages = 0;
492
- cpu->throttle_us_per_full = 0;
574
+ kvm_create_vcpu_internal(cpu, s, kvm_fd);
575
576
trace_kvm_create_vcpu(cpu->cpu_index, vcpu_id, kvm_fd);
577
@@ -508,19 +590,11 @@ int kvm_create_and_park_vcpu(CPUState *cpu)
590
return ret;
591
}
592
511
-static int do_kvm_destroy_vcpu(CPUState *cpu)
593
+static int vcpu_unmap_regions(KVMState *s, CPUState *cpu)
594
{
513
- KVMState *s = kvm_state;
595
int mmap_size;
596
int ret = 0;
597
517
- trace_kvm_destroy_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
518
-
519
- ret = kvm_arch_destroy_vcpu(cpu);
520
- if (ret < 0) {
521
- goto err;
522
- }
523
-
598
mmap_size = kvm_ioctl(s, KVM_GET_VCPU_MMAP_SIZE, 0);
599
if (mmap_size < 0) {
600
ret = mmap_size;
@@ -548,39 +622,47 @@ static int do_kvm_destroy_vcpu(CPUState *cpu)
622
cpu->kvm_dirty_gfns = NULL;
623
}
624
551
- kvm_park_vcpu(cpu);
552
-err:
625
+ err:
626
return ret;
627
}
628
556
-void kvm_destroy_vcpu(CPUState *cpu)
557
-{
558
- if (do_kvm_destroy_vcpu(cpu) < 0) {
559
- error_report("kvm_destroy_vcpu failed");
560
- exit(EXIT_FAILURE);
561
- }
562
-}
563
-
564
-int kvm_init_vcpu(CPUState *cpu, Error **errp)
629
+static int do_kvm_destroy_vcpu(CPUState *cpu)
630
{
631
KVMState *s = kvm_state;
567
- int mmap_size;
568
- int ret;
632
+ int ret = 0;
633
570
- trace_kvm_init_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
634
+ trace_kvm_destroy_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
635
572
- ret = kvm_arch_pre_create_vcpu(cpu, errp);
636
+ ret = kvm_arch_destroy_vcpu(cpu);
637
if (ret < 0) {
638
goto err;
639
}
640
577
- ret = kvm_create_vcpu(cpu);
641
+ /* If I am the CPU that created coalesced_mmio_ring, then discard it */
642
+ if (s->coalesced_mmio_ring == (void *)cpu->kvm_run + PAGE_SIZE) {
643
+ s->coalesced_mmio_ring = NULL;
644
+ }
645
+
646
+ ret = vcpu_unmap_regions(s, cpu);
647
if (ret < 0) {
579
- error_setg_errno(errp, -ret,
580
- "kvm_init_vcpu: kvm_create_vcpu failed (%lu)",
581
- kvm_arch_vcpu_id(cpu));
648
goto err;
649
}
650
+ kvm_park_vcpu(cpu);
651
+err:
652
+ return ret;
653
+}
654
+
655
+void kvm_destroy_vcpu(CPUState *cpu)
656
+{
657
+ if (do_kvm_destroy_vcpu(cpu) < 0) {
658
+ error_report("kvm_destroy_vcpu failed");
659
+ exit(EXIT_FAILURE);
660
+ }
661
+}
662
+
663
+static int map_kvm_run(KVMState *s, CPUState *cpu, Error **errp)
664
+{
665
+ int mmap_size, ret = 0;
666
667
mmap_size = kvm_ioctl(s, KVM_GET_VCPU_MMAP_SIZE, 0);
668
if (mmap_size < 0) {
@@ -605,14 +687,53 @@ int kvm_init_vcpu(CPUState *cpu, Error **errp)
687
(void *)cpu->kvm_run + s->coalesced_mmio * PAGE_SIZE;
688
}
689
690
+ err:
691
+ return ret;
692
+}
693
+
694
+static int map_kvm_dirty_gfns(KVMState *s, CPUState *cpu, Error **errp)
695
+{
696
+ int ret = 0;
697
+ /* Use MAP_SHARED to share pages with the kernel */
698
+ cpu->kvm_dirty_gfns = mmap(NULL, s->kvm_dirty_ring_bytes,
699
+ PROT_READ | PROT_WRITE, MAP_SHARED,
700
+ cpu->kvm_fd,
701
+ PAGE_SIZE * KVM_DIRTY_LOG_PAGE_OFFSET);
702
+ if (cpu->kvm_dirty_gfns == MAP_FAILED) {
703
+ ret = -errno;
704
+ }
705
+
706
+ return ret;
707
+}
708
+
709
+int kvm_init_vcpu(CPUState *cpu, Error **errp)
710
+{
711
+ KVMState *s = kvm_state;
712
+ int ret;
713
+
714
+ trace_kvm_init_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
715
+
716
+ ret = kvm_arch_pre_create_vcpu(cpu, errp);
717
+ if (ret < 0) {
718
+ goto err;
719
+ }
720
+
721
+ ret = kvm_create_vcpu(cpu);
722
+ if (ret < 0) {
723
+ error_setg_errno(errp, -ret,
724
+ "kvm_init_vcpu: kvm_create_vcpu failed (%lu)",
725
+ kvm_arch_vcpu_id(cpu));
726
+ goto err;
727
+ }
728
+
729
+ ret = map_kvm_run(s, cpu, errp);
730
+ if (ret < 0) {
731
+ goto err;
732
+ }
733
+
734
if (s->kvm_dirty_ring_size) {
609
- /* Use MAP_SHARED to share pages with the kernel */
610
- cpu->kvm_dirty_gfns = mmap(NULL, s->kvm_dirty_ring_bytes,
611
- PROT_READ | PROT_WRITE, MAP_SHARED,
612
- cpu->kvm_fd,
613
- PAGE_SIZE * KVM_DIRTY_LOG_PAGE_OFFSET);
614
- if (cpu->kvm_dirty_gfns == MAP_FAILED) {
615
- ret = -errno;
735
+ ret = map_kvm_dirty_gfns(s, cpu, errp);
736
+ if (ret < 0) {
737
goto err;
738
}
739
}
@@ -2710,6 +2831,16 @@ static int kvm_reset_vmfd(MachineState *ms)
2831
}
2832
assert(!err);
2833
2834
+ /*
2835
+ * rebind new vcpu fds with the new kvm fds
2836
+ * These can only be called after kvm_arch_on_vmfd_change()
2837
+ */
2838
+ ret = kvm_rebind_vcpus(&err);
2839
+ if (ret < 0) {
2840
+ return ret;
2841
+ }
2842
+ assert(!err);
2843
+
2844
/* these can be only called after ram_block_rebind() */
2845
memory_listener_register(&kml->listener, &address_space_memory);
2846
memory_listener_register(&kvm_io_listener, &address_space_io);
accel/kvm/trace-events
+1
@@ -15,6 +15,7 @@ kvm_park_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu"
15
kvm_unpark_vcpu(unsigned long arch_cpu_id, const char *msg) "id: %lu %s"
16
kvm_irqchip_commit_routes(void) ""
17
kvm_reset_vmfd(void) ""
18
+kvm_rebind_vcpus(void) ""
19
kvm_irqchip_add_msi_route(char *name, int vector, int virq) "dev %s vector %d virq %d"
20
kvm_irqchip_update_msi_route(int virq) "Updating MSI route virq=%d"
21
kvm_irqchip_release_virq(int virq) "virq %d"