@samitouri / QOSamiQemu / commits / 083ce77fc4

accel/kvm: rebind current VCPUs to the new KVM VM file descriptor upon reset

Confidential guests needs to generate a new KVM file descriptor upon virtual machine reset. Existing VCPUs needs to be reattached to this new KVM VM file descriptor. As a part of this, new VCPU file descriptors against this new KVM VM file descriptor needs to be created and re-initialized. Resources allocated against the old VCPU fds needs to be released. This change makes this happen. Signed-off-by: Ani Sinha <anisinha@redhat.com> Link: https://lore.kernel.org/r/20260225035000.385950-16-anisinha@redhat.com Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>

Ani Sinha committed Feb 25, 2026 at 09:19 UTC 083ce77fc435660330d7497927340020969a29c3
2 files changed +174 -42
accel/kvm/kvm-all.c
+173 -42
@@ -127,6 +127,10 @@ static NotifierList kvm_irqchip_change_notifiers =
127 static NotifierWithReturnList register_vmfd_changed_notifiers =
128 NOTIFIER_WITH_RETURN_LIST_INITIALIZER(register_vmfd_changed_notifiers);
129
130 +static int map_kvm_run(KVMState *s, CPUState *cpu, Error **errp);
131 +static int map_kvm_dirty_gfns(KVMState *s, CPUState *cpu, Error **errp);
132 +static int vcpu_unmap_regions(KVMState *s, CPUState *cpu);
133 +
134 struct KVMResampleFd {
135 int gsi;
136 EventNotifier *resample_event;
@@ -420,6 +424,90 @@ err:
424 return ret;
425 }
426
427 +static void kvm_create_vcpu_internal(CPUState *cpu, KVMState *s, int kvm_fd)
428 +{
429 + cpu->kvm_fd = kvm_fd;
430 + cpu->kvm_state = s;
431 + if (!s->guest_state_protected) {
432 + cpu->vcpu_dirty = true;
433 + }
434 + cpu->dirty_pages = 0;
435 + cpu->throttle_us_per_full = 0;
436 +
437 + return;
438 +}
439 +
440 +static int kvm_rebind_vcpus(Error **errp)
441 +{
442 + CPUState *cpu;
443 + unsigned long vcpu_id;
444 + KVMState *s = kvm_state;
445 + int kvm_fd, ret = 0;
446 +
447 + CPU_FOREACH(cpu) {
448 + vcpu_id = kvm_arch_vcpu_id(cpu);
449 +
450 + if (cpu->kvm_fd) {
451 + close(cpu->kvm_fd);
452 + }
453 +
454 + ret = kvm_arch_destroy_vcpu(cpu);
455 + if (ret < 0) {
456 + goto err;
457 + }
458 +
459 + if (s->coalesced_mmio_ring == (void *)cpu->kvm_run + PAGE_SIZE) {
460 + s->coalesced_mmio_ring = NULL;
461 + }
462 +
463 + ret = vcpu_unmap_regions(s, cpu);
464 + if (ret < 0) {
465 + goto err;
466 + }
467 +
468 + ret = kvm_arch_pre_create_vcpu(cpu, errp);
469 + if (ret < 0) {
470 + goto err;
471 + }
472 +
473 + kvm_fd = kvm_vm_ioctl(s, KVM_CREATE_VCPU, vcpu_id);
474 + if (kvm_fd < 0) {
475 + error_report("KVM_CREATE_VCPU IOCTL failed for vCPU %lu (%s)",
476 + vcpu_id, strerror(kvm_fd));
477 + return kvm_fd;
478 + }
479 +
480 + kvm_create_vcpu_internal(cpu, s, kvm_fd);
481 +
482 + ret = map_kvm_run(s, cpu, errp);
483 + if (ret < 0) {
484 + goto err;
485 + }
486 +
487 + if (s->kvm_dirty_ring_size) {
488 + ret = map_kvm_dirty_gfns(s, cpu, errp);
489 + if (ret < 0) {
490 + goto err;
491 + }
492 + }
493 +
494 + ret = kvm_arch_init_vcpu(cpu);
495 + if (ret < 0) {
496 + error_setg_errno(errp, -ret,
497 + "kvm_init_vcpu: kvm_arch_init_vcpu failed (%lu)",
498 + vcpu_id);
499 + }
500 +
501 + close(cpu->kvm_vcpu_stats_fd);
502 + cpu->kvm_vcpu_stats_fd = kvm_vcpu_ioctl(cpu, KVM_GET_STATS_FD, NULL);
503 + kvm_init_cpu_signals(cpu);
504 + }
505 + trace_kvm_rebind_vcpus();
506 +
507 + err:
508 + return ret;
509 +}
510 +
511 static void kvm_park_vcpu(CPUState *cpu)
512 {
513 struct KVMParkedVcpu *vcpu;
@@ -483,13 +571,7 @@ static int kvm_create_vcpu(CPUState *cpu)
571 }
572 }
573
486 - cpu->kvm_fd = kvm_fd;
487 - cpu->kvm_state = s;
488 - if (!s->guest_state_protected) {
489 - cpu->vcpu_dirty = true;
490 - }
491 - cpu->dirty_pages = 0;
492 - cpu->throttle_us_per_full = 0;
574 + kvm_create_vcpu_internal(cpu, s, kvm_fd);
575
576 trace_kvm_create_vcpu(cpu->cpu_index, vcpu_id, kvm_fd);
577
@@ -508,19 +590,11 @@ int kvm_create_and_park_vcpu(CPUState *cpu)
590 return ret;
591 }
592
511 -static int do_kvm_destroy_vcpu(CPUState *cpu)
593 +static int vcpu_unmap_regions(KVMState *s, CPUState *cpu)
594 {
513 - KVMState *s = kvm_state;
595 int mmap_size;
596 int ret = 0;
597
517 - trace_kvm_destroy_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
518 -
519 - ret = kvm_arch_destroy_vcpu(cpu);
520 - if (ret < 0) {
521 - goto err;
522 - }
523 -
598 mmap_size = kvm_ioctl(s, KVM_GET_VCPU_MMAP_SIZE, 0);
599 if (mmap_size < 0) {
600 ret = mmap_size;
@@ -548,39 +622,47 @@ static int do_kvm_destroy_vcpu(CPUState *cpu)
622 cpu->kvm_dirty_gfns = NULL;
623 }
624
551 - kvm_park_vcpu(cpu);
552 -err:
625 + err:
626 return ret;
627 }
628
556 -void kvm_destroy_vcpu(CPUState *cpu)
557 -{
558 - if (do_kvm_destroy_vcpu(cpu) < 0) {
559 - error_report("kvm_destroy_vcpu failed");
560 - exit(EXIT_FAILURE);
561 - }
562 -}
563 -
564 -int kvm_init_vcpu(CPUState *cpu, Error **errp)
629 +static int do_kvm_destroy_vcpu(CPUState *cpu)
630 {
631 KVMState *s = kvm_state;
567 - int mmap_size;
568 - int ret;
632 + int ret = 0;
633
570 - trace_kvm_init_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
634 + trace_kvm_destroy_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
635
572 - ret = kvm_arch_pre_create_vcpu(cpu, errp);
636 + ret = kvm_arch_destroy_vcpu(cpu);
637 if (ret < 0) {
638 goto err;
639 }
640
577 - ret = kvm_create_vcpu(cpu);
641 + /* If I am the CPU that created coalesced_mmio_ring, then discard it */
642 + if (s->coalesced_mmio_ring == (void *)cpu->kvm_run + PAGE_SIZE) {
643 + s->coalesced_mmio_ring = NULL;
644 + }
645 +
646 + ret = vcpu_unmap_regions(s, cpu);
647 if (ret < 0) {
579 - error_setg_errno(errp, -ret,
580 - "kvm_init_vcpu: kvm_create_vcpu failed (%lu)",
581 - kvm_arch_vcpu_id(cpu));
648 goto err;
649 }
650 + kvm_park_vcpu(cpu);
651 +err:
652 + return ret;
653 +}
654 +
655 +void kvm_destroy_vcpu(CPUState *cpu)
656 +{
657 + if (do_kvm_destroy_vcpu(cpu) < 0) {
658 + error_report("kvm_destroy_vcpu failed");
659 + exit(EXIT_FAILURE);
660 + }
661 +}
662 +
663 +static int map_kvm_run(KVMState *s, CPUState *cpu, Error **errp)
664 +{
665 + int mmap_size, ret = 0;
666
667 mmap_size = kvm_ioctl(s, KVM_GET_VCPU_MMAP_SIZE, 0);
668 if (mmap_size < 0) {
@@ -605,14 +687,53 @@ int kvm_init_vcpu(CPUState *cpu, Error **errp)
687 (void *)cpu->kvm_run + s->coalesced_mmio * PAGE_SIZE;
688 }
689
690 + err:
691 + return ret;
692 +}
693 +
694 +static int map_kvm_dirty_gfns(KVMState *s, CPUState *cpu, Error **errp)
695 +{
696 + int ret = 0;
697 + /* Use MAP_SHARED to share pages with the kernel */
698 + cpu->kvm_dirty_gfns = mmap(NULL, s->kvm_dirty_ring_bytes,
699 + PROT_READ | PROT_WRITE, MAP_SHARED,
700 + cpu->kvm_fd,
701 + PAGE_SIZE * KVM_DIRTY_LOG_PAGE_OFFSET);
702 + if (cpu->kvm_dirty_gfns == MAP_FAILED) {
703 + ret = -errno;
704 + }
705 +
706 + return ret;
707 +}
708 +
709 +int kvm_init_vcpu(CPUState *cpu, Error **errp)
710 +{
711 + KVMState *s = kvm_state;
712 + int ret;
713 +
714 + trace_kvm_init_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu));
715 +
716 + ret = kvm_arch_pre_create_vcpu(cpu, errp);
717 + if (ret < 0) {
718 + goto err;
719 + }
720 +
721 + ret = kvm_create_vcpu(cpu);
722 + if (ret < 0) {
723 + error_setg_errno(errp, -ret,
724 + "kvm_init_vcpu: kvm_create_vcpu failed (%lu)",
725 + kvm_arch_vcpu_id(cpu));
726 + goto err;
727 + }
728 +
729 + ret = map_kvm_run(s, cpu, errp);
730 + if (ret < 0) {
731 + goto err;
732 + }
733 +
734 if (s->kvm_dirty_ring_size) {
609 - /* Use MAP_SHARED to share pages with the kernel */
610 - cpu->kvm_dirty_gfns = mmap(NULL, s->kvm_dirty_ring_bytes,
611 - PROT_READ | PROT_WRITE, MAP_SHARED,
612 - cpu->kvm_fd,
613 - PAGE_SIZE * KVM_DIRTY_LOG_PAGE_OFFSET);
614 - if (cpu->kvm_dirty_gfns == MAP_FAILED) {
615 - ret = -errno;
735 + ret = map_kvm_dirty_gfns(s, cpu, errp);
736 + if (ret < 0) {
737 goto err;
738 }
739 }
@@ -2710,6 +2831,16 @@ static int kvm_reset_vmfd(MachineState *ms)
2831 }
2832 assert(!err);
2833
2834 + /*
2835 + * rebind new vcpu fds with the new kvm fds
2836 + * These can only be called after kvm_arch_on_vmfd_change()
2837 + */
2838 + ret = kvm_rebind_vcpus(&err);
2839 + if (ret < 0) {
2840 + return ret;
2841 + }
2842 + assert(!err);
2843 +
2844 /* these can be only called after ram_block_rebind() */
2845 memory_listener_register(&kml->listener, &address_space_memory);
2846 memory_listener_register(&kvm_io_listener, &address_space_io);
accel/kvm/trace-events
+1
@@ -15,6 +15,7 @@ kvm_park_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu"
15 kvm_unpark_vcpu(unsigned long arch_cpu_id, const char *msg) "id: %lu %s"
16 kvm_irqchip_commit_routes(void) ""
17 kvm_reset_vmfd(void) ""
18 +kvm_rebind_vcpus(void) ""
19 kvm_irqchip_add_msi_route(char *name, int vector, int virq) "dev %s vector %d virq %d"
20 kvm_irqchip_update_msi_route(int virq) "Updating MSI route virq=%d"
21 kvm_irqchip_release_virq(int virq) "virq %d"