]> exis.tech > repos - linux.git/commitdiff
Merge tag 'kvm-x86-pmu-6.20' of https://github.com/kvm-x86/linux into HEAD
authorPaolo Bonzini <pbonzini@redhat.com>
Mon, 9 Feb 2026 18:35:16 +0000 (19:35 +0100)
committerPaolo Bonzini <pbonzini@redhat.com>
Wed, 11 Feb 2026 17:45:40 +0000 (12:45 -0500)
KVM mediated PMU support for 6.20

Add support for mediated PMUs, where KVM gives the guest full ownership of PMU
hardware (contexted switched around the fastpath run loop) and allows direct
access to data MSRs and PMCs (restricted by the vPMU model), but intercepts
access to control registers, e.g. to enforce event filtering and to prevent the
guest from profiling sensitive host state.

To keep overall complexity reasonable, mediated PMU usage is all or nothing
for a given instance of KVM (controlled via module param).  The Mediated PMU
is disabled default, partly to maintain backwards compatilibity for existing
setup, partly because there are tradeoffs when running with a mediated PMU that
may be non-starters for some use cases, e.g. the host loses the ability to
profile guests with mediated PMUs, the fastpath run loop is also a blind spot,
entry/exit transitions are more expensive, etc.

Versus the emulated PMU, where KVM is "just another perf user", the mediated
PMU delivers more accurate profiling and monitoring (no risk of contention and
thus dropped events), with significantly less overhead (fewer exits and faster
emulation/programming of event selectors) E.g. when running Specint-2017 on
a single-socket Sapphire Rapids with 56 cores and no-SMT, and using perf from
within the guest:

  Perf command:
  a. basic-sampling: perf record -F 1000 -e 6-instructions  -a --overwrite
  b. multiplex-sampling: perf record -F 1000 -e 10-instructions -a --overwrite

  Guest performance overhead:
  ---------------------------------------------------------------------------
  | Test case          | emulated vPMU | all passthrough | passthrough with |
  |                    |               |                 | event filters    |
  ---------------------------------------------------------------------------
  | basic-sampling     |   33.62%      |    4.24%        |   6.21%          |
  ---------------------------------------------------------------------------
  | multiplex-sampling |   79.32%      |    7.34%        |   10.45%         |
  ---------------------------------------------------------------------------

18 files changed:
1  2 
Documentation/admin-guide/kernel-parameters.txt
arch/arm64/kvm/arm.c
arch/loongarch/kvm/main.c
arch/x86/events/core.c
arch/x86/include/asm/kvm_host.h
arch/x86/kernel/irq.c
arch/x86/kvm/pmu.c
arch/x86/kvm/svm/nested.c
arch/x86/kvm/svm/svm.c
arch/x86/kvm/vmx/nested.c
arch/x86/kvm/vmx/vmx.c
arch/x86/kvm/vmx/vmx.h
arch/x86/kvm/x86.c
arch/x86/kvm/x86.h
include/linux/kvm_host.h
include/linux/perf_event.h
kernel/events/core.c
virt/kvm/kvm_main.c

index aa0031108bc1da46f32fb2f26f106a4232c104aa,c13a8877f5b378f6261e53bc527fc50e4906705c..8d62a6fdb152d469b76813484ede99a55e7a5af6
@@@ -2917,41 -2917,6 +2917,41 @@@ Kernel parameter
                        for Movable pages.  "nn[KMGTPE]", "nn%", and "mirror"
                        are exclusive, so you cannot specify multiple forms.
  
 +      kfence.burst=   [MM,KFENCE] The number of additional successive
 +                      allocations to be attempted through KFENCE for each
 +                      sample interval.
 +                      Format: <unsigned integer>
 +                      Default: 0
 +
 +      kfence.check_on_panic=
 +                      [MM,KFENCE] Whether to check all KFENCE-managed objects'
 +                      canaries on panic.
 +                      Format: <bool>
 +                      Default: false
 +
 +      kfence.deferrable=
 +                      [MM,KFENCE] Whether to use a deferrable timer to trigger
 +                      allocations. This avoids forcing CPU wake-ups if the
 +                      system is idle, at the risk of a less predictable
 +                      sample interval.
 +                      Format: <bool>
 +                      Default: CONFIG_KFENCE_DEFERRABLE
 +
 +      kfence.sample_interval=
 +                      [MM,KFENCE] KFENCE's sample interval in milliseconds.
 +                      Format: <unsigned integer>
 +                       0 - Disable KFENCE.
 +                      >0 - Enabled KFENCE with given sample interval.
 +                      Default: CONFIG_KFENCE_SAMPLE_INTERVAL
 +
 +      kfence.skip_covered_thresh=
 +                      [MM,KFENCE] If pool utilization reaches this threshold
 +                      (pool usage%), KFENCE limits currently covered
 +                      allocations of the same source from further filling
 +                      up the pool.
 +                      Format: <unsigned integer>
 +                      Default: 75
 +
        kgdbdbgp=       [KGDB,HW,EARLY] kgdb over EHCI usb debug port.
                        Format: <Controller#>[,poll interval]
                        The controller # is the number of the ehci usb debug
  
                        Default is Y (on).
  
+       kvm.enable_pmu=[KVM,X86]
+                       If enabled, KVM will virtualize PMU functionality based
+                       on the virtual CPU model defined by userspace.  This
+                       can be overridden on a per-VM basis via
+                       KVM_CAP_PMU_CAPABILITY.
+                       If disabled, KVM will not virtualize PMU functionality,
+                       e.g. MSRs, PMCs, PMIs, etc., even if userspace defines
+                       a virtual CPU model that contains PMU assets.
+                       Note, KVM's vPMU support implicitly requires running
+                       with an in-kernel local APIC, e.g. to deliver PMIs to
+                       the guest.  Running without an in-kernel local APIC is
+                       not supported, though KVM will allow such a combination
+                       (with severely degraded functionality).
+                       See also enable_mediated_pmu.
+                       Default is Y (on).
        kvm.enable_virt_at_load=[KVM,ARM64,LOONGARCH,MIPS,RISCV,X86]
                        If enabled, KVM will enable virtualization in hardware
                        when KVM is loaded, and disable virtualization when KVM
                        If the value is 0 (the default), KVM will pick a period based
                        on the ratio, such that a page is zapped after 1 hour on average.
  
+       kvm-{amd,intel}.enable_mediated_pmu=[KVM,AMD,INTEL]
+                       If enabled, KVM will provide a mediated virtual PMU,
+                       instead of the default perf-based virtual PMU (if
+                       kvm.enable_pmu is true and PMU is enumerated via the
+                       virtual CPU model).
+                       With a perf-based vPMU, KVM operates as a user of perf,
+                       i.e. emulates guest PMU counters using perf events.
+                       KVM-created perf events are managed by perf as regular
+                       (guest-only) events, e.g. are scheduled in/out, contend
+                       for hardware resources, etc.  Using a perf-based vPMU
+                       allows guest and host usage of the PMU to co-exist, but
+                       incurs non-trivial overhead and can result in silently
+                       dropped guest events (due to resource contention).
+                       With a mediated vPMU, hardware PMU state is context
+                       switched around the world switch to/from the guest.
+                       KVM mediates which events the guest can utilize, but
+                       gives the guest direct access to all other PMU assets
+                       when possible (KVM may intercept some accesses if the
+                       virtual CPU model provides a subset of hardware PMU
+                       functionality).  Using a mediated vPMU significantly
+                       reduces PMU virtualization overhead and eliminates lost
+                       guest events, but is mutually exclusive with using perf
+                       to profile KVM guests and adds latency to most VM-Exits
+                       (to context switch PMU state).
+                       Default is N (off).
        kvm-amd.nested= [KVM,AMD] Control nested virtualization feature in
                        KVM/SVM. Default is 1 (enabled).
  
                        If there are multiple matching configurations changing
                        the same attribute, the last one is used.
  
 +      liveupdate=     [KNL,EARLY]
 +                      Format: <bool>
 +                      Enable Live Update Orchestrator (LUO).
 +                      Default: off.
 +
        load_ramdisk=   [RAM] [Deprecated]
  
        lockd.nlm_grace_period=P  [NFS] Assign grace period.
diff --combined arch/arm64/kvm/arm.c
index 0f40d3da055017debc44065c30009783f3d50ff2,3e6f184d6d042633d45c87cb088c0efb64c826ef..94d5b0b99fd13b34411118aee6b7580213ca307c
@@@ -40,7 -40,6 +40,7 @@@
  #include <asm/kvm_pkvm.h>
  #include <asm/kvm_ptrauth.h>
  #include <asm/sections.h>
 +#include <asm/stacktrace/nvhe.h>
  
  #include <kvm/arm_hypercalls.h>
  #include <kvm/arm_pmu.h>
@@@ -59,51 -58,6 +59,51 @@@ enum kvm_wfx_trap_policy 
  static enum kvm_wfx_trap_policy kvm_wfi_trap_policy __read_mostly = KVM_WFX_NOTRAP_SINGLE_TASK;
  static enum kvm_wfx_trap_policy kvm_wfe_trap_policy __read_mostly = KVM_WFX_NOTRAP_SINGLE_TASK;
  
 +/*
 + * Tracks KVM IOCTLs and their associated KVM capabilities.
 + */
 +struct kvm_ioctl_cap_map {
 +      unsigned int ioctl;
 +      long ext;
 +};
 +
 +/* Make KVM_CAP_NR_VCPUS the reference for features we always supported */
 +#define KVM_CAP_ARM_BASIC     KVM_CAP_NR_VCPUS
 +
 +/*
 + * Sorted by ioctl to allow for potential binary search,
 + * though linear scan is sufficient for this size.
 + */
 +static const struct kvm_ioctl_cap_map vm_ioctl_caps[] = {
 +      { KVM_CREATE_IRQCHIP, KVM_CAP_IRQCHIP },
 +      { KVM_ARM_SET_DEVICE_ADDR, KVM_CAP_ARM_SET_DEVICE_ADDR },
 +      { KVM_ARM_MTE_COPY_TAGS, KVM_CAP_ARM_MTE },
 +      { KVM_SET_DEVICE_ATTR, KVM_CAP_DEVICE_CTRL },
 +      { KVM_GET_DEVICE_ATTR, KVM_CAP_DEVICE_CTRL },
 +      { KVM_HAS_DEVICE_ATTR, KVM_CAP_DEVICE_CTRL },
 +      { KVM_ARM_SET_COUNTER_OFFSET, KVM_CAP_COUNTER_OFFSET },
 +      { KVM_ARM_GET_REG_WRITABLE_MASKS, KVM_CAP_ARM_SUPPORTED_REG_MASK_RANGES },
 +      { KVM_ARM_PREFERRED_TARGET, KVM_CAP_ARM_BASIC },
 +};
 +
 +/*
 + * Set *ext to the capability.
 + * Return 0 if found, or -EINVAL if no IOCTL matches.
 + */
 +long kvm_get_cap_for_kvm_ioctl(unsigned int ioctl, long *ext)
 +{
 +      int i;
 +
 +      for (i = 0; i < ARRAY_SIZE(vm_ioctl_caps); i++) {
 +              if (vm_ioctl_caps[i].ioctl == ioctl) {
 +                      *ext = vm_ioctl_caps[i].ext;
 +                      return 0;
 +              }
 +      }
 +
 +      return -EINVAL;
 +}
 +
  DECLARE_KVM_HYP_PER_CPU(unsigned long, kvm_hyp_vector);
  
  DEFINE_PER_CPU(unsigned long, kvm_arm_hyp_stack_base);
@@@ -133,7 -87,7 +133,7 @@@ int kvm_vm_ioctl_enable_cap(struct kvm 
        if (cap->flags)
                return -EINVAL;
  
 -      if (kvm_vm_is_protected(kvm) && !kvm_pvm_ext_allowed(cap->cap))
 +      if (is_protected_kvm_enabled() && !kvm_pkvm_ext_allowed(kvm, cap->cap))
                return -EINVAL;
  
        switch (cap->cap) {
@@@ -349,7 -303,7 +349,7 @@@ int kvm_vm_ioctl_check_extension(struc
  {
        int r;
  
 -      if (kvm && kvm_vm_is_protected(kvm) && !kvm_pvm_ext_allowed(ext))
 +      if (is_protected_kvm_enabled() && !kvm_pkvm_ext_allowed(kvm, ext))
                return 0;
  
        switch (ext) {
@@@ -615,7 -569,6 +615,7 @@@ static bool kvm_vcpu_should_clear_twi(s
                return kvm_wfi_trap_policy == KVM_WFX_NOTRAP;
  
        return single_task_running() &&
 +             vcpu->kvm->arch.vgic.vgic_model == KVM_DEV_TYPE_ARM_VGIC_V3 &&
               (atomic_read(&vcpu->arch.vgic_cpu.vgic_v3.its_vpe.vlpi_count) ||
                vcpu->kvm->arch.vgic.nassgireq);
  }
@@@ -1940,9 -1893,6 +1940,9 @@@ int kvm_arch_vm_ioctl(struct file *filp
        void __user *argp = (void __user *)arg;
        struct kvm_device_attr attr;
  
 +      if (is_protected_kvm_enabled() && !kvm_pkvm_ioctl_allowed(kvm, ioctl))
 +              return -EINVAL;
 +
        switch (ioctl) {
        case KVM_CREATE_IRQCHIP: {
                int ret;
@@@ -2094,12 -2044,6 +2094,12 @@@ static void __init cpu_prepare_hyp_mode
                params->hcr_el2 = HCR_HOST_NVHE_PROTECTED_FLAGS;
        else
                params->hcr_el2 = HCR_HOST_NVHE_FLAGS;
 +
 +      if (system_supports_mte())
 +              params->hcr_el2 |= HCR_ATA;
 +      else
 +              params->hcr_el2 |= HCR_TID5;
 +
        if (cpus_have_final_cap(ARM64_KVM_HVHE))
                params->hcr_el2 |= HCR_E2H;
        params->vttbr = params->vtcr = 0;
@@@ -2413,7 -2357,7 +2413,7 @@@ static int __init init_subsystems(void
        if (err)
                goto out;
  
-       kvm_register_perf_callbacks(NULL);
+       kvm_register_perf_callbacks();
  
  out:
        if (err)
@@@ -2624,7 -2568,7 +2624,7 @@@ static void pkvm_hyp_init_ptrauth(void
  /* Inits Hyp-mode on all online CPUs */
  static int __init init_hyp_mode(void)
  {
 -      u32 hyp_va_bits;
 +      u32 hyp_va_bits = kvm_hyp_va_bits();
        int cpu;
        int err = -ENOMEM;
  
        /*
         * Allocate Hyp PGD and setup Hyp identity mapping
         */
 -      err = kvm_mmu_init(&hyp_va_bits);
 +      err = kvm_mmu_init(hyp_va_bits);
        if (err)
                goto out_err;
  
index d1c5156e02d860d26f726fb2e34c45901c13d7ea,f62326fe29fa30cc75c5a460c5f5ca80f6aa398b..ac38d0f19dd30efa1d9a96b085ab7c364785442c
@@@ -192,14 -192,6 +192,14 @@@ static void kvm_init_gcsr_flag(void
        set_gcsr_sw_flag(LOONGARCH_CSR_PERFCNTR2);
        set_gcsr_sw_flag(LOONGARCH_CSR_PERFCTRL3);
        set_gcsr_sw_flag(LOONGARCH_CSR_PERFCNTR3);
 +
 +      if (cpu_has_msgint) {
 +              set_gcsr_hw_flag(LOONGARCH_CSR_IPR);
 +              set_gcsr_hw_flag(LOONGARCH_CSR_ISR0);
 +              set_gcsr_hw_flag(LOONGARCH_CSR_ISR1);
 +              set_gcsr_hw_flag(LOONGARCH_CSR_ISR2);
 +              set_gcsr_hw_flag(LOONGARCH_CSR_ISR3);
 +      }
  }
  
  static void kvm_update_vpid(struct kvm_vcpu *vcpu, int cpu)
@@@ -402,7 -394,7 +402,7 @@@ static int kvm_loongarch_env_init(void
        }
  
        kvm_init_gcsr_flag();
-       kvm_register_perf_callbacks(NULL);
+       kvm_register_perf_callbacks();
  
        /* Register LoongArch IPI interrupt controller interface. */
        ret = kvm_loongarch_register_ipi_device();
diff --combined arch/x86/events/core.c
index 576baa9a52c5bdf4687f67c813de1bdf29871ee3,0ecac9495d74c82831d35cf1831e1846a5d448b4..73ed4d753ac5f9ed843a2232e44246bb6683efe5
@@@ -1,7 -1,7 +1,7 @@@
  /*
   * Performance events x86 architecture code
   *
 - *  Copyright (C) 2008 Thomas Gleixner <tglx@linutronix.de>
 + *  Copyright (C) 2008 Linutronix GmbH, Thomas Gleixner <tglx@kernel.org>
   *  Copyright (C) 2008-2009 Red Hat, Inc., Ingo Molnar
   *  Copyright (C) 2009 Jaswinder Singh Rajput
   *  Copyright (C) 2009 Advanced Micro Devices, Inc., Robert Richter
@@@ -30,6 -30,7 +30,7 @@@
  #include <linux/device.h>
  #include <linux/nospec.h>
  #include <linux/static_call.h>
+ #include <linux/kvm_types.h>
  
  #include <asm/apic.h>
  #include <asm/stacktrace.h>
@@@ -56,6 -57,8 +57,8 @@@ DEFINE_PER_CPU(struct cpu_hw_events, cp
        .pmu = &pmu,
  };
  
+ static DEFINE_PER_CPU(bool, guest_lvtpc_loaded);
  DEFINE_STATIC_KEY_FALSE(rdpmc_never_available_key);
  DEFINE_STATIC_KEY_FALSE(rdpmc_always_available_key);
  DEFINE_STATIC_KEY_FALSE(perf_is_hybrid);
@@@ -1760,6 -1763,25 +1763,25 @@@ void perf_events_lapic_init(void
        apic_write(APIC_LVTPC, APIC_DM_NMI);
  }
  
+ #ifdef CONFIG_PERF_GUEST_MEDIATED_PMU
+ void perf_load_guest_lvtpc(u32 guest_lvtpc)
+ {
+       u32 masked = guest_lvtpc & APIC_LVT_MASKED;
+       apic_write(APIC_LVTPC,
+                  APIC_DM_FIXED | PERF_GUEST_MEDIATED_PMI_VECTOR | masked);
+       this_cpu_write(guest_lvtpc_loaded, true);
+ }
+ EXPORT_SYMBOL_FOR_KVM(perf_load_guest_lvtpc);
+ void perf_put_guest_lvtpc(void)
+ {
+       this_cpu_write(guest_lvtpc_loaded, false);
+       apic_write(APIC_LVTPC, APIC_DM_NMI);
+ }
+ EXPORT_SYMBOL_FOR_KVM(perf_put_guest_lvtpc);
+ #endif /* CONFIG_PERF_GUEST_MEDIATED_PMU */
  static int
  perf_event_nmi_handler(unsigned int cmd, struct pt_regs *regs)
  {
        u64 finish_clock;
        int ret;
  
+       /*
+        * Ignore all NMIs when the CPU's LVTPC is configured to route PMIs to
+        * PERF_GUEST_MEDIATED_PMI_VECTOR, i.e. when an NMI time can't be due
+        * to a PMI.  Attempting to handle a PMI while the guest's context is
+        * loaded will generate false positives and clobber guest state.  Note,
+        * the LVTPC is switched to/from the dedicated mediated PMI IRQ vector
+        * while host events are quiesced.
+        */
+       if (this_cpu_read(guest_lvtpc_loaded))
+               return NMI_DONE;
        /*
         * All PMUs/events that share this PMI handler should make sure to
         * increment active_events for their events.
@@@ -3073,11 -3106,12 +3106,12 @@@ void perf_get_x86_pmu_capability(struc
        cap->version            = x86_pmu.version;
        cap->num_counters_gp    = x86_pmu_num_counters(NULL);
        cap->num_counters_fixed = x86_pmu_num_counters_fixed(NULL);
-       cap->bit_width_gp       = x86_pmu.cntval_bits;
-       cap->bit_width_fixed    = x86_pmu.cntval_bits;
+       cap->bit_width_gp       = cap->num_counters_gp ? x86_pmu.cntval_bits : 0;
+       cap->bit_width_fixed    = cap->num_counters_fixed ? x86_pmu.cntval_bits : 0;
        cap->events_mask        = (unsigned int)x86_pmu.events_maskl;
        cap->events_mask_len    = x86_pmu.events_mask_len;
        cap->pebs_ept           = x86_pmu.pebs_ept;
+       cap->mediated           = !!(pmu.capabilities & PERF_PMU_CAP_MEDIATED_VPMU);
  }
  EXPORT_SYMBOL_FOR_KVM(perf_get_x86_pmu_capability);
  
index 94cd4dc0e2a1e68bb903de28bf2ba35cb0a456f0,e72357f64b19b3f8f3eb475c7b9fe86f3a16b3b0..ff07c45e3c731a2833b472faca4262ac4af19a5b
@@@ -195,15 -195,7 +195,15 @@@ enum kvm_reg 
  
        VCPU_EXREG_PDPTR = NR_VCPU_REGS,
        VCPU_EXREG_CR0,
 +      /*
 +       * Alias AMD's ERAPS (not a real register) to CR3 so that common code
 +       * can trigger emulation of the RAP (Return Address Predictor) with
 +       * minimal support required in common code.  Piggyback CR3 as the RAP
 +       * is cleared on writes to CR3, i.e. marking CR3 dirty will naturally
 +       * mark ERAPS dirty as well.
 +       */
        VCPU_EXREG_CR3,
 +      VCPU_EXREG_ERAPS = VCPU_EXREG_CR3,
        VCPU_EXREG_CR4,
        VCPU_EXREG_RFLAGS,
        VCPU_EXREG_SEGMENTS,
@@@ -537,6 -529,7 +537,7 @@@ struct kvm_pmc 
         */
        u64 emulated_counter;
        u64 eventsel;
+       u64 eventsel_hw;
        struct perf_event *perf_event;
        struct kvm_vcpu *vcpu;
        /*
@@@ -565,6 -558,7 +566,7 @@@ struct kvm_pmu 
        unsigned nr_arch_fixed_counters;
        unsigned available_event_types;
        u64 fixed_ctr_ctrl;
+       u64 fixed_ctr_ctrl_hw;
        u64 fixed_ctr_ctrl_rsvd;
        u64 global_ctrl;
        u64 global_status;
@@@ -784,8 -778,6 +786,8 @@@ enum kvm_only_cpuid_leafs 
        CPUID_24_0_EBX,
        CPUID_8000_0021_ECX,
        CPUID_7_1_ECX,
 +      CPUID_1E_1_EAX,
 +      CPUID_24_1_ECX,
        NR_KVM_CPU_CAPS,
  
        NKVMCAPINTS = NR_KVM_CPU_CAPS - NCAPINTS,
@@@ -1232,18 -1224,10 +1234,18 @@@ struct kvm_xen 
  
  enum kvm_irqchip_mode {
        KVM_IRQCHIP_NONE,
 +#ifdef CONFIG_KVM_IOAPIC
        KVM_IRQCHIP_KERNEL,       /* created with KVM_CREATE_IRQCHIP */
 +#endif
        KVM_IRQCHIP_SPLIT,        /* created with KVM_CAP_SPLIT_IRQCHIP */
  };
  
 +enum kvm_suppress_eoi_broadcast_mode {
 +      KVM_SUPPRESS_EOI_BROADCAST_QUIRKED, /* Legacy behavior */
 +      KVM_SUPPRESS_EOI_BROADCAST_ENABLED, /* Enable Suppress EOI broadcast */
 +      KVM_SUPPRESS_EOI_BROADCAST_DISABLED /* Disable Suppress EOI broadcast */
 +};
 +
  struct kvm_x86_msr_filter {
        u8 count;
        bool default_allow:1;
@@@ -1493,7 -1477,6 +1495,7 @@@ struct kvm_arch 
  
        bool x2apic_format;
        bool x2apic_broadcast_quirk_disabled;
 +      enum kvm_suppress_eoi_broadcast_mode suppress_eoi_broadcast_mode;
  
        bool has_mapped_host_mmio;
        bool guest_can_read_msr_platform_info;
  
        bool bus_lock_detection_enabled;
        bool enable_pmu;
+       bool created_mediated_pmu;
  
        u32 notify_window;
        u32 notify_vmexit_flags;
diff --combined arch/x86/kernel/irq.c
index b2fe6181960c3f560a9b736f53506f6c09b5d02b,d56185b49a0e97cc8ea082e3859098cd38ebea83..316730e95fc34617acb37b43881e44eaffb4732f
@@@ -192,6 -192,13 +192,13 @@@ int arch_show_interrupts(struct seq_fil
                           irq_stats(j)->kvm_posted_intr_wakeup_ipis);
        seq_puts(p, "  Posted-interrupt wakeup event\n");
  #endif
+ #ifdef CONFIG_GUEST_PERF_EVENTS
+       seq_printf(p, "%*s: ", prec, "VPMI");
+       for_each_online_cpu(j)
+               seq_printf(p, "%10u ",
+                          irq_stats(j)->perf_guest_mediated_pmis);
+       seq_puts(p, " Perf Guest Mediated PMI\n");
+ #endif
  #ifdef CONFIG_X86_POSTED_MSI
        seq_printf(p, "%*s: ", prec, "PMN");
        for_each_online_cpu(j)
@@@ -349,6 -356,18 +356,18 @@@ DEFINE_IDTENTRY_SYSVEC(sysvec_x86_platf
  }
  #endif
  
+ #ifdef CONFIG_GUEST_PERF_EVENTS
+ /*
+  * Handler for PERF_GUEST_MEDIATED_PMI_VECTOR.
+  */
+ DEFINE_IDTENTRY_SYSVEC(sysvec_perf_guest_mediated_pmi_handler)
+ {
+        apic_eoi();
+        inc_irq_stat(perf_guest_mediated_pmis);
+        perf_guest_handle_mediated_pmi();
+ }
+ #endif
  #if IS_ENABLED(CONFIG_KVM)
  static void dummy_handler(void) {}
  static void (*kvm_posted_intr_wakeup_handler)(void) = dummy_handler;
@@@ -397,7 -416,6 +416,7 @@@ DEFINE_IDTENTRY_SYSVEC_SIMPLE(sysvec_kv
  
  /* Posted Interrupt Descriptors for coalesced MSIs to be posted */
  DEFINE_PER_CPU_ALIGNED(struct pi_desc, posted_msi_pi_desc);
 +static DEFINE_PER_CPU_CACHE_HOT(bool, posted_msi_handler_active);
  
  void intel_posted_msi_init(void)
  {
        this_cpu_write(posted_msi_pi_desc.ndst, destination);
  }
  
 +void intel_ack_posted_msi_irq(struct irq_data *irqd)
 +{
 +      irq_move_irq(irqd);
 +
 +      /*
 +       * Handle the rare case that irq_retrigger() raised the actual
 +       * assigned vector on the target CPU, which means that it was not
 +       * invoked via the posted MSI handler below. In that case APIC EOI
 +       * is required as otherwise the ISR entry becomes stale and lower
 +       * priority interrupts are never going to be delivered after that.
 +       *
 +       * If the posted handler invoked the device interrupt handler then
 +       * the EOI would be premature because it would acknowledge the
 +       * posted vector.
 +       */
 +      if (unlikely(!__this_cpu_read(posted_msi_handler_active)))
 +              apic_eoi();
 +}
 +
  static __always_inline bool handle_pending_pir(unsigned long *pir, struct pt_regs *regs)
  {
        unsigned long pir_copy[NR_PIR_WORDS];
@@@ -466,8 -465,6 +485,8 @@@ DEFINE_IDTENTRY_SYSVEC(sysvec_posted_ms
  
        pid = this_cpu_ptr(&posted_msi_pi_desc);
  
 +      /* Mark the handler active for intel_ack_posted_msi_irq() */
 +      __this_cpu_write(posted_msi_handler_active, true);
        inc_irq_stat(posted_msi_notification_count);
        irq_enter();
  
  
        apic_eoi();
        irq_exit();
 +      __this_cpu_write(posted_msi_handler_active, false);
        set_irq_regs(old_regs);
  }
  #endif /* X86_POSTED_MSI */
diff --combined arch/x86/kvm/pmu.c
index ff20b4102173ae511865ada2f0e7fbb8e52603d7,954622f8f81786a2a2a35f3c804ba0afcd8cc2af..bd6b785cf2612e6069bb9ee8c6c826a1658baba1
@@@ -103,7 -103,7 +103,7 @@@ void kvm_pmu_ops_update(const struct kv
  #undef __KVM_X86_PMU_OP
  }
  
- void kvm_init_pmu_capability(const struct kvm_pmu_ops *pmu_ops)
+ void kvm_init_pmu_capability(struct kvm_pmu_ops *pmu_ops)
  {
        bool is_intel = boot_cpu_data.x86_vendor == X86_VENDOR_INTEL;
        int min_nr_gp_ctrs = pmu_ops->MIN_NR_GP_COUNTERS;
                        enable_pmu = false;
        }
  
+       if (!enable_pmu || !enable_mediated_pmu || !kvm_host_pmu.mediated ||
+           !pmu_ops->is_mediated_pmu_supported(&kvm_host_pmu))
+               enable_mediated_pmu = false;
+       if (!enable_mediated_pmu)
+               pmu_ops->write_global_ctrl = NULL;
        if (!enable_pmu) {
                memset(&kvm_pmu_cap, 0, sizeof(kvm_pmu_cap));
                return;
                perf_get_hw_event_config(PERF_COUNT_HW_BRANCH_INSTRUCTIONS);
  }
  
+ void kvm_handle_guest_mediated_pmi(void)
+ {
+       struct kvm_vcpu *vcpu = kvm_get_running_vcpu();
+       if (WARN_ON_ONCE(!vcpu || !kvm_vcpu_has_mediated_pmu(vcpu)))
+               return;
+       kvm_make_request(KVM_REQ_PMI, vcpu);
+ }
  static inline void __kvm_perf_overflow(struct kvm_pmc *pmc, bool in_pmi)
  {
        struct kvm_pmu *pmu = pmc_to_pmu(pmc);
@@@ -362,6 -379,11 +379,11 @@@ static void pmc_update_sample_period(st
  
  void pmc_write_counter(struct kvm_pmc *pmc, u64 val)
  {
+       if (kvm_vcpu_has_mediated_pmu(pmc->vcpu)) {
+               pmc->counter = val & pmc_bitmask(pmc);
+               return;
+       }
        /*
         * Drop any unconsumed accumulated counts, the WRMSR is a write, not a
         * read-modify-write.  Adjust the counter value so that its value is
@@@ -498,6 -520,25 +520,25 @@@ static bool pmc_is_event_allowed(struc
        return is_fixed_event_allowed(filter, pmc->idx);
  }
  
+ static void kvm_mediated_pmu_refresh_event_filter(struct kvm_pmc *pmc)
+ {
+       bool allowed = pmc_is_event_allowed(pmc);
+       struct kvm_pmu *pmu = pmc_to_pmu(pmc);
+       if (pmc_is_gp(pmc)) {
+               pmc->eventsel_hw &= ~ARCH_PERFMON_EVENTSEL_ENABLE;
+               if (allowed)
+                       pmc->eventsel_hw |= pmc->eventsel &
+                                           ARCH_PERFMON_EVENTSEL_ENABLE;
+       } else {
+               u64 mask = intel_fixed_bits_by_idx(pmc->idx - KVM_FIXED_PMC_BASE_IDX, 0xf);
+               pmu->fixed_ctr_ctrl_hw &= ~mask;
+               if (allowed)
+                       pmu->fixed_ctr_ctrl_hw |= pmu->fixed_ctr_ctrl & mask;
+       }
+ }
  static int reprogram_counter(struct kvm_pmc *pmc)
  {
        struct kvm_pmu *pmu = pmc_to_pmu(pmc);
        bool emulate_overflow;
        u8 fixed_ctr_ctrl;
  
+       if (kvm_vcpu_has_mediated_pmu(pmu_to_vcpu(pmu))) {
+               kvm_mediated_pmu_refresh_event_filter(pmc);
+               return 0;
+       }
        emulate_overflow = pmc_pause_counter(pmc);
  
        if (!pmc_is_globally_enabled(pmc) || !pmc_is_locally_enabled(pmc) ||
@@@ -700,6 -746,46 +746,46 @@@ int kvm_pmu_rdpmc(struct kvm_vcpu *vcpu
        return 0;
  }
  
+ static bool kvm_need_any_pmc_intercept(struct kvm_vcpu *vcpu)
+ {
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       if (!kvm_vcpu_has_mediated_pmu(vcpu))
+               return true;
+       /*
+        * Note!  Check *host* PMU capabilities, not KVM's PMU capabilities, as
+        * KVM's capabilities are constrained based on KVM support, i.e. KVM's
+        * capabilities themselves may be a subset of hardware capabilities.
+        */
+       return pmu->nr_arch_gp_counters != kvm_host_pmu.num_counters_gp ||
+              pmu->nr_arch_fixed_counters != kvm_host_pmu.num_counters_fixed;
+ }
+ bool kvm_need_perf_global_ctrl_intercept(struct kvm_vcpu *vcpu)
+ {
+       return kvm_need_any_pmc_intercept(vcpu) ||
+              !kvm_pmu_has_perf_global_ctrl(vcpu_to_pmu(vcpu));
+ }
+ EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_need_perf_global_ctrl_intercept);
+ bool kvm_need_rdpmc_intercept(struct kvm_vcpu *vcpu)
+ {
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       /*
+        * VMware allows access to these Pseduo-PMCs even when read via RDPMC
+        * in Ring3 when CR4.PCE=0.
+        */
+       if (enable_vmware_backdoor)
+               return true;
+       return kvm_need_any_pmc_intercept(vcpu) ||
+              pmu->counter_bitmask[KVM_PMC_GP] != (BIT_ULL(kvm_host_pmu.bit_width_gp) - 1) ||
+              pmu->counter_bitmask[KVM_PMC_FIXED] != (BIT_ULL(kvm_host_pmu.bit_width_fixed) - 1);
+ }
+ EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_need_rdpmc_intercept);
  void kvm_pmu_deliver_pmi(struct kvm_vcpu *vcpu)
  {
        if (lapic_in_kernel(vcpu)) {
@@@ -795,6 -881,12 +881,12 @@@ int kvm_pmu_set_msr(struct kvm_vcpu *vc
                        pmu->global_ctrl = data;
                        reprogram_counters(pmu, diff);
                }
+               /*
+                * Unconditionally forward writes to vendor code, i.e. to the
+                * VMC{B,S}, as pmu->global_ctrl is per-VCPU, not per-VMC{B,S}.
+                */
+               if (kvm_vcpu_has_mediated_pmu(vcpu))
+                       kvm_pmu_call(write_global_ctrl)(data);
                break;
        case MSR_CORE_PERF_GLOBAL_OVF_CTRL:
                /*
@@@ -835,11 -927,14 +927,14 @@@ static void kvm_pmu_reset(struct kvm_vc
                pmc->counter = 0;
                pmc->emulated_counter = 0;
  
-               if (pmc_is_gp(pmc))
+               if (pmc_is_gp(pmc)) {
                        pmc->eventsel = 0;
+                       pmc->eventsel_hw = 0;
+               }
        }
  
-       pmu->fixed_ctr_ctrl = pmu->global_ctrl = pmu->global_status = 0;
+       pmu->fixed_ctr_ctrl = pmu->fixed_ctr_ctrl_hw = 0;
+       pmu->global_ctrl = pmu->global_status = 0;
  
        kvm_pmu_call(reset)(vcpu);
  }
@@@ -853,7 -948,7 +948,7 @@@ void kvm_pmu_refresh(struct kvm_vcpu *v
  {
        struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
  
 -      if (KVM_BUG_ON(kvm_vcpu_has_run(vcpu), vcpu->kvm))
 +      if (KVM_BUG_ON(!kvm_can_set_cpuid_and_feature_msrs(vcpu), vcpu->kvm))
                return;
  
        /*
         * in the global controls).  Emulate that behavior when refreshing the
         * PMU so that userspace doesn't need to manually set PERF_GLOBAL_CTRL.
         */
-       if (kvm_pmu_has_perf_global_ctrl(pmu) && pmu->nr_arch_gp_counters)
+       if (pmu->nr_arch_gp_counters &&
+           (kvm_pmu_has_perf_global_ctrl(pmu) || kvm_vcpu_has_mediated_pmu(vcpu)))
                pmu->global_ctrl = GENMASK_ULL(pmu->nr_arch_gp_counters - 1, 0);
  
+       if (kvm_vcpu_has_mediated_pmu(vcpu))
+               kvm_pmu_call(write_global_ctrl)(pmu->global_ctrl);
        bitmap_set(pmu->all_valid_pmc_idx, 0, pmu->nr_arch_gp_counters);
        bitmap_set(pmu->all_valid_pmc_idx, KVM_FIXED_PMC_BASE_IDX,
                   pmu->nr_arch_fixed_counters);
@@@ -932,10 -1031,45 +1031,45 @@@ void kvm_pmu_destroy(struct kvm_vcpu *v
        kvm_pmu_reset(vcpu);
  }
  
+ static bool pmc_is_pmi_enabled(struct kvm_pmc *pmc)
+ {
+       u8 fixed_ctr_ctrl;
+       if (pmc_is_gp(pmc))
+               return pmc->eventsel & ARCH_PERFMON_EVENTSEL_INT;
+       fixed_ctr_ctrl = fixed_ctrl_field(pmc_to_pmu(pmc)->fixed_ctr_ctrl,
+                                         pmc->idx - KVM_FIXED_PMC_BASE_IDX);
+       return fixed_ctr_ctrl & INTEL_FIXED_0_ENABLE_PMI;
+ }
  static void kvm_pmu_incr_counter(struct kvm_pmc *pmc)
  {
-       pmc->emulated_counter++;
-       kvm_pmu_request_counter_reprogram(pmc);
+       struct kvm_vcpu *vcpu = pmc->vcpu;
+       /*
+        * For perf-based PMUs, accumulate software-emulated events separately
+        * from pmc->counter, as pmc->counter is offset by the count of the
+        * associated perf event. Request reprogramming, which will consult
+        * both emulated and hardware-generated events to detect overflow.
+        */
+       if (!kvm_vcpu_has_mediated_pmu(vcpu)) {
+               pmc->emulated_counter++;
+               kvm_pmu_request_counter_reprogram(pmc);
+               return;
+       }
+       /*
+        * For mediated PMUs, pmc->counter is updated when the vCPU's PMU is
+        * put, and will be loaded into hardware when the PMU is loaded. Simply
+        * increment the counter and signal overflow if it wraps to zero.
+        */
+       pmc->counter = (pmc->counter + 1) & pmc_bitmask(pmc);
+       if (!pmc->counter) {
+               pmc_to_pmu(pmc)->global_status |= BIT_ULL(pmc->idx);
+               if (pmc_is_pmi_enabled(pmc))
+                       kvm_make_request(KVM_REQ_PMI, vcpu);
+       }
  }
  
  static inline bool cpl_is_matched(struct kvm_pmc *pmc)
@@@ -1148,3 -1282,126 +1282,126 @@@ cleanup
        kfree(filter);
        return r;
  }
+ static __always_inline u32 fixed_counter_msr(u32 idx)
+ {
+       return kvm_pmu_ops.FIXED_COUNTER_BASE + idx * kvm_pmu_ops.MSR_STRIDE;
+ }
+ static __always_inline u32 gp_counter_msr(u32 idx)
+ {
+       return kvm_pmu_ops.GP_COUNTER_BASE + idx * kvm_pmu_ops.MSR_STRIDE;
+ }
+ static __always_inline u32 gp_eventsel_msr(u32 idx)
+ {
+       return kvm_pmu_ops.GP_EVENTSEL_BASE + idx * kvm_pmu_ops.MSR_STRIDE;
+ }
+ static void kvm_pmu_load_guest_pmcs(struct kvm_vcpu *vcpu)
+ {
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       struct kvm_pmc *pmc;
+       u32 i;
+       /*
+        * No need to zero out unexposed GP/fixed counters/selectors since RDPMC
+        * is intercepted if hardware has counters that aren't visible to the
+        * guest (KVM will inject #GP as appropriate).
+        */
+       for (i = 0; i < pmu->nr_arch_gp_counters; i++) {
+               pmc = &pmu->gp_counters[i];
+               if (pmc->counter != rdpmc(i))
+                       wrmsrl(gp_counter_msr(i), pmc->counter);
+               wrmsrl(gp_eventsel_msr(i), pmc->eventsel_hw);
+       }
+       for (i = 0; i < pmu->nr_arch_fixed_counters; i++) {
+               pmc = &pmu->fixed_counters[i];
+               if (pmc->counter != rdpmc(INTEL_PMC_FIXED_RDPMC_BASE | i))
+                       wrmsrl(fixed_counter_msr(i), pmc->counter);
+       }
+ }
+ void kvm_mediated_pmu_load(struct kvm_vcpu *vcpu)
+ {
+       if (!kvm_vcpu_has_mediated_pmu(vcpu) ||
+           KVM_BUG_ON(!lapic_in_kernel(vcpu), vcpu->kvm))
+               return;
+       lockdep_assert_irqs_disabled();
+       perf_load_guest_context();
+       /*
+        * Explicitly clear PERF_GLOBAL_CTRL, as "loading" the guest's context
+        * disables all individual counters (if any were enabled), but doesn't
+        * globally disable the entire PMU.  Loading event selectors and PMCs
+        * with guest values while PERF_GLOBAL_CTRL is non-zero will generate
+        * unexpected events and PMIs.
+        *
+        * VMX will enable/disable counters at VM-Enter/VM-Exit by atomically
+        * loading PERF_GLOBAL_CONTROL.  SVM effectively performs the switch by
+        * configuring all events to be GUEST_ONLY.  Clear PERF_GLOBAL_CONTROL
+        * even for SVM to minimize the damage if a perf event is left enabled,
+        * and to ensure a consistent starting state.
+        */
+       wrmsrq(kvm_pmu_ops.PERF_GLOBAL_CTRL, 0);
+       perf_load_guest_lvtpc(kvm_lapic_get_reg(vcpu->arch.apic, APIC_LVTPC));
+       kvm_pmu_load_guest_pmcs(vcpu);
+       kvm_pmu_call(mediated_load)(vcpu);
+ }
+ static void kvm_pmu_put_guest_pmcs(struct kvm_vcpu *vcpu)
+ {
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       struct kvm_pmc *pmc;
+       u32 i;
+       /*
+        * Clear selectors and counters to ensure hardware doesn't count using
+        * guest controls when the host (perf) restores its state.
+        */
+       for (i = 0; i < pmu->nr_arch_gp_counters; i++) {
+               pmc = &pmu->gp_counters[i];
+               pmc->counter = rdpmc(i);
+               if (pmc->counter)
+                       wrmsrq(gp_counter_msr(i), 0);
+               if (pmc->eventsel_hw)
+                       wrmsrq(gp_eventsel_msr(i), 0);
+       }
+       for (i = 0; i < pmu->nr_arch_fixed_counters; i++) {
+               pmc = &pmu->fixed_counters[i];
+               pmc->counter = rdpmc(INTEL_PMC_FIXED_RDPMC_BASE | i);
+               if (pmc->counter)
+                       wrmsrq(fixed_counter_msr(i), 0);
+       }
+ }
+ void kvm_mediated_pmu_put(struct kvm_vcpu *vcpu)
+ {
+       if (!kvm_vcpu_has_mediated_pmu(vcpu) ||
+           KVM_BUG_ON(!lapic_in_kernel(vcpu), vcpu->kvm))
+               return;
+       lockdep_assert_irqs_disabled();
+       /*
+        * Defer handling of PERF_GLOBAL_CTRL to vendor code.  On Intel, it's
+        * atomically cleared on VM-Exit, i.e. doesn't need to be clear here.
+        */
+       kvm_pmu_call(mediated_put)(vcpu);
+       kvm_pmu_put_guest_pmcs(vcpu);
+       perf_put_guest_lvtpc();
+       perf_put_guest_context();
+ }
index 79cb85b8a15688cfbf888b0e10c83a6e9d470912,9ca8dad9a7f3cad66e3340776ca49cc4159121ff..de90b104a0dd55275a951c4668b47f88fe7d29e4
@@@ -45,6 -45,7 +45,6 @@@ static void nested_svm_inject_npf_exit(
                 * correctly fill in the high bits of exit_info_1.
                 */
                vmcb->control.exit_code = SVM_EXIT_NPF;
 -              vmcb->control.exit_code_hi = 0;
                vmcb->control.exit_info_1 = (1ULL << 32);
                vmcb->control.exit_info_2 = fault->address;
        }
@@@ -193,7 -194,7 +193,7 @@@ void recalc_intercepts(struct vcpu_svm 
   * Hardcode the capacity of the array based on the maximum number of _offsets_.
   * MSRs are batched together, so there are fewer offsets than MSRs.
   */
- static int nested_svm_msrpm_merge_offsets[7] __ro_after_init;
+ static int nested_svm_msrpm_merge_offsets[10] __ro_after_init;
  static int nested_svm_nr_msrpm_merge_offsets __ro_after_init;
  typedef unsigned long nsvm_msrpm_merge_t;
  
@@@ -221,6 -222,22 +221,22 @@@ int __init nested_svm_init_msrpm_merge_
                MSR_IA32_LASTBRANCHTOIP,
                MSR_IA32_LASTINTFROMIP,
                MSR_IA32_LASTINTTOIP,
+               MSR_K7_PERFCTR0,
+               MSR_K7_PERFCTR1,
+               MSR_K7_PERFCTR2,
+               MSR_K7_PERFCTR3,
+               MSR_F15H_PERF_CTR0,
+               MSR_F15H_PERF_CTR1,
+               MSR_F15H_PERF_CTR2,
+               MSR_F15H_PERF_CTR3,
+               MSR_F15H_PERF_CTR4,
+               MSR_F15H_PERF_CTR5,
+               MSR_AMD64_PERF_CNTR_GLOBAL_CTL,
+               MSR_AMD64_PERF_CNTR_GLOBAL_STATUS,
+               MSR_AMD64_PERF_CNTR_GLOBAL_STATUS_CLR,
+               MSR_AMD64_PERF_CNTR_GLOBAL_STATUS_SET,
        };
        int i, j;
  
@@@ -402,19 -419,6 +418,19 @@@ static bool nested_vmcb_check_controls(
        return __nested_vmcb_check_controls(vcpu, ctl);
  }
  
 +/*
 + * If a feature is not advertised to L1, clear the corresponding vmcb12
 + * intercept.
 + */
 +#define __nested_svm_sanitize_intercept(__vcpu, __control, fname, iname)      \
 +do {                                                                          \
 +      if (!guest_cpu_cap_has(__vcpu, X86_FEATURE_##fname))                    \
 +              vmcb12_clr_intercept(__control, INTERCEPT_##iname);             \
 +} while (0)
 +
 +#define nested_svm_sanitize_intercept(__vcpu, __control, name)                        \
 +      __nested_svm_sanitize_intercept(__vcpu, __control, name, name)
 +
  static
  void __nested_copy_vmcb_control_to_cache(struct kvm_vcpu *vcpu,
                                         struct vmcb_ctrl_area_cached *to,
        for (i = 0; i < MAX_INTERCEPT; i++)
                to->intercepts[i] = from->intercepts[i];
  
 +      __nested_svm_sanitize_intercept(vcpu, to, XSAVE, XSETBV);
 +      nested_svm_sanitize_intercept(vcpu, to, INVPCID);
 +      nested_svm_sanitize_intercept(vcpu, to, RDTSCP);
 +      nested_svm_sanitize_intercept(vcpu, to, SKINIT);
 +      nested_svm_sanitize_intercept(vcpu, to, RDPRU);
 +
        to->iopm_base_pa        = from->iopm_base_pa;
        to->msrpm_base_pa       = from->msrpm_base_pa;
        to->tsc_offset          = from->tsc_offset;
        to->tlb_ctl             = from->tlb_ctl;
 +      to->erap_ctl            = from->erap_ctl;
        to->int_ctl             = from->int_ctl;
        to->int_vector          = from->int_vector;
        to->int_state           = from->int_state;
        to->exit_code           = from->exit_code;
 -      to->exit_code_hi        = from->exit_code_hi;
        to->exit_info_1         = from->exit_info_1;
        to->exit_info_2         = from->exit_info_2;
        to->exit_int_info       = from->exit_int_info;
@@@ -681,6 -679,7 +697,6 @@@ static void nested_vmcb02_prepare_save(
        vmcb02->save.rsp = vmcb12->save.rsp;
        vmcb02->save.rip = vmcb12->save.rip;
  
 -      /* These bits will be set properly on the first execution when new_vmc12 is true */
        if (unlikely(new_vmcb12 || vmcb_is_dirty(vmcb12, VMCB_DR))) {
                vmcb02->save.dr7 = svm->nested.save.dr7 | DR7_FIXED_1;
                svm->vcpu.arch.dr6  = svm->nested.save.dr6 | DR6_ACTIVE_LOW;
@@@ -744,8 -743,8 +760,8 @@@ static void nested_vmcb02_prepare_contr
        enter_guest_mode(vcpu);
  
        /*
 -       * Filled at exit: exit_code, exit_code_hi, exit_info_1, exit_info_2,
 -       * exit_int_info, exit_int_info_err, next_rip, insn_len, insn_bytes.
 +       * Filled at exit: exit_code, exit_info_1, exit_info_2, exit_int_info,
 +       * exit_int_info_err, next_rip, insn_len, insn_bytes.
         */
  
        if (guest_cpu_cap_has(vcpu, X86_FEATURE_VGIF) &&
                }
        }
  
 +      /*
 +       * Take ALLOW_LARGER_RAP from vmcb12 even though it should be safe to
 +       * let L2 use a larger RAP since KVM will emulate the necessary clears,
 +       * as it's possible L1 deliberately wants to restrict L2 to the legacy
 +       * RAP size.  Unconditionally clear the RAP on nested VMRUN, as KVM is
 +       * responsible for emulating the host vs. guest tags (L1 is the "host",
 +       * L2 is the "guest").
 +       */
 +      if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS))
 +              vmcb02->control.erap_ctl = (svm->nested.ctl.erap_ctl &
 +                                          ERAP_CONTROL_ALLOW_LARGER_RAP) |
 +                                         ERAP_CONTROL_CLEAR_RAP;
 +
        /*
         * Merge guest and host intercepts - must be called with vcpu in
         * guest-mode to take effect.
@@@ -1015,6 -1001,7 +1031,6 @@@ int nested_svm_vmrun(struct kvm_vcpu *v
        if (!nested_vmcb_check_save(vcpu) ||
            !nested_vmcb_check_controls(vcpu)) {
                vmcb12->control.exit_code    = SVM_EXIT_ERR;
 -              vmcb12->control.exit_code_hi = 0;
                vmcb12->control.exit_info_1  = 0;
                vmcb12->control.exit_info_2  = 0;
                goto out;
@@@ -1047,6 -1034,7 +1063,6 @@@ out_exit_err
        svm->soft_int_injected = false;
  
        svm->vmcb->control.exit_code    = SVM_EXIT_ERR;
 -      svm->vmcb->control.exit_code_hi = 0;
        svm->vmcb->control.exit_info_1  = 0;
        svm->vmcb->control.exit_info_2  = 0;
  
@@@ -1158,10 -1146,11 +1174,10 @@@ int nested_svm_vmexit(struct vcpu_svm *
  
        vmcb12->control.int_state         = vmcb02->control.int_state;
        vmcb12->control.exit_code         = vmcb02->control.exit_code;
 -      vmcb12->control.exit_code_hi      = vmcb02->control.exit_code_hi;
        vmcb12->control.exit_info_1       = vmcb02->control.exit_info_1;
        vmcb12->control.exit_info_2       = vmcb02->control.exit_info_2;
  
 -      if (vmcb12->control.exit_code != SVM_EXIT_ERR)
 +      if (!svm_is_vmrun_failure(vmcb12->control.exit_code))
                nested_save_pending_event_to_vmcb12(svm, vmcb12);
  
        if (guest_cpu_cap_has(vcpu, X86_FEATURE_NRIPS))
  
        kvm_nested_vmexit_handle_ibrs(vcpu);
  
 +      if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS))
 +              vmcb01->control.erap_ctl |= ERAP_CONTROL_CLEAR_RAP;
 +
        svm_switch_vmcb(svm, &svm->vmcb01);
  
        /*
@@@ -1393,8 -1379,6 +1409,8 @@@ void svm_leave_nested(struct kvm_vcpu *
                nested_svm_uninit_mmu_context(vcpu);
                vmcb_mark_all_dirty(svm->vmcb);
  
 +              svm_set_gif(svm, true);
 +
                if (kvm_apicv_activated(vcpu->kvm))
                        kvm_make_request(KVM_REQ_APICV_UPDATE, vcpu);
        }
@@@ -1454,12 -1438,9 +1470,12 @@@ static int nested_svm_intercept_ioio(st
  
  static int nested_svm_intercept(struct vcpu_svm *svm)
  {
 -      u32 exit_code = svm->vmcb->control.exit_code;
 +      u64 exit_code = svm->vmcb->control.exit_code;
        int vmexit = NESTED_EXIT_HOST;
  
 +      if (svm_is_vmrun_failure(exit_code))
 +              return NESTED_EXIT_DONE;
 +
        switch (exit_code) {
        case SVM_EXIT_MSR:
                vmexit = nested_svm_exit_handled_msr(svm);
        case SVM_EXIT_IOIO:
                vmexit = nested_svm_intercept_ioio(svm);
                break;
 -      case SVM_EXIT_EXCP_BASE ... SVM_EXIT_EXCP_BASE + 0x1f: {
 +      case SVM_EXIT_EXCP_BASE ... SVM_EXIT_EXCP_BASE + 0x1f:
                /*
                 * Host-intercepted exceptions have been checked already in
                 * nested_svm_exit_special.  There is nothing to do here,
                 */
                vmexit = NESTED_EXIT_DONE;
                break;
 -      }
 -      case SVM_EXIT_ERR: {
 -              vmexit = NESTED_EXIT_DONE;
 -              break;
 -      }
 -      default: {
 +      default:
                if (vmcb12_is_intercept(&svm->nested.ctl, exit_code))
                        vmexit = NESTED_EXIT_DONE;
 -      }
 +              break;
        }
  
        return vmexit;
@@@ -1526,6 -1512,7 +1542,6 @@@ static void nested_svm_inject_exception
        struct vmcb *vmcb = svm->vmcb;
  
        vmcb->control.exit_code = SVM_EXIT_EXCP_BASE + ex->vector;
 -      vmcb->control.exit_code_hi = 0;
  
        if (ex->has_error_code)
                vmcb->control.exit_info_1 = ex->error_code;
@@@ -1696,11 -1683,11 +1712,11 @@@ static void nested_copy_vmcb_cache_to_c
        dst->tsc_offset           = from->tsc_offset;
        dst->asid                 = from->asid;
        dst->tlb_ctl              = from->tlb_ctl;
 +      dst->erap_ctl             = from->erap_ctl;
        dst->int_ctl              = from->int_ctl;
        dst->int_vector           = from->int_vector;
        dst->int_state            = from->int_state;
        dst->exit_code            = from->exit_code;
 -      dst->exit_code_hi         = from->exit_code_hi;
        dst->exit_info_1          = from->exit_info_1;
        dst->exit_info_2          = from->exit_info_2;
        dst->exit_int_info        = from->exit_int_info;
@@@ -1811,12 -1798,12 +1827,12 @@@ static int svm_set_nested_state(struct 
        /*
         * If in guest mode, vcpu->arch.efer actually refers to the L2 guest's
         * EFER.SVME, but EFER.SVME still has to be 1 for VMRUN to succeed.
 +       * If SVME is disabled, the only valid states are "none" and GIF=1
 +       * (clearing SVME does NOT set GIF, i.e. GIF=0 is allowed).
         */
 -      if (!(vcpu->arch.efer & EFER_SVME)) {
 -              /* GIF=1 and no guest mode are required if SVME=0.  */
 -              if (kvm_state->flags != KVM_STATE_NESTED_GIF_SET)
 -                      return -EINVAL;
 -      }
 +      if (!(vcpu->arch.efer & EFER_SVME) && kvm_state->flags &&
 +          kvm_state->flags != KVM_STATE_NESTED_GIF_SET)
 +              return -EINVAL;
  
        /* SMM temporarily disables SVM, so we cannot be in guest mode.  */
        if (is_smm(vcpu) && (kvm_state->flags & KVM_STATE_NESTED_GUEST_MODE))
         * thus MMU might not be initialized correctly.
         * Set it again to fix this.
         */
 -
        ret = nested_svm_load_cr3(&svm->vcpu, vcpu->arch.cr3,
                                  nested_npt_enabled(svm), false);
 -      if (WARN_ON_ONCE(ret))
 +      if (ret)
                goto out_free;
  
        svm->nested.force_msr_bitmap_recalc = true;
diff --combined arch/x86/kvm/svm/svm.c
index 9ee74c57bd51b4916e8b243fa304e9d126ab8389,5910088fe22ab0c5f0d395a1e9026fe04b1c43c0..8f8bc863e214353284642830db5bd819ed594e5d
@@@ -170,6 -170,8 +170,8 @@@ module_param(intercept_smi, bool, 0444)
  bool vnmi = true;
  module_param(vnmi, bool, 0444);
  
+ module_param(enable_mediated_pmu, bool, 0444);
  static bool svm_gp_erratum_intercept = true;
  
  static u8 rsm_ins_bytes[] = "\x0f\xaa";
@@@ -215,6 -217,7 +217,6 @@@ int svm_set_efer(struct kvm_vcpu *vcpu
        if ((old_efer & EFER_SVME) != (efer & EFER_SVME)) {
                if (!(efer & EFER_SVME)) {
                        svm_leave_nested(vcpu);
 -                      svm_set_gif(svm, true);
                        /* #GP intercept is still needed for vmware backdoor */
                        if (!enable_vmware_backdoor)
                                clr_exception_intercept(svm, GP_VECTOR);
@@@ -729,6 -732,40 +731,40 @@@ void svm_vcpu_free_msrpm(void *msrpm
        __free_pages(virt_to_page(msrpm), get_order(MSRPM_SIZE));
  }
  
+ static void svm_recalc_pmu_msr_intercepts(struct kvm_vcpu *vcpu)
+ {
+       bool intercept = !kvm_vcpu_has_mediated_pmu(vcpu);
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       int i;
+       if (!enable_mediated_pmu)
+               return;
+       /* Legacy counters are always available for AMD CPUs with a PMU. */
+       for (i = 0; i < min(pmu->nr_arch_gp_counters, AMD64_NUM_COUNTERS); i++)
+               svm_set_intercept_for_msr(vcpu, MSR_K7_PERFCTR0 + i,
+                                         MSR_TYPE_RW, intercept);
+       intercept |= !guest_cpu_cap_has(vcpu, X86_FEATURE_PERFCTR_CORE);
+       for (i = 0; i < pmu->nr_arch_gp_counters; i++)
+               svm_set_intercept_for_msr(vcpu, MSR_F15H_PERF_CTR + 2 * i,
+                                         MSR_TYPE_RW, intercept);
+       for ( ; i < kvm_pmu_cap.num_counters_gp; i++)
+               svm_enable_intercept_for_msr(vcpu, MSR_F15H_PERF_CTR + 2 * i,
+                                            MSR_TYPE_RW);
+       intercept = kvm_need_perf_global_ctrl_intercept(vcpu);
+       svm_set_intercept_for_msr(vcpu, MSR_AMD64_PERF_CNTR_GLOBAL_CTL,
+                                 MSR_TYPE_RW, intercept);
+       svm_set_intercept_for_msr(vcpu, MSR_AMD64_PERF_CNTR_GLOBAL_STATUS,
+                                 MSR_TYPE_RW, intercept);
+       svm_set_intercept_for_msr(vcpu, MSR_AMD64_PERF_CNTR_GLOBAL_STATUS_CLR,
+                                 MSR_TYPE_RW, intercept);
+       svm_set_intercept_for_msr(vcpu, MSR_AMD64_PERF_CNTR_GLOBAL_STATUS_SET,
+                                 MSR_TYPE_RW, intercept);
+ }
  static void svm_recalc_msr_intercepts(struct kvm_vcpu *vcpu)
  {
        struct vcpu_svm *svm = to_svm(vcpu);
        if (sev_es_guest(vcpu->kvm))
                sev_es_recalc_msr_intercepts(vcpu);
  
+       svm_recalc_pmu_msr_intercepts(vcpu);
        /*
         * x2APIC intercepts are modified on-demand and cannot be filtered by
         * userspace.
@@@ -995,14 -1034,10 +1033,14 @@@ static void svm_recalc_instruction_inte
                        svm_set_intercept(svm, INTERCEPT_RDTSCP);
        }
  
 +      /*
 +       * No need to toggle VIRTUAL_VMLOAD_VMSAVE_ENABLE_MASK here, it is
 +       * always set if vls is enabled. If the intercepts are set, the bit is
 +       * meaningless anyway.
 +       */
        if (guest_cpuid_is_intel_compatible(vcpu)) {
                svm_set_intercept(svm, INTERCEPT_VMLOAD);
                svm_set_intercept(svm, INTERCEPT_VMSAVE);
 -              svm->vmcb->control.virt_ext &= ~VIRTUAL_VMLOAD_VMSAVE_ENABLE_MASK;
        } else {
                /*
                 * If hardware supports Virtual VMLOAD VMSAVE then enable it
                if (vls) {
                        svm_clr_intercept(svm, INTERCEPT_VMLOAD);
                        svm_clr_intercept(svm, INTERCEPT_VMSAVE);
 -                      svm->vmcb->control.virt_ext |= VIRTUAL_VMLOAD_VMSAVE_ENABLE_MASK;
                }
        }
+       if (kvm_need_rdpmc_intercept(vcpu))
+               svm_set_intercept(svm, INTERCEPT_RDPMC);
+       else
+               svm_clr_intercept(svm, INTERCEPT_RDPMC);
  }
  
  static void svm_recalc_intercepts(struct kvm_vcpu *vcpu)
@@@ -1143,9 -1184,6 +1186,9 @@@ static void init_vmcb(struct kvm_vcpu *
                svm_clr_intercept(svm, INTERCEPT_PAUSE);
        }
  
 +      if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS))
 +              svm->vmcb->control.erap_ctl |= ERAP_CONTROL_ALLOW_LARGER_RAP;
 +
        if (kvm_vcpu_apicv_active(vcpu))
                avic_init_vmcb(svm, vmcb);
  
                svm->vmcb->control.int_ctl |= V_GIF_ENABLE_MASK;
        }
  
 +      if (vls)
 +              svm->vmcb->control.virt_ext |= VIRTUAL_VMLOAD_VMSAVE_ENABLE_MASK;
 +
        if (vcpu->kvm->arch.bus_lock_detection_enabled)
                svm_set_intercept(svm, INTERCEPT_BUSLOCK);
  
@@@ -1870,16 -1905,13 +1913,16 @@@ static int pf_interception(struct kvm_v
                        svm->vmcb->control.insn_len);
  }
  
 +static int svm_check_emulate_instruction(struct kvm_vcpu *vcpu, int emul_type,
 +                                       void *insn, int insn_len);
 +
  static int npf_interception(struct kvm_vcpu *vcpu)
  {
        struct vcpu_svm *svm = to_svm(vcpu);
        int rc;
  
 -      u64 fault_address = svm->vmcb->control.exit_info_2;
        u64 error_code = svm->vmcb->control.exit_info_1;
 +      gpa_t gpa = svm->vmcb->control.exit_info_2;
  
        /*
         * WARN if hardware generates a fault with an error code that collides
        if (WARN_ON_ONCE(error_code & PFERR_SYNTHETIC_MASK))
                error_code &= ~PFERR_SYNTHETIC_MASK;
  
 +      /*
 +       * Expedite fast MMIO kicks if the next RIP is known and KVM is allowed
 +       * emulate a page fault, e.g. skipping the current instruction is wrong
 +       * if the #NPF occurred while vectoring an event.
 +       */
 +      if ((error_code & PFERR_RSVD_MASK) && !is_guest_mode(vcpu)) {
 +              const int emul_type = EMULTYPE_PF | EMULTYPE_NO_DECODE;
 +
 +              if (svm_check_emulate_instruction(vcpu, emul_type, NULL, 0))
 +                      return 1;
 +
 +              if (nrips && svm->vmcb->control.next_rip &&
 +                  !kvm_io_bus_write(vcpu, KVM_FAST_MMIO_BUS, gpa, 0, NULL)) {
 +                      trace_kvm_fast_mmio(gpa);
 +                      return kvm_skip_emulated_instruction(vcpu);
 +              }
 +      }
 +
        if (sev_snp_guest(vcpu->kvm) && (error_code & PFERR_GUEST_ENC_MASK))
                error_code |= PFERR_PRIVATE_ACCESS;
  
 -      trace_kvm_page_fault(vcpu, fault_address, error_code);
 -      rc = kvm_mmu_page_fault(vcpu, fault_address, error_code,
 +      trace_kvm_page_fault(vcpu, gpa, error_code);
 +      rc = kvm_mmu_page_fault(vcpu, gpa, error_code,
                                static_cpu_has(X86_FEATURE_DECODEASSISTS) ?
                                svm->vmcb->control.insn_bytes : NULL,
                                svm->vmcb->control.insn_len);
  
        if (rc > 0 && error_code & PFERR_GUEST_RMP_MASK)
 -              sev_handle_rmp_fault(vcpu, fault_address, error_code);
 +              sev_handle_rmp_fault(vcpu, gpa, error_code);
  
        return rc;
  }
@@@ -2128,13 -2142,12 +2171,13 @@@ static int vmload_vmsave_interception(s
  
        ret = kvm_skip_emulated_instruction(vcpu);
  
 +      /* KVM always performs VMLOAD/VMSAVE on VMCB01 (see __svm_vcpu_run()) */
        if (vmload) {
 -              svm_copy_vmloadsave_state(svm->vmcb, vmcb12);
 +              svm_copy_vmloadsave_state(svm->vmcb01.ptr, vmcb12);
                svm->sysenter_eip_hi = 0;
                svm->sysenter_esp_hi = 0;
        } else {
 -              svm_copy_vmloadsave_state(vmcb12, svm->vmcb);
 +              svm_copy_vmloadsave_state(vmcb12, svm->vmcb01.ptr);
        }
  
        kvm_vcpu_unmap(vcpu, &map);
@@@ -3301,11 -3314,10 +3344,11 @@@ static void dump_vmcb(struct kvm_vcpu *
        pr_err("%-20s%016llx\n", "tsc_offset:", control->tsc_offset);
        pr_err("%-20s%d\n", "asid:", control->asid);
        pr_err("%-20s%d\n", "tlb_ctl:", control->tlb_ctl);
 +      pr_err("%-20s%d\n", "erap_ctl:", control->erap_ctl);
        pr_err("%-20s%08x\n", "int_ctl:", control->int_ctl);
        pr_err("%-20s%08x\n", "int_vector:", control->int_vector);
        pr_err("%-20s%08x\n", "int_state:", control->int_state);
 -      pr_err("%-20s%08x\n", "exit_code:", control->exit_code);
 +      pr_err("%-20s%016llx\n", "exit_code:", control->exit_code);
        pr_err("%-20s%016llx\n", "exit_info1:", control->exit_info_1);
        pr_err("%-20s%016llx\n", "exit_info2:", control->exit_info_2);
        pr_err("%-20s%08x\n", "exit_int_info:", control->exit_int_info);
@@@ -3473,21 -3485,23 +3516,21 @@@ no_vmsa
                sev_free_decrypted_vmsa(vcpu, save);
  }
  
 -static bool svm_check_exit_valid(u64 exit_code)
 +int svm_invoke_exit_handler(struct kvm_vcpu *vcpu, u64 __exit_code)
  {
 -      return (exit_code < ARRAY_SIZE(svm_exit_handlers) &&
 -              svm_exit_handlers[exit_code]);
 -}
 -
 -static int svm_handle_invalid_exit(struct kvm_vcpu *vcpu, u64 exit_code)
 -{
 -      dump_vmcb(vcpu);
 -      kvm_prepare_unexpected_reason_exit(vcpu, exit_code);
 -      return 0;
 -}
 +      u32 exit_code = __exit_code;
  
 -int svm_invoke_exit_handler(struct kvm_vcpu *vcpu, u64 exit_code)
 -{
 -      if (!svm_check_exit_valid(exit_code))
 -              return svm_handle_invalid_exit(vcpu, exit_code);
 +      /*
 +       * SVM uses negative values, i.e. 64-bit values, to indicate that VMRUN
 +       * failed.  Report all such errors to userspace (note, VMEXIT_INVALID,
 +       * a.k.a. SVM_EXIT_ERR, is special cased by svm_handle_exit()).  Skip
 +       * the check when running as a VM, as KVM has historically left garbage
 +       * in bits 63:32, i.e. running KVM-on-KVM would hit false positives if
 +       * the underlying kernel is buggy.
 +       */
 +      if (!cpu_feature_enabled(X86_FEATURE_HYPERVISOR) &&
 +          (u64)exit_code != __exit_code)
 +              goto unexpected_vmexit;
  
  #ifdef CONFIG_MITIGATION_RETPOLINE
        if (exit_code == SVM_EXIT_MSR)
                return sev_handle_vmgexit(vcpu);
  #endif
  #endif
 +      if (exit_code >= ARRAY_SIZE(svm_exit_handlers))
 +              goto unexpected_vmexit;
 +
 +      exit_code = array_index_nospec(exit_code, ARRAY_SIZE(svm_exit_handlers));
 +      if (!svm_exit_handlers[exit_code])
 +              goto unexpected_vmexit;
 +
        return svm_exit_handlers[exit_code](vcpu);
 +
 +unexpected_vmexit:
 +      dump_vmcb(vcpu);
 +      kvm_prepare_unexpected_reason_exit(vcpu, __exit_code);
 +      return 0;
  }
  
  static void svm_get_exit_info(struct kvm_vcpu *vcpu, u32 *reason,
@@@ -3556,6 -3558,7 +3599,6 @@@ static int svm_handle_exit(struct kvm_v
  {
        struct vcpu_svm *svm = to_svm(vcpu);
        struct kvm_run *kvm_run = vcpu->run;
 -      u32 exit_code = svm->vmcb->control.exit_code;
  
        /* SEV-ES guests must use the CR write traps to track CR registers. */
        if (!sev_es_guest(vcpu->kvm)) {
                        return 1;
        }
  
 -      if (svm->vmcb->control.exit_code == SVM_EXIT_ERR) {
 +      if (svm_is_vmrun_failure(svm->vmcb->control.exit_code)) {
                kvm_run->exit_reason = KVM_EXIT_FAIL_ENTRY;
                kvm_run->fail_entry.hardware_entry_failure_reason
                        = svm->vmcb->control.exit_code;
        if (exit_fastpath != EXIT_FASTPATH_NONE)
                return 1;
  
 -      return svm_invoke_exit_handler(vcpu, exit_code);
 +      return svm_invoke_exit_handler(vcpu, svm->vmcb->control.exit_code);
  }
  
  static int pre_svm_run(struct kvm_vcpu *vcpu)
@@@ -4022,13 -4025,6 +4065,13 @@@ static void svm_flush_tlb_gva(struct kv
        invlpga(gva, svm->vmcb->control.asid);
  }
  
 +static void svm_flush_tlb_guest(struct kvm_vcpu *vcpu)
 +{
 +      kvm_register_mark_dirty(vcpu, VCPU_EXREG_ERAPS);
 +
 +      svm_flush_tlb_asid(vcpu);
 +}
 +
  static inline void sync_cr8_to_lapic(struct kvm_vcpu *vcpu)
  {
        struct vcpu_svm *svm = to_svm(vcpu);
@@@ -4287,10 -4283,6 +4330,10 @@@ static __no_kcsan fastpath_t svm_vcpu_r
        }
        svm->vmcb->save.cr2 = vcpu->arch.cr2;
  
 +      if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS) &&
 +          kvm_register_is_dirty(vcpu, VCPU_EXREG_ERAPS))
 +              svm->vmcb->control.erap_ctl |= ERAP_CONTROL_CLEAR_RAP;
 +
        svm_hv_update_vp_id(svm->vmcb, vcpu);
  
        /*
  
                /* Track VMRUNs that have made past consistency checking */
                if (svm->nested.nested_run_pending &&
 -                  svm->vmcb->control.exit_code != SVM_EXIT_ERR)
 +                  !svm_is_vmrun_failure(svm->vmcb->control.exit_code))
                          ++vcpu->stat.nested_run;
  
                svm->nested.nested_run_pending = 0;
        }
  
        svm->vmcb->control.tlb_ctl = TLB_CONTROL_DO_NOTHING;
 +
 +      /*
 +       * Unconditionally mask off the CLEAR_RAP bit, the AND is just as cheap
 +       * as the TEST+Jcc to avoid it.
 +       */
 +      if (cpu_feature_enabled(X86_FEATURE_ERAPS))
 +              svm->vmcb->control.erap_ctl &= ~ERAP_CONTROL_CLEAR_RAP;
 +
        vmcb_mark_all_clean(svm->vmcb);
  
        /* if exit due to PF check for async PF */
  
        vcpu->arch.regs_avail &= ~SVM_REGS_LAZY_LOAD_SET;
  
+       if (!msr_write_intercepted(vcpu, MSR_AMD64_PERF_CNTR_GLOBAL_CTL))
+               rdmsrq(MSR_AMD64_PERF_CNTR_GLOBAL_CTL, vcpu_to_pmu(vcpu)->global_ctrl);
        trace_kvm_exit(vcpu, KVM_ISA_SVM);
  
        svm_complete_interrupts(vcpu);
@@@ -5130,7 -5117,7 +5176,7 @@@ struct kvm_x86_ops svm_x86_ops __initda
        .flush_tlb_all = svm_flush_tlb_all,
        .flush_tlb_current = svm_flush_tlb_current,
        .flush_tlb_gva = svm_flush_tlb_gva,
 -      .flush_tlb_guest = svm_flush_tlb_asid,
 +      .flush_tlb_guest = svm_flush_tlb_guest,
  
        .vcpu_pre_run = svm_vcpu_pre_run,
        .vcpu_run = svm_vcpu_run,
@@@ -5259,7 -5246,7 +5305,7 @@@ static __init void svm_adjust_mmio_mask
  
  static __init void svm_set_cpu_caps(void)
  {
 -      kvm_set_cpu_caps();
 +      kvm_initialize_cpu_caps();
  
        kvm_caps.supported_perf_cap = 0;
  
         */
        kvm_cpu_cap_clear(X86_FEATURE_BUS_LOCK_DETECT);
        kvm_cpu_cap_clear(X86_FEATURE_MSR_IMM);
 +
 +      kvm_setup_xss_caps();
 +      kvm_finalize_cpu_caps();
  }
  
  static __init int svm_hardware_setup(void)
index 881bb914c16491f1a60ae1ec40c15157639c0d7d,1ee1edc8419d059659ad8b77e667fc68615fae86..248635da6766149da25802bd99d39c7f7f5a0a39
@@@ -19,7 -19,6 +19,7 @@@
  #include "trace.h"
  #include "vmx.h"
  #include "smm.h"
 +#include "x86_ops.h"
  
  static bool __read_mostly enable_shadow_vmcs = 1;
  module_param_named(enable_shadow_vmcs, enable_shadow_vmcs, bool, S_IRUGO);
@@@ -86,9 -85,6 +86,9 @@@ static void init_vmcs_shadow_fields(voi
                        pr_err("Missing field from shadow_read_only_field %x\n",
                               field + 1);
  
 +              if (get_vmcs12_field_offset(field) < 0)
 +                      continue;
 +
                clear_bit(field, vmx_vmread_bitmap);
                if (field & 1)
  #ifdef CONFIG_X86_64
                          field <= GUEST_TR_AR_BYTES,
                          "Update vmcs12_write_any() to drop reserved bits from AR_BYTES");
  
 +              if (get_vmcs12_field_offset(field) < 0)
 +                      continue;
 +
                /*
 -               * PML and the preemption timer can be emulated, but the
 -               * processor cannot vmwrite to fields that don't exist
 -               * on bare metal.
 +               * KVM emulates PML and the VMX preemption timer irrespective
 +               * of hardware support, but shadowing their related VMCS fields
 +               * requires hardware support as the CPU will reject VMWRITEs to
 +               * fields that don't exist.
                 */
                switch (field) {
                case GUEST_PML_INDEX:
                        if (!cpu_has_vmx_preemption_timer())
                                continue;
                        break;
 -              case GUEST_INTR_STATUS:
 -                      if (!cpu_has_vmx_apicv())
 -                              continue;
 -                      break;
                default:
                        break;
                }
@@@ -621,6 -617,47 +621,47 @@@ static inline void nested_vmx_set_inter
                                                   msr_bitmap_l0, msr);
  }
  
+ #define nested_vmx_merge_msr_bitmaps(msr, type)       \
+       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1,    \
+                                        msr_bitmap_l0, msr, type)
+ #define nested_vmx_merge_msr_bitmaps_read(msr) \
+       nested_vmx_merge_msr_bitmaps(msr, MSR_TYPE_R)
+ #define nested_vmx_merge_msr_bitmaps_write(msr) \
+       nested_vmx_merge_msr_bitmaps(msr, MSR_TYPE_W)
+ #define nested_vmx_merge_msr_bitmaps_rw(msr) \
+       nested_vmx_merge_msr_bitmaps(msr, MSR_TYPE_RW)
+ static void nested_vmx_merge_pmu_msr_bitmaps(struct kvm_vcpu *vcpu,
+                                            unsigned long *msr_bitmap_l1,
+                                            unsigned long *msr_bitmap_l0)
+ {
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       struct vcpu_vmx *vmx = to_vmx(vcpu);
+       int i;
+       /*
+        * Skip the merges if the vCPU doesn't have a mediated PMU MSR, i.e. if
+        * none of the MSRs can possibly be passed through to L1.
+        */
+       if (!kvm_vcpu_has_mediated_pmu(vcpu))
+               return;
+       for (i = 0; i < pmu->nr_arch_gp_counters; i++) {
+               nested_vmx_merge_msr_bitmaps_rw(MSR_IA32_PERFCTR0 + i);
+               nested_vmx_merge_msr_bitmaps_rw(MSR_IA32_PMC0 + i);
+       }
+       for (i = 0; i < pmu->nr_arch_fixed_counters; i++)
+               nested_vmx_merge_msr_bitmaps_rw(MSR_CORE_PERF_FIXED_CTR0 + i);
+       nested_vmx_merge_msr_bitmaps_rw(MSR_CORE_PERF_GLOBAL_CTRL);
+       nested_vmx_merge_msr_bitmaps_read(MSR_CORE_PERF_GLOBAL_STATUS);
+       nested_vmx_merge_msr_bitmaps_write(MSR_CORE_PERF_GLOBAL_OVF_CTRL);
+ }
  /*
   * Merge L0's and L1's MSR bitmap, return false to indicate that
   * we do not use the hardware.
@@@ -704,23 -741,13 +745,13 @@@ static inline bool nested_vmx_prepare_m
         * other runtime changes to vmcs01's bitmap, e.g. dynamic pass-through.
         */
  #ifdef CONFIG_X86_64
-       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
-                                        MSR_FS_BASE, MSR_TYPE_RW);
-       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
-                                        MSR_GS_BASE, MSR_TYPE_RW);
-       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
-                                        MSR_KERNEL_GS_BASE, MSR_TYPE_RW);
+       nested_vmx_merge_msr_bitmaps_rw(MSR_FS_BASE);
+       nested_vmx_merge_msr_bitmaps_rw(MSR_GS_BASE);
+       nested_vmx_merge_msr_bitmaps_rw(MSR_KERNEL_GS_BASE);
  #endif
-       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
-                                        MSR_IA32_SPEC_CTRL, MSR_TYPE_RW);
-       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
-                                        MSR_IA32_PRED_CMD, MSR_TYPE_W);
-       nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
-                                        MSR_IA32_FLUSH_CMD, MSR_TYPE_W);
+       nested_vmx_merge_msr_bitmaps_rw(MSR_IA32_SPEC_CTRL);
+       nested_vmx_merge_msr_bitmaps_write(MSR_IA32_PRED_CMD);
+       nested_vmx_merge_msr_bitmaps_write(MSR_IA32_FLUSH_CMD);
  
        nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
                                         MSR_IA32_APERF, MSR_TYPE_R);
        nested_vmx_set_intercept_for_msr(vmx, msr_bitmap_l1, msr_bitmap_l0,
                                         MSR_IA32_PL3_SSP, MSR_TYPE_RW);
  
+       nested_vmx_merge_pmu_msr_bitmaps(vcpu, msr_bitmap_l1, msr_bitmap_l0);
        kvm_vcpu_unmap(vcpu, &map);
  
        vmx->nested.force_msr_bitmap_recalc = false;
@@@ -1046,16 -1075,12 +1079,12 @@@ static bool nested_vmx_get_vmexit_msr_v
         * does not include the time taken for emulation of the L2->L1
         * VM-exit in L0, use the more accurate value.
         */
-       if (msr_index == MSR_IA32_TSC) {
-               int i = vmx_find_loadstore_msr_slot(&vmx->msr_autostore.guest,
-                                                   MSR_IA32_TSC);
-               if (i >= 0) {
-                       u64 val = vmx->msr_autostore.guest.val[i].value;
+       if (msr_index == MSR_IA32_TSC && vmx->nested.tsc_autostore_slot >= 0) {
+               int slot = vmx->nested.tsc_autostore_slot;
+               u64 host_tsc = vmx->msr_autostore.val[slot].value;
  
-                       *data = kvm_read_l1_tsc(vcpu, val);
-                       return true;
-               }
+               *data = kvm_read_l1_tsc(vcpu, host_tsc);
+               return true;
        }
  
        if (kvm_emulate_msr_read(vcpu, msr_index, data)) {
@@@ -1134,42 -1159,6 +1163,6 @@@ static bool nested_msr_store_list_has_m
        return false;
  }
  
- static void prepare_vmx_msr_autostore_list(struct kvm_vcpu *vcpu,
-                                          u32 msr_index)
- {
-       struct vcpu_vmx *vmx = to_vmx(vcpu);
-       struct vmx_msrs *autostore = &vmx->msr_autostore.guest;
-       bool in_vmcs12_store_list;
-       int msr_autostore_slot;
-       bool in_autostore_list;
-       int last;
-       msr_autostore_slot = vmx_find_loadstore_msr_slot(autostore, msr_index);
-       in_autostore_list = msr_autostore_slot >= 0;
-       in_vmcs12_store_list = nested_msr_store_list_has_msr(vcpu, msr_index);
-       if (in_vmcs12_store_list && !in_autostore_list) {
-               if (autostore->nr == MAX_NR_LOADSTORE_MSRS) {
-                       /*
-                        * Emulated VMEntry does not fail here.  Instead a less
-                        * accurate value will be returned by
-                        * nested_vmx_get_vmexit_msr_value() by reading KVM's
-                        * internal MSR state instead of reading the value from
-                        * the vmcs02 VMExit MSR-store area.
-                        */
-                       pr_warn_ratelimited(
-                               "Not enough msr entries in msr_autostore.  Can't add msr %x\n",
-                               msr_index);
-                       return;
-               }
-               last = autostore->nr++;
-               autostore->val[last].index = msr_index;
-       } else if (!in_vmcs12_store_list && in_autostore_list) {
-               last = --autostore->nr;
-               autostore->val[msr_autostore_slot] = autostore->val[last];
-       }
- }
  /*
   * Load guest's/host's cr3 at nested entry/exit.  @nested_ept is true if we are
   * emulating VM-Entry into a guest with EPT enabled.  On failure, the expected
@@@ -2337,7 -2326,7 +2330,7 @@@ static void prepare_vmcs02_constant_sta
         * addresses are constant (for vmcs02), the counts can change based
         * on L2's behavior, e.g. switching to/from long mode.
         */
-       vmcs_write64(VM_EXIT_MSR_STORE_ADDR, __pa(vmx->msr_autostore.guest.val));
+       vmcs_write64(VM_EXIT_MSR_STORE_ADDR, __pa(vmx->msr_autostore.val));
        vmcs_write64(VM_EXIT_MSR_LOAD_ADDR, __pa(vmx->msr_autoload.host.val));
        vmcs_write64(VM_ENTRY_MSR_LOAD_ADDR, __pa(vmx->msr_autoload.guest.val));
  
@@@ -2405,6 -2394,7 +2398,6 @@@ static void prepare_vmcs02_early(struc
        exec_control &= ~CPU_BASED_TPR_SHADOW;
        exec_control |= vmcs12->cpu_based_vm_exec_control;
  
 -      vmx->nested.l1_tpr_threshold = -1;
        if (exec_control & CPU_BASED_TPR_SHADOW)
                vmcs_write32(TPR_THRESHOLD, vmcs12->tpr_threshold);
  #ifdef CONFIG_X86_64
@@@ -2669,12 -2659,25 +2662,25 @@@ static void prepare_vmcs02_rare(struct 
        }
  
        /*
-        * Make sure the msr_autostore list is up to date before we set the
-        * count in the vmcs02.
+        * If vmcs12 is configured to save TSC on exit via the auto-store list,
+        * append the MSR to vmcs02's auto-store list so that KVM effectively
+        * reads TSC at the time of VM-Exit from L2.  The saved value will be
+        * propagated to vmcs12's list on nested VM-Exit.
+        *
+        * Don't increment the number of MSRs in the vCPU structure, as saving
+        * TSC is specific to this particular incarnation of vmcb02, i.e. must
+        * not bleed into vmcs01.
         */
-       prepare_vmx_msr_autostore_list(&vmx->vcpu, MSR_IA32_TSC);
+       if (nested_msr_store_list_has_msr(&vmx->vcpu, MSR_IA32_TSC) &&
+           !WARN_ON_ONCE(vmx->msr_autostore.nr >= ARRAY_SIZE(vmx->msr_autostore.val))) {
+               vmx->nested.tsc_autostore_slot = vmx->msr_autostore.nr;
+               vmx->msr_autostore.val[vmx->msr_autostore.nr].index = MSR_IA32_TSC;
  
-       vmcs_write32(VM_EXIT_MSR_STORE_COUNT, vmx->msr_autostore.guest.nr);
+               vmcs_write32(VM_EXIT_MSR_STORE_COUNT, vmx->msr_autostore.nr + 1);
+       } else {
+               vmx->nested.tsc_autostore_slot = -1;
+               vmcs_write32(VM_EXIT_MSR_STORE_COUNT, vmx->msr_autostore.nr);
+       }
        vmcs_write32(VM_EXIT_MSR_LOAD_COUNT, vmx->msr_autoload.host.nr);
        vmcs_write32(VM_ENTRY_MSR_LOAD_COUNT, vmx->msr_autoload.guest.nr);
  
@@@ -3983,6 -3986,28 +3989,6 @@@ static void vmcs12_save_pending_event(s
        }
  }
  
 -
 -void nested_mark_vmcs12_pages_dirty(struct kvm_vcpu *vcpu)
 -{
 -      struct vmcs12 *vmcs12 = get_vmcs12(vcpu);
 -      gfn_t gfn;
 -
 -      /*
 -       * Don't need to mark the APIC access page dirty; it is never
 -       * written to by the CPU during APIC virtualization.
 -       */
 -
 -      if (nested_cpu_has(vmcs12, CPU_BASED_TPR_SHADOW)) {
 -              gfn = vmcs12->virtual_apic_page_addr >> PAGE_SHIFT;
 -              kvm_vcpu_mark_page_dirty(vcpu, gfn);
 -      }
 -
 -      if (nested_cpu_has_posted_intr(vmcs12)) {
 -              gfn = vmcs12->posted_intr_desc_addr >> PAGE_SHIFT;
 -              kvm_vcpu_mark_page_dirty(vcpu, gfn);
 -      }
 -}
 -
  static int vmx_complete_nested_posted_interrupt(struct kvm_vcpu *vcpu)
  {
        struct vcpu_vmx *vmx = to_vmx(vcpu);
                }
        }
  
 -      nested_mark_vmcs12_pages_dirty(vcpu);
 +      kvm_vcpu_map_mark_dirty(vcpu, &vmx->nested.virtual_apic_map);
 +      kvm_vcpu_map_mark_dirty(vcpu, &vmx->nested.pi_desc_map);
        return 0;
  
  mmio_needed:
@@@ -5118,15 -5142,47 +5124,19 @@@ void __nested_vmx_vmexit(struct kvm_vcp
  
        kvm_nested_vmexit_handle_ibrs(vcpu);
  
-       /* Update any VMCS fields that might have changed while L2 ran */
+       /*
+        * Update any VMCS fields that might have changed while vmcs02 was the
+        * active VMCS.  The tracking is per-vCPU, not per-VMCS.
+        */
+       vmcs_write32(VM_EXIT_MSR_STORE_COUNT, vmx->msr_autostore.nr);
        vmcs_write32(VM_EXIT_MSR_LOAD_COUNT, vmx->msr_autoload.host.nr);
        vmcs_write32(VM_ENTRY_MSR_LOAD_COUNT, vmx->msr_autoload.guest.nr);
        vmcs_write64(TSC_OFFSET, vcpu->arch.tsc_offset);
        if (kvm_caps.has_tsc_control)
                vmcs_write64(TSC_MULTIPLIER, vcpu->arch.tsc_scaling_ratio);
  
 -      if (vmx->nested.l1_tpr_threshold != -1)
 -              vmcs_write32(TPR_THRESHOLD, vmx->nested.l1_tpr_threshold);
 -
 -      if (vmx->nested.change_vmcs01_virtual_apic_mode) {
 -              vmx->nested.change_vmcs01_virtual_apic_mode = false;
 -              vmx_set_virtual_apic_mode(vcpu);
 -      }
 -
 -      if (vmx->nested.update_vmcs01_cpu_dirty_logging) {
 -              vmx->nested.update_vmcs01_cpu_dirty_logging = false;
 -              vmx_update_cpu_dirty_logging(vcpu);
 -      }
 -
        nested_put_vmcs12_pages(vcpu);
  
 -      if (vmx->nested.reload_vmcs01_apic_access_page) {
 -              vmx->nested.reload_vmcs01_apic_access_page = false;
 -              kvm_make_request(KVM_REQ_APIC_PAGE_RELOAD, vcpu);
 -      }
 -
 -      if (vmx->nested.update_vmcs01_apicv_status) {
 -              vmx->nested.update_vmcs01_apicv_status = false;
 -              kvm_make_request(KVM_REQ_APICV_UPDATE, vcpu);
 -      }
 -
 -      if (vmx->nested.update_vmcs01_hwapic_isr) {
 -              vmx->nested.update_vmcs01_hwapic_isr = false;
 -              kvm_apic_update_hwapic_isr(vcpu);
 -      }
 -
        if ((vm_exit_reason != -1) &&
            (enable_shadow_vmcs || nested_vmx_is_evmptr12_valid(vmx)))
                vmx->nested.need_vmcs12_to_shadow_sync = true;
@@@ -7027,6 -7083,12 +7037,6 @@@ void nested_vmx_set_vmcs_shadowing_bitm
        }
  }
  
 -/*
 - * Indexing into the vmcs12 uses the VMCS encoding rotated left by 6.  Undo
 - * that madness to get the encoding for comparison.
 - */
 -#define VMCS12_IDX_TO_ENC(idx) ((u16)(((u16)(idx) >> 6) | ((u16)(idx) << 10)))
 -
  static u64 nested_vmx_calc_vmcs_enum_msr(void)
  {
        /*
@@@ -7354,14 -7416,6 +7364,14 @@@ __init int nested_vmx_hardware_setup(in
  {
        int i;
  
 +      /*
 +       * Note!  The set of supported vmcs12 fields is consumed by both VMX
 +       * MSR and shadow VMCS setup.
 +       */
 +      nested_vmx_setup_vmcs12_fields();
 +
 +      nested_vmx_setup_ctls_msrs(&vmcs_config, vmx_capability.ept);
 +
        if (!cpu_has_vmx_shadow_vmcs())
                enable_shadow_vmcs = 0;
        if (enable_shadow_vmcs) {
diff --combined arch/x86/kvm/vmx/vmx.c
index 49f5caa45e1375be915c12de2b211b1404a3dc88,ba1262c3e3ffe73306d760aa8b5253a2a0800389..967b58a8ab9d0d47fb24def7b4ae70bfbe9d5ec0
@@@ -150,6 -150,8 +150,8 @@@ module_param_named(preemption_timer, en
  extern bool __read_mostly allow_smaller_maxphyaddr;
  module_param(allow_smaller_maxphyaddr, bool, S_IRUGO);
  
+ module_param(enable_mediated_pmu, bool, 0444);
  #define KVM_VM_CR0_ALWAYS_OFF (X86_CR0_NW | X86_CR0_CD)
  #define KVM_VM_CR0_ALWAYS_ON_UNRESTRICTED_GUEST X86_CR0_NE
  #define KVM_VM_CR0_ALWAYS_ON                          \
@@@ -1027,7 -1029,7 +1029,7 @@@ static __always_inline void clear_atomi
        vm_exit_controls_clearbit(vmx, exit);
  }
  
- int vmx_find_loadstore_msr_slot(struct vmx_msrs *m, u32 msr)
static int vmx_find_loadstore_msr_slot(struct vmx_msrs *m, u32 msr)
  {
        unsigned int i;
  
        return -ENOENT;
  }
  
- static void clear_atomic_switch_msr(struct vcpu_vmx *vmx, unsigned msr)
+ static void vmx_remove_auto_msr(struct vmx_msrs *m, u32 msr,
+                               unsigned long vmcs_count_field)
  {
        int i;
+       i = vmx_find_loadstore_msr_slot(m, msr);
+       if (i < 0)
+               return;
+       --m->nr;
+       m->val[i] = m->val[m->nr];
+       vmcs_write32(vmcs_count_field, m->nr);
+ }
+ static void clear_atomic_switch_msr(struct vcpu_vmx *vmx, unsigned msr)
+ {
        struct msr_autoload *m = &vmx->msr_autoload;
  
        switch (msr) {
                }
                break;
        }
-       i = vmx_find_loadstore_msr_slot(&m->guest, msr);
-       if (i < 0)
-               goto skip_guest;
-       --m->guest.nr;
-       m->guest.val[i] = m->guest.val[m->guest.nr];
-       vmcs_write32(VM_ENTRY_MSR_LOAD_COUNT, m->guest.nr);
  
- skip_guest:
-       i = vmx_find_loadstore_msr_slot(&m->host, msr);
-       if (i < 0)
-               return;
-       --m->host.nr;
-       m->host.val[i] = m->host.val[m->host.nr];
-       vmcs_write32(VM_EXIT_MSR_LOAD_COUNT, m->host.nr);
+       vmx_remove_auto_msr(&m->guest, msr, VM_ENTRY_MSR_LOAD_COUNT);
+       vmx_remove_auto_msr(&m->host, msr, VM_EXIT_MSR_LOAD_COUNT);
  }
  
  static __always_inline void add_atomic_switch_msr_special(struct vcpu_vmx *vmx,
        vm_exit_controls_setbit(vmx, exit);
  }
  
+ static void vmx_add_auto_msr(struct vmx_msrs *m, u32 msr, u64 value,
+                            unsigned long vmcs_count_field, struct kvm *kvm)
+ {
+       int i;
+       i = vmx_find_loadstore_msr_slot(m, msr);
+       if (i < 0) {
+               if (KVM_BUG_ON(m->nr == MAX_NR_LOADSTORE_MSRS, kvm))
+                       return;
+               i = m->nr++;
+               m->val[i].index = msr;
+               vmcs_write32(vmcs_count_field, m->nr);
+       }
+       m->val[i].value = value;
+ }
  static void add_atomic_switch_msr(struct vcpu_vmx *vmx, unsigned msr,
-                                 u64 guest_val, u64 host_val, bool entry_only)
+                                 u64 guest_val, u64 host_val)
  {
-       int i, j = 0;
        struct msr_autoload *m = &vmx->msr_autoload;
+       struct kvm *kvm = vmx->vcpu.kvm;
  
        switch (msr) {
        case MSR_EFER:
                wrmsrq(MSR_IA32_PEBS_ENABLE, 0);
        }
  
-       i = vmx_find_loadstore_msr_slot(&m->guest, msr);
-       if (!entry_only)
-               j = vmx_find_loadstore_msr_slot(&m->host, msr);
-       if ((i < 0 && m->guest.nr == MAX_NR_LOADSTORE_MSRS) ||
-           (j < 0 &&  m->host.nr == MAX_NR_LOADSTORE_MSRS)) {
-               printk_once(KERN_WARNING "Not enough msr switch entries. "
-                               "Can't add msr %x\n", msr);
-               return;
-       }
-       if (i < 0) {
-               i = m->guest.nr++;
-               vmcs_write32(VM_ENTRY_MSR_LOAD_COUNT, m->guest.nr);
-       }
-       m->guest.val[i].index = msr;
-       m->guest.val[i].value = guest_val;
-       if (entry_only)
-               return;
-       if (j < 0) {
-               j = m->host.nr++;
-               vmcs_write32(VM_EXIT_MSR_LOAD_COUNT, m->host.nr);
-       }
-       m->host.val[j].index = msr;
-       m->host.val[j].value = host_val;
+       vmx_add_auto_msr(&m->guest, msr, guest_val, VM_ENTRY_MSR_LOAD_COUNT, kvm);
+       vmx_add_auto_msr(&m->guest, msr, host_val, VM_EXIT_MSR_LOAD_COUNT, kvm);
  }
  
  static bool update_transition_efer(struct vcpu_vmx *vmx)
                if (!(guest_efer & EFER_LMA))
                        guest_efer &= ~EFER_LME;
                if (guest_efer != kvm_host.efer)
-                       add_atomic_switch_msr(vmx, MSR_EFER,
-                                             guest_efer, kvm_host.efer, false);
+                       add_atomic_switch_msr(vmx, MSR_EFER, guest_efer, kvm_host.efer);
                else
                        clear_atomic_switch_msr(vmx, MSR_EFER);
                return false;
        return true;
  }
  
+ static void vmx_add_autostore_msr(struct vcpu_vmx *vmx, u32 msr)
+ {
+       vmx_add_auto_msr(&vmx->msr_autostore, msr, 0, VM_EXIT_MSR_STORE_COUNT,
+                        vmx->vcpu.kvm);
+ }
+ static void vmx_remove_autostore_msr(struct vcpu_vmx *vmx, u32 msr)
+ {
+       vmx_remove_auto_msr(&vmx->msr_autostore, msr, VM_EXIT_MSR_STORE_COUNT);
+ }
  #ifdef CONFIG_X86_32
  /*
   * On 32-bit kernels, VM exits still load the FS and GS bases from the
@@@ -1594,41 -1600,6 +1600,41 @@@ void vmx_vcpu_put(struct kvm_vcpu *vcpu
        vmx_prepare_switch_to_host(to_vmx(vcpu));
  }
  
 +static void vmx_switch_loaded_vmcs(struct kvm_vcpu *vcpu,
 +                                 struct loaded_vmcs *vmcs)
 +{
 +      struct vcpu_vmx *vmx = to_vmx(vcpu);
 +      int cpu;
 +
 +      cpu = get_cpu();
 +      vmx->loaded_vmcs = vmcs;
 +      vmx_vcpu_load_vmcs(vcpu, cpu);
 +      put_cpu();
 +}
 +
 +static void vmx_load_vmcs01(struct kvm_vcpu *vcpu)
 +{
 +      struct vcpu_vmx *vmx = to_vmx(vcpu);
 +
 +      if (!is_guest_mode(vcpu)) {
 +              WARN_ON_ONCE(vmx->loaded_vmcs != &vmx->vmcs01);
 +              return;
 +      }
 +
 +      WARN_ON_ONCE(vmx->loaded_vmcs != &vmx->nested.vmcs02);
 +      vmx_switch_loaded_vmcs(vcpu, &vmx->vmcs01);
 +}
 +
 +static void vmx_put_vmcs01(struct kvm_vcpu *vcpu)
 +{
 +      if (!is_guest_mode(vcpu))
 +              return;
 +
 +      vmx_switch_loaded_vmcs(vcpu, &to_vmx(vcpu)->nested.vmcs02);
 +}
 +DEFINE_GUARD(vmx_vmcs01, struct kvm_vcpu *,
 +           vmx_load_vmcs01(_T), vmx_put_vmcs01(_T))
 +
  bool vmx_emulation_required(struct kvm_vcpu *vcpu)
  {
        return emulate_invalid_guest_state && !vmx_guest_state_valid(vcpu);
@@@ -2956,23 -2927,8 +2962,23 @@@ int vmx_check_processor_compat(void
        }
        if (nested)
                nested_vmx_setup_ctls_msrs(&vmcs_conf, vmx_cap.ept);
 +
        if (memcmp(&vmcs_config, &vmcs_conf, sizeof(struct vmcs_config))) {
 -              pr_err("Inconsistent VMCS config on CPU %d\n", cpu);
 +              u32 *gold = (void *)&vmcs_config;
 +              u32 *mine = (void *)&vmcs_conf;
 +              int i;
 +
 +              BUILD_BUG_ON(sizeof(struct vmcs_config) % sizeof(u32));
 +
 +              pr_err("VMCS config on CPU %d doesn't match reference config:", cpu);
 +              for (i = 0; i < sizeof(struct vmcs_config) / sizeof(u32); i++) {
 +                      if (gold[i] == mine[i])
 +                              continue;
 +
 +                      pr_cont("\n  Offset %u REF = 0x%08x, CPU%u = 0x%08x, mismatch = 0x%08x",
 +                              i * (int)sizeof(u32), gold[i], cpu, mine[i], gold[i] ^ mine[i]);
 +              }
 +              pr_cont("\n");
                return -EIO;
        }
        return 0;
@@@ -4278,6 -4234,62 +4284,62 @@@ void pt_update_intercept_for_msr(struc
        }
  }
  
+ static void vmx_recalc_pmu_msr_intercepts(struct kvm_vcpu *vcpu)
+ {
+       u64 vm_exit_controls_bits = VM_EXIT_LOAD_IA32_PERF_GLOBAL_CTRL |
+                                   VM_EXIT_SAVE_IA32_PERF_GLOBAL_CTRL;
+       bool has_mediated_pmu = kvm_vcpu_has_mediated_pmu(vcpu);
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       struct vcpu_vmx *vmx = to_vmx(vcpu);
+       bool intercept = !has_mediated_pmu;
+       int i;
+       if (!enable_mediated_pmu)
+               return;
+       if (!cpu_has_save_perf_global_ctrl()) {
+               vm_exit_controls_bits &= ~VM_EXIT_SAVE_IA32_PERF_GLOBAL_CTRL;
+               if (has_mediated_pmu)
+                       vmx_add_autostore_msr(vmx, MSR_CORE_PERF_GLOBAL_CTRL);
+               else
+                       vmx_remove_autostore_msr(vmx, MSR_CORE_PERF_GLOBAL_CTRL);
+       }
+       vm_entry_controls_changebit(vmx, VM_ENTRY_LOAD_IA32_PERF_GLOBAL_CTRL,
+                                   has_mediated_pmu);
+       vm_exit_controls_changebit(vmx, vm_exit_controls_bits, has_mediated_pmu);
+       for (i = 0; i < pmu->nr_arch_gp_counters; i++) {
+               vmx_set_intercept_for_msr(vcpu, MSR_IA32_PERFCTR0 + i,
+                                         MSR_TYPE_RW, intercept);
+               vmx_set_intercept_for_msr(vcpu, MSR_IA32_PMC0 + i, MSR_TYPE_RW,
+                                         intercept || !fw_writes_is_enabled(vcpu));
+       }
+       for ( ; i < kvm_pmu_cap.num_counters_gp; i++) {
+               vmx_set_intercept_for_msr(vcpu, MSR_IA32_PERFCTR0 + i,
+                                         MSR_TYPE_RW, true);
+               vmx_set_intercept_for_msr(vcpu, MSR_IA32_PMC0 + i,
+                                         MSR_TYPE_RW, true);
+       }
+       for (i = 0; i < pmu->nr_arch_fixed_counters; i++)
+               vmx_set_intercept_for_msr(vcpu, MSR_CORE_PERF_FIXED_CTR0 + i,
+                                         MSR_TYPE_RW, intercept);
+       for ( ; i < kvm_pmu_cap.num_counters_fixed; i++)
+               vmx_set_intercept_for_msr(vcpu, MSR_CORE_PERF_FIXED_CTR0 + i,
+                                         MSR_TYPE_RW, true);
+       intercept = kvm_need_perf_global_ctrl_intercept(vcpu);
+       vmx_set_intercept_for_msr(vcpu, MSR_CORE_PERF_GLOBAL_STATUS,
+                                 MSR_TYPE_RW, intercept);
+       vmx_set_intercept_for_msr(vcpu, MSR_CORE_PERF_GLOBAL_CTRL,
+                                 MSR_TYPE_RW, intercept);
+       vmx_set_intercept_for_msr(vcpu, MSR_CORE_PERF_GLOBAL_OVF_CTRL,
+                                 MSR_TYPE_RW, intercept);
+ }
  static void vmx_recalc_msr_intercepts(struct kvm_vcpu *vcpu)
  {
        bool intercept;
                vmx_set_intercept_for_msr(vcpu, MSR_IA32_S_CET, MSR_TYPE_RW, intercept);
        }
  
+       vmx_recalc_pmu_msr_intercepts(vcpu);
        /*
         * x2APIC and LBR MSR intercepts are modified on-demand and cannot be
         * filtered by userspace.
         */
  }
  
+ static void vmx_recalc_instruction_intercepts(struct kvm_vcpu *vcpu)
+ {
+       exec_controls_changebit(to_vmx(vcpu), CPU_BASED_RDPMC_EXITING,
+                               kvm_need_rdpmc_intercept(vcpu));
+ }
  void vmx_recalc_intercepts(struct kvm_vcpu *vcpu)
  {
+       vmx_recalc_instruction_intercepts(vcpu);
        vmx_recalc_msr_intercepts(vcpu);
  }
  
@@@ -4519,6 -4540,16 +4590,16 @@@ void vmx_set_constant_host_state(struc
                vmcs_writel(HOST_SSP, 0);
                vmcs_writel(HOST_INTR_SSP_TABLE, 0);
        }
+       /*
+        * When running a guest with a mediated PMU, guest state is resident in
+        * hardware after VM-Exit.  Zero PERF_GLOBAL_CTRL on exit so that host
+        * activity doesn't bleed into the guest counters.  When running with
+        * an emulated PMU, PERF_GLOBAL_CTRL is dynamically computed on every
+        * entry/exit to merge guest and host PMU usage.
+        */
+       if (enable_mediated_pmu)
+               vmcs_write64(HOST_IA32_PERF_GLOBAL_CTRL, 0);
  }
  
  void set_cr4_guest_host_mask(struct vcpu_vmx *vmx)
@@@ -4586,14 -4617,18 +4667,15 @@@ static u32 vmx_get_initial_vmexit_ctrl(
                                 VM_EXIT_CLEAR_IA32_RTIT_CTL);
        /* Loading of EFER and PERF_GLOBAL_CTRL are toggled dynamically */
        return vmexit_ctrl &
-               ~(VM_EXIT_LOAD_IA32_PERF_GLOBAL_CTRL | VM_EXIT_LOAD_IA32_EFER);
+               ~(VM_EXIT_LOAD_IA32_PERF_GLOBAL_CTRL | VM_EXIT_LOAD_IA32_EFER |
+                 VM_EXIT_SAVE_IA32_PERF_GLOBAL_CTRL);
  }
  
  void vmx_refresh_apicv_exec_ctrl(struct kvm_vcpu *vcpu)
  {
        struct vcpu_vmx *vmx = to_vmx(vcpu);
  
 -      if (is_guest_mode(vcpu)) {
 -              vmx->nested.update_vmcs01_apicv_status = true;
 -              return;
 -      }
 +      guard(vmx_vmcs01)(vcpu);
  
        pin_controls_set(vmx, vmx_pin_based_exec_ctrl(vmx));
  
@@@ -4918,6 -4953,7 +5000,7 @@@ static void init_vmcs(struct vcpu_vmx *
                vmcs_write64(VM_FUNCTION_CONTROL, 0);
  
        vmcs_write32(VM_EXIT_MSR_STORE_COUNT, 0);
+       vmcs_write64(VM_EXIT_MSR_STORE_ADDR, __pa(vmx->msr_autostore.val));
        vmcs_write32(VM_EXIT_MSR_LOAD_COUNT, 0);
        vmcs_write64(VM_EXIT_MSR_LOAD_ADDR, __pa(vmx->msr_autoload.host.val));
        vmcs_write32(VM_ENTRY_MSR_LOAD_COUNT, 0);
@@@ -5350,53 -5386,12 +5433,53 @@@ static bool is_xfd_nm_fault(struct kvm_
               !kvm_is_cr0_bit_set(vcpu, X86_CR0_TS);
  }
  
 +static int vmx_handle_page_fault(struct kvm_vcpu *vcpu, u32 error_code)
 +{
 +      unsigned long cr2 = vmx_get_exit_qual(vcpu);
 +
 +      if (vcpu->arch.apf.host_apf_flags)
 +              goto handle_pf;
 +
 +      /* When using EPT, KVM intercepts #PF only to detect illegal GPAs. */
 +      WARN_ON_ONCE(enable_ept && !allow_smaller_maxphyaddr);
 +
 +      /*
 +       * On SGX2 hardware, EPCM violations are delivered as #PF with the SGX
 +       * flag set in the error code (SGX1 hardware generates #GP(0)).  EPCM
 +       * violations have nothing to do with shadow paging and can never be
 +       * resolved by KVM; always reflect them into the guest.
 +       */
 +      if (error_code & PFERR_SGX_MASK) {
 +              WARN_ON_ONCE(!IS_ENABLED(CONFIG_X86_SGX_KVM) ||
 +                           !cpu_feature_enabled(X86_FEATURE_SGX2));
 +
 +              if (guest_cpu_cap_has(vcpu, X86_FEATURE_SGX2))
 +                      kvm_fixup_and_inject_pf_error(vcpu, cr2, error_code);
 +              else
 +                      kvm_inject_gp(vcpu, 0);
 +              return 1;
 +      }
 +
 +      /*
 +       * If EPT is enabled, fixup and inject the #PF.  KVM intercepts #PFs
 +       * only to set PFERR_RSVD as appropriate (hardware won't set RSVD due
 +       * to the GPA being legal with respect to host.MAXPHYADDR).
 +       */
 +      if (enable_ept) {
 +              kvm_fixup_and_inject_pf_error(vcpu, cr2, error_code);
 +              return 1;
 +      }
 +
 +handle_pf:
 +      return kvm_handle_page_fault(vcpu, error_code, cr2, NULL, 0);
 +}
 +
  static int handle_exception_nmi(struct kvm_vcpu *vcpu)
  {
        struct vcpu_vmx *vmx = to_vmx(vcpu);
        struct kvm_run *kvm_run = vcpu->run;
        u32 intr_info, ex_no, error_code;
 -      unsigned long cr2, dr6;
 +      unsigned long dr6;
        u32 vect_info;
  
        vect_info = vmx->idt_vectoring_info;
                return 0;
        }
  
 -      if (is_page_fault(intr_info)) {
 -              cr2 = vmx_get_exit_qual(vcpu);
 -              if (enable_ept && !vcpu->arch.apf.host_apf_flags) {
 -                      /*
 -                       * EPT will cause page fault only if we need to
 -                       * detect illegal GPAs.
 -                       */
 -                      WARN_ON_ONCE(!allow_smaller_maxphyaddr);
 -                      kvm_fixup_and_inject_pf_error(vcpu, cr2, error_code);
 -                      return 1;
 -              } else
 -                      return kvm_handle_page_fault(vcpu, error_code, cr2, NULL, 0);
 -      }
 +      if (is_page_fault(intr_info))
 +              return vmx_handle_page_fault(vcpu, error_code);
  
        ex_no = intr_info & INTR_INFO_VECTOR_MASK;
  
@@@ -6455,15 -6461,6 +6538,15 @@@ static void vmx_flush_pml_buffer(struc
        vmcs_write16(GUEST_PML_INDEX, PML_HEAD_INDEX);
  }
  
 +static void nested_vmx_mark_all_vmcs12_pages_dirty(struct kvm_vcpu *vcpu)
 +{
 +      struct vcpu_vmx *vmx = to_vmx(vcpu);
 +
 +      kvm_vcpu_map_mark_dirty(vcpu, &vmx->nested.apic_access_page_map);
 +      kvm_vcpu_map_mark_dirty(vcpu, &vmx->nested.virtual_apic_map);
 +      kvm_vcpu_map_mark_dirty(vcpu, &vmx->nested.pi_desc_map);
 +}
 +
  static void vmx_dump_sel(char *name, uint32_t sel)
  {
        pr_err("%s sel=0x%04x, attr=0x%05x, limit=0x%08x, base=0x%016lx\n",
@@@ -6584,7 -6581,7 +6667,7 @@@ void dump_vmcs(struct kvm_vcpu *vcpu
        if (vmcs_read32(VM_ENTRY_MSR_LOAD_COUNT) > 0)
                vmx_dump_msrs("guest autoload", &vmx->msr_autoload.guest);
        if (vmcs_read32(VM_EXIT_MSR_STORE_COUNT) > 0)
-               vmx_dump_msrs("guest autostore", &vmx->msr_autostore.guest);
+               vmx_dump_msrs("autostore", &vmx->msr_autostore);
  
        if (vmentry_ctl & VM_ENTRY_LOAD_CET_STATE)
                pr_err("S_CET = 0x%016lx, SSP = 0x%016lx, SSP TABLE = 0x%016lx\n",
@@@ -6741,7 -6738,7 +6824,7 @@@ static int __vmx_handle_exit(struct kvm
                 * Mark them dirty on every exit from L2 to prevent them from
                 * getting out of sync with dirty tracking.
                 */
 -              nested_mark_vmcs12_pages_dirty(vcpu);
 +              nested_vmx_mark_all_vmcs12_pages_dirty(vcpu);
  
                /*
                 * Synthesize a triple fault if L2 state is invalid.  In normal
@@@ -6878,10 -6875,11 +6961,10 @@@ void vmx_update_cr8_intercept(struct kv
                nested_cpu_has(vmcs12, CPU_BASED_TPR_SHADOW))
                return;
  
 +      guard(vmx_vmcs01)(vcpu);
 +
        tpr_threshold = (irr == -1 || tpr < irr) ? 0 : irr;
 -      if (is_guest_mode(vcpu))
 -              to_vmx(vcpu)->nested.l1_tpr_threshold = tpr_threshold;
 -      else
 -              vmcs_write32(TPR_THRESHOLD, tpr_threshold);
 +      vmcs_write32(TPR_THRESHOLD, tpr_threshold);
  }
  
  void vmx_set_virtual_apic_mode(struct kvm_vcpu *vcpu)
            !cpu_has_vmx_virtualize_x2apic_mode())
                return;
  
 -      /* Postpone execution until vmcs01 is the current VMCS. */
 -      if (is_guest_mode(vcpu)) {
 -              vmx->nested.change_vmcs01_virtual_apic_mode = true;
 -              return;
 -      }
 +      guard(vmx_vmcs01)(vcpu);
  
        sec_exec_control = secondary_exec_controls_get(vmx);
        sec_exec_control &= ~(SECONDARY_EXEC_VIRTUALIZE_APIC_ACCESSES |
                         * only do so if its physical address has changed, but
                         * the guest may have inserted a non-APIC mapping into
                         * the TLB while the APIC access page was disabled.
 +                       *
 +                       * If L2 is active, immediately flush L1's TLB instead
 +                       * of requesting a flush of the current TLB, because
 +                       * the current TLB context is L2's.
                         */
 -                      kvm_make_request(KVM_REQ_TLB_FLUSH_CURRENT, vcpu);
 +                      if (!is_guest_mode(vcpu))
 +                              kvm_make_request(KVM_REQ_TLB_FLUSH_CURRENT, vcpu);
 +                      else if (!enable_ept)
 +                              vpid_sync_context(vmx->vpid);
 +                      else if (VALID_PAGE(vcpu->arch.root_mmu.root.hpa))
 +                              vmx_flush_tlb_ept_root(vcpu->arch.root_mmu.root.hpa);
                }
                break;
        case LAPIC_MODE_X2APIC:
@@@ -6954,8 -6947,11 +7037,8 @@@ void vmx_set_apic_access_page_addr(stru
        kvm_pfn_t pfn;
        bool writable;
  
 -      /* Defer reload until vmcs01 is the current VMCS. */
 -      if (is_guest_mode(vcpu)) {
 -              to_vmx(vcpu)->nested.reload_vmcs01_apic_access_page = true;
 -              return;
 -      }
 +      /* Note, the VIRTUALIZE_APIC_ACCESSES check needs to query vmcs01. */
 +      guard(vmx_vmcs01)(vcpu);
  
        if (!(secondary_exec_controls_get(to_vmx(vcpu)) &
            SECONDARY_EXEC_VIRTUALIZE_APIC_ACCESSES))
@@@ -7016,16 -7012,30 +7099,16 @@@ void vmx_hwapic_isr_update(struct kvm_v
        u16 status;
        u8 old;
  
 -      /*
 -       * If L2 is active, defer the SVI update until vmcs01 is loaded, as SVI
 -       * is only relevant for if and only if Virtual Interrupt Delivery is
 -       * enabled in vmcs12, and if VID is enabled then L2 EOIs affect L2's
 -       * vAPIC, not L1's vAPIC.  KVM must update vmcs01 on the next nested
 -       * VM-Exit, otherwise L1 with run with a stale SVI.
 -       */
 -      if (is_guest_mode(vcpu)) {
 -              /*
 -               * KVM is supposed to forward intercepted L2 EOIs to L1 if VID
 -               * is enabled in vmcs12; as above, the EOIs affect L2's vAPIC.
 -               * Note, userspace can stuff state while L2 is active; assert
 -               * that VID is disabled if and only if the vCPU is in KVM_RUN
 -               * to avoid false positives if userspace is setting APIC state.
 -               */
 -              WARN_ON_ONCE(vcpu->wants_to_run &&
 -                           nested_cpu_has_vid(get_vmcs12(vcpu)));
 -              to_vmx(vcpu)->nested.update_vmcs01_hwapic_isr = true;
 -              return;
 -      }
 -
        if (max_isr == -1)
                max_isr = 0;
  
 +      /*
 +       * Always update SVI in vmcs01, as SVI is only relevant for L2 if and
 +       * only if Virtual Interrupt Delivery is enabled in vmcs12, and if VID
 +       * is enabled then L2 EOIs affect L2's vAPIC, not L1's vAPIC.
 +       */
 +      guard(vmx_vmcs01)(vcpu);
 +
        status = vmcs_read16(GUEST_INTR_STATUS);
        old = status >> 8;
        if (max_isr != old) {
@@@ -7336,6 -7346,9 +7419,9 @@@ static void atomic_switch_perf_msrs(str
        struct perf_guest_switch_msr *msrs;
        struct kvm_pmu *pmu = vcpu_to_pmu(&vmx->vcpu);
  
+       if (kvm_vcpu_has_mediated_pmu(&vmx->vcpu))
+               return;
        pmu->host_cross_mapped_mask = 0;
        if (pmu->pebs_enable & pmu->global_ctrl)
                intel_pmu_cross_mapped_check(pmu);
                        clear_atomic_switch_msr(vmx, msrs[i].msr);
                else
                        add_atomic_switch_msr(vmx, msrs[i].msr, msrs[i].guest,
-                                       msrs[i].host, false);
+                                             msrs[i].host);
+ }
+ static void vmx_refresh_guest_perf_global_control(struct kvm_vcpu *vcpu)
+ {
+       struct kvm_pmu *pmu = vcpu_to_pmu(vcpu);
+       struct vcpu_vmx *vmx = to_vmx(vcpu);
+       if (msr_write_intercepted(vmx, MSR_CORE_PERF_GLOBAL_CTRL))
+               return;
+       if (!cpu_has_save_perf_global_ctrl()) {
+               int slot = vmx_find_loadstore_msr_slot(&vmx->msr_autostore,
+                                                      MSR_CORE_PERF_GLOBAL_CTRL);
+               if (WARN_ON_ONCE(slot < 0))
+                       return;
+               pmu->global_ctrl = vmx->msr_autostore.val[slot].value;
+               vmcs_write64(GUEST_IA32_PERF_GLOBAL_CTRL, pmu->global_ctrl);
+               return;
+       }
+       pmu->global_ctrl = vmcs_read64(GUEST_IA32_PERF_GLOBAL_CTRL);
  }
  
  static void vmx_update_hv_timer(struct kvm_vcpu *vcpu, bool force_immediate_exit)
@@@ -7638,6 -7674,8 +7747,8 @@@ fastpath_t vmx_vcpu_run(struct kvm_vcp
  
        vmx->loaded_vmcs->launched = 1;
  
+       vmx_refresh_guest_perf_global_control(vcpu);
        vmx_recover_nmi_blocking(vmx);
        vmx_complete_interrupts(vmx);
  
@@@ -8031,7 -8069,8 +8142,8 @@@ static __init u64 vmx_get_perf_capabili
        if (boot_cpu_has(X86_FEATURE_PDCM))
                rdmsrq(MSR_IA32_PERF_CAPABILITIES, host_perf_cap);
  
-       if (!cpu_feature_enabled(X86_FEATURE_ARCH_LBR)) {
+       if (!cpu_feature_enabled(X86_FEATURE_ARCH_LBR) &&
+           !enable_mediated_pmu) {
                x86_perf_get_lbr(&vmx_lbr_caps);
  
                /*
  
  static __init void vmx_set_cpu_caps(void)
  {
 -      kvm_set_cpu_caps();
 +      kvm_initialize_cpu_caps();
  
        /* CPUID 0x1 */
        if (nested)
                kvm_cpu_cap_clear(X86_FEATURE_SHSTK);
                kvm_cpu_cap_clear(X86_FEATURE_IBT);
        }
 +
 +      kvm_setup_xss_caps();
 +      kvm_finalize_cpu_caps();
  }
  
  static bool vmx_is_io_intercepted(struct kvm_vcpu *vcpu,
@@@ -8352,7 -8388,10 +8464,7 @@@ void vmx_update_cpu_dirty_logging(struc
        if (WARN_ON_ONCE(!enable_pml))
                return;
  
 -      if (is_guest_mode(vcpu)) {
 -              vmx->nested.update_vmcs01_cpu_dirty_logging = true;
 -              return;
 -      }
 +      guard(vmx_vmcs01)(vcpu);
  
        /*
         * Note, nr_memslots_dirty_logging can be changed concurrent with this
@@@ -8752,14 -8791,16 +8864,14 @@@ __init int vmx_hardware_setup(void
         * can hide/show features based on kvm_cpu_cap_has().
         */
        if (nested) {
 -              nested_vmx_setup_ctls_msrs(&vmcs_config, vmx_capability.ept);
 -
                r = nested_vmx_hardware_setup(kvm_vmx_exit_handlers);
                if (r)
                        return r;
        }
  
        r = alloc_kvm_area();
 -      if (r && nested)
 -              nested_vmx_hardware_unsetup();
 +      if (r)
 +              goto err_kvm_area;
  
        kvm_set_posted_intr_wakeup_handler(pi_wakeup_handler);
  
  
        kvm_caps.inapplicable_quirks &= ~KVM_X86_QUIRK_IGNORE_GUEST_PAT;
  
 +      return 0;
 +
 +err_kvm_area:
 +      if (nested)
 +              nested_vmx_hardware_unsetup();
        return r;
  }
  
diff --combined arch/x86/kvm/vmx/vmx.h
index a926ce43ad400486f2155353d22c25dad799e8a3,3175fedb5a4d2a474c75edba5c3d2b1bc687481a..70bfe81dea540338849acb3e49f01f177dacd7c1
@@@ -131,6 -131,12 +131,6 @@@ struct nested_vmx 
         */
        bool vmcs02_initialized;
  
 -      bool change_vmcs01_virtual_apic_mode;
 -      bool reload_vmcs01_apic_access_page;
 -      bool update_vmcs01_cpu_dirty_logging;
 -      bool update_vmcs01_apicv_status;
 -      bool update_vmcs01_hwapic_isr;
 -
        /*
         * Enlightened VMCS has been enabled. It does not mean that L1 has to
         * use it. However, VMX features available to L1 will be limited based
        u64 pre_vmenter_ssp;
        u64 pre_vmenter_ssp_tbl;
  
 -      /* to migrate it to L1 if L2 writes to L1's CR8 directly */
 -      int l1_tpr_threshold;
 -
        u16 vpid02;
        u16 last_vpid;
  
+       int tsc_autostore_slot;
        struct nested_vmx_msrs msrs;
  
        /* SMM related state */
@@@ -236,9 -246,7 +237,7 @@@ struct vcpu_vmx 
                struct vmx_msrs host;
        } msr_autoload;
  
-       struct msr_autostore {
-               struct vmx_msrs guest;
-       } msr_autostore;
+       struct vmx_msrs msr_autostore;
  
        struct {
                int vm86_active;
@@@ -376,7 -384,6 +375,6 @@@ void vmx_spec_ctrl_restore_host(struct 
  unsigned int __vmx_vcpu_run_flags(struct vcpu_vmx *vmx);
  bool __vmx_vcpu_run(struct vcpu_vmx *vmx, unsigned long *regs,
                    unsigned int flags);
- int vmx_find_loadstore_msr_slot(struct vmx_msrs *m, u32 msr);
  void vmx_ept_load_pdptrs(struct kvm_vcpu *vcpu);
  
  void vmx_set_intercept_for_msr(struct kvm_vcpu *vcpu, u32 msr, int type, bool set);
@@@ -501,7 -508,8 +499,8 @@@ static inline u8 vmx_get_rvi(void
               VM_EXIT_CLEAR_BNDCFGS |                                  \
               VM_EXIT_PT_CONCEAL_PIP |                                 \
               VM_EXIT_CLEAR_IA32_RTIT_CTL |                            \
-              VM_EXIT_LOAD_CET_STATE)
+              VM_EXIT_LOAD_CET_STATE |                                 \
+              VM_EXIT_SAVE_IA32_PERF_GLOBAL_CTRL)
  
  #define KVM_REQUIRED_VMX_PIN_BASED_VM_EXEC_CONTROL                    \
        (PIN_BASED_EXT_INTR_MASK |                                      \
diff --combined arch/x86/kvm/x86.c
index 06f55aa55172cbe7f5177880d280ce318a64155a,4683df775b0a3759ed41f7b9b9404d6eb47734c0..391f4a5ce6dd18fd6aea52e436627463e9dace78
@@@ -121,10 -121,8 +121,10 @@@ static u64 __read_mostly efer_reserved_
  
  #define KVM_CAP_PMU_VALID_MASK KVM_PMU_CAP_DISABLE
  
 -#define KVM_X2APIC_API_VALID_FLAGS (KVM_X2APIC_API_USE_32BIT_IDS | \
 -                                    KVM_X2APIC_API_DISABLE_BROADCAST_QUIRK)
 +#define KVM_X2APIC_API_VALID_FLAGS (KVM_X2APIC_API_USE_32BIT_IDS              | \
 +                                  KVM_X2APIC_API_DISABLE_BROADCAST_QUIRK      | \
 +                                  KVM_X2APIC_ENABLE_SUPPRESS_EOI_BROADCAST    | \
 +                                  KVM_X2APIC_DISABLE_SUPPRESS_EOI_BROADCAST)
  
  static void update_cr8_intercept(struct kvm_vcpu *vcpu);
  static void process_nmi(struct kvm_vcpu *vcpu);
@@@ -185,6 -183,10 +185,10 @@@ bool __read_mostly enable_pmu = true
  EXPORT_SYMBOL_FOR_KVM_INTERNAL(enable_pmu);
  module_param(enable_pmu, bool, 0444);
  
+ /* Enable/disabled mediated PMU virtualization. */
+ bool __read_mostly enable_mediated_pmu;
+ EXPORT_SYMBOL_FOR_KVM_INTERNAL(enable_mediated_pmu);
  bool __read_mostly eager_page_split = true;
  module_param(eager_page_split, bool, 0644);
  
@@@ -2213,6 -2215,9 +2217,9 @@@ EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_emul
  
  fastpath_t handle_fastpath_invd(struct kvm_vcpu *vcpu)
  {
+       if (!kvm_pmu_is_fastpath_emulation_allowed(vcpu))
+               return EXIT_FASTPATH_NONE;
        if (!kvm_emulate_invd(vcpu))
                return EXIT_FASTPATH_EXIT_USERSPACE;
  
@@@ -2269,6 -2274,9 +2276,9 @@@ static inline bool kvm_vcpu_exit_reques
  
  static fastpath_t __handle_fastpath_wrmsr(struct kvm_vcpu *vcpu, u32 msr, u64 data)
  {
+       if (!kvm_pmu_is_fastpath_emulation_allowed(vcpu))
+               return EXIT_FASTPATH_NONE;
        switch (msr) {
        case APIC_BASE_MSR + (APIC_ICR >> 4):
                if (!lapic_in_kernel(vcpu) || !apic_x2apic_mode(vcpu->arch.apic) ||
@@@ -2316,14 -2324,13 +2326,14 @@@ static int do_set_msr(struct kvm_vcpu *
        u64 val;
  
        /*
 -       * Disallow writes to immutable feature MSRs after KVM_RUN.  KVM does
 -       * not support modifying the guest vCPU model on the fly, e.g. changing
 -       * the nVMX capabilities while L2 is running is nonsensical.  Allow
 -       * writes of the same value, e.g. to allow userspace to blindly stuff
 -       * all MSRs when emulating RESET.
 +       * Reject writes to immutable feature MSRs if the vCPU model is frozen,
 +       * as KVM doesn't support modifying the guest vCPU model on the fly,
 +       * e.g. changing the VMX capabilities MSRs while L2 is active is
 +       * nonsensical.  Allow writes of the same value, e.g. so that userspace
 +       * can blindly stuff all MSRs when emulating RESET.
         */
 -      if (kvm_vcpu_has_run(vcpu) && kvm_is_immutable_feature_msr(index) &&
 +      if (!kvm_can_set_cpuid_and_feature_msrs(vcpu) &&
 +          kvm_is_immutable_feature_msr(index) &&
            (do_get_msr(vcpu, index, &val) || *data != val))
                return -EINVAL;
  
@@@ -3944,6 -3951,7 +3954,7 @@@ int kvm_set_msr_common(struct kvm_vcpu 
  
                vcpu->arch.perf_capabilities = data;
                kvm_pmu_refresh(vcpu);
+               kvm_make_request(KVM_REQ_RECALC_INTERCEPTS, vcpu);
                break;
        case MSR_IA32_PRED_CMD: {
                u64 reserved_bits = ~(PRED_CMD_IBPB | PRED_CMD_SBPB);
                break;
        case MSR_KVM_WALL_CLOCK_NEW:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE2))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                vcpu->kvm->arch.wall_clock = data;
                kvm_write_wall_clock(vcpu->kvm, data, 0);
                break;
        case MSR_KVM_WALL_CLOCK:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                vcpu->kvm->arch.wall_clock = data;
                kvm_write_wall_clock(vcpu->kvm, data, 0);
                break;
        case MSR_KVM_SYSTEM_TIME_NEW:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE2))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                kvm_write_system_time(vcpu, data, false, msr_info->host_initiated);
                break;
        case MSR_KVM_SYSTEM_TIME:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                kvm_write_system_time(vcpu, data, true,  msr_info->host_initiated);
                break;
        case MSR_KVM_ASYNC_PF_EN:
                if (!guest_pv_has(vcpu, KVM_FEATURE_ASYNC_PF))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                if (kvm_pv_enable_async_pf(vcpu, data))
                        return 1;
                break;
        case MSR_KVM_ASYNC_PF_INT:
                if (!guest_pv_has(vcpu, KVM_FEATURE_ASYNC_PF_INT))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                if (kvm_pv_enable_async_pf_int(vcpu, data))
                        return 1;
                break;
        case MSR_KVM_ASYNC_PF_ACK:
                if (!guest_pv_has(vcpu, KVM_FEATURE_ASYNC_PF_INT))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
                if (data & 0x1) {
                        /*
                         * Pairs with the smp_mb__after_atomic() in
                break;
        case MSR_KVM_STEAL_TIME:
                if (!guest_pv_has(vcpu, KVM_FEATURE_STEAL_TIME))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                if (unlikely(!sched_info_on()))
                        return 1;
                break;
        case MSR_KVM_PV_EOI_EN:
                if (!guest_pv_has(vcpu, KVM_FEATURE_PV_EOI))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                if (kvm_lapic_set_pv_eoi(vcpu, data, sizeof(u8)))
                        return 1;
  
        case MSR_KVM_POLL_CONTROL:
                if (!guest_pv_has(vcpu, KVM_FEATURE_POLL_CONTROL))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                /* only enable bit supported */
                if (data & (-1ULL << 1))
@@@ -4479,61 -4487,61 +4490,61 @@@ int kvm_get_msr_common(struct kvm_vcpu 
                break;
        case MSR_KVM_WALL_CLOCK:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->kvm->arch.wall_clock;
                break;
        case MSR_KVM_WALL_CLOCK_NEW:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE2))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->kvm->arch.wall_clock;
                break;
        case MSR_KVM_SYSTEM_TIME:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.time;
                break;
        case MSR_KVM_SYSTEM_TIME_NEW:
                if (!guest_pv_has(vcpu, KVM_FEATURE_CLOCKSOURCE2))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.time;
                break;
        case MSR_KVM_ASYNC_PF_EN:
                if (!guest_pv_has(vcpu, KVM_FEATURE_ASYNC_PF))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.apf.msr_en_val;
                break;
        case MSR_KVM_ASYNC_PF_INT:
                if (!guest_pv_has(vcpu, KVM_FEATURE_ASYNC_PF_INT))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.apf.msr_int_val;
                break;
        case MSR_KVM_ASYNC_PF_ACK:
                if (!guest_pv_has(vcpu, KVM_FEATURE_ASYNC_PF_INT))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = 0;
                break;
        case MSR_KVM_STEAL_TIME:
                if (!guest_pv_has(vcpu, KVM_FEATURE_STEAL_TIME))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.st.msr_val;
                break;
        case MSR_KVM_PV_EOI_EN:
                if (!guest_pv_has(vcpu, KVM_FEATURE_PV_EOI))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.pv_eoi.msr_val;
                break;
        case MSR_KVM_POLL_CONTROL:
                if (!guest_pv_has(vcpu, KVM_FEATURE_POLL_CONTROL))
 -                      return 1;
 +                      return KVM_MSR_RET_UNSUPPORTED;
  
                msr_info->data = vcpu->arch.msr_kvm_poll_control;
                break;
@@@ -4934,8 -4942,6 +4945,8 @@@ int kvm_vm_ioctl_check_extension(struc
                break;
        case KVM_CAP_X2APIC_API:
                r = KVM_X2APIC_API_VALID_FLAGS;
 +              if (kvm && !irqchip_split(kvm))
 +                      r &= ~KVM_X2APIC_ENABLE_SUPPRESS_EOI_BROADCAST;
                break;
        case KVM_CAP_NESTED_STATE:
                r = kvm_x86_ops.nested_ops->get_state ?
@@@ -5812,18 -5818,9 +5823,18 @@@ static int kvm_vcpu_ioctl_x86_get_xsave
  static int kvm_vcpu_ioctl_x86_set_xsave(struct kvm_vcpu *vcpu,
                                        struct kvm_xsave *guest_xsave)
  {
 +      union fpregs_state *xstate = (union fpregs_state *)guest_xsave->region;
 +
        if (fpstate_is_confidential(&vcpu->arch.guest_fpu))
                return vcpu->kvm->arch.has_protected_state ? -EINVAL : 0;
  
 +      /*
 +       * For backwards compatibility, do not expect disabled features to be in
 +       * their initial state.  XSTATE_BV[i] must still be cleared whenever
 +       * XFD[i]=1, or XRSTOR would cause a #NM.
 +       */
 +      xstate->xsave.header.xfeatures &= ~vcpu->arch.guest_fpu.fpstate->xfd;
 +
        return fpu_copy_uabi_to_guest_fpstate(&vcpu->arch.guest_fpu,
                                              guest_xsave->region,
                                              kvm_caps.supported_xcr0,
@@@ -6731,7 -6728,7 +6742,7 @@@ int kvm_vm_ioctl_enable_cap(struct kvm 
        case KVM_CAP_SPLIT_IRQCHIP: {
                mutex_lock(&kvm->lock);
                r = -EINVAL;
 -              if (cap->args[0] > MAX_NR_RESERVED_IOAPIC_PINS)
 +              if (cap->args[0] > KVM_MAX_IRQ_ROUTES)
                        goto split_irqchip_unlock;
                r = -EEXIST;
                if (irqchip_in_kernel(kvm))
@@@ -6753,24 -6750,11 +6764,24 @@@ split_irqchip_unlock
                if (cap->args[0] & ~KVM_X2APIC_API_VALID_FLAGS)
                        break;
  
 +              if ((cap->args[0] & KVM_X2APIC_ENABLE_SUPPRESS_EOI_BROADCAST) &&
 +                  (cap->args[0] & KVM_X2APIC_DISABLE_SUPPRESS_EOI_BROADCAST))
 +                      break;
 +
 +              if ((cap->args[0] & KVM_X2APIC_ENABLE_SUPPRESS_EOI_BROADCAST) &&
 +                  !irqchip_split(kvm))
 +                      break;
 +
                if (cap->args[0] & KVM_X2APIC_API_USE_32BIT_IDS)
                        kvm->arch.x2apic_format = true;
                if (cap->args[0] & KVM_X2APIC_API_DISABLE_BROADCAST_QUIRK)
                        kvm->arch.x2apic_broadcast_quirk_disabled = true;
  
 +              if (cap->args[0] & KVM_X2APIC_ENABLE_SUPPRESS_EOI_BROADCAST)
 +                      kvm->arch.suppress_eoi_broadcast_mode = KVM_SUPPRESS_EOI_BROADCAST_ENABLED;
 +              if (cap->args[0] & KVM_X2APIC_DISABLE_SUPPRESS_EOI_BROADCAST)
 +                      kvm->arch.suppress_eoi_broadcast_mode = KVM_SUPPRESS_EOI_BROADCAST_DISABLED;
 +
                r = 0;
                break;
        case KVM_CAP_X86_DISABLE_EXITS:
@@@ -6881,7 -6865,7 +6892,7 @@@ disable_exits_unlock
                        break;
  
                mutex_lock(&kvm->lock);
-               if (!kvm->created_vcpus) {
+               if (!kvm->created_vcpus && !kvm->arch.created_mediated_pmu) {
                        kvm->arch.enable_pmu = !(cap->args[0] & KVM_PMU_CAP_DISABLE);
                        r = 0;
                }
@@@ -9971,23 -9955,6 +9982,23 @@@ static struct notifier_block pvclock_gt
  };
  #endif
  
 +void kvm_setup_xss_caps(void)
 +{
 +      if (!kvm_cpu_cap_has(X86_FEATURE_XSAVES))
 +              kvm_caps.supported_xss = 0;
 +
 +      if (!kvm_cpu_cap_has(X86_FEATURE_SHSTK) &&
 +          !kvm_cpu_cap_has(X86_FEATURE_IBT))
 +              kvm_caps.supported_xss &= ~XFEATURE_MASK_CET_ALL;
 +
 +      if ((kvm_caps.supported_xss & XFEATURE_MASK_CET_ALL) != XFEATURE_MASK_CET_ALL) {
 +              kvm_cpu_cap_clear(X86_FEATURE_SHSTK);
 +              kvm_cpu_cap_clear(X86_FEATURE_IBT);
 +              kvm_caps.supported_xss &= ~XFEATURE_MASK_CET_ALL;
 +      }
 +}
 +EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_setup_xss_caps);
 +
  static inline void kvm_ops_update(struct kvm_x86_init_ops *ops)
  {
        memcpy(&kvm_x86_ops, ops->runtime_ops, sizeof(kvm_x86_ops));
@@@ -10151,7 -10118,8 +10162,8 @@@ int kvm_x86_vendor_init(struct kvm_x86_
                set_hv_tscchange_cb(kvm_hyperv_tsc_notifier);
  #endif
  
-       kvm_register_perf_callbacks(ops->handle_intel_pt_intr);
+       __kvm_register_perf_callbacks(ops->handle_intel_pt_intr,
+                                     enable_mediated_pmu ? kvm_handle_guest_mediated_pmi : NULL);
  
        if (IS_ENABLED(CONFIG_KVM_SW_PROTECTED_VM) && tdp_mmu_enabled)
                kvm_caps.supported_vm_types |= BIT(KVM_X86_SW_PROTECTED_VM);
        if (!tdp_enabled)
                kvm_caps.supported_quirks &= ~KVM_X86_QUIRK_IGNORE_GUEST_PAT;
  
 -      if (!kvm_cpu_cap_has(X86_FEATURE_XSAVES))
 -              kvm_caps.supported_xss = 0;
 -
 -      if (!kvm_cpu_cap_has(X86_FEATURE_SHSTK) &&
 -          !kvm_cpu_cap_has(X86_FEATURE_IBT))
 -              kvm_caps.supported_xss &= ~XFEATURE_MASK_CET_ALL;
 -
 -      if ((kvm_caps.supported_xss & XFEATURE_MASK_CET_ALL) != XFEATURE_MASK_CET_ALL) {
 -              kvm_cpu_cap_clear(X86_FEATURE_SHSTK);
 -              kvm_cpu_cap_clear(X86_FEATURE_IBT);
 -              kvm_caps.supported_xss &= ~XFEATURE_MASK_CET_ALL;
 -      }
 -
        if (kvm_caps.has_tsc_control) {
                /*
                 * Make sure the user can only configure tsc_khz values that
@@@ -10276,7 -10257,7 +10288,7 @@@ static void kvm_pv_kick_cpu_op(struct k
                .dest_id = apicid,
        };
  
 -      kvm_irq_delivery_to_apic(kvm, NULL, &lapic_irq, NULL);
 +      kvm_irq_delivery_to_apic(kvm, NULL, &lapic_irq);
  }
  
  bool kvm_apicv_activated(struct kvm *kvm)
@@@ -11359,6 -11340,8 +11371,8 @@@ static int vcpu_enter_guest(struct kvm_
                run_flags |= KVM_RUN_LOAD_DEBUGCTL;
        vcpu->arch.host_debugctl = debug_ctl;
  
+       kvm_mediated_pmu_load(vcpu);
        guest_timing_enter_irqoff();
  
        /*
  
        kvm_load_host_pkru(vcpu);
  
+       kvm_mediated_pmu_put(vcpu);
        /*
         * Do this here before restoring debug registers on the host.  And
         * since we do this before handling the vmexit, a DR access vmexit
@@@ -11620,7 -11605,8 +11636,7 @@@ static inline int vcpu_block(struct kvm
        if (is_guest_mode(vcpu)) {
                int r = kvm_check_nested_events(vcpu);
  
 -              WARN_ON_ONCE(r == -EBUSY);
 -              if (r < 0)
 +              if (r < 0 && r != -EBUSY)
                        return 0;
        }
  
@@@ -11734,6 -11720,9 +11750,9 @@@ EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_emul
  
  fastpath_t handle_fastpath_hlt(struct kvm_vcpu *vcpu)
  {
+       if (!kvm_pmu_is_fastpath_emulation_allowed(vcpu))
+               return EXIT_FASTPATH_NONE;
        if (!kvm_emulate_halt(vcpu))
                return EXIT_FASTPATH_EXIT_USERSPACE;
  
@@@ -12168,11 -12157,9 +12187,11 @@@ static void __get_sregs2(struct kvm_vcp
                return;
  
        if (is_pae_paging(vcpu)) {
 +              kvm_vcpu_srcu_read_lock(vcpu);
                for (i = 0 ; i < 4 ; i++)
                        sregs2->pdptrs[i] = kvm_pdptr_read(vcpu, i);
                sregs2->flags |= KVM_SREGS2_FLAGS_PDPTRS_VALID;
 +              kvm_vcpu_srcu_read_unlock(vcpu);
        }
  }
  
@@@ -12673,8 -12660,13 +12692,13 @@@ static int sync_regs(struct kvm_vcpu *v
        return 0;
  }
  
+ #define PERF_MEDIATED_PMU_MSG \
+       "Failed to enable mediated vPMU, try disabling system wide perf events and nmi_watchdog.\n"
  int kvm_arch_vcpu_precreate(struct kvm *kvm, unsigned int id)
  {
+       int r;
        if (kvm_check_tsc_unstable() && kvm->created_vcpus)
                pr_warn_once("SMP vm created on host with unstable TSC; "
                             "guest TSC will not be reliable\n");
        if (id >= kvm->arch.max_vcpu_ids)
                return -EINVAL;
  
-       return kvm_x86_call(vcpu_precreate)(kvm);
+       /*
+        * Note, any actions done by .vcpu_create() must be idempotent with
+        * respect to creating multiple vCPUs, and therefore are not undone if
+        * creating a vCPU fails (including failure during pre-create).
+        */
+       r = kvm_x86_call(vcpu_precreate)(kvm);
+       if (r)
+               return r;
+       if (enable_mediated_pmu && kvm->arch.enable_pmu &&
+           !kvm->arch.created_mediated_pmu) {
+               if (irqchip_in_kernel(kvm)) {
+                       r = perf_create_mediated_pmu();
+                       if (r) {
+                               pr_warn_ratelimited(PERF_MEDIATED_PMU_MSG);
+                               return r;
+                       }
+                       kvm->arch.created_mediated_pmu = true;
+               } else {
+                       kvm->arch.enable_pmu = false;
+               }
+       }
+       return 0;
  }
  
  int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu)
@@@ -13332,7 -13346,7 +13378,7 @@@ void kvm_arch_pre_destroy_vm(struct kv
  #endif
  
        kvm_mmu_pre_destroy_vm(kvm);
 -      static_call_cond(kvm_x86_vm_pre_destroy)(kvm);
 +      kvm_x86_call(vm_pre_destroy)(kvm);
  }
  
  void kvm_arch_destroy_vm(struct kvm *kvm)
                __x86_set_memory_region(kvm, TSS_PRIVATE_MEMSLOT, 0, 0);
                mutex_unlock(&kvm->slots_lock);
        }
+       if (kvm->arch.created_mediated_pmu)
+               perf_release_mediated_pmu();
        kvm_destroy_vcpus(kvm);
        kvm_free_msr_filter(srcu_dereference_check(kvm->arch.msr_filter, &kvm->srcu, 1));
  #ifdef CONFIG_KVM_IOAPIC
@@@ -14155,13 -14171,6 +14203,13 @@@ int kvm_handle_invpcid(struct kvm_vcpu 
                        return 1;
                }
  
 +              /*
 +               * When ERAPS is supported, invalidating a specific PCID clears
 +               * the RAP (Return Address Predicator).
 +               */
 +              if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS))
 +                      kvm_register_is_dirty(vcpu, VCPU_EXREG_ERAPS);
 +
                kvm_invalidate_pcid(vcpu, operand.pcid);
                return kvm_skip_emulated_instruction(vcpu);
  
  
                fallthrough;
        case INVPCID_TYPE_ALL_INCL_GLOBAL:
 +              /*
 +               * Don't bother marking VCPU_EXREG_ERAPS dirty, SVM will take
 +               * care of doing so when emulating the full guest TLB flush
 +               * (the RAP is cleared on all implicit TLB flushes).
 +               */
                kvm_make_request(KVM_REQ_TLB_FLUSH_GUEST, vcpu);
                return kvm_skip_emulated_instruction(vcpu);
  
diff --combined arch/x86/kvm/x86.h
index ff20e62d98c63de6eab61b6805937f5fb02967ab,6e1fb1680c0a3df6176ca3e2d7318c99efd7fb3d..94d4f07aaaa09e9ac6c401ca18ee9cf504f69d07
@@@ -172,20 -172,9 +172,20 @@@ static inline void kvm_nested_vmexit_ha
                indirect_branch_prediction_barrier();
  }
  
 -static inline bool kvm_vcpu_has_run(struct kvm_vcpu *vcpu)
 +/*
 + * Disallow modifying CPUID and feature MSRs, which affect the core virtual CPU
 + * model exposed to the guest and virtualized by KVM, if the vCPU has already
 + * run or is in guest mode (L2).  In both cases, KVM has already consumed the
 + * current virtual CPU model, and doesn't support "unwinding" to react to the
 + * new model.
 + *
 + * Note, the only way is_guest_mode() can be true with 'last_vmentry_cpu == -1'
 + * is if userspace sets CPUID and feature MSRs (to enable VMX/SVM), then sets
 + * nested state, and then attempts to set CPUID and/or feature MSRs *again*.
 + */
 +static inline bool kvm_can_set_cpuid_and_feature_msrs(struct kvm_vcpu *vcpu)
  {
 -      return vcpu->arch.last_vmentry_cpu != -1;
 +      return vcpu->arch.last_vmentry_cpu == -1 && !is_guest_mode(vcpu);
  }
  
  static inline void kvm_set_mp_state(struct kvm_vcpu *vcpu, int mp_state)
@@@ -481,9 -470,8 +481,10 @@@ extern struct kvm_caps kvm_caps
  extern struct kvm_host_values kvm_host;
  
  extern bool enable_pmu;
+ extern bool enable_mediated_pmu;
  
 +void kvm_setup_xss_caps(void);
 +
  /*
   * Get a filtered version of KVM's supported XCR0 that strips out dynamic
   * features for which the current process doesn't (yet) have permission to use.
diff --combined include/linux/kvm_host.h
index 021d1fa09e924a5eed7943024da36a0ce002c9e7,8e410d1a63dfc2ac6cd8ba30042ceb2a6956ff5f..c05a79c21745014c46ad2a17e34f73752b426830
@@@ -1381,7 -1381,6 +1381,7 @@@ bool kvm_vcpu_is_visible_gfn(struct kvm
  unsigned long kvm_host_page_size(struct kvm_vcpu *vcpu, gfn_t gfn);
  void mark_page_dirty_in_slot(struct kvm *kvm, const struct kvm_memory_slot *memslot, gfn_t gfn);
  void mark_page_dirty(struct kvm *kvm, gfn_t gfn);
 +void kvm_vcpu_mark_page_dirty(struct kvm_vcpu *vcpu, gfn_t gfn);
  
  int __kvm_vcpu_map(struct kvm_vcpu *vcpu, gpa_t gpa, struct kvm_host_map *map,
                   bool writable);
@@@ -1399,13 -1398,6 +1399,13 @@@ static inline int kvm_vcpu_map_readonly
        return __kvm_vcpu_map(vcpu, gpa, map, false);
  }
  
 +static inline void kvm_vcpu_map_mark_dirty(struct kvm_vcpu *vcpu,
 +                                         struct kvm_host_map *map)
 +{
 +      if (kvm_vcpu_mapped(map))
 +              kvm_vcpu_mark_page_dirty(vcpu, map->gfn);
 +}
 +
  unsigned long kvm_vcpu_gfn_to_hva(struct kvm_vcpu *vcpu, gfn_t gfn);
  unsigned long kvm_vcpu_gfn_to_hva_prot(struct kvm_vcpu *vcpu, gfn_t gfn, bool *writable);
  int kvm_vcpu_read_guest_page(struct kvm_vcpu *vcpu, gfn_t gfn, void *data, int offset,
@@@ -1418,6 -1410,7 +1418,6 @@@ int kvm_vcpu_write_guest_page(struct kv
                              int offset, int len);
  int kvm_vcpu_write_guest(struct kvm_vcpu *vcpu, gpa_t gpa, const void *data,
                         unsigned long len);
 -void kvm_vcpu_mark_page_dirty(struct kvm_vcpu *vcpu, gfn_t gfn);
  
  /**
   * kvm_gpc_init - initialize gfn_to_pfn_cache.
@@@ -1756,10 -1749,17 +1756,17 @@@ static inline bool kvm_arch_intc_initia
  #ifdef CONFIG_GUEST_PERF_EVENTS
  unsigned long kvm_arch_vcpu_get_ip(struct kvm_vcpu *vcpu);
  
- void kvm_register_perf_callbacks(unsigned int (*pt_intr_handler)(void));
+ void __kvm_register_perf_callbacks(unsigned int (*pt_intr_handler)(void),
+                                  void (*mediated_pmi_handler)(void));
+ static inline void kvm_register_perf_callbacks(void)
+ {
+       __kvm_register_perf_callbacks(NULL, NULL);
+ }
  void kvm_unregister_perf_callbacks(void);
  #else
- static inline void kvm_register_perf_callbacks(void *ign) {}
+ static inline void kvm_register_perf_callbacks(void) {}
  static inline void kvm_unregister_perf_callbacks(void) {}
  #endif /* CONFIG_GUEST_PERF_EVENTS */
  
@@@ -2573,7 -2573,7 +2580,7 @@@ int kvm_arch_gmem_prepare(struct kvm *k
   * @gfn: starting GFN to be populated
   * @src: userspace-provided buffer containing data to copy into GFN range
   *       (passed to @post_populate, and incremented on each iteration
 - *       if not NULL)
 + *       if not NULL). Must be page-aligned.
   * @npages: number of pages to copy from userspace-buffer
   * @post_populate: callback to issue for each gmem page that backs the GPA
   *                 range
   * Returns the number of pages that were populated.
   */
  typedef int (*kvm_gmem_populate_cb)(struct kvm *kvm, gfn_t gfn, kvm_pfn_t pfn,
 -                                  void __user *src, int order, void *opaque);
 +                                  struct page *page, void *opaque);
  
  long kvm_gmem_populate(struct kvm *kvm, gfn_t gfn, void __user *src, long npages,
                       kvm_gmem_populate_cb post_populate, void *opaque);
index 9ded2e582c602121a9cfd2d45153390d30f2330e,82e617fad165565b810b9d134f59759282fd569e..48d851fbd8ea583c4737672005e24e3ee31d70b6
@@@ -1,7 -1,7 +1,7 @@@
  /*
   * Performance events:
   *
 - *    Copyright (C) 2008-2009, Thomas Gleixner <tglx@linutronix.de>
 + *    Copyright (C) 2008-2009, Linutronix GmbH, Thomas Gleixner <tglx@kernel.org>
   *    Copyright (C) 2008-2011, Red Hat, Inc., Ingo Molnar
   *    Copyright (C) 2008-2011, Red Hat, Inc., Peter Zijlstra
   *
@@@ -305,6 -305,7 +305,7 @@@ struct perf_event_pmu_context
  #define PERF_PMU_CAP_EXTENDED_HW_TYPE 0x0100
  #define PERF_PMU_CAP_AUX_PAUSE                0x0200
  #define PERF_PMU_CAP_AUX_PREFER_LARGE 0x0400
+ #define PERF_PMU_CAP_MEDIATED_VPMU    0x0800
  
  /**
   * pmu::scope
@@@ -998,6 -999,11 +999,11 @@@ struct perf_event_groups 
        u64                             index;
  };
  
+ struct perf_time_ctx {
+       u64             time;
+       u64             stamp;
+       u64             offset;
+ };
  
  /**
   * struct perf_event_context - event context structure
@@@ -1036,9 -1042,12 +1042,12 @@@ struct perf_event_context 
        /*
         * Context clock, runs when context enabled.
         */
-       u64                             time;
-       u64                             timestamp;
-       u64                             timeoffset;
+       struct perf_time_ctx            time;
+       /*
+        * Context clock, runs when in the guest mode.
+        */
+       struct perf_time_ctx            timeguest;
  
        /*
         * These fields let us detect when two contexts have both
@@@ -1171,9 -1180,8 +1180,8 @@@ struct bpf_perf_event_data_kern 
   * This is a per-cpu dynamically allocated data structure.
   */
  struct perf_cgroup_info {
-       u64                             time;
-       u64                             timestamp;
-       u64                             timeoffset;
+       struct perf_time_ctx            time;
+       struct perf_time_ctx            timeguest;
        int                             active;
  };
  
@@@ -1669,6 -1677,8 +1677,8 @@@ struct perf_guest_info_callbacks 
        unsigned int                    (*state)(void);
        unsigned long                   (*get_ip)(void);
        unsigned int                    (*handle_intel_pt_intr)(void);
+       void                            (*handle_mediated_pmi)(void);
  };
  
  #ifdef CONFIG_GUEST_PERF_EVENTS
@@@ -1678,6 -1688,7 +1688,7 @@@ extern struct perf_guest_info_callback
  DECLARE_STATIC_CALL(__perf_guest_state, *perf_guest_cbs->state);
  DECLARE_STATIC_CALL(__perf_guest_get_ip, *perf_guest_cbs->get_ip);
  DECLARE_STATIC_CALL(__perf_guest_handle_intel_pt_intr, *perf_guest_cbs->handle_intel_pt_intr);
+ DECLARE_STATIC_CALL(__perf_guest_handle_mediated_pmi, *perf_guest_cbs->handle_mediated_pmi);
  
  static inline unsigned int perf_guest_state(void)
  {
@@@ -1694,6 -1705,11 +1705,11 @@@ static inline unsigned int perf_guest_h
        return static_call(__perf_guest_handle_intel_pt_intr)();
  }
  
+ static inline void perf_guest_handle_mediated_pmi(void)
+ {
+       static_call(__perf_guest_handle_mediated_pmi)();
+ }
  extern void perf_register_guest_info_callbacks(struct perf_guest_info_callbacks *cbs);
  extern void perf_unregister_guest_info_callbacks(struct perf_guest_info_callbacks *cbs);
  
@@@ -1914,6 -1930,13 +1930,13 @@@ extern int perf_event_account_interrupt
  extern int perf_event_period(struct perf_event *event, u64 value);
  extern u64 perf_event_pause(struct perf_event *event, bool reset);
  
+ #ifdef CONFIG_PERF_GUEST_MEDIATED_PMU
+ int perf_create_mediated_pmu(void);
+ void perf_release_mediated_pmu(void);
+ void perf_load_guest_context(void);
+ void perf_put_guest_context(void);
+ #endif
  #else /* !CONFIG_PERF_EVENTS: */
  
  static inline void *
diff --combined kernel/events/core.c
index 8cca80094624815eb1dc830a6730b5d9555cdcf0,376fb07d869b8b50f811325f58c6555bb7dd71c7..e320e06c8af6a17166853cc92626ddbdc044468b
@@@ -2,7 -2,7 +2,7 @@@
  /*
   * Performance events core code:
   *
 - *  Copyright (C) 2008 Thomas Gleixner <tglx@linutronix.de>
 + *  Copyright (C) 2008 Linutronix GmbH, Thomas Gleixner <tglx@kernel.org>
   *  Copyright (C) 2008-2011 Red Hat, Inc., Ingo Molnar
   *  Copyright (C) 2008-2011 Red Hat, Inc., Peter Zijlstra
   *  Copyright  Â©  2009 Paul Mackerras, IBM Corp. <paulus@au1.ibm.com>
@@@ -57,6 -57,7 +57,7 @@@
  #include <linux/task_work.h>
  #include <linux/percpu-rwsem.h>
  #include <linux/unwind_deferred.h>
+ #include <linux/kvm_types.h>
  
  #include "internal.h"
  
@@@ -166,6 -167,18 +167,18 @@@ enum event_type_t 
        EVENT_CPU       = 0x10,
        EVENT_CGROUP    = 0x20,
  
+       /*
+        * EVENT_GUEST is set when scheduling in/out events between the host
+        * and a guest with a mediated vPMU.  Among other things, EVENT_GUEST
+        * is used:
+        *
+        * - In for_each_epc() to skip PMUs that don't support events in a
+        *   MEDIATED_VPMU guest, i.e. don't need to be context switched.
+        * - To indicate the start/end point of the events in a guest.  Guest
+        *   running time is deducted for host-only (exclude_guest) events.
+        */
+       EVENT_GUEST     = 0x40,
+       EVENT_FLAGS     = EVENT_CGROUP | EVENT_GUEST,
        /* compound helpers */
        EVENT_ALL         = EVENT_FLEXIBLE | EVENT_PINNED,
        EVENT_TIME_FROZEN = EVENT_TIME | EVENT_FROZEN,
@@@ -458,6 -471,20 +471,20 @@@ static cpumask_var_t perf_online_pkg_ma
  static cpumask_var_t perf_online_sys_mask;
  static struct kmem_cache *perf_event_cache;
  
+ #ifdef CONFIG_PERF_GUEST_MEDIATED_PMU
+ static DEFINE_PER_CPU(bool, guest_ctx_loaded);
+ static __always_inline bool is_guest_mediated_pmu_loaded(void)
+ {
+       return __this_cpu_read(guest_ctx_loaded);
+ }
+ #else
+ static __always_inline bool is_guest_mediated_pmu_loaded(void)
+ {
+       return false;
+ }
+ #endif
  /*
   * perf event paranoia level:
   *  -1 - not paranoid at all
@@@ -779,33 -806,97 +806,97 @@@ do {                                                                    
        ___p;                                                           \
  })
  
- #define for_each_epc(_epc, _ctx, _pmu, _cgroup)                               \
+ static bool perf_skip_pmu_ctx(struct perf_event_pmu_context *pmu_ctx,
+                             enum event_type_t event_type)
+ {
+       if ((event_type & EVENT_CGROUP) && !pmu_ctx->nr_cgroups)
+               return true;
+       if ((event_type & EVENT_GUEST) &&
+           !(pmu_ctx->pmu->capabilities & PERF_PMU_CAP_MEDIATED_VPMU))
+               return true;
+       return false;
+ }
+ #define for_each_epc(_epc, _ctx, _pmu, _event_type)                   \
        list_for_each_entry(_epc, &((_ctx)->pmu_ctx_list), pmu_ctx_entry) \
-               if (_cgroup && !_epc->nr_cgroups)                       \
+               if (perf_skip_pmu_ctx(_epc, _event_type))               \
                        continue;                                       \
                else if (_pmu && _epc->pmu != _pmu)                     \
                        continue;                                       \
                else
  
- static void perf_ctx_disable(struct perf_event_context *ctx, bool cgroup)
+ static void perf_ctx_disable(struct perf_event_context *ctx,
+                            enum event_type_t event_type)
  {
        struct perf_event_pmu_context *pmu_ctx;
  
-       for_each_epc(pmu_ctx, ctx, NULL, cgroup)
+       for_each_epc(pmu_ctx, ctx, NULL, event_type)
                perf_pmu_disable(pmu_ctx->pmu);
  }
  
- static void perf_ctx_enable(struct perf_event_context *ctx, bool cgroup)
+ static void perf_ctx_enable(struct perf_event_context *ctx,
+                           enum event_type_t event_type)
  {
        struct perf_event_pmu_context *pmu_ctx;
  
-       for_each_epc(pmu_ctx, ctx, NULL, cgroup)
+       for_each_epc(pmu_ctx, ctx, NULL, event_type)
                perf_pmu_enable(pmu_ctx->pmu);
  }
  
  static void ctx_sched_out(struct perf_event_context *ctx, struct pmu *pmu, enum event_type_t event_type);
  static void ctx_sched_in(struct perf_event_context *ctx, struct pmu *pmu, enum event_type_t event_type);
  
+ static inline void update_perf_time_ctx(struct perf_time_ctx *time, u64 now, bool adv)
+ {
+       if (adv)
+               time->time += now - time->stamp;
+       time->stamp = now;
+       /*
+        * The above: time' = time + (now - timestamp), can be re-arranged
+        * into: time` = now + (time - timestamp), which gives a single value
+        * offset to compute future time without locks on.
+        *
+        * See perf_event_time_now(), which can be used from NMI context where
+        * it's (obviously) not possible to acquire ctx->lock in order to read
+        * both the above values in a consistent manner.
+        */
+       WRITE_ONCE(time->offset, time->time - time->stamp);
+ }
+ static_assert(offsetof(struct perf_event_context, timeguest) -
+             offsetof(struct perf_event_context, time) ==
+             sizeof(struct perf_time_ctx));
+ #define T_TOTAL               0
+ #define T_GUEST               1
+ static inline u64 __perf_event_time_ctx(struct perf_event *event,
+                                       struct perf_time_ctx *times)
+ {
+       u64 time = times[T_TOTAL].time;
+       if (event->attr.exclude_guest)
+               time -= times[T_GUEST].time;
+       return time;
+ }
+ static inline u64 __perf_event_time_ctx_now(struct perf_event *event,
+                                           struct perf_time_ctx *times,
+                                           u64 now)
+ {
+       if (is_guest_mediated_pmu_loaded() && event->attr.exclude_guest) {
+               /*
+                * (now + times[total].offset) - (now + times[guest].offset) :=
+                * times[total].offset - times[guest].offset
+                */
+               return READ_ONCE(times[T_TOTAL].offset) - READ_ONCE(times[T_GUEST].offset);
+       }
+       return now + READ_ONCE(times[T_TOTAL].offset);
+ }
  #ifdef CONFIG_CGROUP_PERF
  
  static inline bool
@@@ -842,12 -933,16 +933,16 @@@ static inline int is_cgroup_event(struc
        return event->cgrp != NULL;
  }
  
+ static_assert(offsetof(struct perf_cgroup_info, timeguest) -
+             offsetof(struct perf_cgroup_info, time) ==
+             sizeof(struct perf_time_ctx));
  static inline u64 perf_cgroup_event_time(struct perf_event *event)
  {
        struct perf_cgroup_info *t;
  
        t = per_cpu_ptr(event->cgrp->info, event->cpu);
-       return t->time;
+       return __perf_event_time_ctx(event, &t->time);
  }
  
  static inline u64 perf_cgroup_event_time_now(struct perf_event *event, u64 now)
  
        t = per_cpu_ptr(event->cgrp->info, event->cpu);
        if (!__load_acquire(&t->active))
-               return t->time;
-       now += READ_ONCE(t->timeoffset);
-       return now;
+               return __perf_event_time_ctx(event, &t->time);
+       return __perf_event_time_ctx_now(event, &t->time, now);
  }
  
- static inline void __update_cgrp_time(struct perf_cgroup_info *info, u64 now, bool adv)
+ static inline void __update_cgrp_guest_time(struct perf_cgroup_info *info, u64 now, bool adv)
  {
-       if (adv)
-               info->time += now - info->timestamp;
-       info->timestamp = now;
-       /*
-        * see update_context_time()
-        */
-       WRITE_ONCE(info->timeoffset, info->time - info->timestamp);
+       update_perf_time_ctx(&info->timeguest, now, adv);
+ }
+ static inline void update_cgrp_time(struct perf_cgroup_info *info, u64 now)
+ {
+       update_perf_time_ctx(&info->time, now, true);
+       if (is_guest_mediated_pmu_loaded())
+               __update_cgrp_guest_time(info, now, true);
  }
  
  static inline void update_cgrp_time_from_cpuctx(struct perf_cpu_context *cpuctx, bool final)
                        cgrp = container_of(css, struct perf_cgroup, css);
                        info = this_cpu_ptr(cgrp->info);
  
-                       __update_cgrp_time(info, now, true);
+                       update_cgrp_time(info, now);
                        if (final)
                                __store_release(&info->active, 0);
                }
@@@ -908,11 -1004,11 +1004,11 @@@ static inline void update_cgrp_time_fro
         * Do not update time when cgroup is not active
         */
        if (info->active)
-               __update_cgrp_time(info, perf_clock(), true);
+               update_cgrp_time(info, perf_clock());
  }
  
  static inline void
- perf_cgroup_set_timestamp(struct perf_cpu_context *cpuctx)
+ perf_cgroup_set_timestamp(struct perf_cpu_context *cpuctx, bool guest)
  {
        struct perf_event_context *ctx = &cpuctx->ctx;
        struct perf_cgroup *cgrp = cpuctx->cgrp;
        for (css = &cgrp->css; css; css = css->parent) {
                cgrp = container_of(css, struct perf_cgroup, css);
                info = this_cpu_ptr(cgrp->info);
-               __update_cgrp_time(info, ctx->timestamp, false);
-               __store_release(&info->active, 1);
+               if (guest) {
+                       __update_cgrp_guest_time(info, ctx->time.stamp, false);
+               } else {
+                       update_perf_time_ctx(&info->time, ctx->time.stamp, false);
+                       __store_release(&info->active, 1);
+               }
        }
  }
  
@@@ -964,8 -1064,7 +1064,7 @@@ static void perf_cgroup_switch(struct t
                return;
  
        WARN_ON_ONCE(cpuctx->ctx.nr_cgroups == 0);
-       perf_ctx_disable(&cpuctx->ctx, true);
+       perf_ctx_disable(&cpuctx->ctx, EVENT_CGROUP);
  
        ctx_sched_out(&cpuctx->ctx, NULL, EVENT_ALL|EVENT_CGROUP);
        /*
         */
        ctx_sched_in(&cpuctx->ctx, NULL, EVENT_ALL|EVENT_CGROUP);
  
-       perf_ctx_enable(&cpuctx->ctx, true);
+       perf_ctx_enable(&cpuctx->ctx, EVENT_CGROUP);
  }
  
  static int perf_cgroup_ensure_storage(struct perf_event *event,
@@@ -1138,7 -1237,7 +1237,7 @@@ static inline int perf_cgroup_connect(p
  }
  
  static inline void
- perf_cgroup_set_timestamp(struct perf_cpu_context *cpuctx)
+ perf_cgroup_set_timestamp(struct perf_cpu_context *cpuctx, bool guest)
  {
  }
  
@@@ -1550,29 -1649,24 +1649,24 @@@ static void perf_unpin_context(struct p
   */
  static void __update_context_time(struct perf_event_context *ctx, bool adv)
  {
-       u64 now = perf_clock();
        lockdep_assert_held(&ctx->lock);
  
-       if (adv)
-               ctx->time += now - ctx->timestamp;
-       ctx->timestamp = now;
+       update_perf_time_ctx(&ctx->time, perf_clock(), adv);
+ }
  
-       /*
-        * The above: time' = time + (now - timestamp), can be re-arranged
-        * into: time` = now + (time - timestamp), which gives a single value
-        * offset to compute future time without locks on.
-        *
-        * See perf_event_time_now(), which can be used from NMI context where
-        * it's (obviously) not possible to acquire ctx->lock in order to read
-        * both the above values in a consistent manner.
-        */
-       WRITE_ONCE(ctx->timeoffset, ctx->time - ctx->timestamp);
+ static void __update_context_guest_time(struct perf_event_context *ctx, bool adv)
+ {
+       lockdep_assert_held(&ctx->lock);
+       /* must be called after __update_context_time(); */
+       update_perf_time_ctx(&ctx->timeguest, ctx->time.stamp, adv);
  }
  
  static void update_context_time(struct perf_event_context *ctx)
  {
        __update_context_time(ctx, true);
+       if (is_guest_mediated_pmu_loaded())
+               __update_context_guest_time(ctx, true);
  }
  
  static u64 perf_event_time(struct perf_event *event)
        if (is_cgroup_event(event))
                return perf_cgroup_event_time(event);
  
-       return ctx->time;
+       return __perf_event_time_ctx(event, &ctx->time);
  }
  
  static u64 perf_event_time_now(struct perf_event *event, u64 now)
                return perf_cgroup_event_time_now(event, now);
  
        if (!(__load_acquire(&ctx->is_active) & EVENT_TIME))
-               return ctx->time;
+               return __perf_event_time_ctx(event, &ctx->time);
  
-       now += READ_ONCE(ctx->timeoffset);
-       return now;
+       return __perf_event_time_ctx_now(event, &ctx->time, now);
  }
  
  static enum event_type_t get_event_type(struct perf_event *event)
@@@ -2422,20 -2515,23 +2515,23 @@@ group_sched_out(struct perf_event *grou
  }
  
  static inline void
- __ctx_time_update(struct perf_cpu_context *cpuctx, struct perf_event_context *ctx, bool final)
+ __ctx_time_update(struct perf_cpu_context *cpuctx, struct perf_event_context *ctx,
+                 bool final, enum event_type_t event_type)
  {
        if (ctx->is_active & EVENT_TIME) {
                if (ctx->is_active & EVENT_FROZEN)
                        return;
                update_context_time(ctx);
-               update_cgrp_time_from_cpuctx(cpuctx, final);
+               /* vPMU should not stop time */
+               update_cgrp_time_from_cpuctx(cpuctx, !(event_type & EVENT_GUEST) && final);
        }
  }
  
  static inline void
  ctx_time_update(struct perf_cpu_context *cpuctx, struct perf_event_context *ctx)
  {
-       __ctx_time_update(cpuctx, ctx, false);
+       __ctx_time_update(cpuctx, ctx, false, 0);
  }
  
  /*
@@@ -2861,14 -2957,15 +2957,15 @@@ static void task_ctx_sched_out(struct p
  
  static void perf_event_sched_in(struct perf_cpu_context *cpuctx,
                                struct perf_event_context *ctx,
-                               struct pmu *pmu)
+                               struct pmu *pmu,
+                               enum event_type_t event_type)
  {
-       ctx_sched_in(&cpuctx->ctx, pmu, EVENT_PINNED);
+       ctx_sched_in(&cpuctx->ctx, pmu, EVENT_PINNED | event_type);
        if (ctx)
-                ctx_sched_in(ctx, pmu, EVENT_PINNED);
-       ctx_sched_in(&cpuctx->ctx, pmu, EVENT_FLEXIBLE);
+               ctx_sched_in(ctx, pmu, EVENT_PINNED | event_type);
+       ctx_sched_in(&cpuctx->ctx, pmu, EVENT_FLEXIBLE | event_type);
        if (ctx)
-                ctx_sched_in(ctx, pmu, EVENT_FLEXIBLE);
+               ctx_sched_in(ctx, pmu, EVENT_FLEXIBLE | event_type);
  }
  
  /*
@@@ -2902,11 -2999,11 +2999,11 @@@ static void ctx_resched(struct perf_cpu
  
        event_type &= EVENT_ALL;
  
-       for_each_epc(epc, &cpuctx->ctx, pmu, false)
+       for_each_epc(epc, &cpuctx->ctx, pmu, 0)
                perf_pmu_disable(epc->pmu);
  
        if (task_ctx) {
-               for_each_epc(epc, task_ctx, pmu, false)
+               for_each_epc(epc, task_ctx, pmu, 0)
                        perf_pmu_disable(epc->pmu);
  
                task_ctx_sched_out(task_ctx, pmu, event_type);
        else if (event_type & EVENT_PINNED)
                ctx_sched_out(&cpuctx->ctx, pmu, EVENT_FLEXIBLE);
  
-       perf_event_sched_in(cpuctx, task_ctx, pmu);
+       perf_event_sched_in(cpuctx, task_ctx, pmu, 0);
  
-       for_each_epc(epc, &cpuctx->ctx, pmu, false)
+       for_each_epc(epc, &cpuctx->ctx, pmu, 0)
                perf_pmu_enable(epc->pmu);
  
        if (task_ctx) {
-               for_each_epc(epc, task_ctx, pmu, false)
+               for_each_epc(epc, task_ctx, pmu, 0)
                        perf_pmu_enable(epc->pmu);
        }
  }
@@@ -3479,11 -3576,10 +3576,10 @@@ static voi
  ctx_sched_out(struct perf_event_context *ctx, struct pmu *pmu, enum event_type_t event_type)
  {
        struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
+       enum event_type_t active_type = event_type & ~EVENT_FLAGS;
        struct perf_event_pmu_context *pmu_ctx;
        int is_active = ctx->is_active;
-       bool cgroup = event_type & EVENT_CGROUP;
  
-       event_type &= ~EVENT_CGROUP;
  
        lockdep_assert_held(&ctx->lock);
  
         *
         * would only update time for the pinned events.
         */
-       __ctx_time_update(cpuctx, ctx, ctx == &cpuctx->ctx);
+       __ctx_time_update(cpuctx, ctx, ctx == &cpuctx->ctx, event_type);
  
        /*
         * CPU-release for the below ->is_active store,
         * see __load_acquire() in perf_event_time_now()
         */
        barrier();
-       ctx->is_active &= ~event_type;
+       ctx->is_active &= ~active_type;
  
        if (!(ctx->is_active & EVENT_ALL)) {
                /*
                        cpuctx->task_ctx = NULL;
        }
  
-       is_active ^= ctx->is_active; /* changed bits */
+       if (event_type & EVENT_GUEST) {
+               /*
+                * Schedule out all exclude_guest events of PMU
+                * with PERF_PMU_CAP_MEDIATED_VPMU.
+                */
+               is_active = EVENT_ALL;
+               __update_context_guest_time(ctx, false);
+               perf_cgroup_set_timestamp(cpuctx, true);
+               barrier();
+       } else {
+               is_active ^= ctx->is_active; /* changed bits */
+       }
  
-       for_each_epc(pmu_ctx, ctx, pmu, cgroup)
+       for_each_epc(pmu_ctx, ctx, pmu, event_type)
                __pmu_ctx_sched_out(pmu_ctx, is_active);
  }
  
@@@ -3691,7 -3798,7 +3798,7 @@@ perf_event_context_sched_out(struct tas
                raw_spin_lock_nested(&next_ctx->lock, SINGLE_DEPTH_NESTING);
                if (context_equiv(ctx, next_ctx)) {
  
-                       perf_ctx_disable(ctx, false);
+                       perf_ctx_disable(ctx, 0);
  
                        /* PMIs are disabled; ctx->nr_no_switch_fast is stable. */
                        if (local_read(&ctx->nr_no_switch_fast) ||
  
                        perf_ctx_sched_task_cb(ctx, task, false);
  
-                       perf_ctx_enable(ctx, false);
+                       perf_ctx_enable(ctx, 0);
  
                        /*
                         * RCU_INIT_POINTER here is safe because we've not
@@@ -3739,13 -3846,13 +3846,13 @@@ unlock
  
        if (do_switch) {
                raw_spin_lock(&ctx->lock);
-               perf_ctx_disable(ctx, false);
+               perf_ctx_disable(ctx, 0);
  
  inside_switch:
                perf_ctx_sched_task_cb(ctx, task, false);
                task_ctx_sched_out(ctx, NULL, EVENT_ALL);
  
-               perf_ctx_enable(ctx, false);
+               perf_ctx_enable(ctx, 0);
                raw_spin_unlock(&ctx->lock);
        }
  }
@@@ -3992,10 -4099,15 +4099,15 @@@ static inline void group_update_userpag
                event_update_userpage(event);
  }
  
+ struct merge_sched_data {
+       int can_add_hw;
+       enum event_type_t event_type;
+ };
  static int merge_sched_in(struct perf_event *event, void *data)
  {
        struct perf_event_context *ctx = event->ctx;
-       int *can_add_hw = data;
+       struct merge_sched_data *msd = data;
  
        if (event->state <= PERF_EVENT_STATE_OFF)
                return 0;
        if (!event_filter_match(event))
                return 0;
  
-       if (group_can_go_on(event, *can_add_hw)) {
+       /*
+        * Don't schedule in any host events from PMU with
+        * PERF_PMU_CAP_MEDIATED_VPMU, while a guest is running.
+        */
+       if (is_guest_mediated_pmu_loaded() &&
+           event->pmu_ctx->pmu->capabilities & PERF_PMU_CAP_MEDIATED_VPMU &&
+           !(msd->event_type & EVENT_GUEST))
+               return 0;
+       if (group_can_go_on(event, msd->can_add_hw)) {
                if (!group_sched_in(event, ctx))
                        list_add_tail(&event->active_list, get_event_list(event));
        }
  
        if (event->state == PERF_EVENT_STATE_INACTIVE) {
-               *can_add_hw = 0;
+               msd->can_add_hw = 0;
                if (event->attr.pinned) {
                        perf_cgroup_event_disable(event, ctx);
                        perf_event_set_state(event, PERF_EVENT_STATE_ERROR);
  
  static void pmu_groups_sched_in(struct perf_event_context *ctx,
                                struct perf_event_groups *groups,
-                               struct pmu *pmu)
+                               struct pmu *pmu,
+                               enum event_type_t event_type)
  {
-       int can_add_hw = 1;
+       struct merge_sched_data msd = {
+               .can_add_hw = 1,
+               .event_type = event_type,
+       };
        visit_groups_merge(ctx, groups, smp_processor_id(), pmu,
-                          merge_sched_in, &can_add_hw);
+                          merge_sched_in, &msd);
  }
  
  static void __pmu_ctx_sched_in(struct perf_event_pmu_context *pmu_ctx,
        struct perf_event_context *ctx = pmu_ctx->ctx;
  
        if (event_type & EVENT_PINNED)
-               pmu_groups_sched_in(ctx, &ctx->pinned_groups, pmu_ctx->pmu);
+               pmu_groups_sched_in(ctx, &ctx->pinned_groups, pmu_ctx->pmu, event_type);
        if (event_type & EVENT_FLEXIBLE)
-               pmu_groups_sched_in(ctx, &ctx->flexible_groups, pmu_ctx->pmu);
+               pmu_groups_sched_in(ctx, &ctx->flexible_groups, pmu_ctx->pmu, event_type);
  }
  
  static void
  ctx_sched_in(struct perf_event_context *ctx, struct pmu *pmu, enum event_type_t event_type)
  {
        struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
+       enum event_type_t active_type = event_type & ~EVENT_FLAGS;
        struct perf_event_pmu_context *pmu_ctx;
        int is_active = ctx->is_active;
-       bool cgroup = event_type & EVENT_CGROUP;
-       event_type &= ~EVENT_CGROUP;
  
        lockdep_assert_held(&ctx->lock);
  
                return;
  
        if (!(is_active & EVENT_TIME)) {
+               /* EVENT_TIME should be active while the guest runs */
+               WARN_ON_ONCE(event_type & EVENT_GUEST);
                /* start ctx time */
                __update_context_time(ctx, false);
-               perf_cgroup_set_timestamp(cpuctx);
+               perf_cgroup_set_timestamp(cpuctx, false);
                /*
                 * CPU-release for the below ->is_active store,
                 * see __load_acquire() in perf_event_time_now()
                barrier();
        }
  
-       ctx->is_active |= (event_type | EVENT_TIME);
+       ctx->is_active |= active_type | EVENT_TIME;
        if (ctx->task) {
                if (!(is_active & EVENT_ALL))
                        cpuctx->task_ctx = ctx;
                        WARN_ON_ONCE(cpuctx->task_ctx != ctx);
        }
  
-       is_active ^= ctx->is_active; /* changed bits */
+       if (event_type & EVENT_GUEST) {
+               /*
+                * Schedule in the required exclude_guest events of PMU
+                * with PERF_PMU_CAP_MEDIATED_VPMU.
+                */
+               is_active = event_type & EVENT_ALL;
+               /*
+                * Update ctx time to set the new start time for
+                * the exclude_guest events.
+                */
+               update_context_time(ctx);
+               update_cgrp_time_from_cpuctx(cpuctx, false);
+               barrier();
+       } else {
+               is_active ^= ctx->is_active; /* changed bits */
+       }
  
        /*
         * First go through the list and put on any pinned groups
         * in order to give them the best chance of going on.
         */
        if (is_active & EVENT_PINNED) {
-               for_each_epc(pmu_ctx, ctx, pmu, cgroup)
-                       __pmu_ctx_sched_in(pmu_ctx, EVENT_PINNED);
+               for_each_epc(pmu_ctx, ctx, pmu, event_type)
+                       __pmu_ctx_sched_in(pmu_ctx, EVENT_PINNED | (event_type & EVENT_GUEST));
        }
  
        /* Then walk through the lower prio flexible groups */
        if (is_active & EVENT_FLEXIBLE) {
-               for_each_epc(pmu_ctx, ctx, pmu, cgroup)
-                       __pmu_ctx_sched_in(pmu_ctx, EVENT_FLEXIBLE);
+               for_each_epc(pmu_ctx, ctx, pmu, event_type)
+                       __pmu_ctx_sched_in(pmu_ctx, EVENT_FLEXIBLE | (event_type & EVENT_GUEST));
        }
  }
  
@@@ -4114,11 -4255,11 +4255,11 @@@ static void perf_event_context_sched_in
  
        if (cpuctx->task_ctx == ctx) {
                perf_ctx_lock(cpuctx, ctx);
-               perf_ctx_disable(ctx, false);
+               perf_ctx_disable(ctx, 0);
  
                perf_ctx_sched_task_cb(ctx, task, true);
  
-               perf_ctx_enable(ctx, false);
+               perf_ctx_enable(ctx, 0);
                perf_ctx_unlock(cpuctx, ctx);
                goto rcu_unlock;
        }
        if (!ctx->nr_events)
                goto unlock;
  
-       perf_ctx_disable(ctx, false);
+       perf_ctx_disable(ctx, 0);
        /*
         * We want to keep the following priority order:
         * cpu pinned (that don't need to move), task pinned,
         * events, no need to flip the cpuctx's events around.
         */
        if (!RB_EMPTY_ROOT(&ctx->pinned_groups.tree)) {
-               perf_ctx_disable(&cpuctx->ctx, false);
+               perf_ctx_disable(&cpuctx->ctx, 0);
                ctx_sched_out(&cpuctx->ctx, NULL, EVENT_FLEXIBLE);
        }
  
-       perf_event_sched_in(cpuctx, ctx, NULL);
+       perf_event_sched_in(cpuctx, ctx, NULL, 0);
  
        perf_ctx_sched_task_cb(cpuctx->task_ctx, task, true);
  
        if (!RB_EMPTY_ROOT(&ctx->pinned_groups.tree))
-               perf_ctx_enable(&cpuctx->ctx, false);
+               perf_ctx_enable(&cpuctx->ctx, 0);
  
-       perf_ctx_enable(ctx, false);
+       perf_ctx_enable(ctx, 0);
  
  unlock:
        perf_ctx_unlock(cpuctx, ctx);
@@@ -5594,6 -5735,8 +5735,8 @@@ static void __free_event(struct perf_ev
  {
        struct pmu *pmu = event->pmu;
  
+       security_perf_event_free(event);
        if (event->attach_state & PERF_ATTACH_CALLCHAIN)
                put_callchain_buffers();
  
        call_rcu(&event->rcu_head, free_event_rcu);
  }
  
+ static void mediated_pmu_unaccount_event(struct perf_event *event);
  DEFINE_FREE(__free_event, struct perf_event *, if (_T) __free_event(_T))
  
  /* vs perf_event_alloc() success */
@@@ -5656,8 -5801,7 +5801,7 @@@ static void _free_event(struct perf_eve
        irq_work_sync(&event->pending_disable_irq);
  
        unaccount_event(event);
-       security_perf_event_free(event);
+       mediated_pmu_unaccount_event(event);
  
        if (event->rb) {
                /*
@@@ -6180,6 -6324,138 +6324,138 @@@ u64 perf_event_pause(struct perf_event 
  }
  EXPORT_SYMBOL_GPL(perf_event_pause);
  
+ #ifdef CONFIG_PERF_GUEST_MEDIATED_PMU
+ static atomic_t nr_include_guest_events __read_mostly;
+ static atomic_t nr_mediated_pmu_vms __read_mostly;
+ static DEFINE_MUTEX(perf_mediated_pmu_mutex);
+ /* !exclude_guest event of PMU with PERF_PMU_CAP_MEDIATED_VPMU */
+ static inline bool is_include_guest_event(struct perf_event *event)
+ {
+       if ((event->pmu->capabilities & PERF_PMU_CAP_MEDIATED_VPMU) &&
+           !event->attr.exclude_guest)
+               return true;
+       return false;
+ }
+ static int mediated_pmu_account_event(struct perf_event *event)
+ {
+       if (!is_include_guest_event(event))
+               return 0;
+       if (atomic_inc_not_zero(&nr_include_guest_events))
+               return 0;
+       guard(mutex)(&perf_mediated_pmu_mutex);
+       if (atomic_read(&nr_mediated_pmu_vms))
+               return -EOPNOTSUPP;
+       atomic_inc(&nr_include_guest_events);
+       return 0;
+ }
+ static void mediated_pmu_unaccount_event(struct perf_event *event)
+ {
+       if (!is_include_guest_event(event))
+               return;
+       if (WARN_ON_ONCE(!atomic_read(&nr_include_guest_events)))
+               return;
+       atomic_dec(&nr_include_guest_events);
+ }
+ /*
+  * Currently invoked at VM creation to
+  * - Check whether there are existing !exclude_guest events of PMU with
+  *   PERF_PMU_CAP_MEDIATED_VPMU
+  * - Set nr_mediated_pmu_vms to prevent !exclude_guest event creation on
+  *   PMUs with PERF_PMU_CAP_MEDIATED_VPMU
+  *
+  * No impact for the PMU without PERF_PMU_CAP_MEDIATED_VPMU. The perf
+  * still owns all the PMU resources.
+  */
+ int perf_create_mediated_pmu(void)
+ {
+       if (atomic_inc_not_zero(&nr_mediated_pmu_vms))
+               return 0;
+       guard(mutex)(&perf_mediated_pmu_mutex);
+       if (atomic_read(&nr_include_guest_events))
+               return -EBUSY;
+       atomic_inc(&nr_mediated_pmu_vms);
+       return 0;
+ }
+ EXPORT_SYMBOL_FOR_KVM(perf_create_mediated_pmu);
+ void perf_release_mediated_pmu(void)
+ {
+       if (WARN_ON_ONCE(!atomic_read(&nr_mediated_pmu_vms)))
+               return;
+       atomic_dec(&nr_mediated_pmu_vms);
+ }
+ EXPORT_SYMBOL_FOR_KVM(perf_release_mediated_pmu);
+ /* When loading a guest's mediated PMU, schedule out all exclude_guest events. */
+ void perf_load_guest_context(void)
+ {
+       struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
+       lockdep_assert_irqs_disabled();
+       guard(perf_ctx_lock)(cpuctx, cpuctx->task_ctx);
+       if (WARN_ON_ONCE(__this_cpu_read(guest_ctx_loaded)))
+               return;
+       perf_ctx_disable(&cpuctx->ctx, EVENT_GUEST);
+       ctx_sched_out(&cpuctx->ctx, NULL, EVENT_GUEST);
+       if (cpuctx->task_ctx) {
+               perf_ctx_disable(cpuctx->task_ctx, EVENT_GUEST);
+               task_ctx_sched_out(cpuctx->task_ctx, NULL, EVENT_GUEST);
+       }
+       perf_ctx_enable(&cpuctx->ctx, EVENT_GUEST);
+       if (cpuctx->task_ctx)
+               perf_ctx_enable(cpuctx->task_ctx, EVENT_GUEST);
+       __this_cpu_write(guest_ctx_loaded, true);
+ }
+ EXPORT_SYMBOL_GPL(perf_load_guest_context);
+ void perf_put_guest_context(void)
+ {
+       struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
+       lockdep_assert_irqs_disabled();
+       guard(perf_ctx_lock)(cpuctx, cpuctx->task_ctx);
+       if (WARN_ON_ONCE(!__this_cpu_read(guest_ctx_loaded)))
+               return;
+       perf_ctx_disable(&cpuctx->ctx, EVENT_GUEST);
+       if (cpuctx->task_ctx)
+               perf_ctx_disable(cpuctx->task_ctx, EVENT_GUEST);
+       perf_event_sched_in(cpuctx, cpuctx->task_ctx, NULL, EVENT_GUEST);
+       if (cpuctx->task_ctx)
+               perf_ctx_enable(cpuctx->task_ctx, EVENT_GUEST);
+       perf_ctx_enable(&cpuctx->ctx, EVENT_GUEST);
+       __this_cpu_write(guest_ctx_loaded, false);
+ }
+ EXPORT_SYMBOL_GPL(perf_put_guest_context);
+ #else
+ static int mediated_pmu_account_event(struct perf_event *event) { return 0; }
+ static void mediated_pmu_unaccount_event(struct perf_event *event) {}
+ #endif
  /*
   * Holding the top-level event's child_mutex means that any
   * descendant process that has inherited this event will block
@@@ -6548,22 -6824,22 +6824,22 @@@ void perf_event_update_userpage(struct 
                goto unlock;
  
        /*
-        * compute total_time_enabled, total_time_running
-        * based on snapshot values taken when the event
-        * was last scheduled in.
+        * Disable preemption to guarantee consistent time stamps are stored to
+        * the user page.
+        */
+       preempt_disable();
+       /*
+        * Compute total_time_enabled, total_time_running based on snapshot
+        * values taken when the event was last scheduled in.
         *
-        * we cannot simply called update_context_time()
-        * because of locking issue as we can be called in
-        * NMI context
+        * We cannot simply call update_context_time() because doing so would
+        * lead to deadlock when called from NMI context.
         */
        calc_timer_values(event, &now, &enabled, &running);
  
        userpg = rb->user_page;
-       /*
-        * Disable preemption to guarantee consistent time stamps are stored to
-        * the user page.
-        */
-       preempt_disable();
        ++userpg->lock;
        barrier();
        userpg->index = perf_event_index(event);
@@@ -6997,15 -7273,6 +7273,15 @@@ static int perf_mmap_rb(struct vm_area_
                if (data_page_nr(event->rb) != nr_pages)
                        return -EINVAL;
  
 +              /*
 +               * If this event doesn't have mmap_count, we're attempting to
 +               * create an alias of another event's mmap(); this would mean
 +               * both events will end up scribbling the same user_page;
 +               * which makes no sense.
 +               */
 +              if (!refcount_read(&event->mmap_count))
 +                      return -EBUSY;
 +
                if (refcount_inc_not_zero(&event->rb->mmap_count)) {
                        /*
                         * Success -- managed to mmap() the same buffer
@@@ -7383,6 -7650,7 +7659,7 @@@ struct perf_guest_info_callbacks __rcu 
  DEFINE_STATIC_CALL_RET0(__perf_guest_state, *perf_guest_cbs->state);
  DEFINE_STATIC_CALL_RET0(__perf_guest_get_ip, *perf_guest_cbs->get_ip);
  DEFINE_STATIC_CALL_RET0(__perf_guest_handle_intel_pt_intr, *perf_guest_cbs->handle_intel_pt_intr);
+ DEFINE_STATIC_CALL_RET0(__perf_guest_handle_mediated_pmi, *perf_guest_cbs->handle_mediated_pmi);
  
  void perf_register_guest_info_callbacks(struct perf_guest_info_callbacks *cbs)
  {
        if (cbs->handle_intel_pt_intr)
                static_call_update(__perf_guest_handle_intel_pt_intr,
                                   cbs->handle_intel_pt_intr);
+       if (cbs->handle_mediated_pmi)
+               static_call_update(__perf_guest_handle_mediated_pmi,
+                                  cbs->handle_mediated_pmi);
  }
  EXPORT_SYMBOL_GPL(perf_register_guest_info_callbacks);
  
@@@ -7408,8 -7680,8 +7689,8 @@@ void perf_unregister_guest_info_callbac
        rcu_assign_pointer(perf_guest_cbs, NULL);
        static_call_update(__perf_guest_state, (void *)&__static_call_return0);
        static_call_update(__perf_guest_get_ip, (void *)&__static_call_return0);
-       static_call_update(__perf_guest_handle_intel_pt_intr,
-                          (void *)&__static_call_return0);
+       static_call_update(__perf_guest_handle_intel_pt_intr, (void *)&__static_call_return0);
+       static_call_update(__perf_guest_handle_mediated_pmi, (void *)&__static_call_return0);
        synchronize_rcu();
  }
  EXPORT_SYMBOL_GPL(perf_unregister_guest_info_callbacks);
@@@ -7460,7 -7732,7 +7741,7 @@@ static void perf_sample_regs_user(struc
        if (user_mode(regs)) {
                regs_user->abi = perf_reg_abi(current);
                regs_user->regs = regs;
 -      } else if (!(current->flags & (PF_KTHREAD | PF_USER_WORKER))) {
 +      } else if (is_user_task(current)) {
                perf_get_regs_user(regs_user, regs);
        } else {
                regs_user->abi = PERF_SAMPLE_REGS_ABI_NONE;
@@@ -7869,13 -8141,11 +8150,11 @@@ static void perf_output_read(struct per
        u64 read_format = event->attr.read_format;
  
        /*
-        * compute total_time_enabled, total_time_running
-        * based on snapshot values taken when the event
-        * was last scheduled in.
+        * Compute total_time_enabled, total_time_running based on snapshot
+        * values taken when the event was last scheduled in.
         *
-        * we cannot simply called update_context_time()
-        * because of locking issue as we are called in
-        * NMI context
+        * We cannot simply call update_context_time() because doing so would
+        * lead to deadlock when called from NMI context.
         */
        if (read_format & PERF_FORMAT_TOTAL_TIMES)
                calc_timer_values(event, &now, &enabled, &running);
@@@ -8100,7 -8370,7 +8379,7 @@@ static u64 perf_virt_to_phys(u64 virt
                 * Try IRQ-safe get_user_page_fast_only first.
                 * If failed, leave phys_addr as 0.
                 */
 -              if (!(current->flags & (PF_KTHREAD | PF_USER_WORKER))) {
 +              if (is_user_task(current)) {
                        struct page *p;
  
                        pagefault_disable();
@@@ -8215,7 -8485,7 +8494,7 @@@ perf_callchain(struct perf_event *event
  {
        bool kernel = !event->attr.exclude_callchain_kernel;
        bool user   = !event->attr.exclude_callchain_user &&
 -              !(current->flags & (PF_KTHREAD | PF_USER_WORKER));
 +              is_user_task(current);
        /* Disallow cross-task user callchains. */
        bool crosstask = event->ctx->task && event->ctx->task != current;
        bool defer_user = IS_ENABLED(CONFIG_UNWIND_USER) && user &&
@@@ -11915,11 -12185,6 +12194,11 @@@ static void perf_swevent_cancel_hrtimer
        }
  }
  
 +static void perf_swevent_destroy_hrtimer(struct perf_event *event)
 +{
 +      hrtimer_cancel(&event->hw.hrtimer);
 +}
 +
  static void perf_swevent_init_hrtimer(struct perf_event *event)
  {
        struct hw_perf_event *hwc = &event->hw;
                return;
  
        hrtimer_setup(&hwc->hrtimer, perf_swevent_hrtimer, CLOCK_MONOTONIC, HRTIMER_MODE_REL_HARD);
 +      event->destroy = perf_swevent_destroy_hrtimer;
  
        /*
         * Since hrtimers have a fixed rate, we can do a static freq->period
@@@ -12043,7 -12307,7 +12322,7 @@@ static void task_clock_event_update(str
  static void task_clock_event_start(struct perf_event *event, int flags)
  {
        event->hw.state = 0;
-       local64_set(&event->hw.prev_count, event->ctx->time);
+       local64_set(&event->hw.prev_count, event->ctx->time.time);
        perf_swevent_start_hrtimer(event);
  }
  
@@@ -12052,7 -12316,7 +12331,7 @@@ static void task_clock_event_stop(struc
        event->hw.state = PERF_HES_STOPPED;
        perf_swevent_cancel_hrtimer(event);
        if (flags & PERF_EF_UPDATE)
-               task_clock_event_update(event, event->ctx->time);
+               task_clock_event_update(event, event->ctx->time.time);
  }
  
  static int task_clock_event_add(struct perf_event *event, int flags)
@@@ -12072,8 -12336,8 +12351,8 @@@ static void task_clock_event_del(struc
  static void task_clock_event_read(struct perf_event *event)
  {
        u64 now = perf_clock();
-       u64 delta = now - event->ctx->timestamp;
-       u64 time = event->ctx->time + delta;
+       u64 delta = now - event->ctx->time.stamp;
+       u64 time = event->ctx->time.time + delta;
  
        task_clock_event_update(event, time);
  }
@@@ -13155,6 -13419,10 +13434,10 @@@ perf_event_alloc(struct perf_event_att
        if (err)
                return ERR_PTR(err);
  
+       err = mediated_pmu_account_event(event);
+       if (err)
+               return ERR_PTR(err);
        /* symmetric to unaccount_event() in _free_event() */
        account_event(event);
  
diff --combined virt/kvm/kvm_main.c
index 6b1097e7628850d3d89c8a06359aacb56c05cd88,d59cb53af76a5f1ab6014b3ea91113bff225cc2d..61dca8d37abc2f7278b2c70d50886ad9a7d31bd3
@@@ -1749,12 -1749,6 +1749,12 @@@ static void kvm_commit_memory_region(st
                kvm_free_memslot(kvm, old);
                break;
        case KVM_MR_MOVE:
 +              /*
 +               * Moving a guest_memfd memslot isn't supported, and will never
 +               * be supported.
 +               */
 +              WARN_ON_ONCE(old->flags & KVM_MEM_GUEST_MEMFD);
 +              fallthrough;
        case KVM_MR_FLAGS_ONLY:
                /*
                 * Free the dirty bitmap as needed; the below check encompasses
                if (old->dirty_bitmap && !new->dirty_bitmap)
                        kvm_destroy_dirty_bitmap(old);
  
 +              /*
 +               * Unbind the guest_memfd instance as needed; the @new slot has
 +               * already created its own binding.  TODO: Drop the WARN when
 +               * dirty logging guest_memfd memslots is supported.  Until then,
 +               * flags-only changes on guest_memfd slots should be impossible.
 +               */
 +              if (WARN_ON_ONCE(old->flags & KVM_MEM_GUEST_MEMFD))
 +                      kvm_gmem_unbind(old);
 +
                /*
                 * The final quirk.  Free the detached, old slot, but only its
                 * memory, not any metadata.  Metadata, including arch specific
@@@ -2101,7 -2086,7 +2101,7 @@@ static int kvm_set_memory_region(struc
                        return -EINVAL;
                if ((mem->userspace_addr != old->userspace_addr) ||
                    (npages != old->npages) ||
 -                  ((mem->flags ^ old->flags) & KVM_MEM_READONLY))
 +                  ((mem->flags ^ old->flags) & (KVM_MEM_READONLY | KVM_MEM_GUEST_MEMFD)))
                        return -EINVAL;
  
                if (base_gfn != old->base_gfn)
@@@ -3134,7 -3119,7 +3134,7 @@@ int __kvm_vcpu_map(struct kvm_vcpu *vcp
                   bool writable)
  {
        struct kvm_follow_pfn kfp = {
 -              .slot = gfn_to_memslot(vcpu->kvm, gfn),
 +              .slot = kvm_vcpu_gfn_to_memslot(vcpu, gfn),
                .gfn = gfn,
                .flags = writable ? FOLL_WRITE : 0,
                .refcounted_page = &map->pinned_page,
@@@ -6482,11 -6467,15 +6482,15 @@@ static struct perf_guest_info_callback
        .state                  = kvm_guest_state,
        .get_ip                 = kvm_guest_get_ip,
        .handle_intel_pt_intr   = NULL,
+       .handle_mediated_pmi    = NULL,
  };
  
- void kvm_register_perf_callbacks(unsigned int (*pt_intr_handler)(void))
+ void __kvm_register_perf_callbacks(unsigned int (*pt_intr_handler)(void),
+                                  void (*mediated_pmi_handler)(void))
  {
        kvm_guest_cbs.handle_intel_pt_intr = pt_intr_handler;
+       kvm_guest_cbs.handle_mediated_pmi = mediated_pmi_handler;
        perf_register_guest_info_callbacks(&kvm_guest_cbs);
  }
  void kvm_unregister_perf_callbacks(void)