KVM/arm64 changes for 7.3

 - Add support for 'slot' based PMU events, paired with new UAPI that
   compels the user to select a specific PMU implementation

 - Lazy save/restore of vCPU state for pKVM, along with various fixes
   and cleanups to the management of vCPU state between the untrusted
   host and pKVM hypervisor

 - Disable traps of EL1 registers for nested hypervisors when FEAT_NV2p1
   is present, guaranteeing that EL2-specific register bits are stateful
   in the EL1 counterpart

 - Leverage FEAT_NV3 to avoid unnecessary ERET/TLBI traps when the scope
   of those instructions remains 'in host' (i.e. L1 kernel/userspace)

 - Pile of fixes for the management of the VNCR pseudo-TLB, such as
   under-invalidations and races with concurrent TLBIs on other vCPUs

 - Consolidate the non-protected and pKVM view of ICH_VTR_EL2 to a
   runtime-patched constant, allowing the same data to be shared with
   pKVM prior to dropping host privileges

 - Considerable pile of LLM-assisted fixes around the shop but mostly in
   the VGIC, our in-kernel generator of bugs (and sometimes interrupts)
This commit is contained in:
Paolo Bonzini
2026-08-24 12:42:07 -04:00
66 changed files with 1580 additions and 477 deletions

View File

@@ -3524,6 +3524,17 @@ Possible features:
Depends on KVM_CAP_ARM_PSCI_0_2.
- KVM_ARM_VCPU_PMU_V3: Emulate PMUv3 for the CPU.
Depends on KVM_CAP_ARM_PMU_V3.
- KVM_ARM_VCPU_PMU_V3_STRICT: Enable strict PMUv3 UAPI.
Requires KVM_ARM_VCPU_PMU_V3. Depends on KVM_CAP_ARM_PMU_V3_STRICT.
When enabled:
* Userspace must explicitly select a PMU implementation before
initializing the PMU or configuring a PMU event filter
* If the PMU implements FEAT_PMUv3p4, PMMIR_EL1.SLOTS provides the
hardware value of the underlying implementation
* Writes to PMCR_EL0.N via KVM_SET_ONE_REG are ignored
- KVM_ARM_VCPU_PTRAUTH_ADDRESS: Enables Address Pointer authentication
for arm64 only.

View File

@@ -53,8 +53,9 @@ Returns:
======= ======================================================
-EEXIST Interrupt number already used
-ENODEV PMUv3 not supported or GIC not initialized
-ENXIO PMUv3 not supported, missing VCPU feature or interrupt
number not set (non-GICv5 guests, only)
-ENXIO PMUv3 not supported, missing VCPU feature, missing
hardware PMU, or interrupt number not set (non-GICv5
guests, only)
-EBUSY PMUv3 already initialized
======= ======================================================
@@ -62,6 +63,9 @@ Request the initialization of the PMUv3. If using the PMUv3 with an in-kernel
virtual GIC implementation, this must be done after initializing the in-kernel
irqchip.
When the KVM_ARM_VCPU_PMU_V3_STRICT vCPU feature is enabled this must be done
after selecting a hardware PMU.
1.3 ATTRIBUTE: KVM_ARM_VCPU_PMU_V3_FILTER
-----------------------------------------
@@ -108,6 +112,9 @@ hardware event. Filtering event 0x1E (CHAIN) has no effect either, as it
isn't strictly speaking an event. Filtering the cycle counter is possible
using event 0x11 (CPU_CYCLES).
When the KVM_ARM_VCPU_PMU_V3_STRICT vCPU feature is enabled this must be done
after selecting a hardware PMU.
1.4 ATTRIBUTE: KVM_ARM_VCPU_PMU_V3_SET_PMU
------------------------------------------

View File

@@ -968,6 +968,7 @@ struct arm64_ftr_reg *get_arm64_ftr_reg(u32 sys_id);
extern struct arm64_ftr_override id_aa64mmfr0_override;
extern struct arm64_ftr_override id_aa64mmfr1_override;
extern struct arm64_ftr_override id_aa64mmfr2_override;
extern struct arm64_ftr_override id_aa64mmfr4_override;
extern struct arm64_ftr_override id_aa64pfr0_override;
extern struct arm64_ftr_override id_aa64pfr1_override;
extern struct arm64_ftr_override id_aa64zfr0_override;

View File

@@ -287,21 +287,6 @@
GENMASK(19, 18) | \
GENMASK(15, 0))
/*
* Polarity masks for HCRX_EL2, limited to the bits that we know about
* at this point in time. It doesn't mean that we actually *handle*
* them, but that at least those that are not advertised to a guest
* will be RES0 for that guest.
*/
#define __HCRX_EL2_MASK (BIT_ULL(6))
#define __HCRX_EL2_nMASK (GENMASK_ULL(24, 14) | \
GENMASK_ULL(11, 7) | \
GENMASK_ULL(5, 0))
#define __HCRX_EL2_RES0 ~(__HCRX_EL2_nMASK | __HCRX_EL2_MASK)
#define __HCRX_EL2_RES1 ~(__HCRX_EL2_nMASK | \
__HCRX_EL2_MASK | \
__HCRX_EL2_RES0)
/* Hyp Prefetch Fault Address Register (HPFAR/HDFAR) */
#define HPFAR_MASK (~UL(0xf))
/*
@@ -348,4 +333,16 @@
{ PSR_AA32_MODE_UND, "32-bit UND" }, \
{ PSR_AA32_MODE_SYS, "32-bit SYS" }
/*
* ARMv8 Reset Values
*/
#define VCPU_RESET_PSTATE_EL1 (PSR_MODE_EL1h | PSR_A_BIT | PSR_I_BIT | \
PSR_F_BIT | PSR_D_BIT)
#define VCPU_RESET_PSTATE_EL2 (PSR_MODE_EL2h | PSR_A_BIT | PSR_I_BIT | \
PSR_F_BIT | PSR_D_BIT)
#define VCPU_RESET_PSTATE_SVC (PSR_AA32_MODE_SVC | PSR_AA32_A_BIT | \
PSR_AA32_I_BIT | PSR_AA32_F_BIT)
#endif /* __ARM64_KVM_ARM_H__ */

View File

@@ -113,6 +113,7 @@ enum __kvm_host_smccc_func {
__KVM_HOST_SMCCC_FUNC___pkvm_finalize_teardown_vm,
__KVM_HOST_SMCCC_FUNC___pkvm_vcpu_load,
__KVM_HOST_SMCCC_FUNC___pkvm_vcpu_put,
__KVM_HOST_SMCCC_FUNC___pkvm_vcpu_sync_state,
__KVM_HOST_SMCCC_FUNC___pkvm_tlb_flush_vmid,
MARKER(__KVM_HOST_SMCCC_FUNC_MAX)
@@ -214,7 +215,6 @@ struct kvm_nvhe_init_params {
unsigned long hcr_el2;
unsigned long vttbr;
unsigned long vtcr;
unsigned long tmp;
};
/*
@@ -281,7 +281,7 @@ extern int __kvm_vcpu_run(struct kvm_vcpu *vcpu);
extern void __kvm_adjust_pc(struct kvm_vcpu *vcpu);
extern u64 __vgic_v3_get_gic_config(void);
extern bool __vgic_v3_get_gic_config(void);
extern void __vgic_v3_init_lrs(void);
#define __KVM_EXTABLE(from, to) \

View File

@@ -266,6 +266,25 @@ static inline bool vserror_state_is_nested(struct kvm_vcpu *vcpu)
(__vcpu_sys_reg(vcpu, HCRX_EL2) & HCRX_EL2_TMEA);
}
static inline bool kvm_has_nv2(struct kvm *kvm)
{
return (cpus_have_final_cap(ARM64_HAS_NESTED_VIRT) &&
kvm_has_feat(kvm, ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY));
}
static inline bool kvm_has_nv3(struct kvm *kvm)
{
return (cpus_have_final_cap(ARM64_HAS_NV3) &&
kvm_has_feat(kvm, ID_AA64MMFR4_EL1, NV_frac, NV3));
}
static inline bool is_nested_nv3_ctxt(struct kvm_vcpu *vcpu)
{
return (has_vhe() && kvm_has_nv3(vcpu->kvm) && is_nested_ctxt(vcpu) &&
(__vcpu_sys_reg(vcpu, HCR_EL2) & HCR_EL2_NV) &&
(__vcpu_sys_reg(vcpu, HCRX_EL2) & HCRX_EL2_NVTGE));
}
/*
* The layout of SPSR for an AArch32 state is different when observed from an
* AArch64 SPSR_ELx or an AArch32 SPSR_*. This function generates the AArch32
@@ -506,6 +525,12 @@ static inline unsigned long kvm_vcpu_get_mpidr_aff(struct kvm_vcpu *vcpu)
return __vcpu_sys_reg(vcpu, MPIDR_EL1) & MPIDR_HWID_BITMASK;
}
/* In nVHE hyp code, registers are always in memory: use the raw accessors. */
#if defined(__KVM_NVHE_HYPERVISOR__)
#define vcpu_read_sys_reg(v, r) __vcpu_sys_reg(v, r)
#define vcpu_write_sys_reg(v, x, r) __vcpu_assign_sys_reg(v, r, x)
#endif
static inline void kvm_vcpu_set_be(struct kvm_vcpu *vcpu)
{
if (vcpu_mode_is_32bit(vcpu)) {
@@ -617,7 +642,7 @@ static __always_inline void kvm_incr_pc(struct kvm_vcpu *vcpu)
*/
static inline u64 vcpu_sanitised_cptr_el2(const struct kvm_vcpu *vcpu)
{
u64 cptr = __vcpu_sys_reg(vcpu, CPTR_EL2);
u64 cptr = vcpu_read_sys_reg(vcpu, CPTR_EL2);
if (!vcpu_el2_e2h_is_set(vcpu))
cptr = translate_cptr_el2_to_cpacr_el1(cptr);
@@ -686,6 +711,86 @@ static inline void vcpu_set_hcrx(struct kvm_vcpu *vcpu)
if (kvm_has_feat(kvm, ID_AA64ISAR1_EL1, LS64, LS64_V))
vcpu->arch.hcrx_el2 |= HCRX_EL2_EnASR;
/*
* NV3 is a host-specific extension, and we always use
* it when present and that the guest uses NV. It may
* be hidden from the guest though.
*/
if (cpus_have_final_cap(ARM64_HAS_NV3) &&
vcpu_has_nv(vcpu) && vcpu_el2_e2h_is_set(vcpu)) {
vcpu->arch.hcrx_el2 |= HCRX_EL2_NVTGE;
/*
* If the guest is NV2-capable, then we need to see
* all the TLBIs, as configured in HCR_EL2.
* Otherwise, relax the TLBI traps to only TGE=0.
*/
if (!kvm_has_nv2(vcpu->kvm)) {
vcpu->arch.hcrx_el2 |= (HCRX_EL2_NVnTTLB |
HCRX_EL2_NVnTTLBIS);
if (kvm_has_feat(kvm, ID_AA64ISAR0_EL1, TLB, OS))
vcpu->arch.hcrx_el2 |= HCRX_EL2_NVnTTLBOS;
}
}
}
}
/* Reset a vcpu's core registers. */
static inline void kvm_reset_vcpu_core(struct kvm_vcpu *vcpu)
{
u32 pstate;
if (vcpu_el1_is_32bit(vcpu))
pstate = VCPU_RESET_PSTATE_SVC;
else if (vcpu_has_nv(vcpu))
pstate = VCPU_RESET_PSTATE_EL2;
else
pstate = VCPU_RESET_PSTATE_EL1;
/* Reset core registers */
memset(vcpu_gp_regs(vcpu), 0, sizeof(*vcpu_gp_regs(vcpu)));
memset(&vcpu->arch.ctxt.fp_regs, 0, sizeof(vcpu->arch.ctxt.fp_regs));
vcpu->arch.ctxt.spsr_abt = 0;
vcpu->arch.ctxt.spsr_und = 0;
vcpu->arch.ctxt.spsr_irq = 0;
vcpu->arch.ctxt.spsr_fiq = 0;
vcpu_gp_regs(vcpu)->pstate = pstate;
}
/* PSCI reset handling for a vcpu. */
static inline void kvm_reset_vcpu_psci(struct kvm_vcpu *vcpu,
struct vcpu_reset_state *reset_state)
{
unsigned long target_pc = reset_state->pc;
/* Gracefully handle Thumb2 entry point */
if (vcpu_mode_is_32bit(vcpu) && (target_pc & 1)) {
target_pc &= ~1UL;
vcpu_set_thumb(vcpu);
}
/* Propagate caller endianness */
if (reset_state->be)
kvm_vcpu_set_be(vcpu);
*vcpu_pc(vcpu) = target_pc;
/*
* We may come from a state where either a PC update was
* pending (SMC call resulting in PC being increpented to
* skip the SMC) or a pending exception. Make sure we get
* rid of all that, as this cannot be valid out of reset.
*
* Note that clearing the exception mask also clears PC
* updates, but that's an implementation detail, and we
* really want to make it explicit.
*/
vcpu_clear_flag(vcpu, PENDING_EXCEPTION);
vcpu_clear_flag(vcpu, EXCEPT_MASK);
vcpu_clear_flag(vcpu, INCREMENT_PC);
vcpu_set_reg(vcpu, 0, reset_state->r0);
}
#endif /* __ARM64_KVM_EMULATE_H__ */

View File

@@ -39,7 +39,7 @@
#define KVM_MAX_VCPUS VGIC_V3_MAX_CPUS
#define KVM_VCPU_MAX_FEATURES 9
#define KVM_VCPU_MAX_FEATURES 10
#define KVM_VCPU_VALID_FEATURES (BIT(KVM_VCPU_MAX_FEATURES) - 1)
#define KVM_REQ_SLEEP \
@@ -387,6 +387,9 @@ struct kvm_arch {
/* Maximum number of counters for the guest */
u8 nr_pmu_counters;
/* PMMIR_EL1.SLOTS value exposed to the guest. */
u8 pmmir_slots;
/* Hypercall features firmware registers' descriptor */
struct kvm_smccc_features smccc_feat;
struct maple_tree smccc_filter;
@@ -411,8 +414,8 @@ struct kvm_arch {
/* Masks for VNCR-backed and general EL2 sysregs */
struct kvm_sysreg_masks *sysreg_masks;
/* Count the number of VNCR_EL2 currently mapped */
atomic_t vncr_map_count;
/* Count the number of VNCR_EL2 TLBs */
atomic_t vncr_tlb_count;
/*
* For an untrusted host VM, 'pkvm.handle' is used to lookup
@@ -543,6 +546,7 @@ enum vcpu_sysreg {
MDCR_EL2, /* Monitor Debug Configuration Register (EL2) */
CNTHCTL_EL2, /* Counter-timer Hypervisor Control register */
ZCR_EL2, /* SVE Control Register (EL2) */
HCR_EL2, /* Hypervisor Control Register */
/* Any VNCR-capable reg goes after this point */
MARKER(__VNCR_START__),
@@ -571,7 +575,7 @@ enum vcpu_sysreg {
VNCR(TFSR_EL1), /* Tag Fault Status Register (EL1) */
VNCR(VPIDR_EL2),/* Virtualization Processor ID Register */
VNCR(VMPIDR_EL2),/* Virtualization Multiprocessor ID Register */
VNCR(HCR_EL2), /* Hypervisor Configuration Register */
VNCR(NVHCR_EL2),/* NV Hypervisor Configuration Register */
VNCR(HSTR_EL2), /* Hypervisor System Trap Register */
VNCR(VTTBR_EL2),/* Virtualization Translation Table Base Register */
VNCR(VTCR_EL2), /* Virtualization Translation Control Register */
@@ -1051,6 +1055,8 @@ struct kvm_vcpu_arch {
#define INCREMENT_PC __vcpu_single_flag(iflags, BIT(1))
/* Target EL/MODE (not a single flag, but let's abuse the macro) */
#define EXCEPT_MASK __vcpu_single_flag(iflags, GENMASK(3, 1))
/* Host-set: the hyp flushes the non-protected vCPU state in on entry */
#define PKVM_HOST_STATE_DIRTY __vcpu_single_flag(iflags, BIT(4))
/* Helpers to encode exceptions with minimum fuss */
#define __EXCEPT_MASK_VAL unpack_vcpu_flag(EXCEPT_MASK)

View File

@@ -291,6 +291,13 @@ static inline u64 decode_range_tlbi(u64 val, u64 *range, u16 *asid)
base = (val & GENMASK(36, 0)) << shift;
/*
* We only deal with at most 48bit VA/IPA, so 48 is where we
* sign-extend from. Should we support FEAT_L{VP}A* at some point,
* this will need to be revisited.
*/
base = (u64)sign_extend64(base, 48);
if (asid)
*asid = FIELD_GET(TLBIR_ASID_MASK, val);
@@ -298,6 +305,12 @@ static inline u64 decode_range_tlbi(u64 val, u64 *range, u16 *asid)
num = FIELD_GET(GENMASK(43, 39), val);
*range = __TLBI_RANGE_PAGES(num, scale) << shift;
/* Cap the range to the correct half of the address space */
if (!(base & BIT(48)))
*range = min(*range, (BIT(48) - base));
else
*range = min(*range, ~base + 1);
return base;
}
@@ -388,6 +401,8 @@ struct s1_walk_result {
bool failed;
};
#define S1_MMU_DISABLED (-127)
static inline void fail_s1_walk(struct s1_walk_result *wr, u8 fst, bool s1ptw)
{
wr->fst = fst;
@@ -396,6 +411,11 @@ static inline void fail_s1_walk(struct s1_walk_result *wr, u8 fst, bool s1ptw)
wr->failed = true;
}
static inline bool s1_walk_translated(struct s1_walk_result *wr)
{
return wr->level != S1_MMU_DISABLED;
}
int __kvm_translate_va(struct kvm_vcpu *vcpu, struct s1_walk_info *wi,
struct s1_walk_result *wr, u64 va);
int __kvm_find_s1_desc_level(struct kvm_vcpu *vcpu, u64 va, u64 ipa,

View File

@@ -45,6 +45,9 @@ static inline bool kvm_pkvm_ext_allowed(struct kvm *kvm, long ext)
return true;
case KVM_CAP_ARM_MTE:
return false;
case KVM_CAP_ARM_EAGER_SPLIT_CHUNK_SIZE:
case KVM_CAP_ARM_SUPPORTED_BLOCK_SIZES:
return false;
default:
return !kvm || !kvm_vm_is_protected(kvm);
}
@@ -195,7 +198,10 @@ struct pkvm_mapping {
struct rb_node node;
u64 gfn;
u64 pfn;
u64 nr_pages;
struct {
u64 nr_pages:48;
u64 nc:1;
};
u64 __subtree_last; /* Internal member for interval tree */
};

View File

@@ -11,7 +11,7 @@
#define VNCR_VTCR_EL2 0x040
#define VNCR_VMPIDR_EL2 0x050
#define VNCR_CNTVOFF_EL2 0x060
#define VNCR_HCR_EL2 0x078
#define VNCR_NVHCR_EL2 0x078
#define VNCR_HSTR_EL2 0x080
#define VNCR_VPIDR_EL2 0x088
#define VNCR_TPIDR_EL2 0x090

View File

@@ -106,6 +106,7 @@ struct kvm_regs {
#define KVM_ARM_VCPU_PTRAUTH_GENERIC 6 /* VCPU uses generic authentication */
#define KVM_ARM_VCPU_HAS_EL2 7 /* Support nested virtualization */
#define KVM_ARM_VCPU_HAS_EL2_E2H0 8 /* Limit NV support to E2H RES0 */
#define KVM_ARM_VCPU_PMU_V3_STRICT 9 /* No default PMU creation */
struct kvm_vcpu_init {
__u32 target;

View File

@@ -124,7 +124,6 @@ int main(void)
DEFINE(NVHE_INIT_HCR_EL2, offsetof(struct kvm_nvhe_init_params, hcr_el2));
DEFINE(NVHE_INIT_VTTBR, offsetof(struct kvm_nvhe_init_params, vttbr));
DEFINE(NVHE_INIT_VTCR, offsetof(struct kvm_nvhe_init_params, vtcr));
DEFINE(NVHE_INIT_TMP, offsetof(struct kvm_nvhe_init_params, tmp));
#endif
#ifdef CONFIG_CPU_PM
DEFINE(CPU_CTX_SP, offsetof(struct cpu_suspend_ctx, sp));

View File

@@ -272,7 +272,7 @@ has_neoverse_n1_erratum_1542419(const struct arm64_cpu_capabilities *entry,
return is_midr_in_range(&range) && has_dic;
}
static const struct midr_range impdef_pmuv3_cpus[] = {
static const struct midr_range apple_cpus[] = {
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_ICESTORM),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_FIRESTORM),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_ICESTORM_PRO),
@@ -301,7 +301,14 @@ static bool has_impdef_pmuv3(const struct arm64_cpu_capabilities *entry, int sco
if (pmuver != ID_AA64DFR0_EL1_PMUVer_IMP_DEF)
return false;
return is_midr_in_range_list(impdef_pmuv3_cpus);
return is_midr_in_range_list(apple_cpus);
}
static bool has_broken_gic_v3_seis(const struct arm64_cpu_capabilities *entry, int scope)
{
return (is_kernel_in_hyp_mode() &&
is_midr_in_range_list(apple_cpus) &&
(read_sysreg_s(SYS_ICH_VTR_EL2) & ICH_VTR_EL2_SEIS));
}
static void cpu_enable_impdef_pmuv3_traps(const struct arm64_cpu_capabilities *__unused)
@@ -1009,6 +1016,12 @@ const struct arm64_cpu_capabilities arm64_errata[] = {
.matches = has_impdef_pmuv3,
.cpu_enable = cpu_enable_impdef_pmuv3_traps,
},
{
.desc = "Known broken GICv3 SEIS implementation",
.capability = ARM64_WORKAROUND_GICv3_BROKEN_SEIS,
.type = ARM64_CPUCAP_SYSTEM_FEATURE,
.matches = has_broken_gic_v3_seis,
},
{
}
};

View File

@@ -785,6 +785,7 @@ static const struct arm64_ftr_bits ftr_raz[] = {
struct arm64_ftr_override __read_mostly id_aa64mmfr0_override;
struct arm64_ftr_override __read_mostly id_aa64mmfr1_override;
struct arm64_ftr_override __read_mostly id_aa64mmfr2_override;
struct arm64_ftr_override __read_mostly id_aa64mmfr4_override;
struct arm64_ftr_override __read_mostly id_aa64pfr0_override;
struct arm64_ftr_override __read_mostly id_aa64pfr1_override;
struct arm64_ftr_override __read_mostly id_aa64zfr0_override;
@@ -858,7 +859,8 @@ static const struct __ftr_reg_entry {
ARM64_FTR_REG_OVERRIDE(SYS_ID_AA64MMFR2_EL1, ftr_id_aa64mmfr2,
&id_aa64mmfr2_override),
ARM64_FTR_REG(SYS_ID_AA64MMFR3_EL1, ftr_id_aa64mmfr3),
ARM64_FTR_REG(SYS_ID_AA64MMFR4_EL1, ftr_id_aa64mmfr4),
ARM64_FTR_REG_OVERRIDE(SYS_ID_AA64MMFR4_EL1, ftr_id_aa64mmfr4,
&id_aa64mmfr4_override),
/* Op1 = 0, CRn = 10, CRm = 4 */
ARM64_FTR_REG(SYS_MPAMIDR_EL1, ftr_mpamidr),
@@ -2620,6 +2622,20 @@ static const struct arm64_cpu_capabilities arm64_features[] = {
{ /* Sentinel */ }
},
},
{
.desc = "FEAT_NV2p1",
.capability = ARM64_HAS_NV2P1,
.type = ARM64_CPUCAP_SYSTEM_FEATURE,
.matches = has_cpuid_feature,
ARM64_CPUID_FIELDS(ID_AA64MMFR4_EL1, NV_frac, NV2P1)
},
{
.desc = "FEAT_NV3",
.capability = ARM64_HAS_NV3,
.type = ARM64_CPUCAP_SYSTEM_FEATURE,
.matches = has_cpuid_feature,
ARM64_CPUID_FIELDS(ID_AA64MMFR4_EL1, NV_frac, NV3)
},
{
.capability = ARM64_HAS_32BIT_EL0_DO_NOT_USE,
.type = ARM64_CPUCAP_SYSTEM_FEATURE,

View File

@@ -51,6 +51,7 @@ PI_EXPORT_SYM(id_aa64isar2_override);
PI_EXPORT_SYM(id_aa64mmfr0_override);
PI_EXPORT_SYM(id_aa64mmfr1_override);
PI_EXPORT_SYM(id_aa64mmfr2_override);
PI_EXPORT_SYM(id_aa64mmfr4_override);
PI_EXPORT_SYM(id_aa64pfr0_override);
PI_EXPORT_SYM(id_aa64pfr1_override);
PI_EXPORT_SYM(id_aa64smfr0_override);
@@ -92,6 +93,7 @@ KVM_NVHE_ALIAS(spectre_bhb_patch_wa3);
KVM_NVHE_ALIAS(spectre_bhb_patch_clearbhb);
KVM_NVHE_ALIAS(alt_cb_patch_nops);
KVM_NVHE_ALIAS(kvm_compute_ich_hcr_trap_bits);
KVM_NVHE_ALIAS(kvm_patch_ich_vtr_el2);
/* Global kernel state accessed by nVHE hyp code. */
KVM_NVHE_ALIAS(kvm_vgic_global_state);

View File

@@ -106,6 +106,15 @@ static const struct ftr_set_desc mmfr2 __prel64_initconst = {
},
};
static const struct ftr_set_desc mmfr4 __prel64_initconst = {
.name = "id_aa64mmfr4",
.override = &id_aa64mmfr4_override,
.fields = {
FIELD("nv_frac", ID_AA64MMFR4_EL1_NV_frac_SHIFT, NULL),
{}
},
};
static bool __init pfr0_sve_filter(u64 val)
{
/*
@@ -220,6 +229,7 @@ PREL64(const struct ftr_set_desc, reg) regs[] __prel64_initconst = {
{ &mmfr0 },
{ &mmfr1 },
{ &mmfr2 },
{ &mmfr4 },
{ &pfr0 },
{ &pfr1 },
{ &isar1 },

View File

@@ -876,8 +876,14 @@ static void timer_set_traps(struct kvm_vcpu *vcpu, struct timer_map *map)
assign_clear_set_bit(tvt02, CNTHCTL_EL1NVVCT, clr, set);
assign_clear_set_bit(tpt02, CNTHCTL_EL1NVPCT, clr, set);
/* This only happens on VHE, so use the CNTHCTL_EL2 accessor. */
sysreg_clear_set(cnthctl_el2, clr, set);
/*
* This only happens on VHE, so use the CNTHCTL_EL2 accessor, unless
* we are sure CNTKCTL_EL1 is completely stateful with FEAT_NV2p1.
*/
if (!cpus_have_final_cap(ARM64_HAS_NV2P1))
sysreg_clear_set(cnthctl_el2, clr, set);
else
sysreg_clear_set(cntkctl_el1, clr, set);
}
void kvm_timer_vcpu_load(struct kvm_vcpu *vcpu)
@@ -1529,13 +1535,14 @@ static bool timer_irqs_are_valid(struct kvm_vcpu *vcpu)
ctx = vcpu_get_timer(vcpu, i);
irq = timer_irq(ctx);
if (kvm_vgic_set_owner(vcpu, irq, ctx))
break;
/* With GICv5, the default PPI is what you get -- nothing else */
if (vgic_is_v5(vcpu->kvm) && irq != get_vgic_ppi(vcpu->kvm, default_ppi[i]))
break;
if (kvm_vgic_set_owner(vcpu, irq, ctx))
break;
/*
* We know by construction that we only have PPIs, so all values
* are less than 32 for non-GICv5 VGICs. On GICv5, they are

View File

@@ -465,6 +465,7 @@ int kvm_vm_ioctl_check_extension(struct kvm *kvm, long ext)
r = get_num_wrps();
break;
case KVM_CAP_ARM_PMU_V3:
case KVM_CAP_ARM_PMU_V3_STRICT:
r = kvm_supports_guest_pmuv3();
break;
case KVM_CAP_ARM_INJECT_SERROR_ESR:
@@ -748,6 +749,10 @@ void kvm_arch_vcpu_put(struct kvm_vcpu *vcpu)
if (is_protected_kvm_enabled()) {
kvm_call_hyp(__vgic_v3_save_aprs, &vcpu->arch.vgic_cpu.vgic_v3);
kvm_call_hyp_nvhe(__pkvm_vcpu_put);
/* __pkvm_vcpu_put implies a sync of the state */
if (!kvm_vm_is_protected(vcpu->kvm))
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
}
kvm_vcpu_put_debug(vcpu);
@@ -979,6 +984,9 @@ int kvm_arch_vcpu_run_pid_change(struct kvm_vcpu *vcpu)
return ret;
if (is_protected_kvm_enabled()) {
/* Start with the vcpu in a dirty state */
if (!kvm_vm_is_protected(vcpu->kvm))
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
ret = pkvm_create_hyp_vm(kvm);
if (ret)
return ret;
@@ -1576,8 +1584,10 @@ static unsigned long system_supported_vcpu_features(void)
if (!cpus_have_final_cap(ARM64_HAS_32BIT_EL1))
clear_bit(KVM_ARM_VCPU_EL1_32BIT, &features);
if (!kvm_supports_guest_pmuv3())
if (!kvm_supports_guest_pmuv3()) {
clear_bit(KVM_ARM_VCPU_PMU_V3, &features);
clear_bit(KVM_ARM_VCPU_PMU_V3_STRICT, &features);
}
if (!system_supports_sve())
clear_bit(KVM_ARM_VCPU_SVE, &features);
@@ -1618,6 +1628,11 @@ static int kvm_vcpu_init_check_features(struct kvm_vcpu *vcpu,
test_bit(KVM_ARM_VCPU_PTRAUTH_GENERIC, &features))
return -EINVAL;
/* Strict PMUv3 UAPI requires PMUv3. */
if (test_bit(KVM_ARM_VCPU_PMU_V3_STRICT, &features) &&
!test_bit(KVM_ARM_VCPU_PMU_V3, &features))
return -EINVAL;
if (!test_bit(KVM_ARM_VCPU_EL1_32BIT, &features))
return 0;
@@ -1647,10 +1662,13 @@ static int kvm_setup_vcpu(struct kvm_vcpu *vcpu)
int ret = 0;
/*
* When the vCPU has a PMU, but no PMU is set for the guest
* yet, set the default one.
* When the vCPU has a PMU, but no PMU is set for the guest yet, set
* the default one. If KVM_ARM_VCPU_PMU_V3_STRICT is set, no default
* PMU is created, and userspace must select a PMU via
* KVM_ARM_VCPU_PMU_V3_SET_PMU.
*/
if (kvm_vcpu_has_pmu(vcpu) && !kvm->arch.arm_pmu)
if (kvm_vcpu_has_pmu(vcpu) && !kvm->arch.arm_pmu &&
!kvm_vcpu_has_pmuv3_strict(vcpu))
ret = kvm_arm_set_default_pmu(kvm);
/* Prepare for nested if required */

View File

@@ -11,8 +11,6 @@
#include <asm/kvm_mmu.h>
#include <asm/lsui.h>
#define S1_MMU_DISABLED (-127)
static int get_ia_size(struct s1_walk_info *wi)
{
return 64 - wi->txsz;

View File

@@ -225,6 +225,7 @@ struct reg_feat_map_desc {
#define FEAT_HCX ID_AA64MMFR1_EL1, HCX, IMP
#define FEAT_S2PIE ID_AA64MMFR3_EL1, S2PIE, IMP
#define FEAT_GCIE ID_AA64PFR2_EL1, GCIE, IMP
#define FEAT_NV3 ID_AA64MMFR4_EL1, NV_frac, NV3
static bool not_feat_aa64el3(struct kvm *kvm)
{
@@ -904,6 +905,12 @@ static const DECLARE_FEAT_MAP_FGT(hdfgwtr2_desc, hdfgwtr2_masks,
static const struct reg_bits_to_feat_map hcrx_feat_map[] = {
NEEDS_FEAT(HCRX_EL2_NVTGE |
HCRX_EL2_NVnTTLB |
HCRX_EL2_NVnTTLBIS |
HCRX_EL2_NVnTTLBOS,
FEAT_NV3),
NEEDS_FEAT(HCRX_EL2_SRMASKEn, FEAT_SRMASK),
NEEDS_FEAT(HCRX_EL2_PACMEn, feat_pauth_lr),
NEEDS_FEAT(HCRX_EL2_EnFPM, FEAT_FPMR),
NEEDS_FEAT(HCRX_EL2_GCSEn, FEAT_GCS),
@@ -930,10 +937,12 @@ static const struct reg_bits_to_feat_map hcrx_feat_map[] = {
NEEDS_FEAT(HCRX_EL2_EnASR, FEAT_LS64_V),
NEEDS_FEAT(HCRX_EL2_EnALS, FEAT_LS64),
NEEDS_FEAT(HCRX_EL2_EnAS0, FEAT_LS64_ACCDATA),
FORCE_RES0(HCRX_EL2_RES0),
FORCE_RES1(HCRX_EL2_RES1),
};
static const DECLARE_FEAT_MAP(hcrx_desc, __HCRX_EL2,
static const DECLARE_FEAT_MAP(hcrx_desc, HCRX_EL2,
hcrx_feat_map, FEAT_HCX);
static const struct reg_bits_to_feat_map hcr_feat_map[] = {
@@ -1010,6 +1019,9 @@ static const struct reg_bits_to_feat_map hcr_feat_map[] = {
static const DECLARE_FEAT_MAP(hcr_desc, HCR_EL2,
hcr_feat_map, FEAT_AA64EL2);
static const DECLARE_FEAT_MAP(nvhcr_desc, NVHCR_EL2,
hcr_feat_map, FEAT_NV3);
static const struct reg_bits_to_feat_map sctlr2_feat_map[] = {
NEEDS_FEAT(SCTLR2_EL1_NMEA |
SCTLR2_EL1_EASE,
@@ -1384,6 +1396,7 @@ void __init check_feature_map(void)
check_reg_desc(&hdfgwtr2_desc);
check_reg_desc(&hcrx_desc);
check_reg_desc(&hcr_desc);
check_reg_desc(&nvhcr_desc);
check_reg_desc(&sctlr2_desc);
check_reg_desc(&tcr2_el2_desc);
check_reg_desc(&sctlr_el1_desc);
@@ -1579,11 +1592,21 @@ struct resx get_reg_fixed_bits(struct kvm *kvm, enum vcpu_sysreg reg)
break;
case HCRX_EL2:
resx = compute_reg_resx_bits(kvm, &hcrx_desc, 0, 0);
resx.res1 |= __HCRX_EL2_RES1;
break;
case HCR_EL2:
resx = compute_reg_resx_bits(kvm, &hcr_desc, 0, 0);
break;
case NVHCR_EL2:
/*
* Only apply sanitisation if we do have FEAT_NV3.
* Otherwise, the register aliases with HCR_EL2 in VNCR,
* and we're better off relying on data transfers between
* NVHCR_EL2 and HCR_EL2 to sanitise things.
*/
resx = (kvm_has_nv3(kvm) ?
compute_reg_resx_bits(kvm, &nvhcr_desc, 0, 0) :
(typeof(resx)){});
break;
case SCTLR2_EL1:
case SCTLR2_EL2:
resx = compute_reg_resx_bits(kvm, &sctlr2_desc, 0, 0);

View File

@@ -136,6 +136,8 @@ enum cgt_group_id {
CGT_CPTR_TTA,
CGT_MDCR_HPMN,
CGT_HCR_NV_HCRX_nNVTGE,
/* Must be last */
__NR_CGT_GROUP_IDS__
};
@@ -588,6 +590,15 @@ static enum trap_behaviour check_mdcr_hpmn(struct kvm_vcpu *vcpu)
return BEHAVE_HANDLE_LOCALLY;
}
static enum trap_behaviour check_hcr_nv_hcrx_nnvtge(struct kvm_vcpu *vcpu)
{
if ((__vcpu_sys_reg(vcpu, HCR_EL2) & HCR_EL2_NV) &&
!(__vcpu_sys_reg(vcpu, HCRX_EL2) & HCRX_EL2_NVTGE))
return BEHAVE_FORWARD_RW;
return BEHAVE_HANDLE_LOCALLY;
}
#define CCC(id, fn) \
[id - __COMPLEX_CONDITIONS__] = fn
@@ -598,6 +609,7 @@ static const complex_condition_check ccc[] = {
CCC(CGT_CNTHCTL_EL1NVVCT, check_cnthctl_el1nvvct),
CCC(CGT_CPTR_TTA, check_cptr_tta),
CCC(CGT_MDCR_HPMN, check_mdcr_hpmn),
CCC(CGT_HCR_NV_HCRX_nNVTGE, check_hcr_nv_hcrx_nnvtge),
};
/*
@@ -853,6 +865,7 @@ static const struct encoding_to_trap_config encoding_to_cgt[] __initconst = {
SR_TRAP(SYS_SCTLR2_EL2, CGT_HCR_NV),
SR_RANGE_TRAP(SYS_HCR_EL2,
SYS_HCRX_EL2, CGT_HCR_NV),
SR_TRAP(SYS_NVHCR_EL2, CGT_HCR_NV_HCRX_nNVTGE),
SR_TRAP(SYS_SMPRIMAP_EL2, CGT_HCR_NV),
SR_TRAP(SYS_SMCR_EL2, CGT_HCR_NV),
SR_RANGE_TRAP(SYS_TTBR0_EL2,
@@ -2320,7 +2333,6 @@ int __init populate_nv_trap_config(void)
BUILD_BUG_ON(__NR_CGT_GROUP_IDS__ > BIT(TC_CGT_BITS));
BUILD_BUG_ON(__NR_FGT_GROUP_IDS__ > BIT(TC_FGT_BITS));
BUILD_BUG_ON(__NR_FG_FILTER_IDS__ > BIT(TC_FGF_BITS));
BUILD_BUG_ON(__HCRX_EL2_MASK & __HCRX_EL2_nMASK);
for (int i = 0; i < ARRAY_SIZE(encoding_to_cgt); i++) {
const struct encoding_to_trap_config *cgt = &encoding_to_cgt[i];
@@ -2346,10 +2358,6 @@ int __init populate_nv_trap_config(void)
}
}
if (__HCRX_EL2_RES0 != HCRX_EL2_RES0)
kvm_info("Sanitised HCR_EL2_RES0 = %016llx, expecting %016llx\n",
__HCRX_EL2_RES0, HCRX_EL2_RES0);
kvm_info("nv: %ld coarse grained trap handlers\n",
ARRAY_SIZE(encoding_to_cgt));

View File

@@ -486,9 +486,32 @@ int handle_exit(struct kvm_vcpu *vcpu, int exception_index)
}
}
static void handle_exit_pkvm_state(struct kvm_vcpu *vcpu, int exception_index)
{
int exception_code = ARM_EXCEPTION_CODE(exception_index);
if (!is_protected_kvm_enabled() || kvm_vm_is_protected(vcpu->kvm))
return;
/*
* Sync the context back when the host will read (trap) or write
* (SError) it. Preempt-off here, so the loaded hyp vCPU is stable.
*/
if (exception_code == ARM_EXCEPTION_TRAP ||
exception_code == ARM_EXCEPTION_EL1_SERROR ||
ARM_SERROR_PENDING(exception_index)) {
kvm_call_hyp_nvhe(__pkvm_vcpu_sync_state);
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
} else {
vcpu_clear_flag(vcpu, PKVM_HOST_STATE_DIRTY);
}
}
/* For exit types that need handling before we can be preempted */
void handle_exit_early(struct kvm_vcpu *vcpu, int exception_index)
{
handle_exit_pkvm_state(vcpu, exception_index);
if (ARM_SERROR_PENDING(exception_index)) {
if (this_cpu_has_cap(ARM64_HAS_RAS_EXTN)) {
u64 disr = kvm_vcpu_get_disr(vcpu);
@@ -507,10 +530,20 @@ void handle_exit_early(struct kvm_vcpu *vcpu, int exception_index)
kvm_handle_guest_serror(vcpu, kvm_vcpu_get_esr(vcpu));
}
static bool nvhe_hyp_panic_host_s2_disabled(void)
{
return !is_protected_kvm_enabled() ||
IS_ENABLED(CONFIG_PKVM_DISABLE_STAGE2_ON_PANIC);
}
static void print_nvhe_hyp_panic(const char *name, u64 panic_addr)
{
kvm_err("nVHE hyp %s at: [<%016llx>] %pB!\n", name, panic_addr,
(void *)(panic_addr + kaslr_offset()));
/* Kallsyms might not be mapped in the host stage-2 */
if (nvhe_hyp_panic_host_s2_disabled())
kvm_err("nVHE hyp %s at: [<%016llx>] %pB!\n", name, panic_addr,
(void *)(panic_addr + kaslr_offset()));
else
kvm_err("nVHE hyp %s at: %016llx!\n", name, panic_addr);
}
static void kvm_nvhe_report_cfi_failure(u64 panic_addr)
@@ -538,8 +571,7 @@ void __noreturn __cold nvhe_hyp_panic_handler(u64 esr, u64 spsr,
unsigned int line = 0;
/* All hyp bugs, including warnings, are treated as fatal. */
if (!is_protected_kvm_enabled() ||
IS_ENABLED(CONFIG_PKVM_DISABLE_STAGE2_ON_PANIC)) {
if (nvhe_hyp_panic_host_s2_disabled()) {
struct bug_entry *bug = find_bug(elr_in_kimg);
if (bug)

View File

@@ -20,22 +20,6 @@
#error Hypervisor code only!
#endif
static inline u64 __vcpu_read_sys_reg(const struct kvm_vcpu *vcpu, int reg)
{
if (has_vhe())
return vcpu_read_sys_reg(vcpu, reg);
return __vcpu_sys_reg(vcpu, reg);
}
static inline void __vcpu_write_sys_reg(struct kvm_vcpu *vcpu, u64 val, int reg)
{
if (has_vhe())
vcpu_write_sys_reg(vcpu, val, reg);
else
__vcpu_assign_sys_reg(vcpu, reg, val);
}
static void __vcpu_write_spsr(struct kvm_vcpu *vcpu, unsigned long target_mode,
u64 val)
{
@@ -101,14 +85,14 @@ static void enter_exception64(struct kvm_vcpu *vcpu, unsigned long target_mode,
switch (target_mode) {
case PSR_MODE_EL1h:
vbar = __vcpu_read_sys_reg(vcpu, VBAR_EL1);
sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL1);
__vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL1);
vbar = vcpu_read_sys_reg(vcpu, VBAR_EL1);
sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL1);
vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL1);
break;
case PSR_MODE_EL2h:
vbar = __vcpu_read_sys_reg(vcpu, VBAR_EL2);
sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL2);
__vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL2);
vbar = vcpu_read_sys_reg(vcpu, VBAR_EL2);
sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL2);
vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL2);
break;
default:
/* Don't do that */
@@ -185,7 +169,7 @@ static void enter_exception64(struct kvm_vcpu *vcpu, unsigned long target_mode,
*/
static unsigned long get_except32_cpsr(struct kvm_vcpu *vcpu, u32 mode)
{
u32 sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL1);
u32 sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL1);
unsigned long old, new;
old = *vcpu_cpsr(vcpu);
@@ -281,7 +265,7 @@ static void enter_exception32(struct kvm_vcpu *vcpu, u32 mode, u32 vect_offset)
{
unsigned long spsr = *vcpu_cpsr(vcpu);
bool is_thumb = (spsr & PSR_AA32_T_BIT);
u32 sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL1);
u32 sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL1);
u32 return_address;
*vcpu_cpsr(vcpu) = get_except32_cpsr(vcpu, mode);
@@ -305,7 +289,7 @@ static void enter_exception32(struct kvm_vcpu *vcpu, u32 mode, u32 vect_offset)
if (sctlr & (1 << 13))
vect_offset += 0xffff0000;
else /* always have security exceptions */
vect_offset += __vcpu_read_sys_reg(vcpu, VBAR_EL1);
vect_offset += vcpu_read_sys_reg(vcpu, VBAR_EL1);
*vcpu_pc(vcpu) = vect_offset;
}

View File

@@ -108,9 +108,10 @@ static inline void __activate_cptr_traps_vhe(struct kvm_vcpu *vcpu)
* The architecture is a bit crap (what a surprise): an EL2 guest
* writing to CPTR_EL2 via CPACR_EL1 can't set any of TCPAC or TTA,
* as they are RES0 in the guest's view. To work around it, trap the
* sucker using the very same bit it can't set...
* sucker using the very same bit it can't set. FEAT_NV2p1 fixes it.
*/
if (vcpu_el2_e2h_is_set(vcpu) && is_hyp_ctxt(vcpu))
if (!cpus_have_final_cap(ARM64_HAS_NV2P1) &&
vcpu_el2_e2h_is_set(vcpu) && is_hyp_ctxt(vcpu))
val |= CPTR_EL2_TCPAC;
/*
@@ -325,6 +326,24 @@ static inline void __deactivate_traps_mpam(void)
write_sysreg_s(MPAMHCR_HOST_FLAGS, SYS_MPAMHCR_EL2);
}
/*
* Just like for HCR_EL2, we can't let the guest mess with some of the
* basics we rely on in HCRX_EL2. However, the major difference is that
* HCRX_EL2 only affects EL1, and never EL2 (sudden outburst of sanity, I
* guess). So it is always the guest inflicting it on its own guestx.
*
* Things we don't want to let the guest control are:
*
* - TMEA: That's for us to decide how an SEA is routed, not the guest.
*
* - PTTWI: Similarly, it is for us to decide whether Reduced Coherency for
* the PTW is a thing. It really isn't.
*
* - EnIDCP128: We don't allow IMPDEF sysregs -- full stop.
*/
#define NV_HCRX_GUEST_EXCLUDE (HCRX_EL2_TMEA | HCRX_EL2_PTTWI | \
HCRX_EL2_EnIDCP128)
static inline void __activate_traps_common(struct kvm_vcpu *vcpu)
{
struct kvm_cpu_context *hctxt = host_data_ptr(host_ctxt);
@@ -350,8 +369,8 @@ static inline void __activate_traps_common(struct kvm_vcpu *vcpu)
u64 hcrx = vcpu->arch.hcrx_el2;
if (is_nested_ctxt(vcpu)) {
u64 val = __vcpu_sys_reg(vcpu, HCRX_EL2);
hcrx |= val & __HCRX_EL2_MASK;
hcrx &= ~(~val & __HCRX_EL2_nMASK);
hcrx |= (val & ~NV_HCRX_GUEST_EXCLUDE);
hcrx &= ~(~val & ~NV_HCRX_GUEST_EXCLUDE);
}
ctxt_sys_reg(hctxt, HCRX_EL2) = read_sysreg_s(SYS_HCRX_EL2);
@@ -706,22 +725,9 @@ static inline bool handle_tx2_tvm(struct kvm_vcpu *vcpu)
return true;
}
/* Open-coded version of timer_get_offset() to allow for kern_hyp_va() */
static inline u64 hyp_timer_get_offset(struct arch_timer_context *ctxt)
{
u64 offset = 0;
if (ctxt->offset.vm_offset)
offset += *kern_hyp_va(ctxt->offset.vm_offset);
if (ctxt->offset.vcpu_offset)
offset += *kern_hyp_va(ctxt->offset.vcpu_offset);
return offset;
}
static inline u64 compute_counter_value(struct arch_timer_context *ctxt)
{
return arch_timer_read_cntpct_el0() - hyp_timer_get_offset(ctxt);
return arch_timer_read_cntpct_el0() - timer_get_offset(ctxt);
}
static bool kvm_handle_cntxct(struct kvm_vcpu *vcpu)

View File

@@ -172,6 +172,10 @@ static inline void __sysreg_save_el1_state(struct kvm_cpu_context *ctxt)
if (ctxt_has_sctlr2(ctxt))
ctxt_sys_reg(ctxt, SCTLR2_EL1) = read_sysreg_el1(SYS_SCTLR2);
/* Retrieve L2's HCR_EL2, and save it for future use */
if (is_nested_nv3_ctxt(ctxt_to_vcpu(ctxt)))
ctxt_sys_reg(ctxt, NVHCR_EL2) = read_sysreg_s(SYS_NVHCR_EL2);
}
static inline void __sysreg_save_el2_return_state(struct kvm_cpu_context *ctxt)
@@ -285,6 +289,13 @@ static inline void __sysreg_restore_el1_state(struct kvm_cpu_context *ctxt,
if (ctxt_has_sctlr2(ctxt))
write_sysreg_el1(ctxt_sys_reg(ctxt, SCTLR2_EL1), SYS_SCTLR2);
/*
* Publish the L2 view of HCR_EL2 to the HW if L1 is using NV3.
* Otherwise, the data is already in place in the L1's own VNCR.
*/
if (is_nested_nv3_ctxt(ctxt_to_vcpu(ctxt)))
write_sysreg_s(ctxt_sys_reg(ctxt, NVHCR_EL2), SYS_NVHCR_EL2);
}
/* Read the VCPU state's PSTATE, but translate (v)EL2 to EL1. */

View File

@@ -6,11 +6,11 @@
#include <asm/kvm_hyp.h>
#ifdef CONFIG_NVHE_EL2_TRACING
void trace_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc);
u64 trace_clock(void);
void trace_hyp_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc);
u64 trace_hyp_clock(void);
#else
static inline void
trace_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc) { }
static inline u64 trace_clock(void) { return 0; }
trace_hyp_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc) { }
static inline u64 trace_hyp_clock(void) { return 0; }
#endif
#endif

View File

@@ -30,7 +30,7 @@ static u64 __clock_mult_uint128(u64 cyc, u32 mult, u32 shift)
}
/* Does not guarantee no reader on the modified bank. */
void trace_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc)
void trace_hyp_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc)
{
struct clock_data *clock = &trace_clock_data;
u64 bank = clock->cur ^ 1;
@@ -48,7 +48,7 @@ void trace_clock_update(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc)
}
/* Use untrusted host data */
u64 trace_clock(void)
u64 trace_hyp_clock(void)
{
struct clock_data *clock = &trace_clock_data;
u64 bank = smp_load_acquire(&clock->cur);

View File

@@ -106,16 +106,19 @@ SYM_CODE_START_LOCAL(___kvm_hyp_init)
and x2, x1, x2
cbz x2, 1f
// hVHE: Replay the EL2 setup to account for the E2H bit
// TPIDR_EL2 is used to preserve x0 across the macro maze...
/*
* hVHE: Replay the EL2 setup to account for the E2H bit
* TPIDR_EL2 and FAR_EL2 are used to preserve x0 and LR across
* the macro maze...
*/
isb
msr tpidr_el2, x0
str lr, [x0, #NVHE_INIT_TMP]
msr far_el2, lr
bl __kvm_init_el2_state
mrs lr, far_el2
mrs x0, tpidr_el2
ldr lr, [x0, #NVHE_INIT_TMP]
1:
ldr x1, [x0, #NVHE_INIT_TPIDR_EL2]

View File

@@ -7,6 +7,8 @@
#include <hyp/adjust_pc.h>
#include <hyp/switch.h>
#include <linux/irqchip/arm-gic-v3.h>
#include <asm/pgtable-types.h>
#include <asm/kvm_asm.h>
#include <asm/kvm_emulate.h>
@@ -102,16 +104,103 @@ static void fpsimd_sve_sync(struct kvm_vcpu *vcpu)
*host_data_ptr(fp_owner) = FP_STATE_HOST_OWNED;
}
static void flush_hyp_vgic_state(struct pkvm_hyp_vcpu *hyp_vcpu)
{
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
struct vgic_v3_cpu_if *host_cpu_if, *hyp_cpu_if;
unsigned int used_lrs, i;
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
used_lrs = host_cpu_if->used_lrs;
used_lrs = min(used_lrs, hyp_gicv3_nr_lr);
hyp_cpu_if->vgic_hcr = host_cpu_if->vgic_hcr;
/* Should be a one-off */
hyp_cpu_if->vgic_sre = (ICC_SRE_EL1_DIB |
ICC_SRE_EL1_DFB |
ICC_SRE_EL1_SRE);
hyp_cpu_if->used_lrs = used_lrs;
for (i = 0; i < used_lrs; i++)
hyp_cpu_if->vgic_lr[i] = host_cpu_if->vgic_lr[i];
}
static void sync_hyp_vgic_state(struct pkvm_hyp_vcpu *hyp_vcpu)
{
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
struct vgic_v3_cpu_if *host_cpu_if, *hyp_cpu_if;
unsigned int i;
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
host_cpu_if->vgic_hcr = hyp_cpu_if->vgic_hcr;
host_cpu_if->vgic_vmcr = hyp_cpu_if->vgic_vmcr;
for (i = 0; i < hyp_cpu_if->used_lrs; i++)
host_cpu_if->vgic_lr[i] = hyp_cpu_if->vgic_lr[i];
}
static void __copy_vcpu_state(const struct kvm_vcpu *from_vcpu,
struct kvm_vcpu *to_vcpu)
{
int i;
to_vcpu->arch.ctxt.regs = from_vcpu->arch.ctxt.regs;
to_vcpu->arch.ctxt.spsr_abt = from_vcpu->arch.ctxt.spsr_abt;
to_vcpu->arch.ctxt.spsr_und = from_vcpu->arch.ctxt.spsr_und;
to_vcpu->arch.ctxt.spsr_irq = from_vcpu->arch.ctxt.spsr_irq;
to_vcpu->arch.ctxt.spsr_fiq = from_vcpu->arch.ctxt.spsr_fiq;
to_vcpu->arch.ctxt.fp_regs = from_vcpu->arch.ctxt.fp_regs;
/*
* Copy the sysregs, but don't mess with the timer state which
* is directly handled by EL1 and is expected to be preserved.
* enum vcpu_sysreg is sparse: VNCR-mapped registers take values
* derived from their VNCR page offset, so the timer registers do
* not form a contiguous numeric range and must be skipped by name.
*/
for (i = 1; i < NR_SYS_REGS; i++) {
switch (i) {
case CNTVOFF_EL2:
case CNTV_CVAL_EL0:
case CNTV_CTL_EL0:
case CNTP_CVAL_EL0:
case CNTP_CTL_EL0:
continue;
}
to_vcpu->arch.ctxt.sys_regs[i] = from_vcpu->arch.ctxt.sys_regs[i];
}
}
static void sync_hyp_vcpu_state(struct pkvm_hyp_vcpu *hyp_vcpu)
{
__copy_vcpu_state(&hyp_vcpu->vcpu, hyp_vcpu->host_vcpu);
}
static void flush_hyp_vcpu_state(struct pkvm_hyp_vcpu *hyp_vcpu)
{
__copy_vcpu_state(hyp_vcpu->host_vcpu, &hyp_vcpu->vcpu);
}
static void flush_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu)
{
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
hyp_vcpu->vcpu.arch.debug_owner = host_vcpu->arch.debug_owner;
if (kvm_guest_owns_debug_regs(&hyp_vcpu->vcpu))
if (kvm_guest_owns_debug_regs(&hyp_vcpu->vcpu)) {
hyp_vcpu->vcpu.arch.vcpu_debug_state = host_vcpu->arch.vcpu_debug_state;
else if (kvm_host_owns_debug_regs(&hyp_vcpu->vcpu))
} else if (kvm_host_owns_debug_regs(&hyp_vcpu->vcpu)) {
hyp_vcpu->vcpu.arch.external_debug_state = host_vcpu->arch.external_debug_state;
/*
* The world switch loads MDSCR_EL1 from external_mdscr_el1
* (ctxt_mdscr_el1()).
*/
hyp_vcpu->vcpu.arch.external_mdscr_el1 = host_vcpu->arch.external_mdscr_el1;
}
}
static void sync_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu)
@@ -131,7 +220,17 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
fpsimd_sve_flush();
flush_debug_state(hyp_vcpu);
hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt;
/*
* If we deal with a non-protected guest and the state is potentially
* dirty (from a host perspective), copy the state back into the hyp
* vcpu.
*/
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
if (vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY))
flush_hyp_vcpu_state(hyp_vcpu);
} else {
hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt;
}
/* __hyp_running_vcpu must be NULL in a guest context. */
hyp_vcpu->vcpu.arch.ctxt.__hyp_running_vcpu = NULL;
@@ -150,13 +249,7 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
hyp_vcpu->vcpu.arch.vsesr_el2 = host_vcpu->arch.vsesr_el2;
hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3 = host_vcpu->arch.vgic_cpu.vgic_v3;
/* Bound used_lrs by the number of implemented list registers. */
hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3.used_lrs =
min_t(unsigned int,
hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3.used_lrs,
hyp_gicv3_nr_lr);
flush_hyp_vgic_state(hyp_vcpu);
hyp_vcpu->vcpu.arch.pid = host_vcpu->arch.pid;
}
@@ -164,25 +257,26 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
static void sync_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
{
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
struct vgic_v3_cpu_if *hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
struct vgic_v3_cpu_if *host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
unsigned int i;
fpsimd_sve_sync(&hyp_vcpu->vcpu);
sync_debug_state(hyp_vcpu);
host_vcpu->arch.ctxt = hyp_vcpu->vcpu.arch.ctxt;
host_vcpu->arch.hcr_el2 = hyp_vcpu->vcpu.arch.hcr_el2;
if (pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
host_vcpu->arch.ctxt = hyp_vcpu->vcpu.arch.ctxt;
} else {
/*
* PC feeds trace_kvm_exit(), PSTATE.SS the host software-step
* machine, and both run before the next on-demand ctxt sync.
*/
host_vcpu->arch.ctxt.regs.pc = hyp_vcpu->vcpu.arch.ctxt.regs.pc;
host_vcpu->arch.ctxt.regs.pstate = hyp_vcpu->vcpu.arch.ctxt.regs.pstate;
}
host_vcpu->arch.fault = hyp_vcpu->vcpu.arch.fault;
host_vcpu->arch.iflags = hyp_vcpu->vcpu.arch.iflags;
host_cpu_if->vgic_hcr = hyp_cpu_if->vgic_hcr;
host_cpu_if->vgic_vmcr = hyp_cpu_if->vgic_vmcr;
for (i = 0; i < hyp_cpu_if->used_lrs; ++i)
host_cpu_if->vgic_lr[i] = hyp_cpu_if->vgic_lr[i];
sync_hyp_vgic_state(hyp_vcpu);
}
static void handle___pkvm_vcpu_load(struct kvm_cpu_context *host_ctxt)
@@ -210,18 +304,78 @@ static void handle___pkvm_vcpu_put(struct kvm_cpu_context *host_ctxt)
{
struct pkvm_hyp_vcpu *hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
if (hyp_vcpu)
if (hyp_vcpu) {
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu) &&
!vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY)) {
sync_hyp_vcpu_state(hyp_vcpu);
}
pkvm_put_hyp_vcpu(hyp_vcpu);
}
}
static void handle___pkvm_vcpu_sync_state(struct kvm_cpu_context *host_ctxt)
{
struct pkvm_hyp_vcpu *hyp_vcpu;
hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
if (!hyp_vcpu || pkvm_hyp_vcpu_is_protected(hyp_vcpu))
return;
sync_hyp_vcpu_state(hyp_vcpu);
}
static struct kvm_vcpu *__get_host_hyp_vcpus(struct kvm_vcpu *arg,
struct pkvm_hyp_vcpu **hyp_vcpup)
{
struct kvm_vcpu *host_vcpu = kern_hyp_va(arg);
struct pkvm_hyp_vcpu *hyp_vcpu = NULL;
if (unlikely(is_protected_kvm_enabled())) {
hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
if (!hyp_vcpu || hyp_vcpu->host_vcpu != host_vcpu) {
hyp_vcpu = NULL;
host_vcpu = NULL;
}
}
*hyp_vcpup = hyp_vcpu;
return host_vcpu;
}
#define get_host_hyp_vcpus(ctxt, regnr, hyp_vcpup) \
({ \
DECLARE_REG(struct kvm_vcpu *, __vcpu, ctxt, regnr); \
__get_host_hyp_vcpus(__vcpu, hyp_vcpup); \
})
#define get_host_hyp_vcpus_from_vgic_v3_cpu_if(ctxt, regnr, hyp_vcpup) \
({ \
DECLARE_REG(struct vgic_v3_cpu_if *, cif, ctxt, regnr);\
struct kvm_vcpu *__vcpu = container_of(cif, \
struct kvm_vcpu, \
arch.vgic_cpu.vgic_v3); \
\
__get_host_hyp_vcpus(__vcpu, hyp_vcpup); \
})
static void handle___kvm_vcpu_run(struct kvm_cpu_context *host_ctxt)
{
DECLARE_REG(struct kvm_vcpu *, host_vcpu, host_ctxt, 1);
struct pkvm_hyp_vcpu *hyp_vcpu;
struct kvm_vcpu *host_vcpu;
int ret;
if (unlikely(is_protected_kvm_enabled())) {
struct pkvm_hyp_vcpu *hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
host_vcpu = get_host_hyp_vcpus(host_ctxt, 1, &hyp_vcpu);
if (!host_vcpu) {
ret = -EINVAL;
goto out;
}
if (unlikely(hyp_vcpu)) {
/*
* KVM (and pKVM) doesn't support SME guests for now, and
* ensures that SME features aren't enabled in pstate when
@@ -233,23 +387,16 @@ static void handle___kvm_vcpu_run(struct kvm_cpu_context *host_ctxt)
goto out;
}
if (!hyp_vcpu) {
ret = -EINVAL;
goto out;
}
flush_hyp_vcpu(hyp_vcpu);
ret = __kvm_vcpu_run(&hyp_vcpu->vcpu);
sync_hyp_vcpu(hyp_vcpu);
} else {
struct kvm_vcpu *vcpu = kern_hyp_va(host_vcpu);
/* The host is fully trusted, run its vCPU directly. */
fpsimd_lazy_switch_to_guest(vcpu);
ret = __kvm_vcpu_run(vcpu);
fpsimd_lazy_switch_to_host(vcpu);
fpsimd_lazy_switch_to_guest(host_vcpu);
ret = __kvm_vcpu_run(host_vcpu);
fpsimd_lazy_switch_to_host(host_vcpu);
}
out:
cpu_reg(host_ctxt, 1) = ret;
@@ -484,16 +631,63 @@ static void handle___vgic_v3_init_lrs(struct kvm_cpu_context *host_ctxt)
static void handle___vgic_v3_save_aprs(struct kvm_cpu_context *host_ctxt)
{
DECLARE_REG(struct vgic_v3_cpu_if *, cpu_if, host_ctxt, 1);
struct pkvm_hyp_vcpu *hyp_vcpu;
struct kvm_vcpu *host_vcpu;
__vgic_v3_save_aprs(kern_hyp_va(cpu_if));
host_vcpu = get_host_hyp_vcpus_from_vgic_v3_cpu_if(host_ctxt, 1,
&hyp_vcpu);
if (!host_vcpu)
return;
if (unlikely(hyp_vcpu)) {
struct vgic_v3_cpu_if *hyp_cpu_if, *host_cpu_if;
int i;
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
__vgic_v3_save_aprs(hyp_cpu_if);
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
host_cpu_if->vgic_vmcr = hyp_cpu_if->vgic_vmcr;
for (i = 0; i < ARRAY_SIZE(host_cpu_if->vgic_ap0r); i++) {
host_cpu_if->vgic_ap0r[i] = hyp_cpu_if->vgic_ap0r[i];
host_cpu_if->vgic_ap1r[i] = hyp_cpu_if->vgic_ap1r[i];
}
} else {
__vgic_v3_save_aprs(&host_vcpu->arch.vgic_cpu.vgic_v3);
}
}
static void handle___vgic_v3_restore_vmcr_aprs(struct kvm_cpu_context *host_ctxt)
{
DECLARE_REG(struct vgic_v3_cpu_if *, cpu_if, host_ctxt, 1);
struct pkvm_hyp_vcpu *hyp_vcpu;
struct kvm_vcpu *host_vcpu;
__vgic_v3_restore_vmcr_aprs(kern_hyp_va(cpu_if));
host_vcpu = get_host_hyp_vcpus_from_vgic_v3_cpu_if(host_ctxt, 1,
&hyp_vcpu);
if (!host_vcpu)
return;
if (unlikely(hyp_vcpu)) {
struct vgic_v3_cpu_if *hyp_cpu_if, *host_cpu_if;
int i;
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
hyp_cpu_if->vgic_vmcr = host_cpu_if->vgic_vmcr;
/* Should be a one-off */
hyp_cpu_if->vgic_sre = (ICC_SRE_EL1_DIB |
ICC_SRE_EL1_DFB |
ICC_SRE_EL1_SRE);
for (i = 0; i < ARRAY_SIZE(host_cpu_if->vgic_ap0r); i++) {
hyp_cpu_if->vgic_ap0r[i] = host_cpu_if->vgic_ap0r[i];
hyp_cpu_if->vgic_ap1r[i] = host_cpu_if->vgic_ap1r[i];
}
__vgic_v3_restore_vmcr_aprs(hyp_cpu_if);
} else {
__vgic_v3_restore_vmcr_aprs(&host_vcpu->arch.vgic_cpu.vgic_v3);
}
}
static void handle___pkvm_init(struct kvm_cpu_context *host_ctxt)
@@ -761,6 +955,7 @@ static const hcall_t host_hcall[] = {
HANDLE_FUNC(__pkvm_finalize_teardown_vm),
HANDLE_FUNC(__pkvm_vcpu_load),
HANDLE_FUNC(__pkvm_vcpu_put),
HANDLE_FUNC(__pkvm_vcpu_sync_state),
HANDLE_FUNC(__pkvm_tlb_flush_vmid),
};

View File

@@ -261,11 +261,18 @@ static void __apply_guest_page(void *va, size_t size,
static void clean_dcache_guest_page(void *va, size_t size)
{
/* See comment in __clean_dcache_guest_page() */
if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB))
return;
__apply_guest_page(va, size, __clean_dcache_guest_page);
}
static void invalidate_icache_guest_page(void *va, size_t size)
{
if (alternative_has_cap_unlikely(ARM64_HAS_CACHE_DIC))
return;
__apply_guest_page(va, size, __invalidate_icache_guest_page);
}

View File

@@ -433,7 +433,6 @@ static void init_pkvm_hyp_vm(struct kvm *host_kvm, struct pkvm_hyp_vm *hyp_vm,
hyp_vm->host_kvm = host_kvm;
hyp_vm->kvm.created_vcpus = nr_vcpus;
hyp_vm->kvm.arch.pkvm.is_protected = READ_ONCE(host_kvm->arch.pkvm.is_protected);
hyp_vm->kvm.arch.pkvm.is_created = true;
hyp_vm->kvm.arch.flags = 0;
pkvm_init_features_from_host(hyp_vm, host_kvm);
@@ -529,6 +528,20 @@ static int init_pkvm_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu,
hyp_vcpu->vcpu.arch.cflags = READ_ONCE(host_vcpu->arch.cflags);
hyp_vcpu->vcpu.arch.mp_state.mp_state = KVM_MP_STATE_STOPPED;
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
/*
* Timer offsets are pointing to the untrusted KVM copy,
* which is pinned in __pkvm_init_vm() for the VM life time.
* It is worth noting that hyp_vm->host_kvm points to an EL2
* linear map address and timer_get_offset() will use
* kern_hyp_va() which is safe as it is idempotent.
*/
vcpu_vtimer(&hyp_vcpu->vcpu)->offset.vm_offset =
&hyp_vm->host_kvm->arch.timer_data.voffset;
vcpu_ptimer(&hyp_vcpu->vcpu)->offset.vm_offset =
&hyp_vm->host_kvm->arch.timer_data.poffset;
}
ret = pkvm_vcpu_init_sysregs(hyp_vcpu);
if (ret)
goto done;

View File

@@ -257,10 +257,11 @@ static void inject_sync64(struct kvm_vcpu *vcpu, u64 esr)
*vcpu_cpsr(vcpu) = read_sysreg_el2(SYS_SPSR);
/*
* Make sure we have the latest update to VBAR_EL1, as pKVM
* handles traps very early, before sysregs are resync'ed
* Sync VBAR_EL1 and SCTLR_EL1, both read by enter_exception64(),
* as pKVM handles traps before sysregs are resync'ed.
*/
__vcpu_assign_sys_reg(vcpu, VBAR_EL1, read_sysreg_el1(SYS_VBAR));
__vcpu_assign_sys_reg(vcpu, SCTLR_EL1, read_sysreg_el1(SYS_SCTLR));
kvm_pend_exception(vcpu, EXCEPT_AA64_EL1_SYNC);

View File

@@ -45,11 +45,11 @@ void __timer_enable_traps(struct kvm_vcpu *vcpu)
/*
* Disallow physical timer access for the guest
* Physical counter access is allowed if no offset is enforced
* or running protected (we don't offset anything in this case).
* or running a protected VM (we don't offset anything in this case).
*/
clr = CNTHCTL_EL1PCEN;
if (is_protected_kvm_enabled() ||
!kern_hyp_va(vcpu->kvm)->arch.timer_data.poffset)
if (vcpu_is_protected(vcpu) ||
!timer_get_offset(vcpu_ptimer(vcpu)))
set |= CNTHCTL_EL1PCTEN;
else
clr |= CNTHCTL_EL1PCTEN;
@@ -61,9 +61,9 @@ void __timer_enable_traps(struct kvm_vcpu *vcpu)
/*
* Trap the virtual counter/timer if we have a broken cntvoff
* implementation.
* implementation and non zero offset as in timer_set_traps()
*/
if (has_broken_cntvoff())
if (has_broken_cntvoff() && timer_get_offset(vcpu_vtimer(vcpu)))
set |= CNTHCTL_EL1TVT | CNTHCTL_EL1TVCT;
sysreg_clear_set(cnthctl_el2, clr, set);

View File

@@ -35,7 +35,7 @@ static bool hyp_trace_buffer_loaded(struct hyp_trace_buffer *trace_buffer)
void *tracing_reserve_entry(unsigned long length)
{
return simple_ring_buffer_reserve(this_cpu_ptr(trace_buffer.simple_rbs), length,
trace_clock());
trace_hyp_clock());
}
void tracing_commit_entry(void)
@@ -290,7 +290,7 @@ void __tracing_update_clock(u32 mult, u32 shift, u64 epoch_ns, u64 epoch_cyc)
}
/* ...we can now override the old one and swap. */
trace_clock_update(mult, shift, epoch_ns, epoch_cyc);
trace_hyp_clock_update(mult, shift, epoch_ns, epoch_cyc);
}
int __tracing_reset(unsigned int cpu)

View File

@@ -16,9 +16,9 @@
#include "../../vgic/vgic.h"
#define vtr_to_max_lr_idx(v) ((v) & 0xf)
#define vtr_to_nr_pre_bits(v) ((((u32)(v) >> 26) & 7) + 1)
#define vtr_to_nr_apr_regs(v) (1 << (vtr_to_nr_pre_bits(v) - 5))
#define vtr_to_max_lr_idx(v) FIELD_GET(ICH_VTR_EL2_ListRegs, (v))
#define vtr_to_nr_pre_bits(v) (FIELD_GET(ICH_VTR_EL2_PREbits, (v)) + 1)
#define vtr_to_nr_apr_regs(v) BIT(vtr_to_nr_pre_bits(v) - 5)
u64 __gic_v3_get_lr(unsigned int lr)
{
@@ -367,7 +367,7 @@ void __vgic_v3_save_aprs(struct vgic_v3_cpu_if *cpu_if)
u64 val;
u32 nr_pre_bits;
val = read_gicreg(ICH_VTR_EL2);
val = vgic_ich_vtr();
nr_pre_bits = vtr_to_nr_pre_bits(val);
switch (nr_pre_bits) {
@@ -400,7 +400,7 @@ static void __vgic_v3_restore_aprs(struct vgic_v3_cpu_if *cpu_if)
u64 val;
u32 nr_pre_bits;
val = read_gicreg(ICH_VTR_EL2);
val = vgic_ich_vtr();
nr_pre_bits = vtr_to_nr_pre_bits(val);
switch (nr_pre_bits) {
@@ -430,33 +430,19 @@ static void __vgic_v3_restore_aprs(struct vgic_v3_cpu_if *cpu_if)
void __vgic_v3_init_lrs(void)
{
int max_lr_idx = vtr_to_max_lr_idx(read_gicreg(ICH_VTR_EL2));
int max_lr_idx = vtr_to_max_lr_idx(vgic_ich_vtr());
int i;
for (i = 0; i <= max_lr_idx; i++)
__gic_v3_set_lr(0, i);
}
/*
* Return the GIC CPU configuration:
* - [31:0] ICH_VTR_EL2
* - [62:32] RES0
* - [63] MMIO (GICv2) capable
*/
u64 __vgic_v3_get_gic_config(void)
/* Return true if GICv3 is MMIO (GICv2) capable, false otherwise */
bool __vgic_v3_get_gic_config(void)
{
u64 val, sre;
unsigned long flags = 0;
/*
* In compat mode, we cannot access ICC_SRE_EL1 at any EL
* other than EL1 itself; just return the
* ICH_VTR_EL2. ICC_IDR0_EL1 is only implemented on a GICv5
* system, so we first check if we have GICv5 support.
*/
if (cpus_have_final_cap(ARM64_HAS_GICV5_CPUIF))
return read_gicreg(ICH_VTR_EL2);
sre = read_gicreg(ICC_SRE_EL1);
/*
* To check whether we have a MMIO-based (GICv2 compatible)
@@ -497,10 +483,7 @@ u64 __vgic_v3_get_gic_config(void)
isb();
}
val = (val & ICC_SRE_EL1_SRE) ? 0 : (1ULL << 63);
val |= read_gicreg(ICH_VTR_EL2);
return val;
return !(val & ICC_SRE_EL1_SRE);
}
static void __vgic_v3_compat_mode_enable(void)
@@ -540,7 +523,7 @@ void __vgic_v3_restore_vmcr_aprs(struct vgic_v3_cpu_if *cpu_if)
static int __vgic_v3_bpr_min(void)
{
/* See Pseudocode for VPriorityGroup */
return 8 - vtr_to_nr_pre_bits(read_gicreg(ICH_VTR_EL2));
return 8 - vtr_to_nr_pre_bits(vgic_ich_vtr());
}
static int __vgic_v3_get_group(struct kvm_vcpu *vcpu)
@@ -614,7 +597,7 @@ static int __vgic_v3_find_active_lr(struct kvm_vcpu *vcpu, int intid,
static int __vgic_v3_get_highest_active_priority(void)
{
u8 nr_apr_regs = vtr_to_nr_apr_regs(read_gicreg(ICH_VTR_EL2));
u8 nr_apr_regs = vtr_to_nr_apr_regs(vgic_ich_vtr());
u32 hap = 0;
int i;
@@ -707,7 +690,7 @@ static void __vgic_v3_set_active_priority(u8 pri, u32 vmcr, int grp)
static int __vgic_v3_clear_highest_active_priority(void)
{
u8 nr_apr_regs = vtr_to_nr_apr_regs(read_gicreg(ICH_VTR_EL2));
u8 nr_apr_regs = vtr_to_nr_apr_regs(vgic_ich_vtr());
u32 hap = 0;
int i;
@@ -1039,7 +1022,7 @@ static void __vgic_v3_read_ctlr(struct kvm_vcpu *vcpu, u32 vmcr, int rt)
{
u32 vtr, val;
vtr = read_gicreg(ICH_VTR_EL2);
vtr = vgic_ich_vtr();
/* PRIbits */
val = ((vtr >> 29) & 7) << ICC_CTLR_EL1_PRI_BITS_SHIFT;
/* IDbits */

View File

@@ -70,6 +70,12 @@ static u64 __compute_hcr(struct kvm_vcpu *vcpu)
if (!vcpu_el2_e2h_is_set(vcpu))
hcr |= HCR_NV1;
/* Publish the guest's view of HCR_EL2 to the HW */
if (cpus_have_final_cap(ARM64_HAS_NV3) && vcpu_el2_e2h_is_set(vcpu))
write_sysreg_s(__vcpu_sys_reg(vcpu, HCR_EL2), SYS_NVHCR_EL2);
else
__vcpu_assign_sys_reg(vcpu, NVHCR_EL2, __vcpu_sys_reg(vcpu, HCR_EL2));
/*
* Nothing in HCR_EL2 should impact running in hypervisor
* context, apart from bits we have defined as RESx (E2H,
@@ -339,18 +345,24 @@ static bool kvm_hyp_handle_eret(struct kvm_vcpu *vcpu, u64 *exit_code)
u64 esr = kvm_vcpu_get_esr(vcpu);
u64 spsr, elr, mode;
/* With NV3, the fast path is handled in HW */
if (cpus_have_final_cap(ARM64_HAS_NV3) && vcpu_el2_e2h_is_set(vcpu))
return false;
/*
* Going through the whole put/load motions is a waste of time
* if this is a VHE guest hypervisor returning to its own
* userspace, or the hypervisor performing a local exception
* return. No need to save/restore registers, no need to
* switch S2 MMU. Just do the canonical ERET.
* switch S2 MMU. Just do the canonical ERET unless we are in
* nested context.
*
* Unless the trap has to be forwarded further down the line,
* of course...
* Note that this is made possible because KVM itself never traps
* ERET when running an L2. The consequence is that any ERET trap is
* the result of HCR_EL2 or HFGITR_EL2 programming by L1 for its own
* guest, and the exception must be forwarded to L1.
*/
if ((__vcpu_sys_reg(vcpu, HCR_EL2) & HCR_NV) ||
(__vcpu_sys_reg(vcpu, HFGITR_EL2) & HFGITR_EL2_ERET))
if (is_nested_ctxt(vcpu))
return false;
spsr = read_sysreg_el1(SYS_SPSR);
@@ -424,11 +436,15 @@ static bool kvm_hyp_handle_tlbi_el2(struct kvm_vcpu *vcpu, u64 *exit_code)
return false;
/*
* If we have to check for any VNCR mapping being invalidated,
* go back to the slow path for further processing.
* If we have to check for any VNCR TLB being invalidated, go back
* to the slow path for further processing.
*
* The synchronisation betweem TLBI and walk is provided by the
* speculative increment of the TLB counter on walk, and the
* invalidation counter. Yes, this is fiddly.
*/
if (vcpu_el2_e2h_is_set(vcpu) && vcpu_el2_tge_is_set(vcpu) &&
atomic_read(&vcpu->kvm->arch.vncr_map_count))
atomic_read(&vcpu->kvm->arch.vncr_tlb_count))
return false;
__kvm_skip_instr(vcpu);
@@ -441,6 +457,9 @@ static bool kvm_hyp_handle_cpacr_el1(struct kvm_vcpu *vcpu, u64 *exit_code)
u64 esr = kvm_vcpu_get_esr(vcpu);
int rt;
if (cpus_have_final_cap(ARM64_HAS_NV2P1))
return false;
if (!is_hyp_ctxt(vcpu) || esr_sys64_to_sysreg(esr) != SYS_CPACR_EL1)
return false;
@@ -534,19 +553,17 @@ static const exit_handler_fn hyp_exit_handlers[] = {
[0x3F] = kvm_hyp_handle_impdef,
};
static inline bool fixup_guest_exit(struct kvm_vcpu *vcpu, u64 *exit_code)
static void fixup_nv_guest_exit(struct kvm_vcpu *vcpu)
{
synchronize_vcpu_pstate(vcpu);
/*
* If we were in HYP context on entry, adjust the PSTATE view
* so that the usual helpers work correctly. This enforces our
* invariant that the guest's HYP context status is preserved
* across a run.
*/
if (vcpu_has_nv(vcpu) &&
unlikely(host_data_test_flag(VCPU_IN_HYP_CONTEXT))) {
if (unlikely(host_data_test_flag(VCPU_IN_HYP_CONTEXT))) {
u64 mode = *vcpu_cpsr(vcpu) & (PSR_MODE_MASK | PSR_MODE32_BIT);
u64 hcr;
switch (mode) {
case PSR_MODE_EL1t:
@@ -559,11 +576,26 @@ static inline bool fixup_guest_exit(struct kvm_vcpu *vcpu, u64 *exit_code)
*vcpu_cpsr(vcpu) &= ~(PSR_MODE_MASK | PSR_MODE32_BIT);
*vcpu_cpsr(vcpu) |= mode;
/* Publish the latest HCR_EL2 to the emulation */
hcr = (cpus_have_final_cap(ARM64_HAS_NV3) &&
vcpu_el2_e2h_is_set(vcpu)) ?
read_sysreg_s(SYS_NVHCR_EL2) :
__vcpu_sys_reg(vcpu, NVHCR_EL2);
__vcpu_assign_sys_reg(vcpu, HCR_EL2, hcr);
}
/* Apply extreme paranoia! */
BUG_ON(vcpu_has_nv(vcpu) &&
!!host_data_test_flag(VCPU_IN_HYP_CONTEXT) != is_hyp_ctxt(vcpu));
BUG_ON(!!host_data_test_flag(VCPU_IN_HYP_CONTEXT) != is_hyp_ctxt(vcpu));
}
static bool fixup_guest_exit(struct kvm_vcpu *vcpu, u64 *exit_code)
{
synchronize_vcpu_pstate(vcpu);
if (vcpu_has_nv(vcpu))
fixup_nv_guest_exit(vcpu);
return __fixup_guest_exit(vcpu, exit_code, hyp_exit_handlers);
}

View File

@@ -42,10 +42,12 @@ static void __sysreg_save_vel2_state(struct kvm_vcpu *vcpu)
u64 val;
/*
* We don't save CPTR_EL2, as accesses to CPACR_EL1
* are always trapped, ensuring that the in-memory
* copy is always up-to-date. A small blessing...
* Without FEAT_NV2p1, we don't save CPTR_EL2, as accesses
* to CPACR_EL1 are always trapped, ensuring that the
* in-memory copy is always up-to-date. A small blessing...
*/
if (cpus_have_final_cap(ARM64_HAS_NV2P1))
__vcpu_assign_sys_reg(vcpu, CPTR_EL2, read_sysreg_el1(SYS_CPACR));
__vcpu_assign_sys_reg(vcpu, SCTLR_EL2, read_sysreg_el1(SYS_SCTLR));
__vcpu_assign_sys_reg(vcpu, TTBR0_EL2, read_sysreg_el1(SYS_TTBR0));
__vcpu_assign_sys_reg(vcpu, TTBR1_EL2, read_sysreg_el1(SYS_TTBR1));
@@ -67,11 +69,18 @@ static void __sysreg_save_vel2_state(struct kvm_vcpu *vcpu)
* The EL1 view of CNTKCTL_EL1 has a bunch of RES0 bits where
* the interesting CNTHCTL_EL2 bits live. So preserve these
* bits when reading back the guest-visible value.
*
* While NV2p1 fixes some of that, it makes CNTHCTL_EL2.ECV
* even more broken than it already was with NV2.
*/
val = read_sysreg_el1(SYS_CNTKCTL);
val &= CNTKCTL_VALID_BITS;
__vcpu_rmw_sys_reg(vcpu, CNTHCTL_EL2, &=, ~CNTKCTL_VALID_BITS);
__vcpu_rmw_sys_reg(vcpu, CNTHCTL_EL2, |=, val);
if (!cpus_have_final_cap(ARM64_HAS_NV2P1)) {
val &= CNTKCTL_VALID_BITS;
__vcpu_rmw_sys_reg(vcpu, CNTHCTL_EL2, &=, ~CNTKCTL_VALID_BITS);
__vcpu_rmw_sys_reg(vcpu, CNTHCTL_EL2, |=, val);
} else {
__vcpu_assign_sys_reg(vcpu, CNTHCTL_EL2, val);
}
}
__vcpu_assign_sys_reg(vcpu, SP_EL2, read_sysreg(sp_el1));

View File

@@ -2113,11 +2113,14 @@ static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd)
* Permission faults just need to update the existing leaf entry,
* and so normally don't require allocations from the memcache. The
* only exception to this is when dirty logging is enabled at runtime
* and a write fault needs to collapse a block entry into a table.
* and a fault needs to collapse a block entry into a table.
* Under pKVM a permission fault can also collapse pages into a block,
* which needs a fresh mapping object, and the hypervisor requires the
* min-pages memcache even when the install allocates nothing.
*/
memcache = get_mmu_memcache(s2fd->vcpu);
if (!perm_fault || (memslot_is_logging(s2fd->memslot) &&
kvm_is_write_fault(s2fd->vcpu))) {
if (!perm_fault || memslot_is_logging(s2fd->memslot) ||
is_protected_kvm_enabled()) {
ret = topup_mmu_memcache(s2fd->vcpu, memcache);
if (ret)
return ret;

View File

@@ -16,6 +16,7 @@
#include <asm/sysreg.h>
#include "sys_regs.h"
#include "vgic/vgic.h"
struct vncr_tlb {
/* The guest's VNCR_EL2 */
@@ -27,7 +28,7 @@ struct vncr_tlb {
bool hpa_writable;
/* -1 when not mapped on a CPU */
int cpu;
atomic_t cpu;
/*
* true if the TLB is valid. Can only be changed with the
@@ -48,7 +49,7 @@ void kvm_init_nested(struct kvm *kvm)
{
kvm->arch.nested_mmus = NULL;
kvm->arch.nested_mmus_size = 0;
atomic_set(&kvm->arch.vncr_map_count, 0);
atomic_set(&kvm->arch.vncr_tlb_count, 0);
}
static int init_nested_s2_mmu(struct kvm *kvm, struct kvm_s2_mmu *mmu)
@@ -506,7 +507,7 @@ int kvm_walk_nested_s2(struct kvm_vcpu *vcpu, phys_addr_t gipa,
return ret;
}
static unsigned int ttl_to_size(u8 ttl)
static unsigned int __ttl_to_size(u8 ttl)
{
int level = ttl & 3;
int gran = (ttl >> 2) & 3;
@@ -562,10 +563,22 @@ static unsigned int ttl_to_size(u8 ttl)
return max_size;
}
static u8 pgshift_level_to_ttl(u16 shift, u8 level)
static unsigned int ttl_to_size(u8 ttl)
{
return __ttl_to_size(ttl) ?: SZ_1G;
}
static u8 pgshift_level_to_ttl(u16 shift, s8 level)
{
u8 ttl;
/*
* If we don't have a proper level, fallback to the maximum
* size.
*/
if (level < 0)
return 0;
switch(shift) {
case 12:
ttl = TLBI_TTL_TG_4K;
@@ -676,7 +689,11 @@ unsigned long compute_tlb_inval_range(struct kvm_s2_mmu *mmu, u64 val)
ttl = get_guest_mapping_ttl(mmu, addr);
}
max_size = ttl_to_size(ttl);
/*
* Don't use the default 1GB fallback, as we can adapt to the
* max mapping size we allow at S2.
*/
max_size = __ttl_to_size(ttl);
if (!max_size) {
/* Compute the maximum extent of the invalidation */
@@ -879,18 +896,41 @@ void kvm_vcpu_load_hw_mmu(struct kvm_vcpu *vcpu)
}
}
/*
* Unmapping an L1 VNCR can happen concurrently without the mmu lock being
* effective (vcpu_put() vs TLBI handling). The atomic_xchg below ensures
* that only one CPU sets it to -1 while getting a valid CPU number back.
*/
static int unmap_l1_vncr(struct vncr_tlb *vt)
{
int cpu = atomic_xchg_relaxed(&vt->cpu, -1);
if (cpu != -1)
clear_fixmap(vncr_fixmap(cpu));
return cpu;
}
static void this_cpu_reset_vncr_fixmap(struct kvm_vcpu *vcpu)
{
if (!host_data_test_flag(L1_VNCR_MAPPED))
return;
BUG_ON(vcpu->arch.vncr_tlb->cpu != smp_processor_id());
BUG_ON(is_hyp_ctxt(vcpu));
clear_fixmap(vncr_fixmap(vcpu->arch.vncr_tlb->cpu));
vcpu->arch.vncr_tlb->cpu = -1;
/*
* Unconditionally unmap the local VNCR if we have lost the race
* against a concurrent TLBI. Otherwise we could end-up running
* another vcpu with VNCR still mapped if the TLBI thread is
* preempted between the exchange and the clear_fixmap().
*
* Note that we do not care about the TLBI nuking the fixmap behind
* the back of an running vcpu. This will only generate a fault and
* possibly a retranslation.
*/
if (unmap_l1_vncr(vcpu->arch.vncr_tlb) == -1)
clear_fixmap(vncr_fixmap(smp_processor_id()));
host_data_clear_flag(L1_VNCR_MAPPED);
atomic_dec(&vcpu->kvm->arch.vncr_map_count);
}
void kvm_vcpu_put_hw_mmu(struct kvm_vcpu *vcpu)
@@ -978,11 +1018,26 @@ u16 get_asid_by_regime(struct kvm_vcpu *vcpu, enum trans_regime regime)
return asid;
}
static void invalidate_vncr(struct vncr_tlb *vt)
static void invalidate_vncr(struct kvm *kvm, struct vncr_tlb *vt)
{
BUG_ON(!vt->valid);
vt->valid = false;
if (vt->cpu != -1)
clear_fixmap(vncr_fixmap(vt->cpu));
unmap_l1_vncr(vt);
atomic_dec(&kvm->arch.vncr_tlb_count);
}
static bool vncr_tlb_intersects(struct vncr_tlb *vt, u64 addr,
u64 scope_start, u64 scope_size)
{
u64 tlb_size, tlb_start, tlb_end, scope_end;
tlb_size = ttl_to_size(pgshift_level_to_ttl(vt->wi.pgshift, vt->wr.level));
tlb_start = addr & ~(tlb_size - 1);
tlb_end = tlb_start + tlb_size - 1;
scope_end = scope_start + scope_size - 1;
return !(tlb_end < scope_start || tlb_start > scope_end);
}
/*
@@ -1007,19 +1062,15 @@ static void kvm_invalidate_vncr_ipa(struct kvm *kvm, u64 start, u64 end)
if (!kvm_has_feat(kvm, ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY))
return;
kvm_for_each_vncr_tlb(i, vcpu, vt, kvm) {
u64 ipa_start, ipa_end, ipa_size;
ipa_size = ttl_to_size(pgshift_level_to_ttl(vt->wi.pgshift,
vt->wr.level));
ipa_start = vt->wr.pa & ~(ipa_size - 1);
ipa_end = ipa_start + ipa_size;
if (ipa_end <= start || ipa_start >= end)
continue;
invalidate_vncr(vt);
}
/*
* Note that invalidating the VNCR on the back of an MMU notifier
* doesn't require messing with the invalidation counter for a
* parallel walk. The notifier itself will have bumped the counter,
* making sure we rewalk.
*/
kvm_for_each_vncr_tlb(i, vcpu, vt, kvm)
if (vncr_tlb_intersects(vt, vt->wr.pa, start, end - start))
invalidate_vncr(kvm, vt);
}
struct s1e2_tlbi_scope {
@@ -1044,29 +1095,29 @@ static void invalidate_vncr_va(struct kvm *kvm,
lockdep_assert_held_write(&kvm->mmu_lock);
/*
* We might be performing a parallel S1 walk, so bump up the
* invalidation counter even in the absence of an actual VNCR TLB
* invalidation, as this could indicate that the guest has gone
* through a BBM sequence.
*/
kvm->mmu_invalidate_seq++;
smp_wmb();
kvm_for_each_vncr_tlb(i, vcpu, vt, kvm) {
u64 va_start, va_end, va_size;
va_size = ttl_to_size(pgshift_level_to_ttl(vt->wi.pgshift,
vt->wr.level));
va_start = vt->gva & ~(va_size - 1);
va_end = va_start + va_size;
switch (scope->type) {
case TLBI_ALL:
break;
case TLBI_VA:
if (va_end <= scope->va ||
va_start >= (scope->va + scope->size))
if (!vncr_tlb_intersects(vt, vt->gva, scope->va, scope->size))
continue;
if (vt->wr.nG && vt->wr.asid != scope->asid)
continue;
break;
case TLBI_VAA:
if (va_end <= scope->va ||
va_start >= (scope->va + scope->size))
if (!vncr_tlb_intersects(vt, vt->gva, scope->va, scope->size))
continue;
break;
@@ -1076,7 +1127,7 @@ static void invalidate_vncr_va(struct kvm *kvm,
break;
}
invalidate_vncr(vt);
invalidate_vncr(kvm, vt);
}
}
@@ -1126,8 +1177,6 @@ static void compute_s1_tlbi_range(struct kvm_vcpu *vcpu, u32 inst, u64 val,
case OP_TLBI_VALE1OSNXS:
scope->type = TLBI_VA;
scope->size = ttl_to_size(FIELD_GET(TLBI_TTL_MASK, val));
if (!scope->size)
scope->size = SZ_1G;
scope->va = tlbi_va_s1_to_va(val) & ~(scope->size - 1);
scope->asid = FIELD_GET(TLBIR_ASID_MASK, val);
break;
@@ -1154,8 +1203,6 @@ static void compute_s1_tlbi_range(struct kvm_vcpu *vcpu, u32 inst, u64 val,
case OP_TLBI_VAALE1OSNXS:
scope->type = TLBI_VAA;
scope->size = ttl_to_size(FIELD_GET(TLBI_TTL_MASK, val));
if (!scope->size)
scope->size = SZ_1G;
scope->va = tlbi_va_s1_to_va(val) & ~(scope->size - 1);
break;
case OP_TLBI_RVAE2:
@@ -1316,13 +1363,20 @@ void kvm_arch_flush_shadow_all(struct kvm *kvm)
* intersects with the TLBI request, invalidate it, and unmap the page
* from the fixmap. Because we need to look at all the vcpu-private TLBs,
* this requires some wide-ranging locking to ensure that nothing races
* against it. This may require some refcounting to avoid the search when
* no such TLB is present.
* against it. This requires some refcounting to avoid the search when
* no such TLB is present (see below).
*
* - On MMU notifiers, we must invalidate our TLB in a similar way, but
* looking at the IPA instead. The funny part is that there may not be a
* stage-2 mapping for this page if L1 hasn't accessed it using LD/ST
* instructions.
*
* - vncr_tlb_count tracks the number of valid VNCR TLBs VM-wide. This isn't
* the number of *mapped* L1 VNCR pages, which is likely be a subset (and
* by definition, a TLBI handled from L1 runs with the canonical VNCR
* page, not the L1's). The innermost trap handling code checks this to
* find out whether to return to the guest ASAP (no L1 TLBs) or to visit
* this part of the world for some extra invalidation work.
*/
int kvm_vcpu_allocate_vncr_tlb(struct kvm_vcpu *vcpu)
@@ -1377,7 +1431,8 @@ static int kvm_translate_vncr(struct kvm_vcpu *vcpu, bool *is_gmem)
*/
scoped_guard(write_lock, &vcpu->kvm->mmu_lock) {
this_cpu_reset_vncr_fixmap(vcpu);
vt->valid = false;
if (vt->valid)
invalidate_vncr(vcpu->kvm, vt);
vt->wi = (struct s1_walk_info) {
.regime = TR_EL20,
@@ -1391,15 +1446,15 @@ static int kvm_translate_vncr(struct kvm_vcpu *vcpu, bool *is_gmem)
va = read_vncr_el2(vcpu);
mmu_seq = vcpu->kvm->mmu_invalidate_seq;
smp_rmb();
ret = __kvm_translate_va(vcpu, &vt->wi, &vt->wr, va);
if (ret)
return ret;
write_fault = kvm_is_write_fault(vcpu);
mmu_seq = vcpu->kvm->mmu_invalidate_seq;
smp_rmb();
gfn = vt->wr.pa >> PAGE_SHIFT;
memslot = gfn_to_memslot(vcpu->kvm, gfn);
if (!memslot) {
@@ -1447,7 +1502,7 @@ static int kvm_translate_vncr(struct kvm_vcpu *vcpu, bool *is_gmem)
vt->hpa = pfn << PAGE_SHIFT;
vt->hpa_writable = writable;
vt->valid = true;
vt->cpu = -1;
atomic_set(&vt->cpu, -1);
kvm_make_request(KVM_REQ_MAP_L1_VNCR_EL2, vcpu);
kvm_release_faultin_page(vcpu->kvm, page, false, vt->wr.pw && vt->hpa_writable);
@@ -1502,7 +1557,20 @@ int kvm_handle_vncr_abort(struct kvm_vcpu *vcpu)
return -EIO;
}
/*
* Speculatively increment the TLB count to make sure concurrent
* TLBIs will take the slow path, and will interact with the retry
* mechanism. Drop it again on error.
*/
atomic_inc(&vcpu->kvm->arch.vncr_tlb_count);
smp_mb__after_atomic();
ret = kvm_translate_vncr(vcpu, &is_gmem);
if (ret) {
smp_mb__before_atomic();
atomic_dec(&vcpu->kvm->arch.vncr_tlb_count);
}
switch (ret) {
case -EAGAIN:
/* Let's try again... */
@@ -1568,14 +1636,16 @@ static void kvm_map_l1_vncr(struct kvm_vcpu *vcpu)
if (!vt->valid)
return;
/* We cache the MMU state in the TLB. Check that it matches. */
if (!!(vcpu_read_sys_reg(vcpu, SCTLR_EL2) & SCTLR_ELx_M) != s1_walk_translated(&vt->wr))
return;
if (read_vncr_el2(vcpu) != vt->gva)
return;
if (vt->wr.nG && get_asid_by_regime(vcpu, TR_EL20) != vt->wr.asid)
return;
vt->cpu = smp_processor_id();
if (vt->hpa_writable && vt->wr.pw && vt->wr.pr)
prot = PAGE_KERNEL;
else if (vt->wr.pr)
@@ -1590,9 +1660,9 @@ static void kvm_map_l1_vncr(struct kvm_vcpu *vcpu)
* FIXME: WO doesn't work at all, need POE support in the kernel.
*/
if (pgprot_val(prot) != pgprot_val(PAGE_NONE)) {
__set_fixmap(vncr_fixmap(vt->cpu), vt->hpa, prot);
atomic_set(&vt->cpu, smp_processor_id());
__set_fixmap(vncr_fixmap(atomic_read(&vt->cpu)), vt->hpa, prot);
host_data_set_flag(L1_VNCR_MAPPED);
atomic_inc(&vcpu->kvm->arch.vncr_map_count);
}
}
@@ -1728,7 +1798,7 @@ u64 limit_nv_id_reg(struct kvm *kvm, u32 reg, u64 val)
* You get EITHER
*
* - FEAT_VHE without FEAT_E2H0
* - FEAT_NV limited to FEAT_NV2
* - FEAT_NV limited to FEAT_NV2(p1)/NV3
* - HCR_EL2.NV1 being RES0
*
* OR
@@ -1740,7 +1810,13 @@ u64 limit_nv_id_reg(struct kvm *kvm, u32 reg, u64 val)
if (test_bit(KVM_ARM_VCPU_HAS_EL2_E2H0, kvm->arch.vcpu_features)) {
val = 0;
} else {
val = SYS_FIELD_PREP_ENUM(ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY);
val &= ID_AA64MMFR4_EL1_NV_frac;
if (cpus_have_final_cap(ARM64_HAS_NV3))
val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR4_EL1, NV_frac, NV3);
else if (cpus_have_final_cap(ARM64_HAS_NV2P1))
val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR4_EL1, NV_frac, NV2P1);
else
val = SYS_FIELD_PREP_ENUM(ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY);
val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR4_EL1, E2H0, NI_NV1);
}
break;
@@ -1826,6 +1902,10 @@ int kvm_init_nv_sysregs(struct kvm_vcpu *vcpu)
resx = get_reg_fixed_bits(kvm, HCR_EL2);
set_sysreg_masks(kvm, HCR_EL2, resx);
/* NVHCR_EL2 */
resx = get_reg_fixed_bits(kvm, NVHCR_EL2);
set_sysreg_masks(kvm, NVHCR_EL2, resx);
/* HCRX_EL2 */
resx = get_reg_fixed_bits(kvm, HCRX_EL2);
set_sysreg_masks(kvm, HCRX_EL2, resx);
@@ -1906,7 +1986,7 @@ int kvm_init_nv_sysregs(struct kvm_vcpu *vcpu)
/* ICH_HCR_EL2 */
resx.res0 = ICH_HCR_EL2_RES0;
resx.res1 = ICH_HCR_EL2_RES1;
if (!(kvm_vgic_global_state.ich_vtr_el2 & ICH_VTR_EL2_TDS))
if (!(vgic_ich_vtr() & ICH_VTR_EL2_TDS))
resx.res0 |= ICH_HCR_EL2_TDIR;
/* No GICv4 is presented to the guest */
resx.res0 |= ICH_HCR_EL2_DVIM | ICH_HCR_EL2_vSGIEOICount;

View File

@@ -185,7 +185,11 @@ static int __pkvm_create_hyp_vm(struct kvm *kvm)
bool pkvm_hyp_vm_is_created(struct kvm *kvm)
{
return READ_ONCE(kvm->arch.pkvm.is_created);
/*
* Serialised by config_lock/slots_lock, or by VM lifecycle at
* teardown, so a plain read suffices.
*/
return kvm->arch.pkvm.is_created;
}
int pkvm_create_hyp_vm(struct kvm *kvm)
@@ -230,13 +234,6 @@ int pkvm_init_host_vm(struct kvm *kvm, unsigned long type)
int ret;
bool protected = type & KVM_VM_TYPE_ARM_PROTECTED;
if (pkvm_hyp_vm_is_created(kvm))
return -EINVAL;
/* VM is already reserved, no need to proceed. */
if (kvm->arch.pkvm.handle)
return 0;
/* Reserve the VM in hyp and obtain a hyp handle for the VM. */
ret = kvm_call_hyp_nvhe(__pkvm_reserve_vm);
if (ret < 0)
@@ -369,7 +366,7 @@ static int __pkvm_pgtable_stage2_unshare(struct kvm_pgtable *pgt, u64 start, u64
for_each_mapping_in_range_safe(pgt, start, end, mapping) {
ret = kvm_call_hyp_nvhe(__pkvm_host_unshare_guest, handle, mapping->gfn,
mapping->nr_pages);
(u64)mapping->nr_pages);
if (WARN_ON(ret))
return ret;
pkvm_mapping_remove(mapping, &pgt->pkvm_mappings);
@@ -466,13 +463,14 @@ int pkvm_pgtable_stage2_map(struct kvm_pgtable *pgt, u64 addr, u64 size,
size / PAGE_SIZE, prot);
}
if (WARN_ON(ret))
if (ret)
return ret;
swap(mapping, cache->mapping);
mapping->gfn = gfn;
mapping->pfn = pfn;
mapping->nr_pages = size / PAGE_SIZE;
mapping->nc = !!(prot & (KVM_PGTABLE_PROT_DEVICE | KVM_PGTABLE_PROT_NORMAL_NC));
pkvm_mapping_insert(mapping, &pgt->pkvm_mappings);
return ret;
@@ -503,7 +501,7 @@ int pkvm_pgtable_stage2_wrprotect(struct kvm_pgtable *pgt, u64 addr, u64 size)
lockdep_assert_held(&kvm->mmu_lock);
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping) {
ret = kvm_call_hyp_nvhe(__pkvm_host_wrprotect_guest, handle, mapping->gfn,
mapping->nr_pages);
(u64)mapping->nr_pages);
if (WARN_ON(ret))
break;
}
@@ -517,9 +515,15 @@ int pkvm_pgtable_stage2_flush(struct kvm_pgtable *pgt, u64 addr, u64 size)
struct pkvm_mapping *mapping;
lockdep_assert_held(&kvm->mmu_lock);
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping)
__clean_dcache_guest_page(pfn_to_kaddr(mapping->pfn),
PAGE_SIZE * mapping->nr_pages);
if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB))
return 0;
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping) {
if (!mapping->nc)
__clean_dcache_guest_page(pfn_to_kaddr(mapping->pfn),
PAGE_SIZE * mapping->nr_pages);
}
return 0;
}
@@ -537,7 +541,7 @@ bool pkvm_pgtable_stage2_test_clear_young(struct kvm_pgtable *pgt, u64 addr, u64
lockdep_assert_held(&kvm->mmu_lock);
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping)
young |= kvm_call_hyp_nvhe(__pkvm_host_test_clear_young_guest, handle, mapping->gfn,
mapping->nr_pages, mkold);
(u64)mapping->nr_pages, mkold);
return young;
}

View File

@@ -838,9 +838,9 @@ static u64 __compute_pmceid(struct arm_pmu *pmu, bool pmceid1)
return ((u64)hi[pmceid1] << 32) | lo[pmceid1];
}
static u64 compute_pmceid0(struct arm_pmu *pmu)
static u64 compute_pmceid0(struct kvm_vcpu *vcpu)
{
u64 val = __compute_pmceid(pmu, 0);
u64 val = __compute_pmceid(vcpu->kvm->arch.arm_pmu, 0);
/* always support SW_INCR */
val |= BIT(ARMV8_PMUV3_PERFCTR_SW_INCR);
@@ -849,32 +849,33 @@ static u64 compute_pmceid0(struct arm_pmu *pmu)
return val;
}
static u64 compute_pmceid1(struct arm_pmu *pmu)
static u64 compute_pmceid1(struct kvm_vcpu *vcpu)
{
u64 val = __compute_pmceid(pmu, 1);
u64 val = __compute_pmceid(vcpu->kvm->arch.arm_pmu, 1);
/*
* Don't advertise STALL_SLOT*, as PMMIR_EL0 is handled
* as RAZ
* If KVM_ARM_VCPU_PMU_V3_STRICT is not set, PMMIR_EL1 is
* unconditionally RAZ, so don't advertise STALL_SLOT* events.
*/
val &= ~(BIT_ULL(ARMV8_PMUV3_PERFCTR_STALL_SLOT - 32) |
BIT_ULL(ARMV8_PMUV3_PERFCTR_STALL_SLOT_FRONTEND - 32) |
BIT_ULL(ARMV8_PMUV3_PERFCTR_STALL_SLOT_BACKEND - 32));
if (!kvm_vcpu_has_pmuv3_strict(vcpu))
val &= ~(BIT_ULL(ARMV8_PMUV3_PERFCTR_STALL_SLOT - 32) |
BIT_ULL(ARMV8_PMUV3_PERFCTR_STALL_SLOT_FRONTEND - 32) |
BIT_ULL(ARMV8_PMUV3_PERFCTR_STALL_SLOT_BACKEND - 32));
return val;
}
u64 kvm_pmu_get_pmceid(struct kvm_vcpu *vcpu, bool pmceid1)
{
struct arm_pmu *cpu_pmu = vcpu->kvm->arch.arm_pmu;
unsigned long *bmap = vcpu->kvm->arch.pmu_filter;
u64 val, mask = 0;
int base, i, nr_events;
if (!pmceid1) {
val = compute_pmceid0(cpu_pmu);
val = compute_pmceid0(vcpu);
base = 0;
} else {
val = compute_pmceid1(cpu_pmu);
val = compute_pmceid1(vcpu);
base = 32;
}
@@ -938,6 +939,10 @@ int kvm_arm_pmu_v3_enable(struct kvm_vcpu *vcpu)
static int kvm_arm_pmu_v3_init(struct kvm_vcpu *vcpu)
{
/* Only possible when using KVM_ARM_VCPU_PMU_V3_STRICT */
if (!vcpu->kvm->arch.arm_pmu)
return -ENXIO;
if (irqchip_in_kernel(vcpu->kvm)) {
int ret;
@@ -1008,6 +1013,14 @@ u8 kvm_arm_pmu_get_max_counters(struct kvm *kvm)
{
struct arm_pmu *arm_pmu = kvm->arch.arm_pmu;
/*
* Under KVM_ARM_VCPU_PMU_V3_STRICT no PMU exists until userspace sets
* one, so this can be reached before arm_pmu is set. Report no
* counters in that case.
*/
if (!arm_pmu)
return 0;
/*
* PMUv3 requires that all event counters are capable of counting any
* event, though the same may not be true of non-PMUv3 hardware.
@@ -1049,7 +1062,8 @@ static void kvm_arm_set_pmu(struct kvm *kvm, struct arm_pmu *arm_pmu)
}
/**
* kvm_arm_set_default_pmu - No PMU set, get the default one.
* kvm_arm_set_default_pmu - No PMU set and KVM_ARM_VCPU_PMU_V3_STRICT not
* set, get the default one.
* @kvm: The kvm pointer
*
* The observant among you will notice that the supported_cpus
@@ -1092,6 +1106,17 @@ static int kvm_arm_pmu_v3_set_pmu(struct kvm_vcpu *vcpu, int pmu_id)
kvm_arm_set_pmu(kvm, arm_pmu);
cpumask_copy(kvm->arch.supported_cpus, &arm_pmu->supported_cpus);
/*
* Since a specific PMU is explicitly selected,
* PMMIR_EL1.SLOTS is deterministic to the guest.
* If KVM_ARM_VCPU_PMU_V3_STRICT is set, snapshot
* the value to allow the guest to read it.
*/
if (kvm_vcpu_has_pmuv3_strict(vcpu))
kvm->arch.pmmir_slots =
FIELD_GET(ARMV8_PMU_SLOTS,
arm_pmu->reg_pmmir);
ret = 0;
break;
}
@@ -1178,6 +1203,9 @@ int kvm_arm_pmu_v3_set_attr(struct kvm_vcpu *vcpu, struct kvm_device_attr *attr)
if (kvm_vm_has_ran_once(kvm))
return -EBUSY;
if (!kvm->arch.arm_pmu)
return -ENXIO;
if (!kvm->arch.pmu_filter) {
kvm->arch.pmu_filter = bitmap_alloc(nr_events, GFP_KERNEL_ACCOUNT);
if (!kvm->arch.pmu_filter)

View File

@@ -21,16 +21,6 @@
* as described in ARM document number ARM DEN 0022A.
*/
#define AFFINITY_MASK(level) ~((0x1UL << ((level) * MPIDR_LEVEL_BITS)) - 1)
static unsigned long psci_affinity_mask(unsigned long affinity_level)
{
if (affinity_level <= 3)
return MPIDR_HWID_BITMASK & AFFINITY_MASK(affinity_level);
return 0;
}
static unsigned long kvm_psci_vcpu_suspend(struct kvm_vcpu *vcpu)
{
/*
@@ -51,12 +41,6 @@ static unsigned long kvm_psci_vcpu_suspend(struct kvm_vcpu *vcpu)
return PSCI_RET_SUCCESS;
}
static inline bool kvm_psci_valid_affinity(struct kvm_vcpu *vcpu,
unsigned long affinity)
{
return !(affinity & ~MPIDR_HWID_BITMASK);
}
static unsigned long kvm_psci_vcpu_on(struct kvm_vcpu *source_vcpu)
{
struct vcpu_reset_state *reset_state;
@@ -135,7 +119,7 @@ static unsigned long kvm_psci_vcpu_affinity_info(struct kvm_vcpu *vcpu)
return PSCI_RET_INVALID_PARAMS;
/* Determine target affinity mask */
target_affinity_mask = psci_affinity_mask(lowest_affinity_level);
target_affinity_mask = kvm_psci_affinity_mask(lowest_affinity_level);
if (!target_affinity_mask)
return PSCI_RET_INVALID_PARAMS;
@@ -220,18 +204,6 @@ static void kvm_psci_system_suspend(struct kvm_vcpu *vcpu)
run->exit_reason = KVM_EXIT_SYSTEM_EVENT;
}
static void kvm_psci_narrow_to_32bit(struct kvm_vcpu *vcpu)
{
int i;
/*
* Zero the input registers' upper 32 bits. They will be fully
* zeroed on exit, so we're fine changing them in place.
*/
for (i = 1; i < 4; i++)
vcpu_set_reg(vcpu, i, lower_32_bits(vcpu_get_reg(vcpu, i)));
}
static unsigned long kvm_psci_check_allowed_function(struct kvm_vcpu *vcpu, u32 fn)
{
/*

View File

@@ -34,18 +34,6 @@
static u32 __ro_after_init kvm_ipa_limit;
unsigned int __ro_after_init kvm_host_sve_max_vl;
/*
* ARMv8 Reset Values
*/
#define VCPU_RESET_PSTATE_EL1 (PSR_MODE_EL1h | PSR_A_BIT | PSR_I_BIT | \
PSR_F_BIT | PSR_D_BIT)
#define VCPU_RESET_PSTATE_EL2 (PSR_MODE_EL2h | PSR_A_BIT | PSR_I_BIT | \
PSR_F_BIT | PSR_D_BIT)
#define VCPU_RESET_PSTATE_SVC (PSR_AA32_MODE_SVC | PSR_AA32_A_BIT | \
PSR_AA32_I_BIT | PSR_AA32_F_BIT)
unsigned int __ro_after_init kvm_sve_max_vl;
int __init kvm_arm_init_sve(void)
@@ -191,7 +179,6 @@ void kvm_reset_vcpu(struct kvm_vcpu *vcpu)
{
struct vcpu_reset_state reset_state;
bool loaded;
u32 pstate;
spin_lock(&vcpu->arch.mp_state_lock);
reset_state = vcpu->arch.reset_state;
@@ -210,21 +197,8 @@ void kvm_reset_vcpu(struct kvm_vcpu *vcpu)
kvm_vcpu_reset_sve(vcpu);
}
if (vcpu_el1_is_32bit(vcpu))
pstate = VCPU_RESET_PSTATE_SVC;
else if (vcpu_has_nv(vcpu))
pstate = VCPU_RESET_PSTATE_EL2;
else
pstate = VCPU_RESET_PSTATE_EL1;
/* Reset core registers */
memset(vcpu_gp_regs(vcpu), 0, sizeof(*vcpu_gp_regs(vcpu)));
memset(&vcpu->arch.ctxt.fp_regs, 0, sizeof(vcpu->arch.ctxt.fp_regs));
vcpu->arch.ctxt.spsr_abt = 0;
vcpu->arch.ctxt.spsr_und = 0;
vcpu->arch.ctxt.spsr_irq = 0;
vcpu->arch.ctxt.spsr_fiq = 0;
vcpu_gp_regs(vcpu)->pstate = pstate;
kvm_reset_vcpu_core(vcpu);
/* Reset system registers */
kvm_reset_sys_regs(vcpu);
@@ -233,36 +207,8 @@ void kvm_reset_vcpu(struct kvm_vcpu *vcpu)
* Additional reset state handling that PSCI may have imposed on us.
* Must be done after all the sys_reg reset.
*/
if (reset_state.reset) {
unsigned long target_pc = reset_state.pc;
/* Gracefully handle Thumb2 entry point */
if (vcpu_mode_is_32bit(vcpu) && (target_pc & 1)) {
target_pc &= ~1UL;
vcpu_set_thumb(vcpu);
}
/* Propagate caller endianness */
if (reset_state.be)
kvm_vcpu_set_be(vcpu);
*vcpu_pc(vcpu) = target_pc;
/*
* We may come from a state where either a PC update was
* pending (SMC call resulting in PC being increpented to
* skip the SMC) or a pending exception. Make sure we get
* rid of all that, as this cannot be valid out of reset.
*
* Note that clearing the exception mask also clears PC
* updates, but that's an implementation detail, and we
* really want to make it explicit.
*/
vcpu_clear_flag(vcpu, PENDING_EXCEPTION);
vcpu_clear_flag(vcpu, EXCEPT_MASK);
vcpu_clear_flag(vcpu, INCREMENT_PC);
vcpu_set_reg(vcpu, 0, reset_state.r0);
}
if (reset_state.reset)
kvm_reset_vcpu_psci(vcpu, &reset_state);
/* Reset timer */
kvm_timer_vcpu_reset(vcpu);

View File

@@ -183,8 +183,6 @@ static void locate_register(const struct kvm_vcpu *vcpu, enum vcpu_sysreg reg,
switch (reg) {
MAPPED_EL2_SYSREG(SCTLR_EL2, SCTLR_EL1,
translate_sctlr_el2_to_sctlr_el1 );
MAPPED_EL2_SYSREG(CPTR_EL2, CPACR_EL1,
translate_cptr_el2_to_cpacr_el1 );
MAPPED_EL2_SYSREG(TTBR0_EL2, TTBR0_EL1,
translate_ttbr0_el2_to_ttbr0_el1 );
MAPPED_EL2_SYSREG(TTBR1_EL2, TTBR1_EL1, NULL );
@@ -210,6 +208,33 @@ static void locate_register(const struct kvm_vcpu *vcpu, enum vcpu_sysreg reg,
loc->loc = ((is_hyp_ctxt(vcpu) && vcpu_el2_e2h_is_set(vcpu)) ?
SR_LOC_SPECIAL : SR_LOC_MEMORY);
break;
case CPTR_EL2:
/*
* CPTR_EL2 is just as special, and needs a certain amount
* of handholding. It always lives in memory, due to being
* heavily trapped thanks to CPACR_EL1.TCPAC being RES0.
* FEAT_NV2p1 fixes this.
*/
locate_mapped_el2_register(vcpu, CPTR_EL2, CPACR_EL1,
translate_cptr_el2_to_cpacr_el1,
loc);
if (is_hyp_ctxt(vcpu) && vcpu_el2_e2h_is_set(vcpu))
loc->loc = SR_LOC_SPECIAL;
break;
case NVHCR_EL2:
/*
* Yes, NVHCR_EL2 maps to itself when loaded in nested
* context. If you feel like the architecture is double
* backing on itself upside down, you're not alone.
*/
WARN_ON_ONCE(!kvm_has_nv3(vcpu->kvm));
if (is_hyp_ctxt(vcpu)) {
loc->loc = SR_LOC_MEMORY;
} else {
loc->loc = SR_LOC_LOADED | SR_LOC_MAPPED;
loc->map_reg = NVHCR_EL2;
}
break;
default:
loc->loc = locate_direct_register(vcpu, reg);
}
@@ -249,6 +274,7 @@ static u64 read_sr_from_cpu(enum vcpu_sysreg reg)
case DACR32_EL2: val = read_sysreg_s(SYS_DACR32_EL2); break;
case IFSR32_EL2: val = read_sysreg_s(SYS_IFSR32_EL2); break;
case DBGVCR32_EL2: val = read_sysreg_s(SYS_DBGVCR32_EL2); break;
case NVHCR_EL2: val = read_sysreg_s(SYS_NVHCR_EL2); break;
default: WARN_ON_ONCE(1);
}
@@ -287,6 +313,7 @@ static void write_sr_to_cpu(enum vcpu_sysreg reg, u64 val)
case DACR32_EL2: write_sysreg_s(val, SYS_DACR32_EL2); break;
case IFSR32_EL2: write_sysreg_s(val, SYS_IFSR32_EL2); break;
case DBGVCR32_EL2: write_sysreg_s(val, SYS_DBGVCR32_EL2); break;
case NVHCR_EL2: write_sysreg_s(val, SYS_NVHCR_EL2); break;
default: WARN_ON_ONCE(1);
}
}
@@ -311,9 +338,16 @@ u64 vcpu_read_sys_reg(const struct kvm_vcpu *vcpu, enum vcpu_sysreg reg)
switch (reg) {
case CNTHCTL_EL2:
val = read_sysreg_el1(SYS_CNTKCTL);
val &= CNTKCTL_VALID_BITS;
val |= __vcpu_sys_reg(vcpu, reg) & ~CNTKCTL_VALID_BITS;
if (!cpus_have_final_cap(ARM64_HAS_NV2P1)) {
val &= CNTKCTL_VALID_BITS;
val |= __vcpu_sys_reg(vcpu, reg) & ~CNTKCTL_VALID_BITS;
}
return val;
case CPTR_EL2:
if (cpus_have_final_cap(ARM64_HAS_NV2P1))
return read_sysreg_el1(SYS_CPACR);
else
return __vcpu_sys_reg(vcpu, reg);
default:
WARN_ON_ONCE(1);
}
@@ -359,6 +393,9 @@ void vcpu_write_sys_reg(struct kvm_vcpu *vcpu, u64 val, enum vcpu_sysreg reg)
*/
write_sysreg_el1(val, SYS_CNTKCTL);
break;
case CPTR_EL2:
write_sysreg_el1(val, SYS_CPACR);
break;
default:
WARN_ON_ONCE(1);
}
@@ -976,21 +1013,9 @@ static u64 reset_actlr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r)
static u64 reset_mpidr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r)
{
u64 mpidr;
u64 mpidr = kvm_calculate_mpidr(vcpu);
/*
* Map the vcpu_id into the first three affinity level fields of
* the MPIDR. We limit the number of VCPUs in level 0 due to a
* limitation to 16 CPUs in that level in the ICC_SGIxR registers
* of the GICv3 to be able to address each CPU directly when
* sending IPIs.
*/
mpidr = (vcpu->vcpu_id & 0x0f) << MPIDR_LEVEL_SHIFT(0);
mpidr |= ((vcpu->vcpu_id >> 4) & 0xff) << MPIDR_LEVEL_SHIFT(1);
mpidr |= ((vcpu->vcpu_id >> 12) & 0xff) << MPIDR_LEVEL_SHIFT(2);
mpidr |= (1ULL << 31);
vcpu_write_sys_reg(vcpu, mpidr, MPIDR_EL1);
return mpidr;
}
@@ -1367,6 +1392,64 @@ static bool access_pminten(struct kvm_vcpu *vcpu, struct sys_reg_params *p,
return true;
}
static bool access_pmmir(struct kvm_vcpu *vcpu, struct sys_reg_params *p,
const struct sys_reg_desc *r)
{
if (p->is_write)
return write_to_read_only(vcpu, p, r);
/*
* If KVM_ARM_VCPU_PMU_V3_STRICT is set and PMU was explicitly
* selected, the underlying hardware SLOTS value was read into this
* field. Otherwise, it stays 0. All other PMMIR_EL1 fields are RAZ.
*/
p->regval = FIELD_PREP(ARMV8_PMU_SLOTS, vcpu->kvm->arch.pmmir_slots);
return true;
}
static int get_pmmir(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r,
u64 *val)
{
*val = FIELD_PREP(ARMV8_PMU_SLOTS, vcpu->kvm->arch.pmmir_slots);
return 0;
}
static int set_pmmir(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r,
u64 val)
{
struct kvm *kvm = vcpu->kvm;
u8 slots = FIELD_GET(ARMV8_PMU_SLOTS, val);
/*
* Only the SLOTS field is exposed (get_pmmir returns just that field),
* so reject a write that sets any other bit rather than silently
* masking it.
*/
if (val & ~(u64)ARMV8_PMU_SLOTS)
return -EINVAL;
guard(mutex)(&kvm->arch.config_lock);
/*
* Once the VM has started PMMIR_EL1 is immutable. Reject any write
* that does not match the current value.
*/
if (kvm_vm_has_ran_once(kvm))
return slots == kvm->arch.pmmir_slots ? 0 : -EBUSY;
/*
* Only SLOTS = 0 is honored for backwards compatibility with the
* old RAZ behavior. Reject any non-zero write that does not match
* the current value.
*/
if (!slots)
kvm->arch.pmmir_slots = 0;
else if (slots != kvm->arch.pmmir_slots)
return -EINVAL;
return 0;
}
static bool access_pmovs(struct kvm_vcpu *vcpu, struct sys_reg_params *p,
const struct sys_reg_desc *r)
{
@@ -1444,6 +1527,7 @@ static int set_pmcr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r,
*/
if (!kvm_vm_has_ran_once(kvm) &&
!vcpu_has_nv(vcpu) &&
!kvm_vcpu_has_pmuv3_strict(vcpu) &&
new_n <= kvm_arm_pmu_get_max_counters(kvm))
kvm->arch.nr_pmu_counters = new_n;
@@ -2840,6 +2924,16 @@ static unsigned int vncr_el2_visibility(const struct kvm_vcpu *vcpu,
return REG_HIDDEN;
}
static unsigned int nvhcr_el2_visibility(const struct kvm_vcpu *vcpu,
const struct sys_reg_desc *rd)
{
if (el2_visibility(vcpu, rd) == 0 &&
kvm_has_feat(vcpu->kvm, ID_AA64MMFR4_EL1, NV_frac, NV3))
return 0;
return REG_HIDDEN;
}
static unsigned int sctlr2_visibility(const struct kvm_vcpu *vcpu,
const struct sys_reg_desc *rd)
{
@@ -3448,7 +3542,8 @@ static const struct sys_reg_desc sys_reg_descs[] = {
{ PMU_SYS_REG(PMINTENCLR_EL1),
.access = access_pminten, .reg = PMINTENSET_EL1,
.get_user = get_pmreg, .set_user = set_pmreg },
{ SYS_DESC(SYS_PMMIR_EL1), trap_raz_wi },
{ PMU_SYS_REG(PMMIR_EL1), .access = access_pmmir, .reset = NULL,
.get_user = get_pmmir, .set_user = set_pmmir },
{ SYS_DESC(SYS_MAIR_EL1), access_vm_reg, reset_unknown, MAIR_EL1 },
{ SYS_DESC(SYS_PIRE0_EL1), NULL, reset_unknown, PIRE0_EL1,
@@ -3753,6 +3848,8 @@ static const struct sys_reg_desc sys_reg_descs[] = {
sve_el2_visibility),
EL2_REG_VNCR(HCRX_EL2, reset_val, 0),
EL2_REG_FILTERED(NVHCR_EL2, undef_access, reset_val, 0,
nvhcr_el2_visibility),
EL2_REG(TTBR0_EL2, access_rw, reset_val, 0),
EL2_REG(TTBR1_EL2, access_rw, reset_val, 0),
@@ -4057,6 +4154,7 @@ static bool handle_ripas2e1is(struct kvm_vcpu *vcpu, struct sys_reg_params *p,
u32 sys_encoding = sys_insn(p->Op0, p->Op1, p->CRn, p->CRm, p->Op2);
u64 vttbr = vcpu_read_sys_reg(vcpu, VTTBR_EL2);
u64 base, range;
int pa_bits;
if (!kvm_supported_tlbi_ipas2_op(vcpu, sys_encoding))
return undef_access(vcpu, p, r);
@@ -4068,6 +4166,16 @@ static bool handle_ripas2e1is(struct kvm_vcpu *vcpu, struct sys_reg_params *p,
*/
base = decode_range_tlbi(p->regval, &range, NULL);
/*
* Ignore TLBIs that start out of PA_bits range, and cap the
* invalidation to the [base:bit(PA_bits)] interval.
*/
pa_bits = kvm_get_pa_bits(vcpu->kvm);
if (fls64(base) > pa_bits)
return true;
range = min(range, BIT_ULL(pa_bits) - base);
kvm_s2_mmu_iterate_by_vmid(vcpu->kvm, get_vmid(vttbr),
&(union tlbi_info) {
.range = {
@@ -4593,7 +4701,7 @@ static const struct sys_reg_desc cp15_regs[] = {
{ CP15_PMU_SYS_REG(HI, 0, 9, 14, 4), .access = access_pmceid },
{ CP15_PMU_SYS_REG(HI, 0, 9, 14, 5), .access = access_pmceid },
/* PMMIR */
{ CP15_PMU_SYS_REG(DIRECT, 0, 9, 14, 6), .access = trap_raz_wi },
{ CP15_PMU_SYS_REG(DIRECT, 0, 9, 14, 6), .access = access_pmmir },
/* PRRR/MAIR0 */
{ AA32(LO), Op1( 0), CRn(10), CRm( 2), Op2( 0), access_vm_reg, NULL, MAIR_EL1 },
@@ -4861,10 +4969,8 @@ static int kvm_handle_cp_64(struct kvm_vcpu *vcpu,
* Make a 64-bit value out of Rt and Rt2. As we use the same trap
* backends between AArch32 and AArch64, we get away with it.
*/
if (params.is_write) {
params.regval = vcpu_get_reg(vcpu, Rt) & 0xffffffff;
params.regval |= vcpu_get_reg(vcpu, Rt2) << 32;
}
params.regval = vcpu_get_reg(vcpu, Rt) & 0xffffffff;
params.regval |= vcpu_get_reg(vcpu, Rt2) << 32;
/*
* If the table contains a handler, handle the

View File

@@ -222,6 +222,25 @@ find_reg(const struct sys_reg_params *params, const struct sys_reg_desc table[],
return __inline_bsearch((void *)pval, table, num, sizeof(table[0]), match_sys_reg);
}
static inline u64 kvm_calculate_mpidr(const struct kvm_vcpu *vcpu)
{
u64 mpidr;
/*
* Map the vcpu_id into the first three affinity level fields of
* the MPIDR. We limit the number of VCPUs in level 0 due to a
* limitation to 16 CPUs in that level in the ICC_SGIxR registers
* of the GICv3 to be able to address each CPU directly when
* sending IPIs.
*/
mpidr = (vcpu->vcpu_id & 0x0f) << MPIDR_LEVEL_SHIFT(0);
mpidr |= ((vcpu->vcpu_id >> 4) & 0xff) << MPIDR_LEVEL_SHIFT(1);
mpidr |= ((vcpu->vcpu_id >> 12) & 0xff) << MPIDR_LEVEL_SHIFT(2);
mpidr |= (1ULL << 31);
return mpidr;
}
const struct sys_reg_desc *get_reg_by_id(u64 id,
const struct sys_reg_desc table[],
unsigned int num);

View File

@@ -35,12 +35,12 @@ static int set_gic_ctlr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r,
vgic_v3_cpu->num_id_bits = host_id_bits;
host_seis = FIELD_GET(ICH_VTR_EL2_SEIS, kvm_vgic_global_state.ich_vtr_el2);
host_seis = FIELD_GET(ICH_VTR_EL2_SEIS, vgic_ich_vtr());
seis = FIELD_GET(ICC_CTLR_EL1_SEIS_MASK, val);
if (host_seis != seis)
return -EINVAL;
host_a3v = FIELD_GET(ICH_VTR_EL2_A3V, kvm_vgic_global_state.ich_vtr_el2);
host_a3v = FIELD_GET(ICH_VTR_EL2_A3V, vgic_ich_vtr());
a3v = FIELD_GET(ICC_CTLR_EL1_A3V_MASK, val);
if (host_a3v != a3v)
return -EINVAL;
@@ -69,9 +69,9 @@ static int get_gic_ctlr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r,
val |= FIELD_PREP(ICC_CTLR_EL1_ID_BITS_MASK, vgic_v3_cpu->num_id_bits);
val |= FIELD_PREP(ICC_CTLR_EL1_SEIS_MASK,
FIELD_GET(ICH_VTR_EL2_SEIS,
kvm_vgic_global_state.ich_vtr_el2));
vgic_ich_vtr()));
val |= FIELD_PREP(ICC_CTLR_EL1_A3V_MASK,
FIELD_GET(ICH_VTR_EL2_A3V, kvm_vgic_global_state.ich_vtr_el2));
FIELD_GET(ICH_VTR_EL2_A3V, vgic_ich_vtr()));
/*
* The VMCR.CTLR value is in ICC_CTLR_EL1 layout.
* Extract it directly using ICC_CTLR_EL1 reg definitions.

View File

@@ -176,6 +176,7 @@ int kvm_vgic_create(struct kvm *kvm, u32 type)
}
kvm->arch.vgic.vgic_model = 0;
kvm->arch.vgic.in_kernel = false;
goto out_unlock;
}
@@ -210,6 +211,9 @@ static int kvm_vgic_dist_init(struct kvm *kvm, unsigned int nr_spis)
struct kvm_vcpu *vcpu0 = kvm_get_vcpu(kvm, 0);
int i;
if (dist->spis)
return 0;
dist->active_spis = (atomic_t)ATOMIC_INIT(0);
dist->spis = kzalloc_objs(struct vgic_irq, nr_spis, GFP_KERNEL_ACCOUNT);
if (!dist->spis)
@@ -787,7 +791,8 @@ int kvm_vgic_hyp_init(void)
if (has_mask && !gic_kvm_info->maint_irq) {
kvm_err("No vgic maintenance irq\n");
return -ENXIO;
ret = -ENXIO;
goto out_free;
}
/*
@@ -820,6 +825,7 @@ int kvm_vgic_hyp_init(void)
kvm_vgic_global_state.maint_irq = gic_kvm_info->maint_irq;
out_free:
kfree(gic_kvm_info);
gic_kvm_info = NULL;

View File

@@ -2035,15 +2035,16 @@ static u32 compute_next_devid_offset(struct list_head *h,
static u32 compute_next_eventid_offset(struct list_head *h, struct its_ite *ite)
{
struct its_ite *next;
u32 next_offset;
struct its_ite *next = ite;
if (list_is_last(&ite->ite_list, h))
return 0;
next = list_next_entry(ite, ite_list);
next_offset = next->event_id - ite->event_id;
/* Point at the next ITE that vgic_its_save_ite() stores as valid. */
list_for_each_entry_continue(next, h, ite_list) {
if (next->collection)
return min_t(u32, next->event_id - ite->event_id,
VITS_ITE_MAX_EVENTID_OFFSET);
}
return min_t(u32, next_offset, VITS_ITE_MAX_EVENTID_OFFSET);
return 0;
}
/**
@@ -2119,6 +2120,14 @@ static int vgic_its_save_ite(struct vgic_its *its, struct its_device *dev,
u32 next_offset;
u64 val;
/*
* MAPC with V=0 keeps the ITEs mapped but drops their collection,
* and with it the ICID. Save a zeroed entry, which the restore path
* reads back as invalid.
*/
if (!ite->collection)
return vgic_its_write_entry_lock(its, gpa, 0ULL, ite);
next_offset = compute_next_eventid_offset(&dev->itt_head, ite);
val = ((u64)next_offset << KVM_ITS_ITE_NEXT_SHIFT) |
((u64)ite->irq->intid << KVM_ITS_ITE_PINTID_SHIFT) |
@@ -2532,6 +2541,9 @@ static int vgic_its_save_collection_table(struct vgic_its *its)
max_size = GITS_BASER_NR_PAGES(baser) * SZ_64K;
list_for_each_entry(collection, &its->collection_list, coll_list) {
if (!vgic_its_check_id(its, baser, collection->collection_id, NULL))
return -EINVAL;
ret = vgic_its_save_cte(its, collection, gpa);
if (ret)
return ret;

View File

@@ -170,8 +170,9 @@ void vgic_v2_deactivate(struct kvm_vcpu *vcpu, u32 val)
/* Make sure we're in the same context as LR handling */
local_irq_save(flags);
/* Guest-supplied INTID: out of range yields no irq, so ignore it */
irq = vgic_get_vcpu_irq(vcpu, val);
if (WARN_ON_ONCE(!irq))
if (!irq)
goto out;
/* See the corresponding v3 code for the rationale */

View File

@@ -152,7 +152,7 @@ static void vgic_compute_mi_state(struct kvm_vcpu *vcpu, struct mi_state *mi_sta
eisr |= BIT(i);
if (!(lr & ICH_LR_STATE))
elrsr |= BIT(i);
pend |= (lr & ICH_LR_PENDING_BIT);
pend |= (lr & ICH_LR_STATE) == ICH_LR_PENDING_BIT;
}
mi_state->eisr = eisr;

View File

@@ -496,9 +496,9 @@ void vgic_v3_reset(struct kvm_vcpu *vcpu)
}
vcpu->arch.vgic_cpu.num_id_bits = FIELD_GET(ICH_VTR_EL2_IDbits,
kvm_vgic_global_state.ich_vtr_el2);
vgic_ich_vtr());
vcpu->arch.vgic_cpu.num_pri_bits = FIELD_GET(ICH_VTR_EL2_PRIbits,
kvm_vgic_global_state.ich_vtr_el2) + 1;
vgic_ich_vtr()) + 1;
}
void vcpu_set_ich_hcr(struct kvm_vcpu *vcpu)
@@ -617,9 +617,13 @@ int vgic_v3_save_pending_tables(struct kvm *kvm)
bool is_pending;
bool stored;
irq = vgic_get_irq(kvm, index);
if (!irq)
continue;
vcpu = irq->target_vcpu;
if (!vcpu)
continue;
goto put_irq;
pendbase = GICR_PENDBASER_ADDRESS(vcpu->arch.vgic_cpu.pendbaser);
@@ -630,7 +634,7 @@ int vgic_v3_save_pending_tables(struct kvm *kvm)
if (ptr != last_ptr) {
ret = kvm_read_guest_lock(kvm, ptr, &val, 1);
if (ret)
goto out;
goto put_irq;
last_ptr = ptr;
}
@@ -642,7 +646,7 @@ int vgic_v3_save_pending_tables(struct kvm *kvm)
vgic_v4_get_vlpi_state(irq, &is_pending);
if (stored == is_pending)
continue;
goto put_irq;
if (is_pending)
val |= 1 << bit_nr;
@@ -650,6 +654,8 @@ int vgic_v3_save_pending_tables(struct kvm *kvm)
val &= ~(1 << bit_nr);
ret = vgic_write_guest_lock(kvm, ptr, &val, 1);
put_irq:
vgic_put_irq(kvm, irq);
if (ret)
goto out;
}
@@ -815,27 +821,9 @@ static int __init early_gicv4_enable(char *buf)
}
early_param("kvm-arm.vgic_v4_enable", early_gicv4_enable);
static const struct midr_range broken_seis[] = {
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_ICESTORM),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_FIRESTORM),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_ICESTORM_PRO),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_FIRESTORM_PRO),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_ICESTORM_MAX),
MIDR_ALL_VERSIONS(MIDR_APPLE_M1_FIRESTORM_MAX),
MIDR_ALL_VERSIONS(MIDR_APPLE_M2_BLIZZARD),
MIDR_ALL_VERSIONS(MIDR_APPLE_M2_AVALANCHE),
MIDR_ALL_VERSIONS(MIDR_APPLE_M2_BLIZZARD_PRO),
MIDR_ALL_VERSIONS(MIDR_APPLE_M2_AVALANCHE_PRO),
MIDR_ALL_VERSIONS(MIDR_APPLE_M2_BLIZZARD_MAX),
MIDR_ALL_VERSIONS(MIDR_APPLE_M2_AVALANCHE_MAX),
{},
};
static bool vgic_v3_broken_seis(void)
static __always_inline bool vgic_v3_broken_seis(void)
{
return (is_kernel_in_hyp_mode() &&
is_midr_in_range_list(broken_seis) &&
(read_sysreg_s(SYS_ICH_VTR_EL2) & ICH_VTR_EL2_SEIS));
return cpus_have_cap(ARM64_WORKAROUND_GICv3_BROKEN_SEIS);
}
void noinstr kvm_compute_ich_hcr_trap_bits(struct alt_instr *alt,
@@ -882,6 +870,61 @@ void noinstr kvm_compute_ich_hcr_trap_bits(struct alt_instr *alt,
*updptr = cpu_to_le32(insn);
}
void noinstr kvm_patch_ich_vtr_el2(struct alt_instr *alt,
__le32 *origptr, __le32 *updptr,
int nr_inst)
{
struct arm_smccc_res res = {};
u32 insn, oinsn, rd, vtr;
/* No KVM? Nothing to do */
if (!is_hyp_mode_available())
return;
/* No v3, compat, nor the fruity erzatz of a GIC? Bugger off */
if (!cpus_have_cap(ARM64_HAS_GICV5_LEGACY) &&
!cpus_have_cap(ARM64_HAS_GICV3_CPUIF) &&
!vgic_v3_broken_seis())
return;
/*
* At the point where this is called, we are guaranteed that if
* we're running at EL1, then the EL2 stubs are still in place.
*/
if (is_kernel_in_hyp_mode())
res.a1 = read_sysreg_s(SYS_ICH_VTR_EL2);
else
arm_smccc_1_1_hvc(HVC_GET_ICH_VTR_EL2, &res);
if (res.a0 == HVC_STUB_ERR)
return;
vtr = res.a1;
if (vgic_v3_broken_seis())
vtr &= ~ICH_VTR_EL2_SEIS;
/* Compute target register */
oinsn = le32_to_cpu(*origptr);
rd = aarch64_insn_decode_register(AARCH64_INSN_REGTYPE_RD, oinsn);
/* movz rd, #(vtr & 0xffff) */
insn = aarch64_insn_gen_movewide(rd,
(u16)vtr,
0,
AARCH64_INSN_VARIANT_64BIT,
AARCH64_INSN_MOVEWIDE_ZERO);
*updptr++ = cpu_to_le32(insn);
/* movk rd, #((vtr >> 16) & 0xffff), lsl #16 */
insn = aarch64_insn_gen_movewide(rd,
(u16)(vtr >> 16),
16,
AARCH64_INSN_VARIANT_64BIT,
AARCH64_INSN_MOVEWIDE_KEEP);
*updptr++ = cpu_to_le32(insn);
}
void vgic_v3_enable_cpuif_traps(void)
{
u64 traps = vgic_ich_hcr_trap_bits();
@@ -905,12 +948,12 @@ void vgic_v3_enable_cpuif_traps(void)
*/
int vgic_v3_probe(const struct gic_kvm_info *info)
{
u64 ich_vtr_el2 = kvm_call_hyp_ret(__vgic_v3_get_gic_config);
u64 ich_vtr_el2;
bool has_v2;
int ret;
has_v2 = ich_vtr_el2 >> 63;
ich_vtr_el2 = (u32)ich_vtr_el2;
has_v2 = kvm_call_hyp_ret(__vgic_v3_get_gic_config);
ich_vtr_el2 = vgic_ich_vtr();
/*
* The ListRegs field is 5 bits, but there is an architectural
@@ -918,7 +961,6 @@ int vgic_v3_probe(const struct gic_kvm_info *info)
*/
kvm_vgic_global_state.nr_lr = (ich_vtr_el2 & 0xf) + 1;
kvm_vgic_global_state.can_emulate_gicv2 = false;
kvm_vgic_global_state.ich_vtr_el2 = ich_vtr_el2;
/* GICv4 support? */
if (info->has_v4) {
@@ -965,11 +1007,6 @@ int vgic_v3_probe(const struct gic_kvm_info *info)
if (has_v2)
static_branch_enable(&vgic_v3_has_v2_compat);
if (vgic_v3_broken_seis()) {
kvm_info("GICv3 with broken locally generated SEI\n");
kvm_vgic_global_state.ich_vtr_el2 &= ~ICH_VTR_EL2_SEIS;
}
vgic_v3_enable_cpuif_traps();
kvm_vgic_global_state.vctrl_base = NULL;

View File

@@ -40,7 +40,6 @@ static void vgic_v5_get_implemented_ppis(void)
int vgic_v5_probe(const struct gic_kvm_info *info)
{
bool v5_registered = false;
u64 ich_vtr_el2;
int ret;
kvm_vgic_global_state.type = VGIC_V5;
@@ -83,14 +82,12 @@ int vgic_v5_probe(const struct gic_kvm_info *info)
}
kvm_vgic_global_state.has_gcie_v3_compat = true;
ich_vtr_el2 = kvm_call_hyp_ret(__vgic_v3_get_gic_config);
kvm_vgic_global_state.ich_vtr_el2 = (u32)ich_vtr_el2;
/*
* The ListRegs field is 5 bits, but there is an architectural
* maximum of 16 list registers. Just ignore bit 4...
*/
kvm_vgic_global_state.nr_lr = (ich_vtr_el2 & 0xf) + 1;
kvm_vgic_global_state.nr_lr = (vgic_ich_vtr() & 0xf) + 1;
ret = kvm_register_vgic_device(KVM_DEV_TYPE_ARM_VGIC_V3);
if (ret) {

View File

@@ -93,8 +93,9 @@ struct vgic_irq *vgic_get_irq(struct kvm *kvm, u32 intid)
/* SPIs */
if (intid >= VGIC_NR_PRIVATE_IRQS &&
intid < (kvm->arch.vgic.nr_spis + VGIC_NR_PRIVATE_IRQS)) {
intid = array_index_nospec(intid, kvm->arch.vgic.nr_spis + VGIC_NR_PRIVATE_IRQS);
return &kvm->arch.vgic.spis[intid - VGIC_NR_PRIVATE_IRQS];
intid -= VGIC_NR_PRIVATE_IRQS;
intid = array_index_nospec(intid, kvm->arch.vgic.nr_spis);
return &kvm->arch.vgic.spis[intid];
}
/* LPIs */
@@ -117,6 +118,8 @@ struct vgic_irq *vgic_get_vcpu_irq(struct kvm_vcpu *vcpu, u32 intid)
switch (type) {
case KVM_DEV_TYPE_ARM_VGIC_V5:
intid = vgic_v5_get_hwirq_id(intid);
if (intid >= VGIC_V5_NR_PRIVATE_IRQS)
return NULL;
intid = array_index_nospec(intid, VGIC_V5_NR_PRIVATE_IRQS);
break;
default:

View File

@@ -71,11 +71,28 @@
ICH_VTR_EL2_IDbits)
#define KVM_ICH_VTR_EL2_RES1 ICH_VTR_EL2_nV4
void kvm_patch_ich_vtr_el2(struct alt_instr *alt,
__le32 *origptr, __le32 *updptr, int nr_inst);
static inline u64 vgic_ich_vtr(void)
{
u64 vtr;
/* All non-RES0 bits are in the bottom 32bits */
asm volatile(ALTERNATIVE_CB("movz %0, #0\n"
"movk %0, #0, lsl #16\n",
ARM64_ALWAYS_SYSTEM,
kvm_patch_ich_vtr_el2)
: "=r" (vtr));
return vtr;
}
static inline u64 kvm_get_guest_vtr_el2(void)
{
u64 vtr;
vtr = kvm_vgic_global_state.ich_vtr_el2;
vtr = vgic_ich_vtr();
vtr &= ~KVM_ICH_VTR_EL2_RES0;
vtr |= KVM_ICH_VTR_EL2_RES1;

View File

@@ -51,6 +51,8 @@ HAS_LS64_V
HAS_LSUI
HAS_MOPS
HAS_NESTED_VIRT
HAS_NV2P1
HAS_NV3
HAS_BBML2_NOABORT
HAS_PAN
HAS_PMUV3
@@ -121,6 +123,7 @@ WORKAROUND_CAVIUM_TX2_219_TVM
WORKAROUND_CLEAN_CACHE
WORKAROUND_DEVICE_LOAD_ACQUIRE
WORKAROUND_DISABLE_CNP
WORKAROUND_GICv3_BROKEN_SEIS
WORKAROUND_PMUV3_IMPDEF_TRAPS
WORKAROUND_QCOM_FALKOR_E1003
WORKAROUND_QCOM_ORYON_CNTVOFF

View File

@@ -228,7 +228,7 @@ $1 == "EndSysreg" && block_current() == "Sysreg" {
}
# Currently this is effectivey a comment, in future we may want to emit
# defines for the fields.
# defines for the fields. We do emit RESx and UNKN values in any case.
($1 == "Fields" || $1 == "Mapping") && block_current() == "Sysreg" {
expect_fields(2)
@@ -239,9 +239,9 @@ $1 == "EndSysreg" && block_current() == "Sysreg" {
print ""
next_bit = -1
res0 = null
res1 = null
unkn = null
res0 = $2 "_RES0"
res1 = $2 "_RES1"
unkn = $2 "_UNKN"
next
}

View File

@@ -2386,17 +2386,40 @@ EndEnum
EndSysreg
Sysreg ID_AA64MMFR4_EL1 3 0 0 7 4
Res0 63:48
UnsignedEnum 47:44 SRMASK
UnsignedEnum 63:60 MTEFGT
0b0000 NI
0b0001 IMP
EndEnum
UnsignedEnum 59:56 SCRX
0b0000 NI
0b0001 IMP
EndEnum
UnsignedEnum 55:52 TEV
0b0000 NI
0b0001 IMP
EndEnum
UnsignedEnum 51:48 TPS
0b0000 VAL_0000
0b0001 VAL_0001
0b0010 VAL_0010
EndEnum
UnsignedEnum 47:44 SRMASK
0b0000 NI
0b0001 IMP
0b0010 SRMASK2
EndEnum
UnsignedEnum 43:40 TLBID
0b0000 NI
0b0001 IMP
EndEnum
Res0 43:40
UnsignedEnum 39:36 E3DSE
0b0000 NI
0b0001 IMP
EndEnum
Res0 35:32
UnsignedEnum 35:32 EAESR
0b0000 NI
0b0001 IMP
EndEnum
UnsignedEnum 31:28 RMEGDI
0b0000 NI
0b0001 IMP
@@ -2410,6 +2433,7 @@ UnsignedEnum 23:20 NV_frac
0b0000 NV_NV2
0b0001 NV2_ONLY
0b0010 NV2P1
0b0011 NV3
EndEnum
UnsignedEnum 19:16 FGWTE3
0b0000 NI
@@ -4242,6 +4266,9 @@ Field 1 E2TRE
Field 0 E0HTRE
EndSysreg
Sysreg NVHCR_EL2 3 4 1 5 0
Mapping HCR_EL2
EndSysreg
Sysreg HDFGRTR2_EL2 3 4 3 1 0
Res0 63:25
@@ -4521,7 +4548,14 @@ Fields ZCR_ELx
EndSysreg
Sysreg HCRX_EL2 3 4 1 2 2
Res0 63:25
Res0 63:35
Field 34 NVnTTLBOS
Field 33 NVnTTLBIS
Field 32 NVnTTLB
Res0 31:28
Field 27 NVTGE
Field 26 SRMASKEn
Res0 25
Field 24 PACMEn
Field 23 EnFPM
Field 22 GCSEn

View File

@@ -162,20 +162,28 @@ static inline bool has_cntpoff(void)
return (has_vhe() && cpus_have_final_cap(ARM64_HAS_ECV_CNTPOFF));
}
static inline u64 timer_get_offset(struct arch_timer_context *ctxt)
{
u64 offset = 0;
#ifdef __KVM_NVHE_HYPERVISOR__
#define KERN_HYP_VA(x) kern_hyp_va(x)
#else
#define KERN_HYP_VA(x) x
#endif
if (!ctxt)
return 0;
if (ctxt->offset.vm_offset)
offset += *ctxt->offset.vm_offset;
if (ctxt->offset.vcpu_offset)
offset += *ctxt->offset.vcpu_offset;
return offset;
}
#define timer_get_offset(ctxt) \
({ \
struct arch_timer_context *__ctxt = (ctxt); \
u64 off = 0; \
\
if (__ctxt) { \
struct arch_timer_offset *ato = &__ctxt->offset;\
\
if (ato->vm_offset) \
off += *KERN_HYP_VA(ato->vm_offset); \
if (ato->vcpu_offset) \
off += *KERN_HYP_VA(ato->vcpu_offset); \
} \
\
off; \
})
static inline void timer_set_offset(struct arch_timer_context *ctxt, u64 offset)
{

View File

@@ -75,6 +75,9 @@ void kvm_vcpu_pmu_resync_el0(void);
#define kvm_vcpu_has_pmu(vcpu) \
(vcpu_has_feature(vcpu, KVM_ARM_VCPU_PMU_V3))
#define kvm_vcpu_has_pmuv3_strict(vcpu) \
(vcpu_has_feature(vcpu, KVM_ARM_VCPU_PMU_V3_STRICT))
/*
* Updates the vcpu's view of the pmu events for this cpu.
* Must be called before every vcpu run after disabling interrupts, to ensure
@@ -160,6 +163,7 @@ static inline u64 kvm_pmu_get_pmceid(struct kvm_vcpu *vcpu, bool pmceid1)
}
#define kvm_vcpu_has_pmu(vcpu) ({ false; })
#define kvm_vcpu_has_pmuv3_strict(vcpu) ({ false; })
static inline void kvm_pmu_update_vcpu_events(struct kvm_vcpu *vcpu) {}
static inline void kvm_vcpu_pmu_restore_guest(struct kvm_vcpu *vcpu) {}
static inline void kvm_vcpu_pmu_restore_host(struct kvm_vcpu *vcpu) {}

View File

@@ -38,6 +38,33 @@ static inline int kvm_psci_version(struct kvm_vcpu *vcpu)
return KVM_ARM_PSCI_0_1;
}
/* Narrow the PSCI register arguments (r1 to r3) to 32 bits. */
static inline void kvm_psci_narrow_to_32bit(struct kvm_vcpu *vcpu)
{
int i;
/*
* Zero the input registers' upper 32 bits. They will be fully
* zeroed on exit, so we're fine changing them in place.
*/
for (i = 1; i < 4; i++)
vcpu_set_reg(vcpu, i, lower_32_bits(vcpu_get_reg(vcpu, i)));
}
static inline bool kvm_psci_valid_affinity(struct kvm_vcpu *vcpu,
unsigned long affinity)
{
return !(affinity & ~MPIDR_HWID_BITMASK);
}
static inline unsigned long kvm_psci_affinity_mask(unsigned long affinity_level)
{
if (affinity_level <= 3)
return MPIDR_HWID_BITMASK &
~((0x1UL << (affinity_level * MPIDR_LEVEL_BITS)) - 1);
return 0;
}
int kvm_psci_call(struct kvm_vcpu *vcpu);

View File

@@ -65,6 +65,8 @@
switch (t) { \
case KVM_DEV_TYPE_ARM_VGIC_V5: \
__ret = is_v5_type(GICV5_HWIRQ_TYPE_PPI, (i)); \
__ret &= FIELD_GET(GICV5_HWIRQ_ID, (i)) < \
VGIC_V5_NR_PRIVATE_IRQS; \
break; \
default: \
__ret = (i) >= VGIC_NR_SGIS; \
@@ -176,8 +178,6 @@ struct vgic_global {
/* GICv3 compat mode on a GICv5 host */
bool has_gcie_v3_compat;
u32 ich_vtr_el2;
/* GICv5 PPI capabilities */
struct {
DECLARE_BITMAP(impl_ppi_mask, VGIC_V5_NR_PRIVATE_IRQS);

View File

@@ -997,6 +997,7 @@ struct kvm_enable_cap {
#define KVM_CAP_S390_KEYOP 247
#define KVM_CAP_S390_VSIE_ESAMODE 248
#define KVM_CAP_S390_HPAGE_2G 249
#define KVM_CAP_ARM_PMU_V3_STRICT 250
struct kvm_irq_routing_irqchip {
__u32 irqchip;

View File

@@ -185,6 +185,7 @@ TEST_GEN_PROGS_arm64 += arm64/psci_test
TEST_GEN_PROGS_arm64 += arm64/sea_to_user
TEST_GEN_PROGS_arm64 += arm64/set_id_regs
TEST_GEN_PROGS_arm64 += arm64/smccc_filter
TEST_GEN_PROGS_arm64 += arm64/stage2_block_transitions
TEST_GEN_PROGS_arm64 += arm64/vcpu_width_config
TEST_GEN_PROGS_arm64 += arm64/vgic_init
TEST_GEN_PROGS_arm64 += arm64/vgic_irq

View File

@@ -527,6 +527,46 @@ void test_single_step_from_userspace(int test_cnt)
kvm_vm_free(vm);
}
static void guest_code_wp(void)
{
write_data = 'x';
GUEST_DONE();
}
/*
* A userspace hardware watchpoint (KVM_GUESTDBG_USE_HW) must fire and report
* the accessed address in debug.arch.far, exercising the watchpoint exit path.
*/
static void test_watchpoint_from_userspace(void)
{
struct kvm_guest_debug debug = {};
struct kvm_vcpu *vcpu;
struct kvm_run *run;
struct kvm_vm *vm;
vm = vm_create_with_one_vcpu(&vcpu, guest_code_wp);
run = vcpu->run;
debug.control = KVM_GUESTDBG_ENABLE | KVM_GUESTDBG_USE_HW;
debug.arch.dbg_wcr[0] = DBGWCR_LEN8 | DBGWCR_RD | DBGWCR_WR |
DBGWCR_EL1 | DBGWCR_E;
/*
* BAS = 0xff (LEN8) requires a doubleword-aligned DBGWVR; FAR still
* reports the exact accessed byte.
*/
debug.arch.dbg_wvr[0] = PC(write_data) & ~7UL;
vcpu_guest_debug_set(vcpu, &debug);
vcpu_run(vcpu);
TEST_ASSERT(run->exit_reason == KVM_EXIT_DEBUG,
"Expected KVM_EXIT_DEBUG, got %u", run->exit_reason);
TEST_ASSERT((u64)run->debug.arch.far == PC(write_data),
"Watchpoint FAR 0x%lx != accessed address 0x%lx",
(u64)run->debug.arch.far, PC(write_data));
kvm_vm_free(vm);
}
/*
* Run debug testing using the various breakpoint#, watchpoint# and
* context-aware breakpoint# with the given ID_AA64DFR0_EL1 configuration.
@@ -600,6 +640,7 @@ int main(int argc, char *argv[])
test_guest_debug_exceptions_all(aa64dfr0);
test_single_step_from_userspace(ss_iteration);
test_watchpoint_from_userspace();
return 0;
}

View File

@@ -67,6 +67,7 @@ static struct feature_id_reg feat_id_regs[] = {
REG_FEAT(VDISR_EL2, ID_AA64PFR0_EL1, RAS, IMP),
REG_FEAT(VSESR_EL2, ID_AA64PFR0_EL1, RAS, IMP),
REG_FEAT(VNCR_EL2, ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY),
REG_FEAT(NVHCR_EL2, ID_AA64MMFR4_EL1, NV_frac, NV3),
REG_FEAT(CNTHV_CTL_EL2, ID_AA64MMFR1_EL1, VH, IMP),
REG_FEAT(CNTHV_CVAL_EL2,ID_AA64MMFR1_EL1, VH, IMP),
REG_FEAT(ZCR_EL2, ID_AA64PFR0_EL1, SVE, IMP),
@@ -532,6 +533,7 @@ static __u64 base_regs[] = {
static __u64 pmu_regs[] = {
ARM64_SYS_REG(3, 0, 9, 14, 1), /* PMINTENSET_EL1 */
ARM64_SYS_REG(3, 0, 9, 14, 2), /* PMINTENCLR_EL1 */
ARM64_SYS_REG(3, 0, 9, 14, 6), /* PMMIR_EL1 */
ARM64_SYS_REG(3, 3, 9, 12, 0), /* PMCR_EL0 */
ARM64_SYS_REG(3, 3, 9, 12, 1), /* PMCNTENSET_EL0 */
ARM64_SYS_REG(3, 3, 9, 12, 2), /* PMCNTENCLR_EL0 */
@@ -770,6 +772,7 @@ static __u64 el2_regs[] = {
SYS_REG(SP_EL2),
SYS_REG(VDISR_EL2),
SYS_REG(VSESR_EL2),
SYS_REG(NVHCR_EL2),
};
static __u64 el2_e2h0_regs[] = {

View File

@@ -0,0 +1,226 @@
// SPDX-License-Identifier: GPL-2.0-only
/*
* Copyright (c) 2026 Google LLC
* Author: Fuad Tabba <fuad.tabba@linux.dev>
*
* stage2_block_transitions - Exercise stage-2 block/page granularity changes
* that dirty logging forces at fault time, and assert the guest completes.
*
* Both scenarios need the fault handler to allocate at fault time (a fresh
* mapping and/or page-table pages while holding mmu_lock), so a fault path
* that fails to stage that memory manifests as a KVM_RUN error or, worse, a
* host crash. The asserted property is host-agnostic: the guest runs the
* sequence to completion and every KVM_RUN succeeds. On a pKVM host, where a
* non-protected guest's stage-2 faults are serviced by the pkvm_pgtable_*()
* backend, the same sequences also guard that backend's fault-time staging.
*
* Scenario 1 - block collapse on dirty-logging disable:
* A write under dirty logging installs a 4K page; GET_DIRTY_LOG
* re-write-protects it; logging is disabled; a second write takes a
* permission fault that collapses the page into a hugetlb-backed block,
* which requires a fresh mapping object under mmu_lock.
*
* Scenario 2 - block split under dirty logging:
* Several hugetlb-backed blocks are faulted in as non-executable blocks,
* dirty logging is enabled (write-protect only), then the guest executes
* into each block. Each instruction fetch takes an execute permission
* fault that must split the block into pages during logging, draining
* page-table pages. Skipped on CTR_EL0.DIC hardware, where mappings are
* made executable eagerly and the execute fault never occurs.
*/
#include <linux/bitfield.h>
#include <linux/bitmap.h>
#include <linux/mman.h>
#include <linux/sizes.h>
#include <sys/mman.h>
#include <asm/sysreg.h>
#include "kvm_util.h"
#include "processor.h"
#include "test_util.h"
#include "ucall.h"
#define DATA_SLOT 1
#define TEST_GVA 0xc0000000UL
#define BLOCK_SIZE SZ_2M
/* AArch64 "ret" (ret x30): a self-contained, returnable executable payload. */
#define RET_INSN 0xd65f03c0U
/*
* A non-protected guest's per-VM stage-2 pool is seeded only with the PGD
* donation, which stage-2 init immediately consumes, so the page-table budget
* for a fault that does not top up is just the handful (~2x the stage-2 min
* pages) of memcache leftovers. Executing into this many distinct blocks
* demands far more than that budget: a fault path that tops up on every fault
* completes all of them, one that skips non-write faults runs out mid-sequence.
*/
#define NR_BLOCKS 16
/* Scenario 2 guest -> host sync stages. */
#define STAGE_SKIP_DIC 1
#define STAGE_BLOCKS_READY 2
static void collapse_guest_code(u64 gva)
{
u64 *data = (u64 *)gva;
/* Under dirty logging: install a 4K writable page. */
WRITE_ONCE(*data, 0x1);
GUEST_SYNC(1);
/* Logging disabled: a permission fault collapses the page into a block. */
WRITE_ONCE(*data, 0x2);
GUEST_SYNC(2);
GUEST_DONE();
}
static void test_block_collapse(void)
{
struct kvm_vcpu *vcpu;
unsigned long *bmap;
struct kvm_vm *vm;
struct ucall uc;
size_t npages;
u64 gpa;
vm = vm_create_with_one_vcpu(&vcpu, collapse_guest_code);
npages = BLOCK_SIZE / vm->page_size;
gpa = (vm_compute_max_gfn(vm) * vm->page_size) - BLOCK_SIZE;
gpa = align_down(gpa, BLOCK_SIZE);
vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS_HUGETLB_2MB, gpa,
DATA_SLOT, npages, KVM_MEM_LOG_DIRTY_PAGES);
virt_map(vm, TEST_GVA, gpa, npages);
vcpu_args_set(vcpu, 1, TEST_GVA);
bmap = bitmap_zalloc(BLOCK_SIZE / getpagesize());
vcpu_run(vcpu);
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC && uc.args[1] == 1,
"Expected first sync, got cmd %lu arg %lu", uc.cmd, uc.args[1]);
/* GET_DIRTY_LOG re-write-protects the dirtied page; then stop logging. */
kvm_vm_get_dirty_log(vm, DATA_SLOT, bmap);
vm_mem_region_set_flags(vm, DATA_SLOT, 0);
/* The collapsing permission fault: a broken fault path faults here. */
vcpu_run(vcpu);
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC && uc.args[1] == 2,
"Expected second sync, got cmd %lu arg %lu", uc.cmd, uc.args[1]);
vcpu_run(vcpu);
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_DONE,
"Expected done, got cmd %lu", uc.cmd);
free(bmap);
kvm_vm_free(vm);
}
static void guest_sync_insn(u64 va)
{
/* Make the just-written instruction coherent for execution (!DIC). */
asm volatile("dc cvau, %0\n"
"dsb ish\n"
"ic ivau, %0\n"
"dsb ish\n"
"isb\n"
:: "r" (va) : "memory");
}
static void split_guest_code(u64 base_gva, u64 nblocks)
{
u64 i, va;
if (FIELD_GET(CTR_EL0_DIC_MASK, read_sysreg(ctr_el0))) {
GUEST_SYNC(STAGE_SKIP_DIC);
GUEST_DONE();
return;
}
/* Fault in each block (non-executable) and stage an executable payload. */
for (i = 0; i < nblocks; i++) {
va = base_gva + i * BLOCK_SIZE;
WRITE_ONCE(*(u32 *)va, RET_INSN);
guest_sync_insn(va);
}
GUEST_SYNC(STAGE_BLOCKS_READY);
/* Logging is now on: executing into each block splits it into pages. */
for (i = 0; i < nblocks; i++) {
va = base_gva + i * BLOCK_SIZE;
((void (*)(void))va)();
}
GUEST_DONE();
}
static void test_exec_split_drain(void)
{
struct kvm_vcpu *vcpu;
struct kvm_vm *vm;
struct ucall uc;
size_t npages;
u64 gpa;
vm = vm_create_with_one_vcpu(&vcpu, split_guest_code);
npages = NR_BLOCKS * (BLOCK_SIZE / vm->page_size);
gpa = (vm_compute_max_gfn(vm) * vm->page_size) - NR_BLOCKS * BLOCK_SIZE;
gpa = align_down(gpa, BLOCK_SIZE);
vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS_HUGETLB_2MB, gpa,
DATA_SLOT, npages, 0);
virt_map(vm, TEST_GVA, gpa, npages);
vcpu_args_set(vcpu, 2, TEST_GVA, (u64)NR_BLOCKS);
vcpu_run(vcpu);
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC,
"Expected sync, got cmd %lu", uc.cmd);
if (uc.args[1] == STAGE_SKIP_DIC) {
ksft_print_msg("SKIP block split: CTR_EL0.DIC == 1\n");
kvm_vm_free(vm);
return;
}
TEST_ASSERT(uc.args[1] == STAGE_BLOCKS_READY,
"Expected blocks-ready sync, got arg %lu", uc.args[1]);
/* Write-protect the blocks; the guest then splits them by executing. */
vm_mem_region_set_flags(vm, DATA_SLOT, KVM_MEM_LOG_DIRTY_PAGES);
vcpu_run(vcpu);
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_DONE,
"Expected done, got cmd %lu", uc.cmd);
kvm_vm_free(vm);
}
/*
* The explicit-size hugetlb backing hard-fails region creation if the pages
* are not already reserved, so probe here and skip rather than abort. The
* peak reservation is scenario 2's; the two scenarios run and free in turn.
*/
static void require_hugepages(size_t bytes)
{
void *mem = mmap(NULL, bytes, PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB | MAP_HUGE_2MB,
-1, 0);
if (mem == MAP_FAILED)
ksft_exit_skip("Need %zu bytes of reserved 2M hugepages\n", bytes);
munmap(mem, bytes);
}
int main(void)
{
require_hugepages(NR_BLOCKS * BLOCK_SIZE);
test_block_collapse();
test_exec_split_drain();
ksft_print_msg("All ok!\n");
return 0;
}