mirror of
https://github.com/linux-msm/laptops-kernel.git
synced 2026-08-13 14:19:53 -07:00
Merge branch 'kvm-arm64/pkvm-7.3' into kvmarm/next
* kvm-arm64/pkvm-7.3:
: pKVM updates for 7.3
:
: - Avoid name collision on trace_clock() when CONFIG_NVHE_EL2_TRACING is
: disabled (Mostafa Saleh)
:
: - Clean up state tracking for whether the EL2 shadow VM has been
: created (Fuad Tabba)
:
: - Synchronize SCTLR_EL1 when injecting an exception to use current
: PAN/SSBS state (Fuad Tabba)
:
: - Avoid unnecessary cache maintenance when I/D-cache are known to be
: coherent in pKVM (Mostafa Saleh)
:
: - Lazy vCPU context save/restore for pKVM (Fuad Tabba)
:
: - Various fixes to the stage-2 MMU for pKVM (Fuad Tabba)
KVM: arm64: selftests: Add stage-2 block transition test
KVM: arm64: Don't advertise eager page splitting under pKVM
KVM: arm64: Don't WARN on pKVM stage-2 map failures
KVM: arm64: Skip pKVM stage-2 flush when FWB is enabled
KVM: arm64: Top up stage-2 memcache for dirty logging faults
KVM: arm64: Top up the memcache for pKVM permission faults
KVM: arm64: Skip cache maintenance for non-cacheable pKVM mappings
KVM: arm64: Implement lazy vCPU state sync for non-protected guests
KVM: arm64: Add primitives to flush/sync the VGIC state at EL2
KVM: arm64: Minimise EL2's exposure of host VGIC state during world switch
KVM: arm64: Add host and hypervisor vCPU lookup primitives
KVM: arm64: Move PSCI helper functions to a shared header
KVM: arm64: Factor out reusable vCPU reset helpers
KVM: arm64: Make vcpu_{read,write}_sys_reg available to HYP code
KVM: arm64: Extract MPIDR computation into a shared header
KVM: arm64: selftests: Add a userspace watchpoint test
KVM: arm64: Flush external_mdscr_el1 to the pKVM hyp vCPU
KVM: arm64: Optimize protected mode with FWB and DIC
Signed-off-by: Oliver Upton <oupton@kernel.org>
This commit is contained in:
@@ -348,4 +348,16 @@
|
||||
{ PSR_AA32_MODE_UND, "32-bit UND" }, \
|
||||
{ PSR_AA32_MODE_SYS, "32-bit SYS" }
|
||||
|
||||
/*
|
||||
* ARMv8 Reset Values
|
||||
*/
|
||||
#define VCPU_RESET_PSTATE_EL1 (PSR_MODE_EL1h | PSR_A_BIT | PSR_I_BIT | \
|
||||
PSR_F_BIT | PSR_D_BIT)
|
||||
|
||||
#define VCPU_RESET_PSTATE_EL2 (PSR_MODE_EL2h | PSR_A_BIT | PSR_I_BIT | \
|
||||
PSR_F_BIT | PSR_D_BIT)
|
||||
|
||||
#define VCPU_RESET_PSTATE_SVC (PSR_AA32_MODE_SVC | PSR_AA32_A_BIT | \
|
||||
PSR_AA32_I_BIT | PSR_AA32_F_BIT)
|
||||
|
||||
#endif /* __ARM64_KVM_ARM_H__ */
|
||||
|
||||
@@ -113,6 +113,7 @@ enum __kvm_host_smccc_func {
|
||||
__KVM_HOST_SMCCC_FUNC___pkvm_finalize_teardown_vm,
|
||||
__KVM_HOST_SMCCC_FUNC___pkvm_vcpu_load,
|
||||
__KVM_HOST_SMCCC_FUNC___pkvm_vcpu_put,
|
||||
__KVM_HOST_SMCCC_FUNC___pkvm_vcpu_sync_state,
|
||||
__KVM_HOST_SMCCC_FUNC___pkvm_tlb_flush_vmid,
|
||||
|
||||
MARKER(__KVM_HOST_SMCCC_FUNC_MAX)
|
||||
|
||||
@@ -506,6 +506,12 @@ static inline unsigned long kvm_vcpu_get_mpidr_aff(struct kvm_vcpu *vcpu)
|
||||
return __vcpu_sys_reg(vcpu, MPIDR_EL1) & MPIDR_HWID_BITMASK;
|
||||
}
|
||||
|
||||
/* In nVHE hyp code, registers are always in memory: use the raw accessors. */
|
||||
#if defined(__KVM_NVHE_HYPERVISOR__)
|
||||
#define vcpu_read_sys_reg(v, r) __vcpu_sys_reg(v, r)
|
||||
#define vcpu_write_sys_reg(v, x, r) __vcpu_assign_sys_reg(v, r, x)
|
||||
#endif
|
||||
|
||||
static inline void kvm_vcpu_set_be(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
if (vcpu_mode_is_32bit(vcpu)) {
|
||||
@@ -688,4 +694,61 @@ static inline void vcpu_set_hcrx(struct kvm_vcpu *vcpu)
|
||||
vcpu->arch.hcrx_el2 |= HCRX_EL2_EnASR;
|
||||
}
|
||||
}
|
||||
|
||||
/* Reset a vcpu's core registers. */
|
||||
static inline void kvm_reset_vcpu_core(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
u32 pstate;
|
||||
|
||||
if (vcpu_el1_is_32bit(vcpu))
|
||||
pstate = VCPU_RESET_PSTATE_SVC;
|
||||
else if (vcpu_has_nv(vcpu))
|
||||
pstate = VCPU_RESET_PSTATE_EL2;
|
||||
else
|
||||
pstate = VCPU_RESET_PSTATE_EL1;
|
||||
|
||||
/* Reset core registers */
|
||||
memset(vcpu_gp_regs(vcpu), 0, sizeof(*vcpu_gp_regs(vcpu)));
|
||||
memset(&vcpu->arch.ctxt.fp_regs, 0, sizeof(vcpu->arch.ctxt.fp_regs));
|
||||
vcpu->arch.ctxt.spsr_abt = 0;
|
||||
vcpu->arch.ctxt.spsr_und = 0;
|
||||
vcpu->arch.ctxt.spsr_irq = 0;
|
||||
vcpu->arch.ctxt.spsr_fiq = 0;
|
||||
vcpu_gp_regs(vcpu)->pstate = pstate;
|
||||
}
|
||||
|
||||
/* PSCI reset handling for a vcpu. */
|
||||
static inline void kvm_reset_vcpu_psci(struct kvm_vcpu *vcpu,
|
||||
struct vcpu_reset_state *reset_state)
|
||||
{
|
||||
unsigned long target_pc = reset_state->pc;
|
||||
|
||||
/* Gracefully handle Thumb2 entry point */
|
||||
if (vcpu_mode_is_32bit(vcpu) && (target_pc & 1)) {
|
||||
target_pc &= ~1UL;
|
||||
vcpu_set_thumb(vcpu);
|
||||
}
|
||||
|
||||
/* Propagate caller endianness */
|
||||
if (reset_state->be)
|
||||
kvm_vcpu_set_be(vcpu);
|
||||
|
||||
*vcpu_pc(vcpu) = target_pc;
|
||||
|
||||
/*
|
||||
* We may come from a state where either a PC update was
|
||||
* pending (SMC call resulting in PC being increpented to
|
||||
* skip the SMC) or a pending exception. Make sure we get
|
||||
* rid of all that, as this cannot be valid out of reset.
|
||||
*
|
||||
* Note that clearing the exception mask also clears PC
|
||||
* updates, but that's an implementation detail, and we
|
||||
* really want to make it explicit.
|
||||
*/
|
||||
vcpu_clear_flag(vcpu, PENDING_EXCEPTION);
|
||||
vcpu_clear_flag(vcpu, EXCEPT_MASK);
|
||||
vcpu_clear_flag(vcpu, INCREMENT_PC);
|
||||
vcpu_set_reg(vcpu, 0, reset_state->r0);
|
||||
}
|
||||
|
||||
#endif /* __ARM64_KVM_EMULATE_H__ */
|
||||
|
||||
@@ -1054,6 +1054,8 @@ struct kvm_vcpu_arch {
|
||||
#define INCREMENT_PC __vcpu_single_flag(iflags, BIT(1))
|
||||
/* Target EL/MODE (not a single flag, but let's abuse the macro) */
|
||||
#define EXCEPT_MASK __vcpu_single_flag(iflags, GENMASK(3, 1))
|
||||
/* Host-set: the hyp flushes the non-protected vCPU state in on entry */
|
||||
#define PKVM_HOST_STATE_DIRTY __vcpu_single_flag(iflags, BIT(4))
|
||||
|
||||
/* Helpers to encode exceptions with minimum fuss */
|
||||
#define __EXCEPT_MASK_VAL unpack_vcpu_flag(EXCEPT_MASK)
|
||||
|
||||
@@ -45,6 +45,9 @@ static inline bool kvm_pkvm_ext_allowed(struct kvm *kvm, long ext)
|
||||
return true;
|
||||
case KVM_CAP_ARM_MTE:
|
||||
return false;
|
||||
case KVM_CAP_ARM_EAGER_SPLIT_CHUNK_SIZE:
|
||||
case KVM_CAP_ARM_SUPPORTED_BLOCK_SIZES:
|
||||
return false;
|
||||
default:
|
||||
return !kvm || !kvm_vm_is_protected(kvm);
|
||||
}
|
||||
@@ -195,7 +198,10 @@ struct pkvm_mapping {
|
||||
struct rb_node node;
|
||||
u64 gfn;
|
||||
u64 pfn;
|
||||
u64 nr_pages;
|
||||
struct {
|
||||
u64 nr_pages:48;
|
||||
u64 nc:1;
|
||||
};
|
||||
u64 __subtree_last; /* Internal member for interval tree */
|
||||
};
|
||||
|
||||
|
||||
@@ -736,6 +736,10 @@ void kvm_arch_vcpu_put(struct kvm_vcpu *vcpu)
|
||||
if (is_protected_kvm_enabled()) {
|
||||
kvm_call_hyp(__vgic_v3_save_aprs, &vcpu->arch.vgic_cpu.vgic_v3);
|
||||
kvm_call_hyp_nvhe(__pkvm_vcpu_put);
|
||||
|
||||
/* __pkvm_vcpu_put implies a sync of the state */
|
||||
if (!kvm_vm_is_protected(vcpu->kvm))
|
||||
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
|
||||
}
|
||||
|
||||
kvm_vcpu_put_debug(vcpu);
|
||||
@@ -967,6 +971,9 @@ int kvm_arch_vcpu_run_pid_change(struct kvm_vcpu *vcpu)
|
||||
return ret;
|
||||
|
||||
if (is_protected_kvm_enabled()) {
|
||||
/* Start with the vcpu in a dirty state */
|
||||
if (!kvm_vm_is_protected(vcpu->kvm))
|
||||
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
|
||||
ret = pkvm_create_hyp_vm(kvm);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
@@ -486,9 +486,32 @@ int handle_exit(struct kvm_vcpu *vcpu, int exception_index)
|
||||
}
|
||||
}
|
||||
|
||||
static void handle_exit_pkvm_state(struct kvm_vcpu *vcpu, int exception_index)
|
||||
{
|
||||
int exception_code = ARM_EXCEPTION_CODE(exception_index);
|
||||
|
||||
if (!is_protected_kvm_enabled() || kvm_vm_is_protected(vcpu->kvm))
|
||||
return;
|
||||
|
||||
/*
|
||||
* Sync the context back when the host will read (trap) or write
|
||||
* (SError) it. Preempt-off here, so the loaded hyp vCPU is stable.
|
||||
*/
|
||||
if (exception_code == ARM_EXCEPTION_TRAP ||
|
||||
exception_code == ARM_EXCEPTION_EL1_SERROR ||
|
||||
ARM_SERROR_PENDING(exception_index)) {
|
||||
kvm_call_hyp_nvhe(__pkvm_vcpu_sync_state);
|
||||
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
|
||||
} else {
|
||||
vcpu_clear_flag(vcpu, PKVM_HOST_STATE_DIRTY);
|
||||
}
|
||||
}
|
||||
|
||||
/* For exit types that need handling before we can be preempted */
|
||||
void handle_exit_early(struct kvm_vcpu *vcpu, int exception_index)
|
||||
{
|
||||
handle_exit_pkvm_state(vcpu, exception_index);
|
||||
|
||||
if (ARM_SERROR_PENDING(exception_index)) {
|
||||
if (this_cpu_has_cap(ARM64_HAS_RAS_EXTN)) {
|
||||
u64 disr = kvm_vcpu_get_disr(vcpu);
|
||||
|
||||
@@ -20,22 +20,6 @@
|
||||
#error Hypervisor code only!
|
||||
#endif
|
||||
|
||||
static inline u64 __vcpu_read_sys_reg(const struct kvm_vcpu *vcpu, int reg)
|
||||
{
|
||||
if (has_vhe())
|
||||
return vcpu_read_sys_reg(vcpu, reg);
|
||||
|
||||
return __vcpu_sys_reg(vcpu, reg);
|
||||
}
|
||||
|
||||
static inline void __vcpu_write_sys_reg(struct kvm_vcpu *vcpu, u64 val, int reg)
|
||||
{
|
||||
if (has_vhe())
|
||||
vcpu_write_sys_reg(vcpu, val, reg);
|
||||
else
|
||||
__vcpu_assign_sys_reg(vcpu, reg, val);
|
||||
}
|
||||
|
||||
static void __vcpu_write_spsr(struct kvm_vcpu *vcpu, unsigned long target_mode,
|
||||
u64 val)
|
||||
{
|
||||
@@ -101,14 +85,14 @@ static void enter_exception64(struct kvm_vcpu *vcpu, unsigned long target_mode,
|
||||
|
||||
switch (target_mode) {
|
||||
case PSR_MODE_EL1h:
|
||||
vbar = __vcpu_read_sys_reg(vcpu, VBAR_EL1);
|
||||
sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL1);
|
||||
__vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL1);
|
||||
vbar = vcpu_read_sys_reg(vcpu, VBAR_EL1);
|
||||
sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL1);
|
||||
vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL1);
|
||||
break;
|
||||
case PSR_MODE_EL2h:
|
||||
vbar = __vcpu_read_sys_reg(vcpu, VBAR_EL2);
|
||||
sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL2);
|
||||
__vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL2);
|
||||
vbar = vcpu_read_sys_reg(vcpu, VBAR_EL2);
|
||||
sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL2);
|
||||
vcpu_write_sys_reg(vcpu, *vcpu_pc(vcpu), ELR_EL2);
|
||||
break;
|
||||
default:
|
||||
/* Don't do that */
|
||||
@@ -185,7 +169,7 @@ static void enter_exception64(struct kvm_vcpu *vcpu, unsigned long target_mode,
|
||||
*/
|
||||
static unsigned long get_except32_cpsr(struct kvm_vcpu *vcpu, u32 mode)
|
||||
{
|
||||
u32 sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL1);
|
||||
u32 sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL1);
|
||||
unsigned long old, new;
|
||||
|
||||
old = *vcpu_cpsr(vcpu);
|
||||
@@ -281,7 +265,7 @@ static void enter_exception32(struct kvm_vcpu *vcpu, u32 mode, u32 vect_offset)
|
||||
{
|
||||
unsigned long spsr = *vcpu_cpsr(vcpu);
|
||||
bool is_thumb = (spsr & PSR_AA32_T_BIT);
|
||||
u32 sctlr = __vcpu_read_sys_reg(vcpu, SCTLR_EL1);
|
||||
u32 sctlr = vcpu_read_sys_reg(vcpu, SCTLR_EL1);
|
||||
u32 return_address;
|
||||
|
||||
*vcpu_cpsr(vcpu) = get_except32_cpsr(vcpu, mode);
|
||||
@@ -305,7 +289,7 @@ static void enter_exception32(struct kvm_vcpu *vcpu, u32 mode, u32 vect_offset)
|
||||
if (sctlr & (1 << 13))
|
||||
vect_offset += 0xffff0000;
|
||||
else /* always have security exceptions */
|
||||
vect_offset += __vcpu_read_sys_reg(vcpu, VBAR_EL1);
|
||||
vect_offset += vcpu_read_sys_reg(vcpu, VBAR_EL1);
|
||||
|
||||
*vcpu_pc(vcpu) = vect_offset;
|
||||
}
|
||||
|
||||
@@ -7,6 +7,8 @@
|
||||
#include <hyp/adjust_pc.h>
|
||||
#include <hyp/switch.h>
|
||||
|
||||
#include <linux/irqchip/arm-gic-v3.h>
|
||||
|
||||
#include <asm/pgtable-types.h>
|
||||
#include <asm/kvm_asm.h>
|
||||
#include <asm/kvm_emulate.h>
|
||||
@@ -102,16 +104,103 @@ static void fpsimd_sve_sync(struct kvm_vcpu *vcpu)
|
||||
*host_data_ptr(fp_owner) = FP_STATE_HOST_OWNED;
|
||||
}
|
||||
|
||||
static void flush_hyp_vgic_state(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
{
|
||||
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
|
||||
struct vgic_v3_cpu_if *host_cpu_if, *hyp_cpu_if;
|
||||
unsigned int used_lrs, i;
|
||||
|
||||
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
|
||||
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
|
||||
|
||||
used_lrs = host_cpu_if->used_lrs;
|
||||
used_lrs = min(used_lrs, hyp_gicv3_nr_lr);
|
||||
|
||||
hyp_cpu_if->vgic_hcr = host_cpu_if->vgic_hcr;
|
||||
/* Should be a one-off */
|
||||
hyp_cpu_if->vgic_sre = (ICC_SRE_EL1_DIB |
|
||||
ICC_SRE_EL1_DFB |
|
||||
ICC_SRE_EL1_SRE);
|
||||
hyp_cpu_if->used_lrs = used_lrs;
|
||||
|
||||
for (i = 0; i < used_lrs; i++)
|
||||
hyp_cpu_if->vgic_lr[i] = host_cpu_if->vgic_lr[i];
|
||||
}
|
||||
|
||||
static void sync_hyp_vgic_state(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
{
|
||||
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
|
||||
struct vgic_v3_cpu_if *host_cpu_if, *hyp_cpu_if;
|
||||
unsigned int i;
|
||||
|
||||
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
|
||||
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
|
||||
|
||||
host_cpu_if->vgic_hcr = hyp_cpu_if->vgic_hcr;
|
||||
host_cpu_if->vgic_vmcr = hyp_cpu_if->vgic_vmcr;
|
||||
|
||||
for (i = 0; i < hyp_cpu_if->used_lrs; i++)
|
||||
host_cpu_if->vgic_lr[i] = hyp_cpu_if->vgic_lr[i];
|
||||
}
|
||||
|
||||
static void __copy_vcpu_state(const struct kvm_vcpu *from_vcpu,
|
||||
struct kvm_vcpu *to_vcpu)
|
||||
{
|
||||
int i;
|
||||
|
||||
to_vcpu->arch.ctxt.regs = from_vcpu->arch.ctxt.regs;
|
||||
to_vcpu->arch.ctxt.spsr_abt = from_vcpu->arch.ctxt.spsr_abt;
|
||||
to_vcpu->arch.ctxt.spsr_und = from_vcpu->arch.ctxt.spsr_und;
|
||||
to_vcpu->arch.ctxt.spsr_irq = from_vcpu->arch.ctxt.spsr_irq;
|
||||
to_vcpu->arch.ctxt.spsr_fiq = from_vcpu->arch.ctxt.spsr_fiq;
|
||||
to_vcpu->arch.ctxt.fp_regs = from_vcpu->arch.ctxt.fp_regs;
|
||||
|
||||
/*
|
||||
* Copy the sysregs, but don't mess with the timer state which
|
||||
* is directly handled by EL1 and is expected to be preserved.
|
||||
* enum vcpu_sysreg is sparse: VNCR-mapped registers take values
|
||||
* derived from their VNCR page offset, so the timer registers do
|
||||
* not form a contiguous numeric range and must be skipped by name.
|
||||
*/
|
||||
for (i = 1; i < NR_SYS_REGS; i++) {
|
||||
switch (i) {
|
||||
case CNTVOFF_EL2:
|
||||
case CNTV_CVAL_EL0:
|
||||
case CNTV_CTL_EL0:
|
||||
case CNTP_CVAL_EL0:
|
||||
case CNTP_CTL_EL0:
|
||||
continue;
|
||||
}
|
||||
to_vcpu->arch.ctxt.sys_regs[i] = from_vcpu->arch.ctxt.sys_regs[i];
|
||||
}
|
||||
}
|
||||
|
||||
static void sync_hyp_vcpu_state(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
{
|
||||
__copy_vcpu_state(&hyp_vcpu->vcpu, hyp_vcpu->host_vcpu);
|
||||
}
|
||||
|
||||
static void flush_hyp_vcpu_state(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
{
|
||||
__copy_vcpu_state(hyp_vcpu->host_vcpu, &hyp_vcpu->vcpu);
|
||||
}
|
||||
|
||||
static void flush_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
{
|
||||
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
|
||||
|
||||
hyp_vcpu->vcpu.arch.debug_owner = host_vcpu->arch.debug_owner;
|
||||
|
||||
if (kvm_guest_owns_debug_regs(&hyp_vcpu->vcpu))
|
||||
if (kvm_guest_owns_debug_regs(&hyp_vcpu->vcpu)) {
|
||||
hyp_vcpu->vcpu.arch.vcpu_debug_state = host_vcpu->arch.vcpu_debug_state;
|
||||
else if (kvm_host_owns_debug_regs(&hyp_vcpu->vcpu))
|
||||
} else if (kvm_host_owns_debug_regs(&hyp_vcpu->vcpu)) {
|
||||
hyp_vcpu->vcpu.arch.external_debug_state = host_vcpu->arch.external_debug_state;
|
||||
/*
|
||||
* The world switch loads MDSCR_EL1 from external_mdscr_el1
|
||||
* (ctxt_mdscr_el1()).
|
||||
*/
|
||||
hyp_vcpu->vcpu.arch.external_mdscr_el1 = host_vcpu->arch.external_mdscr_el1;
|
||||
}
|
||||
}
|
||||
|
||||
static void sync_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
@@ -131,7 +220,17 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
fpsimd_sve_flush();
|
||||
flush_debug_state(hyp_vcpu);
|
||||
|
||||
hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt;
|
||||
/*
|
||||
* If we deal with a non-protected guest and the state is potentially
|
||||
* dirty (from a host perspective), copy the state back into the hyp
|
||||
* vcpu.
|
||||
*/
|
||||
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
|
||||
if (vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY))
|
||||
flush_hyp_vcpu_state(hyp_vcpu);
|
||||
} else {
|
||||
hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt;
|
||||
}
|
||||
|
||||
/* __hyp_running_vcpu must be NULL in a guest context. */
|
||||
hyp_vcpu->vcpu.arch.ctxt.__hyp_running_vcpu = NULL;
|
||||
@@ -150,13 +249,7 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
|
||||
hyp_vcpu->vcpu.arch.vsesr_el2 = host_vcpu->arch.vsesr_el2;
|
||||
|
||||
hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3 = host_vcpu->arch.vgic_cpu.vgic_v3;
|
||||
|
||||
/* Bound used_lrs by the number of implemented list registers. */
|
||||
hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3.used_lrs =
|
||||
min_t(unsigned int,
|
||||
hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3.used_lrs,
|
||||
hyp_gicv3_nr_lr);
|
||||
flush_hyp_vgic_state(hyp_vcpu);
|
||||
|
||||
hyp_vcpu->vcpu.arch.pid = host_vcpu->arch.pid;
|
||||
}
|
||||
@@ -164,25 +257,26 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
static void sync_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
|
||||
{
|
||||
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
|
||||
struct vgic_v3_cpu_if *hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
|
||||
struct vgic_v3_cpu_if *host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
|
||||
unsigned int i;
|
||||
|
||||
fpsimd_sve_sync(&hyp_vcpu->vcpu);
|
||||
sync_debug_state(hyp_vcpu);
|
||||
|
||||
host_vcpu->arch.ctxt = hyp_vcpu->vcpu.arch.ctxt;
|
||||
|
||||
host_vcpu->arch.hcr_el2 = hyp_vcpu->vcpu.arch.hcr_el2;
|
||||
if (pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
|
||||
host_vcpu->arch.ctxt = hyp_vcpu->vcpu.arch.ctxt;
|
||||
} else {
|
||||
/*
|
||||
* PC feeds trace_kvm_exit(), PSTATE.SS the host software-step
|
||||
* machine, and both run before the next on-demand ctxt sync.
|
||||
*/
|
||||
host_vcpu->arch.ctxt.regs.pc = hyp_vcpu->vcpu.arch.ctxt.regs.pc;
|
||||
host_vcpu->arch.ctxt.regs.pstate = hyp_vcpu->vcpu.arch.ctxt.regs.pstate;
|
||||
}
|
||||
|
||||
host_vcpu->arch.fault = hyp_vcpu->vcpu.arch.fault;
|
||||
|
||||
host_vcpu->arch.iflags = hyp_vcpu->vcpu.arch.iflags;
|
||||
|
||||
host_cpu_if->vgic_hcr = hyp_cpu_if->vgic_hcr;
|
||||
host_cpu_if->vgic_vmcr = hyp_cpu_if->vgic_vmcr;
|
||||
for (i = 0; i < hyp_cpu_if->used_lrs; ++i)
|
||||
host_cpu_if->vgic_lr[i] = hyp_cpu_if->vgic_lr[i];
|
||||
sync_hyp_vgic_state(hyp_vcpu);
|
||||
}
|
||||
|
||||
static void handle___pkvm_vcpu_load(struct kvm_cpu_context *host_ctxt)
|
||||
@@ -210,18 +304,78 @@ static void handle___pkvm_vcpu_put(struct kvm_cpu_context *host_ctxt)
|
||||
{
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
|
||||
|
||||
if (hyp_vcpu)
|
||||
if (hyp_vcpu) {
|
||||
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
|
||||
|
||||
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu) &&
|
||||
!vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY)) {
|
||||
sync_hyp_vcpu_state(hyp_vcpu);
|
||||
}
|
||||
|
||||
pkvm_put_hyp_vcpu(hyp_vcpu);
|
||||
}
|
||||
}
|
||||
|
||||
static void handle___pkvm_vcpu_sync_state(struct kvm_cpu_context *host_ctxt)
|
||||
{
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu;
|
||||
|
||||
hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
|
||||
if (!hyp_vcpu || pkvm_hyp_vcpu_is_protected(hyp_vcpu))
|
||||
return;
|
||||
|
||||
sync_hyp_vcpu_state(hyp_vcpu);
|
||||
}
|
||||
|
||||
static struct kvm_vcpu *__get_host_hyp_vcpus(struct kvm_vcpu *arg,
|
||||
struct pkvm_hyp_vcpu **hyp_vcpup)
|
||||
{
|
||||
struct kvm_vcpu *host_vcpu = kern_hyp_va(arg);
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu = NULL;
|
||||
|
||||
if (unlikely(is_protected_kvm_enabled())) {
|
||||
hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
|
||||
|
||||
if (!hyp_vcpu || hyp_vcpu->host_vcpu != host_vcpu) {
|
||||
hyp_vcpu = NULL;
|
||||
host_vcpu = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
*hyp_vcpup = hyp_vcpu;
|
||||
return host_vcpu;
|
||||
}
|
||||
|
||||
#define get_host_hyp_vcpus(ctxt, regnr, hyp_vcpup) \
|
||||
({ \
|
||||
DECLARE_REG(struct kvm_vcpu *, __vcpu, ctxt, regnr); \
|
||||
__get_host_hyp_vcpus(__vcpu, hyp_vcpup); \
|
||||
})
|
||||
|
||||
#define get_host_hyp_vcpus_from_vgic_v3_cpu_if(ctxt, regnr, hyp_vcpup) \
|
||||
({ \
|
||||
DECLARE_REG(struct vgic_v3_cpu_if *, cif, ctxt, regnr);\
|
||||
struct kvm_vcpu *__vcpu = container_of(cif, \
|
||||
struct kvm_vcpu, \
|
||||
arch.vgic_cpu.vgic_v3); \
|
||||
\
|
||||
__get_host_hyp_vcpus(__vcpu, hyp_vcpup); \
|
||||
})
|
||||
|
||||
static void handle___kvm_vcpu_run(struct kvm_cpu_context *host_ctxt)
|
||||
{
|
||||
DECLARE_REG(struct kvm_vcpu *, host_vcpu, host_ctxt, 1);
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu;
|
||||
struct kvm_vcpu *host_vcpu;
|
||||
int ret;
|
||||
|
||||
if (unlikely(is_protected_kvm_enabled())) {
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu = pkvm_get_loaded_hyp_vcpu();
|
||||
host_vcpu = get_host_hyp_vcpus(host_ctxt, 1, &hyp_vcpu);
|
||||
|
||||
if (!host_vcpu) {
|
||||
ret = -EINVAL;
|
||||
goto out;
|
||||
}
|
||||
|
||||
if (unlikely(hyp_vcpu)) {
|
||||
/*
|
||||
* KVM (and pKVM) doesn't support SME guests for now, and
|
||||
* ensures that SME features aren't enabled in pstate when
|
||||
@@ -233,23 +387,16 @@ static void handle___kvm_vcpu_run(struct kvm_cpu_context *host_ctxt)
|
||||
goto out;
|
||||
}
|
||||
|
||||
if (!hyp_vcpu) {
|
||||
ret = -EINVAL;
|
||||
goto out;
|
||||
}
|
||||
|
||||
flush_hyp_vcpu(hyp_vcpu);
|
||||
|
||||
ret = __kvm_vcpu_run(&hyp_vcpu->vcpu);
|
||||
|
||||
sync_hyp_vcpu(hyp_vcpu);
|
||||
} else {
|
||||
struct kvm_vcpu *vcpu = kern_hyp_va(host_vcpu);
|
||||
|
||||
/* The host is fully trusted, run its vCPU directly. */
|
||||
fpsimd_lazy_switch_to_guest(vcpu);
|
||||
ret = __kvm_vcpu_run(vcpu);
|
||||
fpsimd_lazy_switch_to_host(vcpu);
|
||||
fpsimd_lazy_switch_to_guest(host_vcpu);
|
||||
ret = __kvm_vcpu_run(host_vcpu);
|
||||
fpsimd_lazy_switch_to_host(host_vcpu);
|
||||
}
|
||||
out:
|
||||
cpu_reg(host_ctxt, 1) = ret;
|
||||
@@ -484,16 +631,63 @@ static void handle___vgic_v3_init_lrs(struct kvm_cpu_context *host_ctxt)
|
||||
|
||||
static void handle___vgic_v3_save_aprs(struct kvm_cpu_context *host_ctxt)
|
||||
{
|
||||
DECLARE_REG(struct vgic_v3_cpu_if *, cpu_if, host_ctxt, 1);
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu;
|
||||
struct kvm_vcpu *host_vcpu;
|
||||
|
||||
__vgic_v3_save_aprs(kern_hyp_va(cpu_if));
|
||||
host_vcpu = get_host_hyp_vcpus_from_vgic_v3_cpu_if(host_ctxt, 1,
|
||||
&hyp_vcpu);
|
||||
if (!host_vcpu)
|
||||
return;
|
||||
|
||||
if (unlikely(hyp_vcpu)) {
|
||||
struct vgic_v3_cpu_if *hyp_cpu_if, *host_cpu_if;
|
||||
int i;
|
||||
|
||||
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
|
||||
__vgic_v3_save_aprs(hyp_cpu_if);
|
||||
|
||||
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
|
||||
host_cpu_if->vgic_vmcr = hyp_cpu_if->vgic_vmcr;
|
||||
for (i = 0; i < ARRAY_SIZE(host_cpu_if->vgic_ap0r); i++) {
|
||||
host_cpu_if->vgic_ap0r[i] = hyp_cpu_if->vgic_ap0r[i];
|
||||
host_cpu_if->vgic_ap1r[i] = hyp_cpu_if->vgic_ap1r[i];
|
||||
}
|
||||
} else {
|
||||
__vgic_v3_save_aprs(&host_vcpu->arch.vgic_cpu.vgic_v3);
|
||||
}
|
||||
}
|
||||
|
||||
static void handle___vgic_v3_restore_vmcr_aprs(struct kvm_cpu_context *host_ctxt)
|
||||
{
|
||||
DECLARE_REG(struct vgic_v3_cpu_if *, cpu_if, host_ctxt, 1);
|
||||
struct pkvm_hyp_vcpu *hyp_vcpu;
|
||||
struct kvm_vcpu *host_vcpu;
|
||||
|
||||
__vgic_v3_restore_vmcr_aprs(kern_hyp_va(cpu_if));
|
||||
host_vcpu = get_host_hyp_vcpus_from_vgic_v3_cpu_if(host_ctxt, 1,
|
||||
&hyp_vcpu);
|
||||
if (!host_vcpu)
|
||||
return;
|
||||
|
||||
if (unlikely(hyp_vcpu)) {
|
||||
struct vgic_v3_cpu_if *hyp_cpu_if, *host_cpu_if;
|
||||
int i;
|
||||
|
||||
hyp_cpu_if = &hyp_vcpu->vcpu.arch.vgic_cpu.vgic_v3;
|
||||
host_cpu_if = &host_vcpu->arch.vgic_cpu.vgic_v3;
|
||||
|
||||
hyp_cpu_if->vgic_vmcr = host_cpu_if->vgic_vmcr;
|
||||
/* Should be a one-off */
|
||||
hyp_cpu_if->vgic_sre = (ICC_SRE_EL1_DIB |
|
||||
ICC_SRE_EL1_DFB |
|
||||
ICC_SRE_EL1_SRE);
|
||||
for (i = 0; i < ARRAY_SIZE(host_cpu_if->vgic_ap0r); i++) {
|
||||
hyp_cpu_if->vgic_ap0r[i] = host_cpu_if->vgic_ap0r[i];
|
||||
hyp_cpu_if->vgic_ap1r[i] = host_cpu_if->vgic_ap1r[i];
|
||||
}
|
||||
|
||||
__vgic_v3_restore_vmcr_aprs(hyp_cpu_if);
|
||||
} else {
|
||||
__vgic_v3_restore_vmcr_aprs(&host_vcpu->arch.vgic_cpu.vgic_v3);
|
||||
}
|
||||
}
|
||||
|
||||
static void handle___pkvm_init(struct kvm_cpu_context *host_ctxt)
|
||||
@@ -761,6 +955,7 @@ static const hcall_t host_hcall[] = {
|
||||
HANDLE_FUNC(__pkvm_finalize_teardown_vm),
|
||||
HANDLE_FUNC(__pkvm_vcpu_load),
|
||||
HANDLE_FUNC(__pkvm_vcpu_put),
|
||||
HANDLE_FUNC(__pkvm_vcpu_sync_state),
|
||||
HANDLE_FUNC(__pkvm_tlb_flush_vmid),
|
||||
};
|
||||
|
||||
|
||||
@@ -261,11 +261,18 @@ static void __apply_guest_page(void *va, size_t size,
|
||||
|
||||
static void clean_dcache_guest_page(void *va, size_t size)
|
||||
{
|
||||
/* See comment in __clean_dcache_guest_page() */
|
||||
if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB))
|
||||
return;
|
||||
|
||||
__apply_guest_page(va, size, __clean_dcache_guest_page);
|
||||
}
|
||||
|
||||
static void invalidate_icache_guest_page(void *va, size_t size)
|
||||
{
|
||||
if (alternative_has_cap_unlikely(ARM64_HAS_CACHE_DIC))
|
||||
return;
|
||||
|
||||
__apply_guest_page(va, size, __invalidate_icache_guest_page);
|
||||
}
|
||||
|
||||
|
||||
@@ -2113,11 +2113,14 @@ static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd)
|
||||
* Permission faults just need to update the existing leaf entry,
|
||||
* and so normally don't require allocations from the memcache. The
|
||||
* only exception to this is when dirty logging is enabled at runtime
|
||||
* and a write fault needs to collapse a block entry into a table.
|
||||
* and a fault needs to collapse a block entry into a table.
|
||||
* Under pKVM a permission fault can also collapse pages into a block,
|
||||
* which needs a fresh mapping object, and the hypervisor requires the
|
||||
* min-pages memcache even when the install allocates nothing.
|
||||
*/
|
||||
memcache = get_mmu_memcache(s2fd->vcpu);
|
||||
if (!perm_fault || (memslot_is_logging(s2fd->memslot) &&
|
||||
kvm_is_write_fault(s2fd->vcpu))) {
|
||||
if (!perm_fault || memslot_is_logging(s2fd->memslot) ||
|
||||
is_protected_kvm_enabled()) {
|
||||
ret = topup_mmu_memcache(s2fd->vcpu, memcache);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
+14
-7
@@ -366,7 +366,7 @@ static int __pkvm_pgtable_stage2_unshare(struct kvm_pgtable *pgt, u64 start, u64
|
||||
|
||||
for_each_mapping_in_range_safe(pgt, start, end, mapping) {
|
||||
ret = kvm_call_hyp_nvhe(__pkvm_host_unshare_guest, handle, mapping->gfn,
|
||||
mapping->nr_pages);
|
||||
(u64)mapping->nr_pages);
|
||||
if (WARN_ON(ret))
|
||||
return ret;
|
||||
pkvm_mapping_remove(mapping, &pgt->pkvm_mappings);
|
||||
@@ -463,13 +463,14 @@ int pkvm_pgtable_stage2_map(struct kvm_pgtable *pgt, u64 addr, u64 size,
|
||||
size / PAGE_SIZE, prot);
|
||||
}
|
||||
|
||||
if (WARN_ON(ret))
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
swap(mapping, cache->mapping);
|
||||
mapping->gfn = gfn;
|
||||
mapping->pfn = pfn;
|
||||
mapping->nr_pages = size / PAGE_SIZE;
|
||||
mapping->nc = !!(prot & (KVM_PGTABLE_PROT_DEVICE | KVM_PGTABLE_PROT_NORMAL_NC));
|
||||
pkvm_mapping_insert(mapping, &pgt->pkvm_mappings);
|
||||
|
||||
return ret;
|
||||
@@ -500,7 +501,7 @@ int pkvm_pgtable_stage2_wrprotect(struct kvm_pgtable *pgt, u64 addr, u64 size)
|
||||
lockdep_assert_held(&kvm->mmu_lock);
|
||||
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping) {
|
||||
ret = kvm_call_hyp_nvhe(__pkvm_host_wrprotect_guest, handle, mapping->gfn,
|
||||
mapping->nr_pages);
|
||||
(u64)mapping->nr_pages);
|
||||
if (WARN_ON(ret))
|
||||
break;
|
||||
}
|
||||
@@ -514,9 +515,15 @@ int pkvm_pgtable_stage2_flush(struct kvm_pgtable *pgt, u64 addr, u64 size)
|
||||
struct pkvm_mapping *mapping;
|
||||
|
||||
lockdep_assert_held(&kvm->mmu_lock);
|
||||
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping)
|
||||
__clean_dcache_guest_page(pfn_to_kaddr(mapping->pfn),
|
||||
PAGE_SIZE * mapping->nr_pages);
|
||||
|
||||
if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB))
|
||||
return 0;
|
||||
|
||||
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping) {
|
||||
if (!mapping->nc)
|
||||
__clean_dcache_guest_page(pfn_to_kaddr(mapping->pfn),
|
||||
PAGE_SIZE * mapping->nr_pages);
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -534,7 +541,7 @@ bool pkvm_pgtable_stage2_test_clear_young(struct kvm_pgtable *pgt, u64 addr, u64
|
||||
lockdep_assert_held(&kvm->mmu_lock);
|
||||
for_each_mapping_in_range_safe(pgt, addr, addr + size, mapping)
|
||||
young |= kvm_call_hyp_nvhe(__pkvm_host_test_clear_young_guest, handle, mapping->gfn,
|
||||
mapping->nr_pages, mkold);
|
||||
(u64)mapping->nr_pages, mkold);
|
||||
|
||||
return young;
|
||||
}
|
||||
|
||||
+1
-29
@@ -21,16 +21,6 @@
|
||||
* as described in ARM document number ARM DEN 0022A.
|
||||
*/
|
||||
|
||||
#define AFFINITY_MASK(level) ~((0x1UL << ((level) * MPIDR_LEVEL_BITS)) - 1)
|
||||
|
||||
static unsigned long psci_affinity_mask(unsigned long affinity_level)
|
||||
{
|
||||
if (affinity_level <= 3)
|
||||
return MPIDR_HWID_BITMASK & AFFINITY_MASK(affinity_level);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static unsigned long kvm_psci_vcpu_suspend(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
/*
|
||||
@@ -51,12 +41,6 @@ static unsigned long kvm_psci_vcpu_suspend(struct kvm_vcpu *vcpu)
|
||||
return PSCI_RET_SUCCESS;
|
||||
}
|
||||
|
||||
static inline bool kvm_psci_valid_affinity(struct kvm_vcpu *vcpu,
|
||||
unsigned long affinity)
|
||||
{
|
||||
return !(affinity & ~MPIDR_HWID_BITMASK);
|
||||
}
|
||||
|
||||
static unsigned long kvm_psci_vcpu_on(struct kvm_vcpu *source_vcpu)
|
||||
{
|
||||
struct vcpu_reset_state *reset_state;
|
||||
@@ -135,7 +119,7 @@ static unsigned long kvm_psci_vcpu_affinity_info(struct kvm_vcpu *vcpu)
|
||||
return PSCI_RET_INVALID_PARAMS;
|
||||
|
||||
/* Determine target affinity mask */
|
||||
target_affinity_mask = psci_affinity_mask(lowest_affinity_level);
|
||||
target_affinity_mask = kvm_psci_affinity_mask(lowest_affinity_level);
|
||||
if (!target_affinity_mask)
|
||||
return PSCI_RET_INVALID_PARAMS;
|
||||
|
||||
@@ -220,18 +204,6 @@ static void kvm_psci_system_suspend(struct kvm_vcpu *vcpu)
|
||||
run->exit_reason = KVM_EXIT_SYSTEM_EVENT;
|
||||
}
|
||||
|
||||
static void kvm_psci_narrow_to_32bit(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
int i;
|
||||
|
||||
/*
|
||||
* Zero the input registers' upper 32 bits. They will be fully
|
||||
* zeroed on exit, so we're fine changing them in place.
|
||||
*/
|
||||
for (i = 1; i < 4; i++)
|
||||
vcpu_set_reg(vcpu, i, lower_32_bits(vcpu_get_reg(vcpu, i)));
|
||||
}
|
||||
|
||||
static unsigned long kvm_psci_check_allowed_function(struct kvm_vcpu *vcpu, u32 fn)
|
||||
{
|
||||
/*
|
||||
|
||||
+3
-57
@@ -34,18 +34,6 @@
|
||||
static u32 __ro_after_init kvm_ipa_limit;
|
||||
unsigned int __ro_after_init kvm_host_sve_max_vl;
|
||||
|
||||
/*
|
||||
* ARMv8 Reset Values
|
||||
*/
|
||||
#define VCPU_RESET_PSTATE_EL1 (PSR_MODE_EL1h | PSR_A_BIT | PSR_I_BIT | \
|
||||
PSR_F_BIT | PSR_D_BIT)
|
||||
|
||||
#define VCPU_RESET_PSTATE_EL2 (PSR_MODE_EL2h | PSR_A_BIT | PSR_I_BIT | \
|
||||
PSR_F_BIT | PSR_D_BIT)
|
||||
|
||||
#define VCPU_RESET_PSTATE_SVC (PSR_AA32_MODE_SVC | PSR_AA32_A_BIT | \
|
||||
PSR_AA32_I_BIT | PSR_AA32_F_BIT)
|
||||
|
||||
unsigned int __ro_after_init kvm_sve_max_vl;
|
||||
|
||||
int __init kvm_arm_init_sve(void)
|
||||
@@ -191,7 +179,6 @@ void kvm_reset_vcpu(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
struct vcpu_reset_state reset_state;
|
||||
bool loaded;
|
||||
u32 pstate;
|
||||
|
||||
spin_lock(&vcpu->arch.mp_state_lock);
|
||||
reset_state = vcpu->arch.reset_state;
|
||||
@@ -210,21 +197,8 @@ void kvm_reset_vcpu(struct kvm_vcpu *vcpu)
|
||||
kvm_vcpu_reset_sve(vcpu);
|
||||
}
|
||||
|
||||
if (vcpu_el1_is_32bit(vcpu))
|
||||
pstate = VCPU_RESET_PSTATE_SVC;
|
||||
else if (vcpu_has_nv(vcpu))
|
||||
pstate = VCPU_RESET_PSTATE_EL2;
|
||||
else
|
||||
pstate = VCPU_RESET_PSTATE_EL1;
|
||||
|
||||
/* Reset core registers */
|
||||
memset(vcpu_gp_regs(vcpu), 0, sizeof(*vcpu_gp_regs(vcpu)));
|
||||
memset(&vcpu->arch.ctxt.fp_regs, 0, sizeof(vcpu->arch.ctxt.fp_regs));
|
||||
vcpu->arch.ctxt.spsr_abt = 0;
|
||||
vcpu->arch.ctxt.spsr_und = 0;
|
||||
vcpu->arch.ctxt.spsr_irq = 0;
|
||||
vcpu->arch.ctxt.spsr_fiq = 0;
|
||||
vcpu_gp_regs(vcpu)->pstate = pstate;
|
||||
kvm_reset_vcpu_core(vcpu);
|
||||
|
||||
/* Reset system registers */
|
||||
kvm_reset_sys_regs(vcpu);
|
||||
@@ -233,36 +207,8 @@ void kvm_reset_vcpu(struct kvm_vcpu *vcpu)
|
||||
* Additional reset state handling that PSCI may have imposed on us.
|
||||
* Must be done after all the sys_reg reset.
|
||||
*/
|
||||
if (reset_state.reset) {
|
||||
unsigned long target_pc = reset_state.pc;
|
||||
|
||||
/* Gracefully handle Thumb2 entry point */
|
||||
if (vcpu_mode_is_32bit(vcpu) && (target_pc & 1)) {
|
||||
target_pc &= ~1UL;
|
||||
vcpu_set_thumb(vcpu);
|
||||
}
|
||||
|
||||
/* Propagate caller endianness */
|
||||
if (reset_state.be)
|
||||
kvm_vcpu_set_be(vcpu);
|
||||
|
||||
*vcpu_pc(vcpu) = target_pc;
|
||||
|
||||
/*
|
||||
* We may come from a state where either a PC update was
|
||||
* pending (SMC call resulting in PC being increpented to
|
||||
* skip the SMC) or a pending exception. Make sure we get
|
||||
* rid of all that, as this cannot be valid out of reset.
|
||||
*
|
||||
* Note that clearing the exception mask also clears PC
|
||||
* updates, but that's an implementation detail, and we
|
||||
* really want to make it explicit.
|
||||
*/
|
||||
vcpu_clear_flag(vcpu, PENDING_EXCEPTION);
|
||||
vcpu_clear_flag(vcpu, EXCEPT_MASK);
|
||||
vcpu_clear_flag(vcpu, INCREMENT_PC);
|
||||
vcpu_set_reg(vcpu, 0, reset_state.r0);
|
||||
}
|
||||
if (reset_state.reset)
|
||||
kvm_reset_vcpu_psci(vcpu, &reset_state);
|
||||
|
||||
/* Reset timer */
|
||||
kvm_timer_vcpu_reset(vcpu);
|
||||
|
||||
@@ -976,21 +976,9 @@ static u64 reset_actlr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r)
|
||||
|
||||
static u64 reset_mpidr(struct kvm_vcpu *vcpu, const struct sys_reg_desc *r)
|
||||
{
|
||||
u64 mpidr;
|
||||
u64 mpidr = kvm_calculate_mpidr(vcpu);
|
||||
|
||||
/*
|
||||
* Map the vcpu_id into the first three affinity level fields of
|
||||
* the MPIDR. We limit the number of VCPUs in level 0 due to a
|
||||
* limitation to 16 CPUs in that level in the ICC_SGIxR registers
|
||||
* of the GICv3 to be able to address each CPU directly when
|
||||
* sending IPIs.
|
||||
*/
|
||||
mpidr = (vcpu->vcpu_id & 0x0f) << MPIDR_LEVEL_SHIFT(0);
|
||||
mpidr |= ((vcpu->vcpu_id >> 4) & 0xff) << MPIDR_LEVEL_SHIFT(1);
|
||||
mpidr |= ((vcpu->vcpu_id >> 12) & 0xff) << MPIDR_LEVEL_SHIFT(2);
|
||||
mpidr |= (1ULL << 31);
|
||||
vcpu_write_sys_reg(vcpu, mpidr, MPIDR_EL1);
|
||||
|
||||
return mpidr;
|
||||
}
|
||||
|
||||
|
||||
@@ -222,6 +222,25 @@ find_reg(const struct sys_reg_params *params, const struct sys_reg_desc table[],
|
||||
return __inline_bsearch((void *)pval, table, num, sizeof(table[0]), match_sys_reg);
|
||||
}
|
||||
|
||||
static inline u64 kvm_calculate_mpidr(const struct kvm_vcpu *vcpu)
|
||||
{
|
||||
u64 mpidr;
|
||||
|
||||
/*
|
||||
* Map the vcpu_id into the first three affinity level fields of
|
||||
* the MPIDR. We limit the number of VCPUs in level 0 due to a
|
||||
* limitation to 16 CPUs in that level in the ICC_SGIxR registers
|
||||
* of the GICv3 to be able to address each CPU directly when
|
||||
* sending IPIs.
|
||||
*/
|
||||
mpidr = (vcpu->vcpu_id & 0x0f) << MPIDR_LEVEL_SHIFT(0);
|
||||
mpidr |= ((vcpu->vcpu_id >> 4) & 0xff) << MPIDR_LEVEL_SHIFT(1);
|
||||
mpidr |= ((vcpu->vcpu_id >> 12) & 0xff) << MPIDR_LEVEL_SHIFT(2);
|
||||
mpidr |= (1ULL << 31);
|
||||
|
||||
return mpidr;
|
||||
}
|
||||
|
||||
const struct sys_reg_desc *get_reg_by_id(u64 id,
|
||||
const struct sys_reg_desc table[],
|
||||
unsigned int num);
|
||||
|
||||
@@ -38,6 +38,33 @@ static inline int kvm_psci_version(struct kvm_vcpu *vcpu)
|
||||
return KVM_ARM_PSCI_0_1;
|
||||
}
|
||||
|
||||
/* Narrow the PSCI register arguments (r1 to r3) to 32 bits. */
|
||||
static inline void kvm_psci_narrow_to_32bit(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
int i;
|
||||
|
||||
/*
|
||||
* Zero the input registers' upper 32 bits. They will be fully
|
||||
* zeroed on exit, so we're fine changing them in place.
|
||||
*/
|
||||
for (i = 1; i < 4; i++)
|
||||
vcpu_set_reg(vcpu, i, lower_32_bits(vcpu_get_reg(vcpu, i)));
|
||||
}
|
||||
|
||||
static inline bool kvm_psci_valid_affinity(struct kvm_vcpu *vcpu,
|
||||
unsigned long affinity)
|
||||
{
|
||||
return !(affinity & ~MPIDR_HWID_BITMASK);
|
||||
}
|
||||
|
||||
static inline unsigned long kvm_psci_affinity_mask(unsigned long affinity_level)
|
||||
{
|
||||
if (affinity_level <= 3)
|
||||
return MPIDR_HWID_BITMASK &
|
||||
~((0x1UL << (affinity_level * MPIDR_LEVEL_BITS)) - 1);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int kvm_psci_call(struct kvm_vcpu *vcpu);
|
||||
|
||||
|
||||
@@ -179,6 +179,7 @@ TEST_GEN_PROGS_arm64 += arm64/psci_test
|
||||
TEST_GEN_PROGS_arm64 += arm64/sea_to_user
|
||||
TEST_GEN_PROGS_arm64 += arm64/set_id_regs
|
||||
TEST_GEN_PROGS_arm64 += arm64/smccc_filter
|
||||
TEST_GEN_PROGS_arm64 += arm64/stage2_block_transitions
|
||||
TEST_GEN_PROGS_arm64 += arm64/vcpu_width_config
|
||||
TEST_GEN_PROGS_arm64 += arm64/vgic_init
|
||||
TEST_GEN_PROGS_arm64 += arm64/vgic_irq
|
||||
|
||||
@@ -527,6 +527,46 @@ void test_single_step_from_userspace(int test_cnt)
|
||||
kvm_vm_free(vm);
|
||||
}
|
||||
|
||||
static void guest_code_wp(void)
|
||||
{
|
||||
write_data = 'x';
|
||||
GUEST_DONE();
|
||||
}
|
||||
|
||||
/*
|
||||
* A userspace hardware watchpoint (KVM_GUESTDBG_USE_HW) must fire and report
|
||||
* the accessed address in debug.arch.far, exercising the watchpoint exit path.
|
||||
*/
|
||||
static void test_watchpoint_from_userspace(void)
|
||||
{
|
||||
struct kvm_guest_debug debug = {};
|
||||
struct kvm_vcpu *vcpu;
|
||||
struct kvm_run *run;
|
||||
struct kvm_vm *vm;
|
||||
|
||||
vm = vm_create_with_one_vcpu(&vcpu, guest_code_wp);
|
||||
run = vcpu->run;
|
||||
|
||||
debug.control = KVM_GUESTDBG_ENABLE | KVM_GUESTDBG_USE_HW;
|
||||
debug.arch.dbg_wcr[0] = DBGWCR_LEN8 | DBGWCR_RD | DBGWCR_WR |
|
||||
DBGWCR_EL1 | DBGWCR_E;
|
||||
/*
|
||||
* BAS = 0xff (LEN8) requires a doubleword-aligned DBGWVR; FAR still
|
||||
* reports the exact accessed byte.
|
||||
*/
|
||||
debug.arch.dbg_wvr[0] = PC(write_data) & ~7UL;
|
||||
vcpu_guest_debug_set(vcpu, &debug);
|
||||
|
||||
vcpu_run(vcpu);
|
||||
TEST_ASSERT(run->exit_reason == KVM_EXIT_DEBUG,
|
||||
"Expected KVM_EXIT_DEBUG, got %u", run->exit_reason);
|
||||
TEST_ASSERT((u64)run->debug.arch.far == PC(write_data),
|
||||
"Watchpoint FAR 0x%lx != accessed address 0x%lx",
|
||||
(u64)run->debug.arch.far, PC(write_data));
|
||||
|
||||
kvm_vm_free(vm);
|
||||
}
|
||||
|
||||
/*
|
||||
* Run debug testing using the various breakpoint#, watchpoint# and
|
||||
* context-aware breakpoint# with the given ID_AA64DFR0_EL1 configuration.
|
||||
@@ -600,6 +640,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
test_guest_debug_exceptions_all(aa64dfr0);
|
||||
test_single_step_from_userspace(ss_iteration);
|
||||
test_watchpoint_from_userspace();
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
// SPDX-License-Identifier: GPL-2.0-only
|
||||
/*
|
||||
* Copyright (c) 2026 Google LLC
|
||||
* Author: Fuad Tabba <fuad.tabba@linux.dev>
|
||||
*
|
||||
* stage2_block_transitions - Exercise stage-2 block/page granularity changes
|
||||
* that dirty logging forces at fault time, and assert the guest completes.
|
||||
*
|
||||
* Both scenarios need the fault handler to allocate at fault time (a fresh
|
||||
* mapping and/or page-table pages while holding mmu_lock), so a fault path
|
||||
* that fails to stage that memory manifests as a KVM_RUN error or, worse, a
|
||||
* host crash. The asserted property is host-agnostic: the guest runs the
|
||||
* sequence to completion and every KVM_RUN succeeds. On a pKVM host, where a
|
||||
* non-protected guest's stage-2 faults are serviced by the pkvm_pgtable_*()
|
||||
* backend, the same sequences also guard that backend's fault-time staging.
|
||||
*
|
||||
* Scenario 1 - block collapse on dirty-logging disable:
|
||||
* A write under dirty logging installs a 4K page; GET_DIRTY_LOG
|
||||
* re-write-protects it; logging is disabled; a second write takes a
|
||||
* permission fault that collapses the page into a hugetlb-backed block,
|
||||
* which requires a fresh mapping object under mmu_lock.
|
||||
*
|
||||
* Scenario 2 - block split under dirty logging:
|
||||
* Several hugetlb-backed blocks are faulted in as non-executable blocks,
|
||||
* dirty logging is enabled (write-protect only), then the guest executes
|
||||
* into each block. Each instruction fetch takes an execute permission
|
||||
* fault that must split the block into pages during logging, draining
|
||||
* page-table pages. Skipped on CTR_EL0.DIC hardware, where mappings are
|
||||
* made executable eagerly and the execute fault never occurs.
|
||||
*/
|
||||
#include <linux/bitfield.h>
|
||||
#include <linux/bitmap.h>
|
||||
#include <linux/mman.h>
|
||||
#include <linux/sizes.h>
|
||||
#include <sys/mman.h>
|
||||
|
||||
#include <asm/sysreg.h>
|
||||
|
||||
#include "kvm_util.h"
|
||||
#include "processor.h"
|
||||
#include "test_util.h"
|
||||
#include "ucall.h"
|
||||
|
||||
#define DATA_SLOT 1
|
||||
#define TEST_GVA 0xc0000000UL
|
||||
#define BLOCK_SIZE SZ_2M
|
||||
|
||||
/* AArch64 "ret" (ret x30): a self-contained, returnable executable payload. */
|
||||
#define RET_INSN 0xd65f03c0U
|
||||
|
||||
/*
|
||||
* A non-protected guest's per-VM stage-2 pool is seeded only with the PGD
|
||||
* donation, which stage-2 init immediately consumes, so the page-table budget
|
||||
* for a fault that does not top up is just the handful (~2x the stage-2 min
|
||||
* pages) of memcache leftovers. Executing into this many distinct blocks
|
||||
* demands far more than that budget: a fault path that tops up on every fault
|
||||
* completes all of them, one that skips non-write faults runs out mid-sequence.
|
||||
*/
|
||||
#define NR_BLOCKS 16
|
||||
|
||||
/* Scenario 2 guest -> host sync stages. */
|
||||
#define STAGE_SKIP_DIC 1
|
||||
#define STAGE_BLOCKS_READY 2
|
||||
|
||||
static void collapse_guest_code(u64 gva)
|
||||
{
|
||||
u64 *data = (u64 *)gva;
|
||||
|
||||
/* Under dirty logging: install a 4K writable page. */
|
||||
WRITE_ONCE(*data, 0x1);
|
||||
GUEST_SYNC(1);
|
||||
|
||||
/* Logging disabled: a permission fault collapses the page into a block. */
|
||||
WRITE_ONCE(*data, 0x2);
|
||||
GUEST_SYNC(2);
|
||||
|
||||
GUEST_DONE();
|
||||
}
|
||||
|
||||
static void test_block_collapse(void)
|
||||
{
|
||||
struct kvm_vcpu *vcpu;
|
||||
unsigned long *bmap;
|
||||
struct kvm_vm *vm;
|
||||
struct ucall uc;
|
||||
size_t npages;
|
||||
u64 gpa;
|
||||
|
||||
vm = vm_create_with_one_vcpu(&vcpu, collapse_guest_code);
|
||||
npages = BLOCK_SIZE / vm->page_size;
|
||||
|
||||
gpa = (vm_compute_max_gfn(vm) * vm->page_size) - BLOCK_SIZE;
|
||||
gpa = align_down(gpa, BLOCK_SIZE);
|
||||
|
||||
vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS_HUGETLB_2MB, gpa,
|
||||
DATA_SLOT, npages, KVM_MEM_LOG_DIRTY_PAGES);
|
||||
virt_map(vm, TEST_GVA, gpa, npages);
|
||||
vcpu_args_set(vcpu, 1, TEST_GVA);
|
||||
|
||||
bmap = bitmap_zalloc(BLOCK_SIZE / getpagesize());
|
||||
|
||||
vcpu_run(vcpu);
|
||||
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC && uc.args[1] == 1,
|
||||
"Expected first sync, got cmd %lu arg %lu", uc.cmd, uc.args[1]);
|
||||
|
||||
/* GET_DIRTY_LOG re-write-protects the dirtied page; then stop logging. */
|
||||
kvm_vm_get_dirty_log(vm, DATA_SLOT, bmap);
|
||||
vm_mem_region_set_flags(vm, DATA_SLOT, 0);
|
||||
|
||||
/* The collapsing permission fault: a broken fault path faults here. */
|
||||
vcpu_run(vcpu);
|
||||
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC && uc.args[1] == 2,
|
||||
"Expected second sync, got cmd %lu arg %lu", uc.cmd, uc.args[1]);
|
||||
|
||||
vcpu_run(vcpu);
|
||||
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_DONE,
|
||||
"Expected done, got cmd %lu", uc.cmd);
|
||||
|
||||
free(bmap);
|
||||
kvm_vm_free(vm);
|
||||
}
|
||||
|
||||
static void guest_sync_insn(u64 va)
|
||||
{
|
||||
/* Make the just-written instruction coherent for execution (!DIC). */
|
||||
asm volatile("dc cvau, %0\n"
|
||||
"dsb ish\n"
|
||||
"ic ivau, %0\n"
|
||||
"dsb ish\n"
|
||||
"isb\n"
|
||||
:: "r" (va) : "memory");
|
||||
}
|
||||
|
||||
static void split_guest_code(u64 base_gva, u64 nblocks)
|
||||
{
|
||||
u64 i, va;
|
||||
|
||||
if (FIELD_GET(CTR_EL0_DIC_MASK, read_sysreg(ctr_el0))) {
|
||||
GUEST_SYNC(STAGE_SKIP_DIC);
|
||||
GUEST_DONE();
|
||||
return;
|
||||
}
|
||||
|
||||
/* Fault in each block (non-executable) and stage an executable payload. */
|
||||
for (i = 0; i < nblocks; i++) {
|
||||
va = base_gva + i * BLOCK_SIZE;
|
||||
WRITE_ONCE(*(u32 *)va, RET_INSN);
|
||||
guest_sync_insn(va);
|
||||
}
|
||||
GUEST_SYNC(STAGE_BLOCKS_READY);
|
||||
|
||||
/* Logging is now on: executing into each block splits it into pages. */
|
||||
for (i = 0; i < nblocks; i++) {
|
||||
va = base_gva + i * BLOCK_SIZE;
|
||||
((void (*)(void))va)();
|
||||
}
|
||||
|
||||
GUEST_DONE();
|
||||
}
|
||||
|
||||
static void test_exec_split_drain(void)
|
||||
{
|
||||
struct kvm_vcpu *vcpu;
|
||||
struct kvm_vm *vm;
|
||||
struct ucall uc;
|
||||
size_t npages;
|
||||
u64 gpa;
|
||||
|
||||
vm = vm_create_with_one_vcpu(&vcpu, split_guest_code);
|
||||
npages = NR_BLOCKS * (BLOCK_SIZE / vm->page_size);
|
||||
|
||||
gpa = (vm_compute_max_gfn(vm) * vm->page_size) - NR_BLOCKS * BLOCK_SIZE;
|
||||
gpa = align_down(gpa, BLOCK_SIZE);
|
||||
|
||||
vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS_HUGETLB_2MB, gpa,
|
||||
DATA_SLOT, npages, 0);
|
||||
virt_map(vm, TEST_GVA, gpa, npages);
|
||||
vcpu_args_set(vcpu, 2, TEST_GVA, (u64)NR_BLOCKS);
|
||||
|
||||
vcpu_run(vcpu);
|
||||
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_SYNC,
|
||||
"Expected sync, got cmd %lu", uc.cmd);
|
||||
if (uc.args[1] == STAGE_SKIP_DIC) {
|
||||
ksft_print_msg("SKIP block split: CTR_EL0.DIC == 1\n");
|
||||
kvm_vm_free(vm);
|
||||
return;
|
||||
}
|
||||
TEST_ASSERT(uc.args[1] == STAGE_BLOCKS_READY,
|
||||
"Expected blocks-ready sync, got arg %lu", uc.args[1]);
|
||||
|
||||
/* Write-protect the blocks; the guest then splits them by executing. */
|
||||
vm_mem_region_set_flags(vm, DATA_SLOT, KVM_MEM_LOG_DIRTY_PAGES);
|
||||
|
||||
vcpu_run(vcpu);
|
||||
TEST_ASSERT(get_ucall(vcpu, &uc) == UCALL_DONE,
|
||||
"Expected done, got cmd %lu", uc.cmd);
|
||||
|
||||
kvm_vm_free(vm);
|
||||
}
|
||||
|
||||
/*
|
||||
* The explicit-size hugetlb backing hard-fails region creation if the pages
|
||||
* are not already reserved, so probe here and skip rather than abort. The
|
||||
* peak reservation is scenario 2's; the two scenarios run and free in turn.
|
||||
*/
|
||||
static void require_hugepages(size_t bytes)
|
||||
{
|
||||
void *mem = mmap(NULL, bytes, PROT_READ | PROT_WRITE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB | MAP_HUGE_2MB,
|
||||
-1, 0);
|
||||
|
||||
if (mem == MAP_FAILED)
|
||||
ksft_exit_skip("Need %zu bytes of reserved 2M hugepages\n", bytes);
|
||||
munmap(mem, bytes);
|
||||
}
|
||||
|
||||
int main(void)
|
||||
{
|
||||
require_hugepages(NR_BLOCKS * BLOCK_SIZE);
|
||||
|
||||
test_block_collapse();
|
||||
test_exec_split_drain();
|
||||
|
||||
ksft_print_msg("All ok!\n");
|
||||
return 0;
|
||||
}
|
||||
Reference in New Issue
Block a user