use alloc::{format, sync::Arc};
use crate::{
AsVCpuTask, AxVmResult, GuestPhysAddr, StopReason, VCpuTask, VmStatus, VmVcpuState,
arch::{ArchOps, CurrentArch, VcpuRunAction},
ax_err_type,
runtime::{VCpuRef, VMRef, sub_running_vm_count},
vm::VmRuntimeHandle,
};
const KERNEL_STACK_SIZE: usize = 0x40000;
fn wait(vm_vcpus: &VmRuntimeHandle) {
vm_vcpus.wait();
}
fn wait_for<F>(vm_vcpus: &VmRuntimeHandle, condition: F)
where
F: Fn() -> bool,
{
vm_vcpus.wait_until(condition);
}
pub(crate) fn notify_primary_vcpu(vm_id: usize) {
let Some(vm) = crate::get_vm_by_id(vm_id) else {
warn!("VM[{vm_id}] not found while notifying primary vCPU");
return;
};
if let Err(err) = vm.with_runtime(|runtime| {
runtime.notify_one();
Ok(())
}) {
warn!("VM[{vm_id}] vCPU runtime not found: {err:?}");
}
}
pub(crate) fn notify_all_vcpus(vm_id: usize) {
if let Some(vm) = crate::get_vm_by_id(vm_id) {
let _ = vm.with_runtime(|runtime| {
runtime.notify_all();
Ok(())
});
}
}
pub(crate) fn queue_interrupt(vm_id: usize, vcpu_id: usize, vector: usize) -> AxVmResult {
let vm = crate::get_vm_by_id(vm_id)
.ok_or_else(|| ax_err_type!(NotFound, format!("VM[{vm_id}] not found")))?;
if !matches!(vm.status(), VmStatus::Running | VmStatus::Paused) {
return Err(ax_err_type!(
BadState,
format!("VM[{vm_id}] is not accepting interrupts")
));
}
let cpu_id = vm.with_runtime(|runtime| runtime.queue_interrupt(vcpu_id, vector))?;
vm.with_runtime(|runtime| {
runtime.notify_all();
Ok(())
})?;
crate::host::task::send_ipi(cpu_id);
Ok(())
}
#[expect(
dead_code,
reason = "only the LoongArch IRQ backend queues physical interrupts"
)]
pub(crate) fn queue_external_interrupt(
vm_id: usize,
vcpu_id: usize,
vector: usize,
physical_irq: usize,
) -> AxVmResult {
let vm = crate::get_vm_by_id(vm_id)
.ok_or_else(|| ax_err_type!(NotFound, format!("VM[{vm_id}] not found")))?;
if !matches!(vm.status(), VmStatus::Running | VmStatus::Paused) {
return Err(ax_err_type!(
BadState,
format!("VM[{vm_id}] is not accepting interrupts")
));
}
let cpu_id =
vm.with_runtime(|runtime| runtime.queue_external_interrupt(vcpu_id, vector, physical_irq))?;
vm.with_runtime(|runtime| {
runtime.notify_all();
Ok(())
})?;
crate::host::task::send_ipi(cpu_id);
Ok(())
}
pub(crate) fn inject_pending_interrupts<A: ArchOps>(
vm_id: usize,
vcpu_id: usize,
vcpu: &crate::vm::AxVCpuRef<A::VCpu>,
) {
let Some(vm) = crate::get_vm_by_id(vm_id) else {
warn!("VM[{vm_id}] not found, cannot drain VCpu[{vcpu_id}] interrupts");
return;
};
let Ok(interrupts) = vm.with_runtime(|runtime| Ok(runtime.drain_pending_interrupts(vcpu_id)))
else {
warn!("VM[{vm_id}] vCPU runtime not found, cannot drain VCpu[{vcpu_id}] interrupts");
return;
};
for interrupt in interrupts {
A::inject_pending_interrupt(&vm, vcpu, interrupt);
}
}
pub(crate) fn cleanup_vm_vcpus(vm_id: usize) {
if let Some(vm) = crate::get_vm_by_id(vm_id)
&& let Err(err) = vm.with_runtime(|runtime| {
runtime.join_all_vcpu_tasks(vm_id);
Ok(())
})
{
warn!("VM[{vm_id}] vCPU runtime cleanup skipped: {err:?}");
}
}
fn mark_vcpu_running(vm: &VMRef) {
let _ = vm.with_runtime(|runtime| {
runtime.mark_vcpu_running();
Ok(())
});
}
#[cfg(test)]
type CpuOnStartAckLock<T> = std::sync::Mutex<T>;
#[cfg(not(test))]
type CpuOnStartAckLock<T> = ax_kspin::SpinNoIrq<T>;
#[allow(dead_code)]
pub(crate) struct CpuOnStartAck {
inner: CpuOnStartAckLock<CpuOnStartAckInner>,
}
struct CpuOnStartAckInner {
started: bool,
cancelled: bool,
result: Option<crate::AxVmResult>,
}
#[allow(dead_code)]
impl CpuOnStartAck {
pub(crate) fn new() -> Self {
Self {
inner: CpuOnStartAckLock::new(CpuOnStartAckInner {
started: false,
cancelled: false,
result: None,
}),
}
}
pub(crate) fn begin_startup(&self) -> bool {
let mut inner = self.lock_inner();
if inner.cancelled {
false
} else {
inner.started = true;
true
}
}
pub(crate) fn cancel_before_startup(&self) -> bool {
let mut inner = self.lock_inner();
if inner.started || inner.result.is_some() {
false
} else {
inner.cancelled = true;
true
}
}
pub(crate) fn is_cancelled(&self) -> bool {
self.lock_inner().cancelled
}
pub(crate) fn complete(&self, result: crate::AxVmResult) {
self.lock_inner().result = Some(result);
}
pub(crate) fn is_complete(&self) -> bool {
self.lock_inner().result.is_some()
}
pub(crate) fn take_result(&self) -> Option<crate::AxVmResult> {
self.lock_inner().result.take()
}
#[cfg(test)]
fn lock_inner(&self) -> impl core::ops::DerefMut<Target = CpuOnStartAckInner> + '_ {
self.inner.lock().unwrap()
}
#[cfg(not(test))]
fn lock_inner(&self) -> impl core::ops::DerefMut<Target = CpuOnStartAckInner> + '_ {
self.inner.lock()
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[allow(dead_code)]
pub(crate) enum VcpuOnError {
AlreadyOn,
OnPending,
StartFailed,
}
#[allow(dead_code)]
pub(crate) fn vcpu_on(
vm: VMRef,
vcpu_id: usize,
entry_point: GuestPhysAddr,
arg: usize,
) -> Result<(), VcpuOnError> {
let vcpu = vm
.vcpu_list()
.get(vcpu_id)
.cloned()
.ok_or(VcpuOnError::StartFailed)?;
match vcpu.state() {
VmVcpuState::Free => {}
VmVcpuState::Starting => return Err(VcpuOnError::OnPending),
VmVcpuState::Ready | VmVcpuState::Running => return Err(VcpuOnError::AlreadyOn),
_ => return Err(VcpuOnError::StartFailed),
}
vcpu.reserve_for_cpu_on()
.map_err(|_| VcpuOnError::OnPending)?;
let start_result = (|| {
let runtime = vm
.with_runtime(|runtime| Ok(runtime.clone()))
.map_err(|_| VcpuOnError::StartFailed)?;
if runtime.has_vcpu_task(vcpu_id) {
return Err(VcpuOnError::StartFailed);
}
vcpu.set_entry(entry_point)
.map_err(|_| VcpuOnError::StartFailed)?;
CurrentArch::set_vcpu_on_args(&vcpu, vcpu_id, arg);
let ack = Arc::new(CpuOnStartAck::new());
runtime
.insert_cpu_on_start_ack(vcpu_id, ack.clone())
.map_err(|_| VcpuOnError::StartFailed)?;
let vcpu_task = alloc_vcpu_task(&vm, vcpu.clone());
if runtime.add_vcpu_task(vcpu_id, vcpu_task).is_err() {
runtime.remove_cpu_on_start_ack(vcpu_id);
return Err(VcpuOnError::StartFailed);
}
runtime.notify_all();
runtime.wait_until(|| ack.is_complete() || !vm.running());
if !ack.is_complete() && !vm.running() {
if ack.cancel_before_startup() {
runtime.notify_all();
if let Some(task) = runtime.remove_vcpu_task(vcpu_id) {
let _ = task.join();
}
runtime.remove_cpu_on_start_ack(vcpu_id);
return Err(VcpuOnError::StartFailed);
}
runtime.wait_until(|| ack.is_complete());
}
let result = ack.take_result().unwrap_or_else(|| {
Err(ax_err_type!(
BadState,
format!("vCPU {vcpu_id} CPU_ON startup did not complete")
))
});
runtime.remove_cpu_on_start_ack(vcpu_id);
if result.is_err() {
runtime.remove_vcpu_task(vcpu_id);
return Err(VcpuOnError::StartFailed);
}
Ok(())
})();
if start_result.is_err() && vcpu.state() == VmVcpuState::Starting {
vcpu.rollback_cpu_on();
}
start_result
}
#[allow(dead_code)]
pub(crate) fn alloc_vcpu_task(vm: &VMRef, vcpu: VCpuRef) -> crate::AxTaskRef {
crate::host::task::spawn_task(build_vcpu_task(vm, vcpu))
}
fn spawn_deferred_reset_task(vm_id: usize) {
let reset_task = crate::TaskInner::new(
move || {
if let Err(err) = crate::runtime::reset_vm(vm_id) {
warn!("VM[{vm_id}] deferred reset failed: {err:?}");
crate::host::task::wait_queue_wake(&super::VMM, 1);
}
},
format!("VM[{vm_id}]-reset"),
KERNEL_STACK_SIZE,
);
crate::host::task::spawn_task(reset_task);
}
pub(crate) fn build_vcpu_task(vm: &VMRef, vcpu: VCpuRef) -> crate::TaskInner {
info!("Spawning task for VM[{}] VCpu[{}]", vm.id(), vcpu.id());
let mut vcpu_task = crate::TaskInner::new(
vcpu_run,
format!("VM[{}]-VCpu[{}]", vm.id(), vcpu.id()),
KERNEL_STACK_SIZE,
);
if let Some(phys_cpu_set) = vcpu.phys_cpu_set() {
vcpu_task.set_cpumask(crate::host::task::cpu_mask_from_raw_bits(
vcpu_task_cpu_mask(vm.id(), vcpu.id(), phys_cpu_set),
));
}
let inner = VCpuTask::new(vm, vcpu);
*vcpu_task.task_ext_mut() = Some(crate::AxTaskExt::from_impl(inner));
info!(
"VCpu task {} created {:?}",
vcpu_task.id_name(),
vcpu_task.cpumask()
);
vcpu_task
}
fn vcpu_task_cpu_mask(vm_id: usize, vcpu_id: usize, requested_mask: usize) -> usize {
let enabled_mask = crate::percpu::enabled_cpu_mask();
if enabled_mask == 0 {
warn!(
"VM[{vm_id}] VCpu[{vcpu_id}] has no initialized host CPU mask; using requested mask \
{requested_mask:#x}"
);
return requested_mask;
}
let initialized_requested_mask = requested_mask & enabled_mask;
if initialized_requested_mask != 0 {
if initialized_requested_mask != requested_mask {
warn!(
"VM[{vm_id}] VCpu[{vcpu_id}] requested host CPU mask {requested_mask:#x}, but \
only {initialized_requested_mask:#x} is initialized for AxVM"
);
}
return initialized_requested_mask;
}
let fallback_mask = enabled_mask.isolate_lowest_one();
warn!(
"VM[{vm_id}] VCpu[{vcpu_id}] requested host CPU mask {requested_mask:#x}, but none of \
those CPUs initialized AxVM; using initialized host CPU mask {fallback_mask:#x}"
);
fallback_mask
}
fn vcpu_run() {
let curr = crate::host::task::current_task();
let vm = curr.as_vcpu_task().vm();
let vcpu = curr.as_vcpu_task().vcpu.clone();
let vm_id = vm.id();
let vcpu_id = vcpu.id();
let Ok(runtime) = vm.with_runtime(|runtime| Ok(runtime.clone())) else {
warn!("VM[{vm_id}] vCPU runtime not found, VCpu[{vcpu_id}] exiting");
return;
};
info!("VM[{}] VCpu[{}] waiting for running", vm.id(), vcpu.id());
let cpu_on_start_ack = runtime.cpu_on_start_ack(vcpu_id);
wait_for(&runtime, || {
vm.running()
|| cpu_on_start_ack
.as_ref()
.is_some_and(|ack| ack.is_cancelled())
});
if let Some(ack) = &cpu_on_start_ack {
if !ack.begin_startup() {
ack.complete(Err(ax_err_type!(
BadState,
format!("vCPU {vcpu_id} CPU_ON startup was cancelled")
)));
runtime.notify_all();
return;
}
match vcpu.bind_after_cpu_on_or_rollback() {
Ok(()) => {
CurrentArch::before_first_run(&vm, &vcpu);
runtime.publish_cpu_on_start_success(ack);
runtime.notify_all();
}
Err(err) => {
ack.complete(Err(err));
runtime.notify_all();
runtime.remove_cpu_on_start_ack(vcpu_id);
runtime.remove_vcpu_task(vcpu_id);
return;
}
}
} else {
CurrentArch::before_first_run(&vm, &vcpu);
mark_vcpu_running(&vm);
}
info!("VM[{}] VCpu[{}] running...", vm.id(), vcpu.id());
loop {
CurrentArch::before_vcpu_run(&vm, &vcpu);
match CurrentArch::run_vcpu(&vm, &vcpu) {
Ok(VcpuRunAction {
exits_vcpu: true, ..
}) => {
if let Err(err) = vcpu.power_off_after_cpu_off() {
warn!("VM[{vm_id}] VCpu[{vcpu_id}] CPU_OFF cleanup failed: {err:?}");
}
runtime.remove_vcpu_task(vcpu_id);
if !runtime.consume_cpu_off_reservation(vcpu_id) {
let _ = runtime.mark_vcpu_exiting();
}
break;
}
Ok(VcpuRunAction {
resets_vm: true, ..
}) => {
if runtime.request_deferred_reset()
&& let Err(err) = vm.stop(StopReason::Forced)
{
if vm.stopping() {
warn!("VM[{vm_id}] reset requested while VM is already stopping: {err:?}");
} else {
let _ = runtime.take_deferred_reset_request();
warn!("VM[{vm_id}] failed to request deferred reset stop: {err:?}");
if let Err(stop_err) = vm.stop(StopReason::Fault(format!("{err:?}"))) {
warn!(
"VM[{vm_id}] shutdown after reset request failure failed: \
{stop_err:?}"
);
}
}
}
notify_all_vcpus(vm_id);
}
Ok(VcpuRunAction {
stop_reason: Some(reason),
..
}) => {
if let Err(err) = vm.stop(reason) {
warn!("VM[{vm_id}] shutdown failed: {err:?}");
}
notify_all_vcpus(vm_id);
}
Ok(VcpuRunAction {
waits_for_event: true,
..
}) => wait(&runtime),
Ok(VcpuRunAction { .. }) => {}
Err(err) => {
error!("VM[{vm_id}] run VCpu[{vcpu_id}] get error {err:?}");
if let Err(err) = vm.stop(StopReason::Fault(format!("{err:?}"))) {
warn!("VM[{vm_id}] shutdown failed after vCPU error: {err:?}");
}
notify_all_vcpus(vm_id);
}
}
if vm.suspending() {
debug!(
"VM[{}] VCpu[{}] is suspended, waiting for resume...",
vm_id, vcpu_id
);
wait_for(&runtime, || !vm.suspending());
info!("VM[{}] VCpu[{}] resumed from suspend", vm_id, vcpu_id);
continue;
}
if vm.stopping() {
warn!(
"VM[{}] VCpu[{}] stopping because of VM stopping",
vm_id, vcpu_id
);
if runtime.mark_vcpu_exiting() {
let reset_after_stop = runtime.take_deferred_reset_request();
info!("VM[{vm_id}] VCpu[{vcpu_id}] last VCpu exiting, decreasing running VM count");
if let Err(err) = vm.finish_stop() {
warn!("VM[{vm_id}] finish stop failed: {err:?}");
}
info!("VM[{}] state changed to Stopped", vm_id);
CurrentArch::on_last_vcpu_exit(&vm);
sub_running_vm_count(1);
if reset_after_stop {
spawn_deferred_reset_task(vm_id);
} else {
crate::host::task::wait_queue_wake(&super::VMM, 1);
}
}
break;
}
}
info!("VM[{}] VCpu[{}] exiting...", vm_id, vcpu_id);
}
#[cfg(test)]
mod cpu_on_start_ack_tests {
use super::*;
#[test]
fn cpu_on_start_ack_cancel_before_startup_blocks_late_startup() {
let ack = CpuOnStartAck::new();
assert!(ack.cancel_before_startup());
assert!(ack.is_cancelled());
assert!(!ack.begin_startup());
ack.complete(Err(ax_err_type!(
BadState,
"vCPU 1 CPU_ON startup was cancelled"
)));
assert!(ack.is_complete());
assert!(ack.take_result().unwrap().is_err());
}
}