virt/generic.rs
1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4mod partition_memory_map;
5
6pub use partition_memory_map::PartitionHostAccess;
7pub use partition_memory_map::PartitionMemoryMap;
8pub use vm_topology::processor::VpIndex;
9
10use crate::CpuidLeaf;
11use crate::PartitionCapabilities;
12use crate::io::CpuIo;
13use crate::irqcon::ControlGic;
14use crate::irqcon::IoApicRouting;
15use crate::irqcon::MsiRequest;
16use crate::irqfd::IrqFd;
17use crate::x86::DebugState;
18use crate::x86::HardwareBreakpoint;
19use guestmem::DoorbellRegistration;
20use guestmem::GuestMemory;
21use guestmem::GuestMemoryBackingError;
22use hvdef::Vtl;
23use inspect::Inspect;
24use inspect::InspectMut;
25use memory_range::MemoryRange;
26use pci_core::msi::SignalMsi;
27use std::cell::Cell;
28use std::convert::Infallible;
29use std::fmt::Debug;
30use std::future::Future;
31use std::future::poll_fn;
32use std::pin::pin;
33use std::sync::Arc;
34use std::sync::atomic::AtomicBool;
35use std::sync::atomic::Ordering;
36use std::task::Poll;
37use std::task::Waker;
38use vm_topology::memory::MemoryLayout;
39use vm_topology::processor::ProcessorTopology;
40use vmcore::reference_time::ReferenceTimeSource;
41use vmcore::vmtime::VmTimeSource;
42use vmcore::vpci_msi::MapVpciInterrupt;
43use vmcore::vpci_msi::MsiAddressData;
44use vmcore::vpci_msi::RegisterInterruptError;
45use vmcore::vpci_msi::VpciInterruptParameters;
46
47/// Platform capabilities detected from the hypervisor before partition
48/// creation. On x86 there are currently no pre-partition queries.
49#[cfg(guest_arch = "x86_64")]
50#[derive(Debug, Clone, Default)]
51pub struct PlatformInfo {}
52
53/// Platform capabilities detected from the hypervisor before partition
54/// creation.
55#[cfg(guest_arch = "aarch64")]
56#[derive(Debug, Clone)]
57pub struct PlatformInfo {
58 /// The platform PMU GSIV (GIC INTID), if available.
59 pub platform_gsiv: Option<u32>,
60 /// Whether the hypervisor supports GICv3. When `false`, only
61 /// GICv2 is available (e.g., Raspberry Pi 5 with GIC-400).
62 pub supports_gic_v3: bool,
63 /// Whether the hypervisor supports an in-kernel GICv3 ITS for
64 /// MSI delivery via LPIs. When `true`, the topology can include
65 /// a `GicItsInfo` and the backend will create/manage the ITS device.
66 pub supports_its: bool,
67 /// How the physical SMMU implementation selects the IOVA range reserved
68 /// for device-assignment MSI writes.
69 pub device_assignment_msi_iova: DeviceAssignmentMsiIova,
70}
71
72/// Selection policy for the device-assignment MSI IOVA reservation.
73#[cfg(guest_arch = "aarch64")]
74#[derive(Debug, Clone, Copy)]
75pub enum DeviceAssignmentMsiIova {
76 /// Device assignment does not expose an MSI IOVA reservation contract.
77 Unsupported,
78 /// The physical SMMU driver requires this exact range.
79 Fixed(MemoryRange),
80 /// The VMM selects the range and passes its base to the physical SMMU
81 /// implementation during partition creation.
82 Configurable,
83}
84
85/// A hypervisor backend capable of creating partitions.
86///
87/// # Recognized features
88///
89/// The `recognizes_*` methods report whether the backend acts on an optional
90/// partition request rather than silently ignoring it: it either honors the
91/// request or fails partition creation with a specific error. They let the code
92/// assembling a [`ProtoPartitionConfig`] reject a request up front when the
93/// backend has no concept of it, instead of the request being quietly dropped.
94/// Recognition is *not* a promise that the request succeeds — the backend may
95/// still reject it in combination with another feature, or fail later during
96/// partition creation. Each method defaults to `false`, so a new optional
97/// feature is unrecognized everywhere until a backend overrides its method.
98pub trait Hypervisor: 'static {
99 /// The prototype partition type.
100 type ProtoPartition<'a>: ProtoPartition<Partition = Self::Partition>;
101 /// The partition type.
102 type Partition;
103 /// The error type when creating the partition.
104 type Error: std::error::Error + Send + Sync + 'static;
105
106 /// Returns platform capabilities detected from the hypervisor.
107 ///
108 /// This is called before partition creation to query platform-specific
109 /// information needed for topology construction and firmware table
110 /// generation.
111 fn platform_info(&self) -> PlatformInfo;
112
113 /// Whether the backend recognizes a request to expose hardware
114 /// virtualization (VMX/SVM) to the guest so it can run its own hypervisor.
115 /// See the [`Hypervisor`] trait docs on recognized features.
116 fn recognizes_nested_virt(&self) -> bool {
117 false
118 }
119
120 /// Returns a new prototype partition from the given configuration.
121 fn new_partition<'a>(
122 &'a mut self,
123 config: ProtoPartitionConfig<'a>,
124 ) -> Result<Self::ProtoPartition<'a>, Self::Error>;
125}
126
127/// Isolation type for a partition.
128#[derive(Eq, PartialEq, Debug, Copy, Clone, Inspect)]
129pub enum IsolationType {
130 /// No isolation.
131 None,
132 /// Hypervisor based isolation.
133 Vbs,
134 /// Secure nested paging (AMD SEV-SNP) - hardware based isolation.
135 Snp,
136 /// Trust domain extensions (Intel TDX) - hardware based isolation.
137 Tdx,
138 /// Confidential Compute Architecture (ARM CCA) - hardware based isolation.
139 Cca,
140}
141
142impl IsolationType {
143 /// Returns true if the isolation type is not `None`.
144 pub fn is_isolated(&self) -> bool {
145 !matches!(self, Self::None)
146 }
147
148 /// Returns whether the isolation type is hardware-backed.
149 pub fn is_hardware_isolated(&self) -> bool {
150 matches!(self, Self::Snp | Self::Tdx | Self::Cca)
151 }
152}
153
154/// An unexpected isolation type was provided.
155#[derive(Debug)]
156pub struct UnexpectedIsolationType;
157
158impl IsolationType {
159 pub const fn from_hv(
160 value: hvdef::HvPartitionIsolationType,
161 ) -> Result<Self, UnexpectedIsolationType> {
162 match value {
163 hvdef::HvPartitionIsolationType::NONE => Ok(IsolationType::None),
164 hvdef::HvPartitionIsolationType::VBS => Ok(IsolationType::Vbs),
165 hvdef::HvPartitionIsolationType::SNP => Ok(IsolationType::Snp),
166 hvdef::HvPartitionIsolationType::TDX => Ok(IsolationType::Tdx),
167 hvdef::HvPartitionIsolationType::CCA => Ok(IsolationType::Cca),
168 _ => Err(UnexpectedIsolationType),
169 }
170 }
171
172 pub const fn to_hv(self) -> hvdef::HvPartitionIsolationType {
173 match self {
174 IsolationType::None => hvdef::HvPartitionIsolationType::NONE,
175 IsolationType::Vbs => hvdef::HvPartitionIsolationType::VBS,
176 IsolationType::Snp => hvdef::HvPartitionIsolationType::SNP,
177 IsolationType::Tdx => hvdef::HvPartitionIsolationType::TDX,
178 IsolationType::Cca => hvdef::HvPartitionIsolationType::CCA,
179 }
180 }
181}
182
183/// Page visibility types for isolated partitions.
184#[derive(Eq, PartialEq, Debug, Copy, Clone, Inspect)]
185pub enum PageVisibility {
186 /// The guest has exclusive access to the page, and no access from the host.
187 Exclusive,
188 /// The page has shared access with the guest and host.
189 Shared,
190}
191
192/// Initial page import type for isolated partitions.
193#[derive(Eq, PartialEq, Debug, Copy, Clone, Inspect)]
194pub enum InitialPageImportType {
195 /// A measured page with exclusive guest access.
196 Normal,
197 /// An unmeasured page with exclusive guest access.
198 NormalUnmeasured,
199 /// A page shared between the guest and host.
200 Shared,
201 /// A virtual processor context page.
202 VpContext,
203 /// An SNP secrets page.
204 Secrets,
205 /// An SNP CPUID page.
206 Cpuid,
207 /// An SNP CPUID extended state page.
208 CpuidExtendedState,
209}
210
211impl InitialPageImportType {
212 /// Returns the visibility implied by this import type.
213 pub fn page_visibility(self) -> PageVisibility {
214 match self {
215 Self::Shared => PageVisibility::Shared,
216 Self::Normal
217 | Self::NormalUnmeasured
218 | Self::VpContext
219 | Self::Secrets
220 | Self::Cpuid
221 | Self::CpuidExtendedState => PageVisibility::Exclusive,
222 }
223 }
224}
225
226/// Initial page import metadata for isolated partitions.
227#[derive(Eq, PartialEq, Debug, Clone)]
228pub struct InitialPageImport {
229 /// The guest physical range being imported.
230 pub range: MemoryRange,
231 /// The hypervisor-facing import type for this range.
232 pub import_type: InitialPageImportType,
233 /// Loader-provided debug tag identifying the source of this range.
234 pub tag: &'static str,
235}
236
237/// An opaque SNP virtual processor context.
238#[derive(Eq, PartialEq, Debug, Clone)]
239pub struct SnpVpContext {
240 /// The guest physical address associated with the context.
241 pub gpa: u64,
242 /// The virtual processor described by the context.
243 pub vp_index: VpIndex,
244 /// The complete 4-KiB VMSA page.
245 pub page: Box<[u8; 4096]>,
246}
247
248/// SNP ID block and authentication data supplied by an IGVM file.
249#[derive(Eq, PartialEq, Debug, Clone)]
250pub struct SnpIdBlock {
251 /// Whether the author key is enabled.
252 pub author_key_enabled: u8,
253 /// The launch digest supplied by the IGVM file.
254 pub launch_digest: [u8; 48],
255 /// The guest family identifier.
256 pub family_id: [u8; 16],
257 /// The guest image identifier.
258 pub image_id: [u8; 16],
259 /// The ID-block format version.
260 pub version: u32,
261 /// The guest security version number.
262 pub guest_svn: u32,
263 /// The ID-key algorithm.
264 pub id_key_algorithm: u32,
265 /// The author-key algorithm.
266 pub author_key_algorithm: u32,
267 /// The ID-block signature.
268 pub id_key_signature: x86defs::snp::SnpIdBlockSignature,
269 /// The ID public key.
270 pub id_public_key: x86defs::snp::SnpIdBlockPublicKey,
271 /// The author-key signature.
272 pub author_key_signature: x86defs::snp::SnpIdBlockSignature,
273 /// The author public key.
274 pub author_public_key: x86defs::snp::SnpIdBlockPublicKey,
275}
276
277/// Backend-neutral SNP launch configuration combining IGVM metadata and host parameters.
278#[derive(Eq, PartialEq, Debug, Clone)]
279pub struct SnpConfig {
280 /// Optional host-provided data included in SNP launch finish.
281 pub host_data: Option<[u8; 32]>,
282 /// The SNP guest policy.
283 pub policy: u64,
284 /// The highest VTL requested by the selected IGVM platform.
285 pub highest_vtl: u8,
286 /// The shared GPA boundary requested by the selected IGVM platform.
287 pub shared_gpa_boundary: u64,
288 /// Whether the IGVM contains relocation metadata.
289 pub has_relocation: bool,
290 /// Opaque virtual processor contexts in file order.
291 pub vp_contexts: Vec<SnpVpContext>,
292 /// Optional ID block and authentication data.
293 pub id_block: Option<SnpIdBlock>,
294}
295
296/// SNP boot configuration needed before a backend creates a partition.
297#[derive(Eq, PartialEq, Debug, Clone)]
298pub enum SnpPartitionConfig {
299 /// A loader-generated Linux direct-boot VMSA.
300 DirectBoot {
301 /// Enables restricted interrupt injection in the partition and VMSA.
302 restricted_injection: bool,
303 },
304 /// Launch configuration extracted from an IGVM file.
305 Igvm(Box<SnpConfig>),
306}
307
308/// Isolation configuration needed before a backend creates a partition.
309#[derive(Eq, PartialEq, Debug, Clone)]
310pub enum ProtoPartitionIsolation {
311 /// No isolation.
312 None,
313 /// Hypervisor-based isolation.
314 Vbs,
315 /// AMD SEV-SNP with explicit boot configuration.
316 Snp(SnpPartitionConfig),
317 /// Intel Trust Domain Extensions.
318 Tdx,
319 /// Arm Confidential Compute Architecture.
320 Cca,
321}
322
323impl ProtoPartitionIsolation {
324 /// Returns the simple isolation classification.
325 pub fn isolation_type(&self) -> IsolationType {
326 match self {
327 Self::None => IsolationType::None,
328 Self::Vbs => IsolationType::Vbs,
329 Self::Snp(_) => IsolationType::Snp,
330 Self::Tdx => IsolationType::Tdx,
331 Self::Cca => IsolationType::Cca,
332 }
333 }
334
335 /// Returns whether the partition is isolated.
336 pub fn is_isolated(&self) -> bool {
337 self.isolation_type().is_isolated()
338 }
339}
340
341/// Prototype partition creation configuration.
342pub struct ProtoPartitionConfig<'a> {
343 /// The set of VPs to create.
344 pub processor_topology: &'a ProcessorTopology,
345 /// Microsoft hypervisor guest interface configuration.
346 pub hv_config: Option<HvConfig>,
347 /// VM time access.
348 pub vmtime: &'a VmTimeSource,
349 /// Isolation type and optional backend configuration for this partition.
350 pub isolation: ProtoPartitionIsolation,
351 /// Expose hardware virtualization (VMX/SVM) to the guest so that it can run
352 /// its own hypervisor.
353 ///
354 /// The code assembling this config must only set this when the chosen
355 /// backend recognizes it via [`Hypervisor::recognizes_nested_virt`]; a
356 /// backend that receives an unrecognized request may silently ignore it.
357 pub nested_virt: bool,
358 /// Device-assignment MSI IOVA reservation selected for this partition.
359 #[cfg(guest_arch = "aarch64")]
360 pub device_assignment_msi_iova_range: Option<MemoryRange>,
361}
362
363/// Partition creation configuration.
364pub struct PartitionConfig<'a> {
365 /// The guest memory layout.
366 pub mem_layout: &'a MemoryLayout,
367 /// Guest memory access.
368 pub guest_memory: &'a GuestMemory,
369 /// Cpuid leaves to add to the default CPUID results.
370 pub cpuid: &'a [CpuidLeaf],
371 /// The offset of the VTL0 alias map. This maps VTL0's view of memory into
372 /// VTL2 at the specified offset (which must be a power of 2).
373 pub vtl0_alias_map: Option<u64>,
374 /// An optional resolver used to prepare guest-memory backing on demand when
375 /// the partition delivers memory-access faults back to the VMM.
376 ///
377 /// This is set only when the backend reports
378 /// [`ProtoPartition::supports_memory_fault_resolution`]. The backend calls
379 /// it from its memory-fault handler to commit lazily-backed pages and to
380 /// learn the (possibly widened) GPA range to map; the backend retains the
381 /// final per-page safety decision over the returned range.
382 pub fault_resolver: Option<Arc<dyn ResolveMemoryFault>>,
383}
384
385/// Prepares guest-memory backing to resolve a memory-access fault, and reports
386/// the GPA range the partition should map in response.
387///
388/// This is implemented by the memory backing and called by hypervisor backends
389/// (e.g. WHP) that forward guest memory-access faults to the VMM. It lets the
390/// backing commit lazily-backed pages and opportunistically widen the mapped
391/// range to a large page (soft large pages), while the backend keeps the final
392/// per-page safety decision over the returned range.
393pub trait ResolveMemoryFault: Send + Sync {
394 /// Prepares backing for the faulting range `fault` and returns the GPA range
395 /// the partition should map.
396 ///
397 /// The caller passes the range it needs backed (expressed in whatever page
398 /// granularity the backend uses), so this layer never needs to know the
399 /// guest page size. The returned range is always a superset of `fault`,
400 /// clamped to a single uniform RAM region. It is widened (e.g. to 2 MB) only
401 /// on the first fault of a large-page-eligible region that fully contains
402 /// `fault`; otherwise `fault` is returned unchanged. Subsequent faults of an
403 /// already-attempted region are not widened.
404 fn resolve(
405 &self,
406 fault: MemoryRange,
407 write: bool,
408 ) -> Result<MemoryRange, GuestMemoryBackingError>;
409}
410
411/// Trait for a prototype partition, one that is partially created but still
412/// needs final configuration.
413///
414/// This is separate from the partition so that it can be queried to determine
415/// the final partition configuration.
416pub trait ProtoPartition {
417 /// The partition type.
418 type Partition: Partition;
419 /// The VP binder type.
420 type ProcessorBinder: 'static + BindProcessor + Send;
421 /// The error type when creating the partition.
422 type Error: std::error::Error + Send + Sync + 'static;
423
424 /// The maximum physical address width that processors and devices for this
425 /// partition can access.
426 ///
427 /// This may be smaller than what is reported to the guest via architectural
428 /// interfaces by default, and it may be larger or smaller than what the VMM
429 /// ultimately chooses to report to the guest.
430 fn max_physical_address_size(&self) -> u8;
431
432 /// Whether the partition delivers guest-memory-access faults back to the
433 /// VMM and resolves them through a [`ResolveMemoryFault`] supplied in
434 /// [`PartitionConfig::fault_resolver`].
435 ///
436 /// Defaults to `false`. A backend that forwards memory faults to the VMM
437 /// (e.g. WHP) overrides this to `true`. The code assembling
438 /// [`PartitionConfig`] uses it to decide whether to supply a resolver, and
439 /// the memory backing uses it to select a lazy commit strategy.
440 fn supports_memory_fault_resolution(&self) -> bool {
441 false
442 }
443
444 /// Constructs the full partition.
445 fn build(
446 self,
447 config: PartitionConfig<'_>,
448 ) -> Result<(Self::Partition, Vec<Self::ProcessorBinder>), Self::Error>;
449}
450
451/// Trait used to bind a processor to the current thread.
452pub trait BindProcessor {
453 /// The processor object.
454 type Processor<'a>: Processor
455 where
456 Self: 'a;
457
458 /// A binding error.
459 type Error: std::error::Error + Send + Sync + 'static;
460
461 /// Binds the processor to the current thread.
462 fn bind(&mut self) -> Result<Self::Processor<'_>, Self::Error>;
463}
464
465/// Policy for the partition when mapping VTL0 memory late.
466#[derive(Eq, PartialEq, Debug, Copy, Clone)]
467pub enum LateMapVtl0MemoryPolicy {
468 /// Halt execution of the VP if VTL0 memory is accessed.
469 Halt,
470 /// Log the error but emulate the access with the instruction emulator.
471 Log,
472 /// Inject an exception into the guest.
473 InjectException,
474}
475
476/// Which ranges VTL2 is allowed to access before VTL0 ram is mapped.
477#[derive(Debug, Clone)]
478pub enum LateMapVtl0AllowedRanges {
479 /// Ask the memory layout what the vtl2_ram ranges are.
480 MemoryLayout,
481 /// These specific ranges are allowed.
482 Ranges(Vec<MemoryRange>),
483}
484
485/// Config used to determine late mapping VTL0 memory.
486#[derive(Debug, Clone)]
487pub struct LateMapVtl0MemoryConfig {
488 /// What ranges VTL2 are allowed to access before VTL0 memory is mapped.
489 /// Generally this consists of the ranges representing VTL2 ram.
490 pub allowed_ranges: LateMapVtl0AllowedRanges,
491 /// The policy for the partition mapping VTL0 memory late.
492 pub policy: LateMapVtl0MemoryPolicy,
493}
494
495/// VTL2 configuration.
496#[derive(Debug)]
497pub struct Vtl2Config {
498 /// If set, map VTL0 memory late after VTL2 has started. The current
499 /// heuristic is to defer mapping VTL0 memory until the first
500 /// [`hvdef::HypercallCode::HvCallModifyVtlProtectionMask`] hypercall is
501 /// made.
502 ///
503 /// Accesses before memory is mapped is determined by the specified config.
504 pub late_map_vtl0_memory: Option<LateMapVtl0MemoryConfig>,
505}
506
507/// Hypervisor configuration.
508#[derive(Debug)]
509pub struct HvConfig {
510 /// Allow device assignment on the partition.
511 pub allow_device_assignment: bool,
512 /// Enable VTL2 support if set. Additional options are described by
513 /// [Vtl2Config].
514 pub vtl2: Option<Vtl2Config>,
515}
516
517/// Source of the initial virtual processor state.
518#[derive(Debug, Clone, Copy, PartialEq, Eq)]
519pub enum InitialVpStateSource {
520 /// The partition unit writes the loader-produced register state.
521 Registers,
522 /// The state is supplied through an imported isolation context.
523 ImportedContext,
524}
525
526/// Methods for manipulating a VM partition.
527pub trait Partition: 'static + Hv1 + Inspect + Send + Sync {
528 /// Returns the source of the initial virtual processor state.
529 fn initial_vp_state_source(&self) -> InitialVpStateSource;
530
531 /// Returns a trait object for initial page imports during the initial start
532 /// flow.
533 fn supports_initial_page_acceptance(
534 &self,
535 ) -> Option<&dyn AcceptInitialPages<Error = <Self as Hv1>::Error>> {
536 None
537 }
538
539 /// Returns a trait object to reset the partition, if supported.
540 fn supports_reset(&self) -> Option<&dyn ResetPartition<Error = <Self as Hv1>::Error>>;
541
542 /// Returns an interface to control partition time, if supported.
543 ///
544 /// Partitions exposing this interface start with time frozen.
545 fn supports_time_control(&self) -> Option<&dyn PartitionTimeControl> {
546 None
547 }
548
549 /// Returns a trait object to reset VTL state, if supported.
550 fn supports_vtl_scrub(&self) -> Option<&dyn ScrubVtl<Error = <Self as Hv1>::Error>> {
551 None
552 }
553
554 /// Returns an interface for registering MMIO doorbells for this partition.
555 ///
556 /// Not all partitions support this.
557 fn doorbell_registration(
558 self: &Arc<Self>,
559 minimum_vtl: Vtl,
560 ) -> Option<Arc<dyn DoorbellRegistration>> {
561 let _ = minimum_vtl;
562 None
563 }
564
565 /// Requests an MSI for the specified VTL.
566 ///
567 /// On x86, the MSI format is the architectural APIC format.
568 ///
569 /// On ARM64, the MSI format is currently not defined, since we only support
570 /// Hyper-V-style VMs (which use synthetic MSIs via VPCI). In the future, we
571 /// may want to support either or both SPI- and ITS+LPI-based MSIs.
572 fn request_msi(&self, vtl: Vtl, request: MsiRequest);
573
574 /// Returns an MSI interrupt target for this partition, which can be used to
575 /// create MSI interrupts.
576 ///
577 /// Not all partitions support this.
578 fn as_signal_msi(&self, vtl: Vtl) -> Option<Arc<dyn SignalMsi>> {
579 let _ = vtl;
580 None
581 }
582
583 /// Returns an irqfd routing interface for this partition.
584 ///
585 /// irqfd allows the kernel to inject MSIs directly into the guest when an
586 /// eventfd is signaled, without a userspace transition. This is used for
587 /// device passthrough with VFIO.
588 ///
589 /// Not all partitions support this.
590 fn irqfd(&self) -> Option<Arc<dyn IrqFd>> {
591 None
592 }
593
594 /// Get the partition capabilities for this partition.
595 fn caps(&self) -> &PartitionCapabilities;
596
597 /// Forces the run_vp call to yield to the scheduler (i.e. return
598 /// Poll::Pending).
599 fn request_yield(&self, vp_index: VpIndex);
600}
601
602/// X86-specific partition methods.
603pub trait X86Partition: Partition {
604 /// Gets the IO-APIC routing control for VTL0.
605 fn ioapic_routing(&self) -> Arc<dyn IoApicRouting>;
606
607 /// Pulses the specified APIC's local interrupt line (0 or 1).
608 fn pulse_lint(&self, vp_index: VpIndex, vtl: Vtl, lint: u8);
609}
610
611/// ARM64-specific partition methods.
612pub trait Aarch64Partition: Partition {
613 /// Returns an interface for accessing the GIC interrupt controller for `vtl`.
614 fn control_gic(&self, vtl: Vtl) -> Arc<dyn ControlGic>;
615}
616
617/// Extension trait for accepting initial pages.
618pub trait AcceptInitialPages {
619 type Error: std::error::Error;
620
621 /// Accepts initial pages on behalf of the guest.
622 ///
623 /// This can only be used during the load path during partition start to
624 /// accept pages on behalf of the guest that were set as part of the load
625 /// process. The host virtstack cannot accept pages on behalf of the guest
626 /// once it has started running.
627 fn accept_initial_pages(&self, pages: &[InitialPageImport]) -> Result<(), Self::Error>;
628}
629
630/// Controls the passage of partition time independently of VP execution.
631///
632/// This controls backend time, not the software device clock in
633/// [`vmcore::vmtime`]. Stopping VPs alone does not freeze time. A full VM stop
634/// must freeze time after stopping all VPs, and resume must thaw time before
635/// running any VP. Temporary VP stops need not freeze time.
636///
637/// State access while frozen must observe frozen time, and restoring time
638/// must not thaw it. Implementations may provide this behavior in software.
639/// These transitions are infallible lifecycle operations; a backend must
640/// treat an unexpected failure as fatal rather than return with unknown time
641/// state.
642pub trait PartitionTimeControl {
643 /// Freezes partition time until [`Self::thaw_time`] is called.
644 ///
645 /// The caller must ensure that all VPs are stopped. Violating this
646 /// precondition has backend-specific behavior; implementations need not
647 /// check it. This is a no-op if time is already frozen.
648 fn freeze_time(&self);
649
650 /// Resumes partition time, including any time state replaced by reset or
651 /// VTL scrub.
652 ///
653 /// The caller must ensure that all VPs are stopped. Violating this
654 /// precondition has backend-specific behavior; implementations need not
655 /// check it. This is a no-op for time that is already running.
656 fn thaw_time(&self);
657}
658
659/// Extension trait for resetting the partition.
660pub trait ResetPartition {
661 type Error: std::error::Error;
662
663 /// Resets the partition, restoring all partition state to the initial
664 /// state.
665 ///
666 /// The caller must ensure that no VPs are running when this is called.
667 /// If the partition supports [`PartitionTimeControl`], time must be frozen
668 /// and remains frozen after reset.
669 ///
670 /// This resets partition-level (VM-wide) state. After this completes,
671 /// the caller dispatches [`Processor::reset`] to each VP's thread to
672 /// reset per-VP state (registers, APIC, synic message queues, etc.).
673 ///
674 /// If this fails, the partition is in a bad state and cannot be resumed
675 /// until a subsequent reset call succeeds.
676 fn reset(&self) -> Result<(), Self::Error>;
677}
678
679/// Extension trait for scrubbing higher VTL state while leaving lower VTLs
680/// untouched.
681pub trait ScrubVtl {
682 type Error: std::error::Error;
683
684 /// Scrubs partition and VP state for `vtl`. This is useful for servicing
685 /// and restarting a higher VTL without touching the lower VTL.
686 ///
687 /// The caller must ensure that no VPs are running when this is called.
688 /// A scrub may freeze the target VTL's time. After restoring its VP state,
689 /// call [`PartitionTimeControl::thaw_time`] before resuming execution.
690 ///
691 /// This scrubs partition-level state. After this completes, the caller
692 /// dispatches [`Processor::scrub`] to each VP's thread to scrub per-VP
693 /// state for the specified VTL.
694 ///
695 /// Note that this does not reset page protections. This is necessary
696 /// because there may be devices assigned to lower VTLs, and they should not
697 /// be able to DMA to higher VTL memory during servicing.
698 fn scrub(&self, vtl: Vtl) -> Result<(), Self::Error>;
699}
700
701/// Provides access to partition state for save, restore, and reset.
702///
703/// This is not part of [`Partition`] because some scenarios do not require such
704/// access.
705pub trait PartitionAccessState {
706 type StateAccess<'a>: crate::vm::AccessVmState
707 where
708 Self: 'a;
709
710 /// Returns an object to access VM state for the specified VTL.
711 fn access_state(&self, vtl: Vtl) -> Self::StateAccess<'_>;
712}
713
714/// Change memory protections for lower VTLs. This can be used to share memory
715/// with a lower VTL or make memory accesses trigger an intercept. This is
716/// intended for dynamic state as initial memory protections are applied at VM
717/// start.
718pub trait VtlMemoryProtection {
719 /// Sets lower VTL permissions on a physical page.
720 ///
721 /// TODO: To remain generic may want to replace hvdef::HvMapGpaFlags with
722 /// something else.
723 fn modify_vtl_page_setting(&self, pfn: u64, flags: hvdef::HvMapGpaFlags) -> anyhow::Result<()>;
724}
725
726pub trait Processor: InspectMut {
727 type StateAccess<'a>: crate::vp::AccessVpState
728 where
729 Self: 'a;
730
731 /// Sets the debug state: conditions under which the VP should exit for
732 /// debugging the guest. This including single stepping and hardware
733 /// breakpoints.
734 ///
735 /// TODO: generalize for non-x86 architectures.
736 fn set_debug_state(
737 &mut self,
738 vtl: Vtl,
739 state: Option<&DebugState>,
740 ) -> Result<(), <Self::StateAccess<'_> as crate::vp::AccessVpState>::Error>;
741
742 /// Runs the VP.
743 ///
744 /// Although this is an async function, it may block synchronously until
745 /// [`Partition::request_yield`] is called for this VP. Then its future must
746 /// return [`Poll::Pending`] at least once.
747 ///
748 /// Returns when an error occurs, the VP halts, or the VP is requested to
749 /// stop via `stop`.
750 #[expect(async_fn_in_trait)] // don't need or want Send bound
751 async fn run_vp(
752 &mut self,
753 stop: StopVp<'_>,
754 dev: &impl CpuIo,
755 ) -> Result<Infallible, VpHaltReason>;
756
757 /// Without running the VP, flushes any asynchronous requests from other
758 /// processors or objects that might affect this state, so that the object
759 /// can be saved/restored correctly.
760 fn flush_async_requests(&mut self);
761
762 /// Returns whether the specified VTL can be inspected on this processor.
763 ///
764 /// VTL0 is always inspectable.
765 fn vtl_inspectable(&self, vtl: Vtl) -> bool {
766 vtl == Vtl::Vtl0
767 }
768
769 /// Resets per-VP state after a partition-level reset.
770 ///
771 /// Called on each VP's thread while VPs are stopped, after
772 /// [`ResetPartition::reset`] has completed.
773 ///
774 /// The default implementation panics. Backends that support
775 /// [`ResetPartition`] must override this.
776 #[expect(unreachable_code)]
777 fn reset(&mut self) -> Result<(), impl std::error::Error + Send + Sync + 'static> {
778 Ok::<(), Infallible>(unimplemented!(
779 "Processor::reset not implemented for this backend"
780 ))
781 }
782
783 /// Scrubs per-VP state for a specific VTL.
784 ///
785 /// Called on each VP's thread while VPs are stopped, after
786 /// [`ScrubVtl::scrub`] has completed.
787 ///
788 /// The default implementation panics. Backends that support
789 /// [`ScrubVtl`] must override this.
790 #[expect(unreachable_code)]
791 fn scrub(&mut self, _vtl: Vtl) -> Result<(), impl std::error::Error + Send + Sync + 'static> {
792 Ok::<(), Infallible>(unimplemented!(
793 "Processor::scrub not implemented for this backend"
794 ))
795 }
796
797 fn access_state(&mut self, vtl: Vtl) -> Self::StateAccess<'_>;
798}
799
800/// A source for [`StopVp`].
801pub struct StopVpSource {
802 stop: Cell<bool>,
803 waker: Cell<Option<Waker>>,
804}
805
806impl StopVpSource {
807 /// Creates a new source.
808 pub fn new() -> Self {
809 Self {
810 stop: Cell::new(false),
811 waker: Cell::new(None),
812 }
813 }
814
815 /// Returns an object to wait for stops.
816 pub fn checker(&self) -> StopVp<'_> {
817 StopVp { source: self }
818 }
819
820 /// Initiates a VP stop.
821 ///
822 /// After this, calls to [`StopVp::check`] or [`StopVp::until_stop`] will
823 /// fail.
824 pub fn stop(&self) {
825 self.stop.set(true);
826 if let Some(waker) = self.waker.take() {
827 waker.wake();
828 }
829 }
830
831 /// Returns whether [`Self::stop`] has been called.
832 pub fn is_stopping(&self) -> bool {
833 self.stop.get()
834 }
835}
836
837/// Object to check for VP stop requests.
838pub struct StopVp<'a> {
839 source: &'a StopVpSource,
840}
841
842/// An error result that the VP stopped due to request.
843#[derive(Debug)]
844pub struct VpStopped(());
845
846impl StopVp<'_> {
847 /// Returns `Err(VpStopped(_))` if the VP should stop.
848 pub fn check(&self) -> Result<(), VpStopped> {
849 if self.source.stop.get() {
850 Err(VpStopped(()))
851 } else {
852 Ok(())
853 }
854 }
855
856 /// Runs `fut` until it completes or the VP should stop.
857 pub async fn until_stop<Fut: Future>(&mut self, fut: Fut) -> Result<Fut::Output, VpStopped> {
858 let mut fut = pin!(fut);
859 poll_fn(|cx| match fut.as_mut().poll(cx) {
860 Poll::Ready(r) => Poll::Ready(Ok(r)),
861 Poll::Pending => {
862 self.check()?;
863 self.source.waker.set(Some(cx.waker().clone()));
864 Poll::Pending
865 }
866 })
867 .await
868 }
869}
870
871/// An object that can be polled to see if a yield has been requested.
872#[derive(Debug)]
873pub struct NeedsYield {
874 yield_requested: AtomicBool,
875}
876
877impl NeedsYield {
878 /// Creates a new object.
879 pub fn new() -> Self {
880 Self {
881 yield_requested: false.into(),
882 }
883 }
884
885 /// Requests a yield.
886 ///
887 /// Returns whether a signal is necessary to ensure that the task yields
888 /// soon.
889 pub fn request_yield(&self) -> bool {
890 !self.yield_requested.swap(true, Ordering::Release)
891 }
892
893 /// Yields execution to the executor if `request_yield` has been called
894 /// since the last call to `maybe_yield`.
895 pub async fn maybe_yield(&self) {
896 poll_fn(|cx| {
897 if self.yield_requested.load(Ordering::Acquire) {
898 // Wake this task again to ensure it runs again.
899 cx.waker().wake_by_ref();
900 self.yield_requested.store(false, Ordering::Relaxed);
901 Poll::Pending
902 } else {
903 Poll::Ready(())
904 }
905 })
906 .await
907 }
908}
909
910/// The reason that [`Processor::run_vp`] returned.
911#[derive(Debug)]
912pub enum VpHaltReason {
913 /// The processor was requested to stop.
914 Stop(VpStopped),
915 /// The processor task should be restarted, possibly on a different thread.
916 Cancel,
917 /// The processor initiated a power off.
918 PowerOff,
919 /// The processor initiated a reboot.
920 Reset,
921 /// The processor initiated a hibernation.
922 Hibernate,
923 /// The processor triple faulted.
924 TripleFault {
925 /// The faulting VTL.
926 // FUTURE: move VTL state into `AccessVpState``.
927 vtl: Vtl,
928 },
929 /// Debugger single step.
930 SingleStep,
931 /// Debugger hardware breakpoint.
932 HwBreak(HardwareBreakpoint),
933}
934
935impl From<VpStopped> for VpHaltReason {
936 fn from(stop: VpStopped) -> Self {
937 Self::Stop(stop)
938 }
939}
940
941pub trait PartitionMemoryMapper {
942 /// Returns a memory mapper for the partition backing `vtl`.
943 fn memory_mapper(&self, vtl: Vtl) -> Arc<dyn PartitionMemoryMap>;
944
945 /// Returns an interface for acquiring host access to memory.
946 fn host_access(&self) -> Option<Arc<dyn PartitionHostAccess>> {
947 None
948 }
949}
950
951pub trait Hv1 {
952 type Error: std::error::Error + Send + Sync + 'static;
953 type Device: MapVpciInterrupt + SignalMsi;
954
955 fn reference_time_source(&self) -> Option<ReferenceTimeSource>;
956
957 fn new_virtual_device(
958 &self,
959 ) -> Option<&dyn DeviceBuilder<Device = Self::Device, Error = Self::Error>>;
960
961 /// Returns the partition's synic port access, or an error if the
962 /// backend cannot support synic in its current configuration.
963 fn synic(&self) -> anyhow::Result<Arc<dyn vmcore::synic::SynicPortAccess>>;
964}
965
966pub trait DeviceBuilder: Hv1 {
967 fn build(&self, vtl: Vtl, device_id: u64) -> Result<Self::Device, Self::Error>;
968}
969
970pub enum UnimplementedDevice {}
971
972impl MapVpciInterrupt for UnimplementedDevice {
973 async fn register_interrupt(
974 &self,
975 _vector_count: u32,
976 _params: &VpciInterruptParameters<'_>,
977 ) -> Result<MsiAddressData, RegisterInterruptError> {
978 match *self {}
979 }
980
981 async fn unregister_interrupt(&self, _address: u64, _data: u32) {
982 match *self {}
983 }
984}
985
986impl SignalMsi for UnimplementedDevice {
987 fn signal_msi(&self, _devid: Option<u32>, _address: u64, _data: u32) {
988 match *self {}
989 }
990}
991
992/// MNF support routines for the emulator
993pub trait EmulatorMonitorSupport {
994 /// Check if the specified write is inside the monitor page, and signal the associated
995 /// connection ID if it is.
996 #[must_use]
997 fn check_write(&self, gpa: u64, bytes: &[u8]) -> bool;
998
999 /// Check if the specified read is inside the monitor page, and fill the provided buffer
1000 /// if it is.
1001 #[must_use]
1002 fn check_read(&self, gpa: u64, bytes: &mut [u8]) -> bool;
1003}