Skip to main content

virt/
generic.rs

1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4mod partition_memory_map;
5
6pub use partition_memory_map::PartitionHostAccess;
7pub use partition_memory_map::PartitionMemoryMap;
8pub use vm_topology::processor::VpIndex;
9
10use crate::CpuidLeaf;
11use crate::PartitionCapabilities;
12use crate::io::CpuIo;
13use crate::irqcon::ControlGic;
14use crate::irqcon::IoApicRouting;
15use crate::irqcon::MsiRequest;
16use crate::irqfd::IrqFd;
17use crate::x86::DebugState;
18use crate::x86::HardwareBreakpoint;
19use guestmem::DoorbellRegistration;
20use guestmem::GuestMemory;
21use guestmem::GuestMemoryBackingError;
22use hvdef::Vtl;
23use inspect::Inspect;
24use inspect::InspectMut;
25use memory_range::MemoryRange;
26use pci_core::msi::SignalMsi;
27use std::cell::Cell;
28use std::convert::Infallible;
29use std::fmt::Debug;
30use std::future::Future;
31use std::future::poll_fn;
32use std::pin::pin;
33use std::sync::Arc;
34use std::sync::atomic::AtomicBool;
35use std::sync::atomic::Ordering;
36use std::task::Poll;
37use std::task::Waker;
38use vm_topology::memory::MemoryLayout;
39use vm_topology::processor::ProcessorTopology;
40use vmcore::reference_time::ReferenceTimeSource;
41use vmcore::vmtime::VmTimeSource;
42use vmcore::vpci_msi::MapVpciInterrupt;
43use vmcore::vpci_msi::MsiAddressData;
44use vmcore::vpci_msi::RegisterInterruptError;
45use vmcore::vpci_msi::VpciInterruptParameters;
46
47/// Platform capabilities detected from the hypervisor before partition
48/// creation. On x86 there are currently no pre-partition queries.
49#[cfg(guest_arch = "x86_64")]
50#[derive(Debug, Clone, Default)]
51pub struct PlatformInfo {}
52
53/// Platform capabilities detected from the hypervisor before partition
54/// creation.
55#[cfg(guest_arch = "aarch64")]
56#[derive(Debug, Clone)]
57pub struct PlatformInfo {
58    /// The platform PMU GSIV (GIC INTID), if available.
59    pub platform_gsiv: Option<u32>,
60    /// Whether the hypervisor supports GICv3. When `false`, only
61    /// GICv2 is available (e.g., Raspberry Pi 5 with GIC-400).
62    pub supports_gic_v3: bool,
63    /// Whether the hypervisor supports an in-kernel GICv3 ITS for
64    /// MSI delivery via LPIs. When `true`, the topology can include
65    /// a `GicItsInfo` and the backend will create/manage the ITS device.
66    pub supports_its: bool,
67    /// How the physical SMMU implementation selects the IOVA range reserved
68    /// for device-assignment MSI writes.
69    pub device_assignment_msi_iova: DeviceAssignmentMsiIova,
70}
71
72/// Selection policy for the device-assignment MSI IOVA reservation.
73#[cfg(guest_arch = "aarch64")]
74#[derive(Debug, Clone, Copy)]
75pub enum DeviceAssignmentMsiIova {
76    /// Device assignment does not expose an MSI IOVA reservation contract.
77    Unsupported,
78    /// The physical SMMU driver requires this exact range.
79    Fixed(MemoryRange),
80    /// The VMM selects the range and passes its base to the physical SMMU
81    /// implementation during partition creation.
82    Configurable,
83}
84
85/// A hypervisor backend capable of creating partitions.
86///
87/// # Recognized features
88///
89/// The `recognizes_*` methods report whether the backend acts on an optional
90/// partition request rather than silently ignoring it: it either honors the
91/// request or fails partition creation with a specific error. They let the code
92/// assembling a [`ProtoPartitionConfig`] reject a request up front when the
93/// backend has no concept of it, instead of the request being quietly dropped.
94/// Recognition is *not* a promise that the request succeeds — the backend may
95/// still reject it in combination with another feature, or fail later during
96/// partition creation. Each method defaults to `false`, so a new optional
97/// feature is unrecognized everywhere until a backend overrides its method.
98pub trait Hypervisor: 'static {
99    /// The prototype partition type.
100    type ProtoPartition<'a>: ProtoPartition<Partition = Self::Partition>;
101    /// The partition type.
102    type Partition;
103    /// The error type when creating the partition.
104    type Error: std::error::Error + Send + Sync + 'static;
105
106    /// Returns platform capabilities detected from the hypervisor.
107    ///
108    /// This is called before partition creation to query platform-specific
109    /// information needed for topology construction and firmware table
110    /// generation.
111    fn platform_info(&self) -> PlatformInfo;
112
113    /// Whether the backend recognizes a request to expose hardware
114    /// virtualization (VMX/SVM) to the guest so it can run its own hypervisor.
115    /// See the [`Hypervisor`] trait docs on recognized features.
116    fn recognizes_nested_virt(&self) -> bool {
117        false
118    }
119
120    /// Returns a new prototype partition from the given configuration.
121    fn new_partition<'a>(
122        &'a mut self,
123        config: ProtoPartitionConfig<'a>,
124    ) -> Result<Self::ProtoPartition<'a>, Self::Error>;
125}
126
127/// Isolation type for a partition.
128#[derive(Eq, PartialEq, Debug, Copy, Clone, Inspect)]
129pub enum IsolationType {
130    /// No isolation.
131    None,
132    /// Hypervisor based isolation.
133    Vbs,
134    /// Secure nested paging (AMD SEV-SNP) - hardware based isolation.
135    Snp,
136    /// Trust domain extensions (Intel TDX) - hardware based isolation.
137    Tdx,
138    /// Confidential Compute Architecture (ARM CCA) - hardware based isolation.
139    Cca,
140}
141
142impl IsolationType {
143    /// Returns true if the isolation type is not `None`.
144    pub fn is_isolated(&self) -> bool {
145        !matches!(self, Self::None)
146    }
147
148    /// Returns whether the isolation type is hardware-backed.
149    pub fn is_hardware_isolated(&self) -> bool {
150        matches!(self, Self::Snp | Self::Tdx | Self::Cca)
151    }
152}
153
154/// An unexpected isolation type was provided.
155#[derive(Debug)]
156pub struct UnexpectedIsolationType;
157
158impl IsolationType {
159    pub const fn from_hv(
160        value: hvdef::HvPartitionIsolationType,
161    ) -> Result<Self, UnexpectedIsolationType> {
162        match value {
163            hvdef::HvPartitionIsolationType::NONE => Ok(IsolationType::None),
164            hvdef::HvPartitionIsolationType::VBS => Ok(IsolationType::Vbs),
165            hvdef::HvPartitionIsolationType::SNP => Ok(IsolationType::Snp),
166            hvdef::HvPartitionIsolationType::TDX => Ok(IsolationType::Tdx),
167            hvdef::HvPartitionIsolationType::CCA => Ok(IsolationType::Cca),
168            _ => Err(UnexpectedIsolationType),
169        }
170    }
171
172    pub const fn to_hv(self) -> hvdef::HvPartitionIsolationType {
173        match self {
174            IsolationType::None => hvdef::HvPartitionIsolationType::NONE,
175            IsolationType::Vbs => hvdef::HvPartitionIsolationType::VBS,
176            IsolationType::Snp => hvdef::HvPartitionIsolationType::SNP,
177            IsolationType::Tdx => hvdef::HvPartitionIsolationType::TDX,
178            IsolationType::Cca => hvdef::HvPartitionIsolationType::CCA,
179        }
180    }
181}
182
183/// Page visibility types for isolated partitions.
184#[derive(Eq, PartialEq, Debug, Copy, Clone, Inspect)]
185pub enum PageVisibility {
186    /// The guest has exclusive access to the page, and no access from the host.
187    Exclusive,
188    /// The page has shared access with the guest and host.
189    Shared,
190}
191
192/// Initial page import type for isolated partitions.
193#[derive(Eq, PartialEq, Debug, Copy, Clone, Inspect)]
194pub enum InitialPageImportType {
195    /// A measured page with exclusive guest access.
196    Normal,
197    /// An unmeasured page with exclusive guest access.
198    NormalUnmeasured,
199    /// A page shared between the guest and host.
200    Shared,
201    /// A virtual processor context page.
202    VpContext,
203    /// An SNP secrets page.
204    Secrets,
205    /// An SNP CPUID page.
206    Cpuid,
207    /// An SNP CPUID extended state page.
208    CpuidExtendedState,
209}
210
211impl InitialPageImportType {
212    /// Returns the visibility implied by this import type.
213    pub fn page_visibility(self) -> PageVisibility {
214        match self {
215            Self::Shared => PageVisibility::Shared,
216            Self::Normal
217            | Self::NormalUnmeasured
218            | Self::VpContext
219            | Self::Secrets
220            | Self::Cpuid
221            | Self::CpuidExtendedState => PageVisibility::Exclusive,
222        }
223    }
224}
225
226/// Initial page import metadata for isolated partitions.
227#[derive(Eq, PartialEq, Debug, Clone)]
228pub struct InitialPageImport {
229    /// The guest physical range being imported.
230    pub range: MemoryRange,
231    /// The hypervisor-facing import type for this range.
232    pub import_type: InitialPageImportType,
233    /// Loader-provided debug tag identifying the source of this range.
234    pub tag: &'static str,
235}
236
237/// An opaque SNP virtual processor context.
238#[derive(Eq, PartialEq, Debug, Clone)]
239pub struct SnpVpContext {
240    /// The guest physical address associated with the context.
241    pub gpa: u64,
242    /// The virtual processor described by the context.
243    pub vp_index: VpIndex,
244    /// The complete 4-KiB VMSA page.
245    pub page: Box<[u8; 4096]>,
246}
247
248/// SNP ID block and authentication data supplied by an IGVM file.
249#[derive(Eq, PartialEq, Debug, Clone)]
250pub struct SnpIdBlock {
251    /// Whether the author key is enabled.
252    pub author_key_enabled: u8,
253    /// The launch digest supplied by the IGVM file.
254    pub launch_digest: [u8; 48],
255    /// The guest family identifier.
256    pub family_id: [u8; 16],
257    /// The guest image identifier.
258    pub image_id: [u8; 16],
259    /// The ID-block format version.
260    pub version: u32,
261    /// The guest security version number.
262    pub guest_svn: u32,
263    /// The ID-key algorithm.
264    pub id_key_algorithm: u32,
265    /// The author-key algorithm.
266    pub author_key_algorithm: u32,
267    /// The ID-block signature.
268    pub id_key_signature: x86defs::snp::SnpIdBlockSignature,
269    /// The ID public key.
270    pub id_public_key: x86defs::snp::SnpIdBlockPublicKey,
271    /// The author-key signature.
272    pub author_key_signature: x86defs::snp::SnpIdBlockSignature,
273    /// The author public key.
274    pub author_public_key: x86defs::snp::SnpIdBlockPublicKey,
275}
276
277/// Backend-neutral SNP launch configuration combining IGVM metadata and host parameters.
278#[derive(Eq, PartialEq, Debug, Clone)]
279pub struct SnpConfig {
280    /// Optional host-provided data included in SNP launch finish.
281    pub host_data: Option<[u8; 32]>,
282    /// The SNP guest policy.
283    pub policy: u64,
284    /// The highest VTL requested by the selected IGVM platform.
285    pub highest_vtl: u8,
286    /// The shared GPA boundary requested by the selected IGVM platform.
287    pub shared_gpa_boundary: u64,
288    /// Whether the IGVM contains relocation metadata.
289    pub has_relocation: bool,
290    /// Opaque virtual processor contexts in file order.
291    pub vp_contexts: Vec<SnpVpContext>,
292    /// Optional ID block and authentication data.
293    pub id_block: Option<SnpIdBlock>,
294}
295
296/// SNP boot configuration needed before a backend creates a partition.
297#[derive(Eq, PartialEq, Debug, Clone)]
298pub enum SnpPartitionConfig {
299    /// A loader-generated Linux direct-boot VMSA.
300    DirectBoot {
301        /// Enables restricted interrupt injection in the partition and VMSA.
302        restricted_injection: bool,
303    },
304    /// Launch configuration extracted from an IGVM file.
305    Igvm(Box<SnpConfig>),
306}
307
308/// Isolation configuration needed before a backend creates a partition.
309#[derive(Eq, PartialEq, Debug, Clone)]
310pub enum ProtoPartitionIsolation {
311    /// No isolation.
312    None,
313    /// Hypervisor-based isolation.
314    Vbs,
315    /// AMD SEV-SNP with explicit boot configuration.
316    Snp(SnpPartitionConfig),
317    /// Intel Trust Domain Extensions.
318    Tdx,
319    /// Arm Confidential Compute Architecture.
320    Cca,
321}
322
323impl ProtoPartitionIsolation {
324    /// Returns the simple isolation classification.
325    pub fn isolation_type(&self) -> IsolationType {
326        match self {
327            Self::None => IsolationType::None,
328            Self::Vbs => IsolationType::Vbs,
329            Self::Snp(_) => IsolationType::Snp,
330            Self::Tdx => IsolationType::Tdx,
331            Self::Cca => IsolationType::Cca,
332        }
333    }
334
335    /// Returns whether the partition is isolated.
336    pub fn is_isolated(&self) -> bool {
337        self.isolation_type().is_isolated()
338    }
339}
340
341/// Prototype partition creation configuration.
342pub struct ProtoPartitionConfig<'a> {
343    /// The set of VPs to create.
344    pub processor_topology: &'a ProcessorTopology,
345    /// Microsoft hypervisor guest interface configuration.
346    pub hv_config: Option<HvConfig>,
347    /// VM time access.
348    pub vmtime: &'a VmTimeSource,
349    /// Isolation type and optional backend configuration for this partition.
350    pub isolation: ProtoPartitionIsolation,
351    /// Expose hardware virtualization (VMX/SVM) to the guest so that it can run
352    /// its own hypervisor.
353    ///
354    /// The code assembling this config must only set this when the chosen
355    /// backend recognizes it via [`Hypervisor::recognizes_nested_virt`]; a
356    /// backend that receives an unrecognized request may silently ignore it.
357    pub nested_virt: bool,
358    /// Device-assignment MSI IOVA reservation selected for this partition.
359    #[cfg(guest_arch = "aarch64")]
360    pub device_assignment_msi_iova_range: Option<MemoryRange>,
361}
362
363/// Partition creation configuration.
364pub struct PartitionConfig<'a> {
365    /// The guest memory layout.
366    pub mem_layout: &'a MemoryLayout,
367    /// Guest memory access.
368    pub guest_memory: &'a GuestMemory,
369    /// Cpuid leaves to add to the default CPUID results.
370    pub cpuid: &'a [CpuidLeaf],
371    /// The offset of the VTL0 alias map. This maps VTL0's view of memory into
372    /// VTL2 at the specified offset (which must be a power of 2).
373    pub vtl0_alias_map: Option<u64>,
374    /// An optional resolver used to prepare guest-memory backing on demand when
375    /// the partition delivers memory-access faults back to the VMM.
376    ///
377    /// This is set only when the backend reports
378    /// [`ProtoPartition::supports_memory_fault_resolution`]. The backend calls
379    /// it from its memory-fault handler to commit lazily-backed pages and to
380    /// learn the (possibly widened) GPA range to map; the backend retains the
381    /// final per-page safety decision over the returned range.
382    pub fault_resolver: Option<Arc<dyn ResolveMemoryFault>>,
383}
384
385/// Prepares guest-memory backing to resolve a memory-access fault, and reports
386/// the GPA range the partition should map in response.
387///
388/// This is implemented by the memory backing and called by hypervisor backends
389/// (e.g. WHP) that forward guest memory-access faults to the VMM. It lets the
390/// backing commit lazily-backed pages and opportunistically widen the mapped
391/// range to a large page (soft large pages), while the backend keeps the final
392/// per-page safety decision over the returned range.
393pub trait ResolveMemoryFault: Send + Sync {
394    /// Prepares backing for the faulting range `fault` and returns the GPA range
395    /// the partition should map.
396    ///
397    /// The caller passes the range it needs backed (expressed in whatever page
398    /// granularity the backend uses), so this layer never needs to know the
399    /// guest page size. The returned range is always a superset of `fault`,
400    /// clamped to a single uniform RAM region. It is widened (e.g. to 2 MB) only
401    /// on the first fault of a large-page-eligible region that fully contains
402    /// `fault`; otherwise `fault` is returned unchanged. Subsequent faults of an
403    /// already-attempted region are not widened.
404    fn resolve(
405        &self,
406        fault: MemoryRange,
407        write: bool,
408    ) -> Result<MemoryRange, GuestMemoryBackingError>;
409}
410
411/// Trait for a prototype partition, one that is partially created but still
412/// needs final configuration.
413///
414/// This is separate from the partition so that it can be queried to determine
415/// the final partition configuration.
416pub trait ProtoPartition {
417    /// The partition type.
418    type Partition: Partition;
419    /// The VP binder type.
420    type ProcessorBinder: 'static + BindProcessor + Send;
421    /// The error type when creating the partition.
422    type Error: std::error::Error + Send + Sync + 'static;
423
424    /// The maximum physical address width that processors and devices for this
425    /// partition can access.
426    ///
427    /// This may be smaller than what is reported to the guest via architectural
428    /// interfaces by default, and it may be larger or smaller than what the VMM
429    /// ultimately chooses to report to the guest.
430    fn max_physical_address_size(&self) -> u8;
431
432    /// Whether the partition delivers guest-memory-access faults back to the
433    /// VMM and resolves them through a [`ResolveMemoryFault`] supplied in
434    /// [`PartitionConfig::fault_resolver`].
435    ///
436    /// Defaults to `false`. A backend that forwards memory faults to the VMM
437    /// (e.g. WHP) overrides this to `true`. The code assembling
438    /// [`PartitionConfig`] uses it to decide whether to supply a resolver, and
439    /// the memory backing uses it to select a lazy commit strategy.
440    fn supports_memory_fault_resolution(&self) -> bool {
441        false
442    }
443
444    /// Constructs the full partition.
445    fn build(
446        self,
447        config: PartitionConfig<'_>,
448    ) -> Result<(Self::Partition, Vec<Self::ProcessorBinder>), Self::Error>;
449}
450
451/// Trait used to bind a processor to the current thread.
452pub trait BindProcessor {
453    /// The processor object.
454    type Processor<'a>: Processor
455    where
456        Self: 'a;
457
458    /// A binding error.
459    type Error: std::error::Error + Send + Sync + 'static;
460
461    /// Binds the processor to the current thread.
462    fn bind(&mut self) -> Result<Self::Processor<'_>, Self::Error>;
463}
464
465/// Policy for the partition when mapping VTL0 memory late.
466#[derive(Eq, PartialEq, Debug, Copy, Clone)]
467pub enum LateMapVtl0MemoryPolicy {
468    /// Halt execution of the VP if VTL0 memory is accessed.
469    Halt,
470    /// Log the error but emulate the access with the instruction emulator.
471    Log,
472    /// Inject an exception into the guest.
473    InjectException,
474}
475
476/// Which ranges VTL2 is allowed to access before VTL0 ram is mapped.
477#[derive(Debug, Clone)]
478pub enum LateMapVtl0AllowedRanges {
479    /// Ask the memory layout what the vtl2_ram ranges are.
480    MemoryLayout,
481    /// These specific ranges are allowed.
482    Ranges(Vec<MemoryRange>),
483}
484
485/// Config used to determine late mapping VTL0 memory.
486#[derive(Debug, Clone)]
487pub struct LateMapVtl0MemoryConfig {
488    /// What ranges VTL2 are allowed to access before VTL0 memory is mapped.
489    /// Generally this consists of the ranges representing VTL2 ram.
490    pub allowed_ranges: LateMapVtl0AllowedRanges,
491    /// The policy for the partition mapping VTL0 memory late.
492    pub policy: LateMapVtl0MemoryPolicy,
493}
494
495/// VTL2 configuration.
496#[derive(Debug)]
497pub struct Vtl2Config {
498    /// If set, map VTL0 memory late after VTL2 has started. The current
499    /// heuristic is to defer mapping VTL0 memory until the first
500    /// [`hvdef::HypercallCode::HvCallModifyVtlProtectionMask`] hypercall is
501    /// made.
502    ///
503    /// Accesses before memory is mapped is determined by the specified config.
504    pub late_map_vtl0_memory: Option<LateMapVtl0MemoryConfig>,
505}
506
507/// Hypervisor configuration.
508#[derive(Debug)]
509pub struct HvConfig {
510    /// Allow device assignment on the partition.
511    pub allow_device_assignment: bool,
512    /// Enable VTL2 support if set. Additional options are described by
513    /// [Vtl2Config].
514    pub vtl2: Option<Vtl2Config>,
515}
516
517/// Source of the initial virtual processor state.
518#[derive(Debug, Clone, Copy, PartialEq, Eq)]
519pub enum InitialVpStateSource {
520    /// The partition unit writes the loader-produced register state.
521    Registers,
522    /// The state is supplied through an imported isolation context.
523    ImportedContext,
524}
525
526/// Methods for manipulating a VM partition.
527pub trait Partition: 'static + Hv1 + Inspect + Send + Sync {
528    /// Returns the source of the initial virtual processor state.
529    fn initial_vp_state_source(&self) -> InitialVpStateSource;
530
531    /// Returns a trait object for initial page imports during the initial start
532    /// flow.
533    fn supports_initial_page_acceptance(
534        &self,
535    ) -> Option<&dyn AcceptInitialPages<Error = <Self as Hv1>::Error>> {
536        None
537    }
538
539    /// Returns a trait object to reset the partition, if supported.
540    fn supports_reset(&self) -> Option<&dyn ResetPartition<Error = <Self as Hv1>::Error>>;
541
542    /// Returns an interface to control partition time, if supported.
543    ///
544    /// Partitions exposing this interface start with time frozen.
545    fn supports_time_control(&self) -> Option<&dyn PartitionTimeControl> {
546        None
547    }
548
549    /// Returns a trait object to reset VTL state, if supported.
550    fn supports_vtl_scrub(&self) -> Option<&dyn ScrubVtl<Error = <Self as Hv1>::Error>> {
551        None
552    }
553
554    /// Returns an interface for registering MMIO doorbells for this partition.
555    ///
556    /// Not all partitions support this.
557    fn doorbell_registration(
558        self: &Arc<Self>,
559        minimum_vtl: Vtl,
560    ) -> Option<Arc<dyn DoorbellRegistration>> {
561        let _ = minimum_vtl;
562        None
563    }
564
565    /// Requests an MSI for the specified VTL.
566    ///
567    /// On x86, the MSI format is the architectural APIC format.
568    ///
569    /// On ARM64, the MSI format is currently not defined, since we only support
570    /// Hyper-V-style VMs (which use synthetic MSIs via VPCI). In the future, we
571    /// may want to support either or both SPI- and ITS+LPI-based MSIs.
572    fn request_msi(&self, vtl: Vtl, request: MsiRequest);
573
574    /// Returns an MSI interrupt target for this partition, which can be used to
575    /// create MSI interrupts.
576    ///
577    /// Not all partitions support this.
578    fn as_signal_msi(&self, vtl: Vtl) -> Option<Arc<dyn SignalMsi>> {
579        let _ = vtl;
580        None
581    }
582
583    /// Returns an irqfd routing interface for this partition.
584    ///
585    /// irqfd allows the kernel to inject MSIs directly into the guest when an
586    /// eventfd is signaled, without a userspace transition. This is used for
587    /// device passthrough with VFIO.
588    ///
589    /// Not all partitions support this.
590    fn irqfd(&self) -> Option<Arc<dyn IrqFd>> {
591        None
592    }
593
594    /// Get the partition capabilities for this partition.
595    fn caps(&self) -> &PartitionCapabilities;
596
597    /// Forces the run_vp call to yield to the scheduler (i.e. return
598    /// Poll::Pending).
599    fn request_yield(&self, vp_index: VpIndex);
600}
601
602/// X86-specific partition methods.
603pub trait X86Partition: Partition {
604    /// Gets the IO-APIC routing control for VTL0.
605    fn ioapic_routing(&self) -> Arc<dyn IoApicRouting>;
606
607    /// Pulses the specified APIC's local interrupt line (0 or 1).
608    fn pulse_lint(&self, vp_index: VpIndex, vtl: Vtl, lint: u8);
609}
610
611/// ARM64-specific partition methods.
612pub trait Aarch64Partition: Partition {
613    /// Returns an interface for accessing the GIC interrupt controller for `vtl`.
614    fn control_gic(&self, vtl: Vtl) -> Arc<dyn ControlGic>;
615}
616
617/// Extension trait for accepting initial pages.
618pub trait AcceptInitialPages {
619    type Error: std::error::Error;
620
621    /// Accepts initial pages on behalf of the guest.
622    ///
623    /// This can only be used during the load path during partition start to
624    /// accept pages on behalf of the guest that were set as part of the load
625    /// process. The host virtstack cannot accept pages on behalf of the guest
626    /// once it has started running.
627    fn accept_initial_pages(&self, pages: &[InitialPageImport]) -> Result<(), Self::Error>;
628}
629
630/// Controls the passage of partition time independently of VP execution.
631///
632/// This controls backend time, not the software device clock in
633/// [`vmcore::vmtime`]. Stopping VPs alone does not freeze time. A full VM stop
634/// must freeze time after stopping all VPs, and resume must thaw time before
635/// running any VP. Temporary VP stops need not freeze time.
636///
637/// State access while frozen must observe frozen time, and restoring time
638/// must not thaw it. Implementations may provide this behavior in software.
639/// These transitions are infallible lifecycle operations; a backend must
640/// treat an unexpected failure as fatal rather than return with unknown time
641/// state.
642pub trait PartitionTimeControl {
643    /// Freezes partition time until [`Self::thaw_time`] is called.
644    ///
645    /// The caller must ensure that all VPs are stopped. Violating this
646    /// precondition has backend-specific behavior; implementations need not
647    /// check it. This is a no-op if time is already frozen.
648    fn freeze_time(&self);
649
650    /// Resumes partition time, including any time state replaced by reset or
651    /// VTL scrub.
652    ///
653    /// The caller must ensure that all VPs are stopped. Violating this
654    /// precondition has backend-specific behavior; implementations need not
655    /// check it. This is a no-op for time that is already running.
656    fn thaw_time(&self);
657}
658
659/// Extension trait for resetting the partition.
660pub trait ResetPartition {
661    type Error: std::error::Error;
662
663    /// Resets the partition, restoring all partition state to the initial
664    /// state.
665    ///
666    /// The caller must ensure that no VPs are running when this is called.
667    /// If the partition supports [`PartitionTimeControl`], time must be frozen
668    /// and remains frozen after reset.
669    ///
670    /// This resets partition-level (VM-wide) state. After this completes,
671    /// the caller dispatches [`Processor::reset`] to each VP's thread to
672    /// reset per-VP state (registers, APIC, synic message queues, etc.).
673    ///
674    /// If this fails, the partition is in a bad state and cannot be resumed
675    /// until a subsequent reset call succeeds.
676    fn reset(&self) -> Result<(), Self::Error>;
677}
678
679/// Extension trait for scrubbing higher VTL state while leaving lower VTLs
680/// untouched.
681pub trait ScrubVtl {
682    type Error: std::error::Error;
683
684    /// Scrubs partition and VP state for `vtl`. This is useful for servicing
685    /// and restarting a higher VTL without touching the lower VTL.
686    ///
687    /// The caller must ensure that no VPs are running when this is called.
688    /// A scrub may freeze the target VTL's time. After restoring its VP state,
689    /// call [`PartitionTimeControl::thaw_time`] before resuming execution.
690    ///
691    /// This scrubs partition-level state. After this completes, the caller
692    /// dispatches [`Processor::scrub`] to each VP's thread to scrub per-VP
693    /// state for the specified VTL.
694    ///
695    /// Note that this does not reset page protections. This is necessary
696    /// because there may be devices assigned to lower VTLs, and they should not
697    /// be able to DMA to higher VTL memory during servicing.
698    fn scrub(&self, vtl: Vtl) -> Result<(), Self::Error>;
699}
700
701/// Provides access to partition state for save, restore, and reset.
702///
703/// This is not part of [`Partition`] because some scenarios do not require such
704/// access.
705pub trait PartitionAccessState {
706    type StateAccess<'a>: crate::vm::AccessVmState
707    where
708        Self: 'a;
709
710    /// Returns an object to access VM state for the specified VTL.
711    fn access_state(&self, vtl: Vtl) -> Self::StateAccess<'_>;
712}
713
714/// Change memory protections for lower VTLs. This can be used to share memory
715/// with a lower VTL or make memory accesses trigger an intercept. This is
716/// intended for dynamic state as initial memory protections are applied at VM
717/// start.
718pub trait VtlMemoryProtection {
719    /// Sets lower VTL permissions on a physical page.
720    ///
721    /// TODO: To remain generic may want to replace hvdef::HvMapGpaFlags with
722    ///       something else.
723    fn modify_vtl_page_setting(&self, pfn: u64, flags: hvdef::HvMapGpaFlags) -> anyhow::Result<()>;
724}
725
726pub trait Processor: InspectMut {
727    type StateAccess<'a>: crate::vp::AccessVpState
728    where
729        Self: 'a;
730
731    /// Sets the debug state: conditions under which the VP should exit for
732    /// debugging the guest. This including single stepping and hardware
733    /// breakpoints.
734    ///
735    /// TODO: generalize for non-x86 architectures.
736    fn set_debug_state(
737        &mut self,
738        vtl: Vtl,
739        state: Option<&DebugState>,
740    ) -> Result<(), <Self::StateAccess<'_> as crate::vp::AccessVpState>::Error>;
741
742    /// Runs the VP.
743    ///
744    /// Although this is an async function, it may block synchronously until
745    /// [`Partition::request_yield`] is called for this VP. Then its future must
746    /// return [`Poll::Pending`] at least once.
747    ///
748    /// Returns when an error occurs, the VP halts, or the VP is requested to
749    /// stop via `stop`.
750    #[expect(async_fn_in_trait)] // don't need or want Send bound
751    async fn run_vp(
752        &mut self,
753        stop: StopVp<'_>,
754        dev: &impl CpuIo,
755    ) -> Result<Infallible, VpHaltReason>;
756
757    /// Without running the VP, flushes any asynchronous requests from other
758    /// processors or objects that might affect this state, so that the object
759    /// can be saved/restored correctly.
760    fn flush_async_requests(&mut self);
761
762    /// Returns whether the specified VTL can be inspected on this processor.
763    ///
764    /// VTL0 is always inspectable.
765    fn vtl_inspectable(&self, vtl: Vtl) -> bool {
766        vtl == Vtl::Vtl0
767    }
768
769    /// Resets per-VP state after a partition-level reset.
770    ///
771    /// Called on each VP's thread while VPs are stopped, after
772    /// [`ResetPartition::reset`] has completed.
773    ///
774    /// The default implementation panics. Backends that support
775    /// [`ResetPartition`] must override this.
776    #[expect(unreachable_code)]
777    fn reset(&mut self) -> Result<(), impl std::error::Error + Send + Sync + 'static> {
778        Ok::<(), Infallible>(unimplemented!(
779            "Processor::reset not implemented for this backend"
780        ))
781    }
782
783    /// Scrubs per-VP state for a specific VTL.
784    ///
785    /// Called on each VP's thread while VPs are stopped, after
786    /// [`ScrubVtl::scrub`] has completed.
787    ///
788    /// The default implementation panics. Backends that support
789    /// [`ScrubVtl`] must override this.
790    #[expect(unreachable_code)]
791    fn scrub(&mut self, _vtl: Vtl) -> Result<(), impl std::error::Error + Send + Sync + 'static> {
792        Ok::<(), Infallible>(unimplemented!(
793            "Processor::scrub not implemented for this backend"
794        ))
795    }
796
797    fn access_state(&mut self, vtl: Vtl) -> Self::StateAccess<'_>;
798}
799
800/// A source for [`StopVp`].
801pub struct StopVpSource {
802    stop: Cell<bool>,
803    waker: Cell<Option<Waker>>,
804}
805
806impl StopVpSource {
807    /// Creates a new source.
808    pub fn new() -> Self {
809        Self {
810            stop: Cell::new(false),
811            waker: Cell::new(None),
812        }
813    }
814
815    /// Returns an object to wait for stops.
816    pub fn checker(&self) -> StopVp<'_> {
817        StopVp { source: self }
818    }
819
820    /// Initiates a VP stop.
821    ///
822    /// After this, calls to [`StopVp::check`] or [`StopVp::until_stop`] will
823    /// fail.
824    pub fn stop(&self) {
825        self.stop.set(true);
826        if let Some(waker) = self.waker.take() {
827            waker.wake();
828        }
829    }
830
831    /// Returns whether [`Self::stop`] has been called.
832    pub fn is_stopping(&self) -> bool {
833        self.stop.get()
834    }
835}
836
837/// Object to check for VP stop requests.
838pub struct StopVp<'a> {
839    source: &'a StopVpSource,
840}
841
842/// An error result that the VP stopped due to request.
843#[derive(Debug)]
844pub struct VpStopped(());
845
846impl StopVp<'_> {
847    /// Returns `Err(VpStopped(_))` if the VP should stop.
848    pub fn check(&self) -> Result<(), VpStopped> {
849        if self.source.stop.get() {
850            Err(VpStopped(()))
851        } else {
852            Ok(())
853        }
854    }
855
856    /// Runs `fut` until it completes or the VP should stop.
857    pub async fn until_stop<Fut: Future>(&mut self, fut: Fut) -> Result<Fut::Output, VpStopped> {
858        let mut fut = pin!(fut);
859        poll_fn(|cx| match fut.as_mut().poll(cx) {
860            Poll::Ready(r) => Poll::Ready(Ok(r)),
861            Poll::Pending => {
862                self.check()?;
863                self.source.waker.set(Some(cx.waker().clone()));
864                Poll::Pending
865            }
866        })
867        .await
868    }
869}
870
871/// An object that can be polled to see if a yield has been requested.
872#[derive(Debug)]
873pub struct NeedsYield {
874    yield_requested: AtomicBool,
875}
876
877impl NeedsYield {
878    /// Creates a new object.
879    pub fn new() -> Self {
880        Self {
881            yield_requested: false.into(),
882        }
883    }
884
885    /// Requests a yield.
886    ///
887    /// Returns whether a signal is necessary to ensure that the task yields
888    /// soon.
889    pub fn request_yield(&self) -> bool {
890        !self.yield_requested.swap(true, Ordering::Release)
891    }
892
893    /// Yields execution to the executor if `request_yield` has been called
894    /// since the last call to `maybe_yield`.
895    pub async fn maybe_yield(&self) {
896        poll_fn(|cx| {
897            if self.yield_requested.load(Ordering::Acquire) {
898                // Wake this task again to ensure it runs again.
899                cx.waker().wake_by_ref();
900                self.yield_requested.store(false, Ordering::Relaxed);
901                Poll::Pending
902            } else {
903                Poll::Ready(())
904            }
905        })
906        .await
907    }
908}
909
910/// The reason that [`Processor::run_vp`] returned.
911#[derive(Debug)]
912pub enum VpHaltReason {
913    /// The processor was requested to stop.
914    Stop(VpStopped),
915    /// The processor task should be restarted, possibly on a different thread.
916    Cancel,
917    /// The processor initiated a power off.
918    PowerOff,
919    /// The processor initiated a reboot.
920    Reset,
921    /// The processor initiated a hibernation.
922    Hibernate,
923    /// The processor triple faulted.
924    TripleFault {
925        /// The faulting VTL.
926        // FUTURE: move VTL state into `AccessVpState``.
927        vtl: Vtl,
928    },
929    /// Debugger single step.
930    SingleStep,
931    /// Debugger hardware breakpoint.
932    HwBreak(HardwareBreakpoint),
933}
934
935impl From<VpStopped> for VpHaltReason {
936    fn from(stop: VpStopped) -> Self {
937        Self::Stop(stop)
938    }
939}
940
941pub trait PartitionMemoryMapper {
942    /// Returns a memory mapper for the partition backing `vtl`.
943    fn memory_mapper(&self, vtl: Vtl) -> Arc<dyn PartitionMemoryMap>;
944
945    /// Returns an interface for acquiring host access to memory.
946    fn host_access(&self) -> Option<Arc<dyn PartitionHostAccess>> {
947        None
948    }
949}
950
951pub trait Hv1 {
952    type Error: std::error::Error + Send + Sync + 'static;
953    type Device: MapVpciInterrupt + SignalMsi;
954
955    fn reference_time_source(&self) -> Option<ReferenceTimeSource>;
956
957    fn new_virtual_device(
958        &self,
959    ) -> Option<&dyn DeviceBuilder<Device = Self::Device, Error = Self::Error>>;
960
961    /// Returns the partition's synic port access, or an error if the
962    /// backend cannot support synic in its current configuration.
963    fn synic(&self) -> anyhow::Result<Arc<dyn vmcore::synic::SynicPortAccess>>;
964}
965
966pub trait DeviceBuilder: Hv1 {
967    fn build(&self, vtl: Vtl, device_id: u64) -> Result<Self::Device, Self::Error>;
968}
969
970pub enum UnimplementedDevice {}
971
972impl MapVpciInterrupt for UnimplementedDevice {
973    async fn register_interrupt(
974        &self,
975        _vector_count: u32,
976        _params: &VpciInterruptParameters<'_>,
977    ) -> Result<MsiAddressData, RegisterInterruptError> {
978        match *self {}
979    }
980
981    async fn unregister_interrupt(&self, _address: u64, _data: u32) {
982        match *self {}
983    }
984}
985
986impl SignalMsi for UnimplementedDevice {
987    fn signal_msi(&self, _devid: Option<u32>, _address: u64, _data: u32) {
988        match *self {}
989    }
990}
991
992/// MNF support routines for the emulator
993pub trait EmulatorMonitorSupport {
994    /// Check if the specified write is inside the monitor page, and signal the associated
995    /// connection ID if it is.
996    #[must_use]
997    fn check_write(&self, gpa: u64, bytes: &[u8]) -> bool;
998
999    /// Check if the specified read is inside the monitor page, and fill the provided buffer
1000    /// if it is.
1001    #[must_use]
1002    fn check_read(&self, gpa: u64, bytes: &mut [u8]) -> bool;
1003}