Skip to main content

membacking/mapping_manager/
va_mapper.rs

1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4//! Implements the VA mapper, which maintains a linear virtual address space for
5//! all memory mapped into a partition.
6//!
7//! VA mappers come in two modes:
8//!
9//! - **Eager**: mappings are pushed by the mapping manager when they are added
10//!   and replayed when the mapper is created. Page faults on file-backed ranges
11//!   fail immediately — the mapping should already be established. This is the
12//!   right mode for the VP process, where hypervisors like KVM do not forward
13//!   page faults back to the VMM.
14//!
15//! - **Lazy**: mappings are not pushed proactively. Instead, page faults
16//!   trigger an on-demand request to the mapping manager, which finds the
17//!   backing mapping and pushes it to the mapper via Rpc. This avoids the cost
18//!   of notifying processes that rarely access certain mappings (e.g.,
19//!   device-emulation processes with virtio-fs DAX).
20//!
21//! In both modes, private memory ranges are committed up front (Windows) or
22//! handled transparently by the kernel (Linux).
23//!
24//! On Windows, the **primary** (local) mapper's writable THP-eligible guest RAM
25//! (private *and* shared/section) additionally uses a "deferred protect" scheme
26//! for soft large pages: the range is committed/mapped read-only, and the first
27//! write fault upgrades a full 2 MB window to read-write and prefetches it (via
28//! `page_fault` for host-side writes such as the loader, or `resolve` for guest
29//! writes). Faulting a uniform 2 MB region in one operation gives the OS the
30//! opportunity to back it with a large page (which the hypervisor can map as a
31//! 2 MB SLAT entry) instead of the fragmented small pages that result from
32//! dribbled per-page writes. Non-primary (device/DMA) mappers use plain 4 KB
33//! read-write pages.
34//!
35//! When such a range is also *prefetched*, it is populated eagerly at build
36//! time instead: it stays read-write (the build-time populate cannot access a
37//! read-only mapping) and its per-window first-fault bitmap starts fully set, so
38//! `resolve` treats every window as already attempted.
39
40// UNSAFETY: Implementing the unsafe GuestMemoryAccess trait by calling unsafe
41// low level memory manipulation functions.
42#![expect(unsafe_code)]
43
44// Soft large pages are a Windows-only optimization; other targets get an
45// uninhabited stub with the same interface (`SoftLp::new` returns `None`) so the
46// fault paths compile without per-item `cfg`s.
47#[cfg_attr(not(windows), path = "va_mapper/soft_lp_stub.rs")]
48mod soft_lp;
49
50use self::soft_lp::SoftLp;
51use super::manager::DmaRegionProvider;
52use super::manager::MapperId;
53use super::manager::MapperRequest;
54use super::manager::MappingBacking;
55use super::manager::MappingError;
56use super::manager::MappingParams;
57use super::manager::MappingRequest;
58use super::manager::MemoryPolicy;
59use crate::RemoteProcess;
60use futures::executor::block_on;
61use guestmem::GuestMemoryAccess;
62use guestmem::GuestMemoryBackingError;
63use guestmem::GuestMemoryErrorKind;
64use guestmem::GuestMemorySharing;
65use guestmem::PageFaultAction;
66use guestmem::PageFaultError;
67use inspect::Inspect;
68use inspect_counters::SharedCounter;
69use memory_range::MemoryRange;
70use mesh::error::RemoteError;
71use mesh::rpc::RpcError;
72use mesh::rpc::RpcSend;
73use parking_lot::Mutex;
74use parking_lot::RwLock;
75use range_map_vec::RangeMap;
76use sparse_mmap::SparseMapping;
77use std::ptr::NonNull;
78use std::sync::Arc;
79use std::sync::OnceLock;
80use std::sync::atomic::AtomicBool;
81use std::sync::atomic::Ordering;
82use std::thread::JoinHandle;
83use thiserror::Error;
84use virt::ResolveMemoryFault;
85#[cfg(windows)]
86use windows_sys::Win32::System::Memory::PAGE_READONLY;
87#[cfg(windows)]
88use windows_sys::Win32::System::Memory::PAGE_READWRITE;
89#[cfg(windows)]
90use windows_sys::Win32::System::Memory::SECTION_MAP_READ;
91#[cfg(windows)]
92use windows_sys::Win32::System::Memory::SECTION_MAP_WRITE;
93
94#[derive(Debug, Error)]
95#[error("unexpected page fault")]
96struct UnexpectedPageFault;
97
98/// The role of a [`VaMapper`].
99///
100/// Exactly one mapper per VM is [`Primary`](Self::Primary): the loader's write
101/// target and the partition's fault resolver, and the only mapper for which soft
102/// large pages (Windows) are worthwhile, since its host backing drives the
103/// guest's SLAT. All other guest-memory access — `guest_memory()` in any
104/// process, DMA mappers, and remote partition-backing mappers — is
105/// [`Secondary`](Self::Secondary) and uses plain read-write 4 KB pages.
106///
107/// This is a role, not a location: it is set explicitly at construction rather
108/// than inferred from whether the mapping is local, so a remote primary mapper
109/// or a local secondary mapper (e.g. a device process's own local mapper) is
110/// handled correctly.
111#[derive(Debug, Copy, Clone, PartialEq, Eq)]
112pub(crate) enum MapperRole {
113    /// The single loader/partition mapper; eligible for soft large pages.
114    Primary {
115        /// Whether the partition delivers guest-memory-access faults to the
116        /// VMM. Soft large pages map a window read-only and rely on the
117        /// resulting write fault being resolved to raise it, so they are only
118        /// enabled when this is set. Carried on the `Primary` variant because it
119        /// is meaningless for a secondary mapper.
120        supports_memory_fault_resolution: bool,
121    },
122    /// Any other mapper; plain read-write 4 KB pages.
123    Secondary,
124}
125
126/// Properties recorded for each active guest-memory mapping, used to answer
127/// per-address queries (private vs. shared, soft-large-page state) without a
128/// static snapshot of the RAM layout.
129#[derive(Debug)]
130struct MappingProps {
131    /// Backed by private anonymous memory (committed up front) rather than a
132    /// shared file/section mapping.
133    private: bool,
134    /// General per-mapping fault counters, always present. See [`FaultStats`].
135    stats: FaultStats,
136    /// Soft-large-page (Windows THP) state, or `None` when the scheme does not
137    /// apply (non-primary/device mappers, read-only or non-THP ranges, and every
138    /// non-Windows host). See the [`soft_lp`] module.
139    soft_lp: Option<SoftLp>,
140}
141
142/// Per-mapping fault counters, exposed via `Inspect`.
143///
144/// Recorded for every mapping regardless of host OS, role, or backing, so
145/// general fault accounting is available even on mappings that never use soft
146/// large pages. Kept per mapping — one set of counters per backing, and thus
147/// per NUMA node — so they scale for large multi-NUMA-node VMs rather than
148/// contending on a single global counter. The counters are plain atomics
149/// (`SharedCounter`), bumped in place under the mapping-index read lock.
150#[derive(Debug, Default, Inspect)]
151struct FaultStats {
152    /// Guest memory faults resolved for this mapping.
153    guest_faults: SharedCounter,
154}
155
156/// A virtual address space mapper for guest memory.
157///
158/// Maintains a reserved VA range and maps file-backed or anonymous memory
159/// into it as directed by the mapping manager.
160pub struct VaMapper {
161    inner: Arc<MapperInner>,
162    id: MapperId,
163    process: Option<RemoteProcess>,
164    _thread: JoinHandle<()>,
165}
166
167impl std::fmt::Debug for VaMapper {
168    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
169        f.debug_struct("VaMapper")
170            .field("inner", &self.inner)
171            .field("_thread", &self._thread)
172            .finish()
173    }
174}
175
176impl Drop for VaMapper {
177    fn drop(&mut self) {
178        // Do not join the mapper thread here. The mapping manager must process
179        // this request before the mapper request channel closes, and joining in
180        // Drop could deadlock if the manager task needs the current executor to
181        // make progress. Once the manager removes its sender, the mapper thread
182        // exits naturally.
183        self.inner
184            .req_send
185            .send(MappingRequest::RemoveMapper(self.id));
186    }
187}
188
189impl Inspect for VaMapper {
190    /// Contributes each mapping's counters to the shared `mappings` node, keyed
191    /// by GPA range, so the stats sit alongside the mapping they describe (the
192    /// mapping-manager entry with the same range merges with this one). Every
193    /// mapping contributes a `faults` child (general fault accounting); only
194    /// soft-large-page mappings (the primary mapper's writable THP ranges on
195    /// Windows) additionally contribute a `soft_large_pages` child.
196    fn inspect(&self, req: inspect::Request<'_>) {
197        req.respond().field(
198            "mappings",
199            inspect::adhoc(|req| {
200                let mut resp = req.respond();
201                let mappings = self.inner.mappings.read();
202                for (range, props) in mappings.iter() {
203                    let range = MemoryRange::new(*range.start()..*range.end() + 1);
204                    resp.field(
205                        &range.to_string(),
206                        inspect::adhoc(|req| {
207                            let mut resp = req.respond();
208                            resp.field("faults", &props.stats);
209                            if let Some(sl) = &props.soft_lp {
210                                resp.field("soft_large_pages", sl);
211                            }
212                        }),
213                    );
214                }
215            }),
216        );
217    }
218}
219
220#[derive(Debug)]
221struct MapperInner {
222    mapping: SparseMapping,
223    /// Waiters for lazy mapping requests. `None` after the mapper task exits.
224    waiters: Mutex<Option<Vec<MapWaiter>>>,
225    /// Index of active mappings recorded as they are established, keyed by GPA.
226    /// Written by the mapper task on map/unmap and read by the page-fault and
227    /// fault-resolution paths. Replaces a static snapshot of the RAM layout, so
228    /// hot-added ranges populate it like any other mapping.
229    mappings: RwLock<RangeMap<u64, MappingProps>>,
230    /// Whether this mapper receives mappings eagerly (pushed by the
231    /// mapping manager) or lazily (on demand via page faults).
232    /// Set by the mapping manager task after replay is complete.
233    ///
234    /// `Relaxed` ordering is sufficient: this flag is only read by the
235    /// page-fault handler to decide between eager-fail and lazy-request
236    /// paths. A stale `false` (lazy) is harmless — the lazy path
237    /// succeeds because the mapping is already established. The flag
238    /// is eventually updated after `SetEager` is processed.
239    eager: AtomicBool,
240    /// Whether this is the **primary** mapper — the one the partition and the
241    /// loader run against. Soft large pages (Windows) are only worthwhile here,
242    /// since this is the mapping whose host backing drives the guest's SLAT.
243    /// Secondary mappers use plain read-write 4 KB pages. See [`MapperRole`].
244    primary: bool,
245    /// Whether the partition delivers guest-memory-access faults to the VMM.
246    /// Soft large pages map a window read-only and rely on the resulting write
247    /// fault being resolved to raise it, so without fault resolution they would
248    /// wedge on the first guest write; the primary mapper only enables them when
249    /// this is set.
250    supports_memory_fault_resolution: bool,
251    req_send: mesh::Sender<MappingRequest>,
252    /// Maintains a weak reference to avoid a reference cycle with the partition.
253    host_access: OnceLock<HostAccess>,
254}
255
256#[derive(Clone)]
257struct HostAccess(std::sync::Weak<dyn virt::PartitionHostAccess>);
258
259impl std::fmt::Debug for HostAccess {
260    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
261        f.write_str("HostAccess")
262    }
263}
264
265/// A pending lazy mapping request.
266#[derive(Debug)]
267struct MapWaiter {
268    range: MemoryRange,
269    writable: bool,
270    done: mesh::OneshotSender<bool>,
271}
272
273impl MapWaiter {
274    /// Check whether the established mapping satisfies this waiter.
275    /// Returns `Some(true)` if fully satisfied, `Some(false)` if the
276    /// mapping doesn't meet requirements (e.g., read-only when write
277    /// needed), or `None` if the waiter still has remaining range.
278    fn complete(&mut self, range: MemoryRange, writable: Option<bool>) -> Option<bool> {
279        if range.contains_addr(self.range.start()) {
280            if writable.is_none() || (self.writable && writable == Some(false)) {
281                return Some(false);
282            }
283            let new_start = self.range.end().min(range.end());
284            let remaining = MemoryRange::new(new_start..self.range.end());
285            if remaining.is_empty() {
286                return Some(true);
287            }
288            tracing::debug!(%remaining, "waiting for more");
289            self.range = remaining;
290        }
291        None
292    }
293}
294
295struct MapperTask {
296    inner: Arc<MapperInner>,
297}
298
299impl MapperTask {
300    async fn run(mut self, mut req_recv: mesh::Receiver<MapperRequest>) {
301        while let Ok(req) = req_recv.recv().await {
302            match req {
303                MapperRequest::Unmap(rpc) => rpc.handle_sync(|range| {
304                    tracing::debug!(%range, "invalidate received");
305                    self.inner
306                        .mapping
307                        .unmap(range.start() as usize, range.len() as usize)
308                        .expect("invalidate request should be valid");
309                    self.inner.remove_mapping(range);
310                }),
311                MapperRequest::MapEager(rpc) => {
312                    rpc.handle_failable_sync(|params| {
313                        tracing::debug!(range = %params.range, "eager mapping received");
314                        self.map(params)
315                    });
316                }
317                MapperRequest::MapLazy(params) => {
318                    tracing::debug!(range = %params.range, "lazy mapping received");
319                    let (range, writable) = (params.range, params.writable);
320                    match self.map(params) {
321                        Ok(()) => self.wake_waiters(range, Some(writable)),
322                        Err(e) => {
323                            tracing::error!(
324                                error = &e as &dyn std::error::Error,
325                                %range,
326                                "failed to map file for range"
327                            );
328                            self.wake_waiters(range, None);
329                        }
330                    }
331                }
332                MapperRequest::NoMapping(range) => {
333                    // Wake up waiters. They'll see a failure when they try
334                    // to access the VA.
335                    tracing::debug!(%range, "no mapping received for range");
336                    self.wake_waiters(range, None);
337                }
338                MapperRequest::SetEager(rpc) => rpc.handle_sync(|()| {
339                    tracing::debug!("mapper upgraded to eager");
340                    self.inner.eager.store(true, Ordering::Relaxed);
341                }),
342            }
343        }
344        // Don't allow more waiters.
345        *self.inner.waiters.lock() = None;
346        // Invalidate everything.
347        let _ = self.inner.mapping.unmap(0, self.inner.mapping.len());
348    }
349
350    /// Establishes a mapping in the VA space, dispatching on how it is backed.
351    fn map(&self, params: MappingParams) -> Result<(), MappingError> {
352        // Soft large pages apply only to writable THP-eligible RAM on the primary
353        // mapper (Windows); `SoftLp::new` returns `None` otherwise. See the
354        // `soft_lp` module. They also depend on the partition delivering write
355        // faults to raise deferred-protect windows, so don't even build one when
356        // the partition can't resolve faults.
357        let soft_lp = if self.inner.supports_memory_fault_resolution {
358            SoftLp::new(
359                params.range,
360                &params.policy,
361                params.writable,
362                self.inner.primary,
363            )
364        } else {
365            None
366        };
367
368        // Deferred protect (map read-only, raise on the first write fault) drives
369        // the lazy soft-LP path; prefetched ranges are populated read-write
370        // eagerly at build time instead.
371        let deferred_protect = soft_lp.as_ref().is_some_and(SoftLp::deferred_protect);
372
373        let private = match &params.backing {
374            MappingBacking::File {
375                mappable,
376                file_offset,
377            } => {
378                self.map_file(&params, mappable, *file_offset, deferred_protect)?;
379                false
380            }
381            MappingBacking::Private => {
382                self.map_private(&params, deferred_protect)?;
383                true
384            }
385        };
386        self.inner.record_mapping(
387            params.range,
388            MappingProps {
389                private,
390                stats: FaultStats::default(),
391                soft_lp,
392            },
393        );
394        Ok(())
395    }
396
397    /// Maps a file-backed region into the VA space, applying NUMA policy where
398    /// supported.
399    fn map_file(
400        &self,
401        params: &MappingParams,
402        mappable: &super::mappable::Mappable,
403        file_offset: u64,
404        deferred_protect: bool,
405    ) -> Result<(), MappingError> {
406        let &MappingParams {
407            range,
408            backing: _,
409            writable,
410            mapping_type: _,
411            policy:
412                MemoryPolicy {
413                    numa_node,
414                    transparent_hugepages,
415                    prefetch: _,
416                },
417        } = params;
418        // A deferred-protect range is mapped read-write (so the view has write
419        // access and its pages can be raised back to read-write on the first
420        // write fault) and then immediately protected down to read-only just
421        // below. Mapping the view read-only up front instead would create a view
422        // whose pages cannot be raised to read-write later (`VirtualProtect`
423        // fails with ERROR_INVALID_PARAMETER). Deferred protect implies
424        // `writable`.
425        #[cfg(windows)]
426        let (protect, access) = (
427            if writable {
428                PAGE_READWRITE
429            } else {
430                PAGE_READONLY
431            },
432            if writable {
433                SECTION_MAP_READ | SECTION_MAP_WRITE
434            } else {
435                SECTION_MAP_READ
436            },
437        );
438        // `deferred_protect` is only consulted on Windows below; keep it live on
439        // other targets so the shared parameter doesn't warn.
440        let _ = deferred_protect;
441        let map_result = cfg_select! {
442            windows => {
443                self.inner.mapping.map_view_of_file_access(
444                    range.start() as usize,
445                    range.len() as usize,
446                    mappable,
447                    file_offset,
448                    protect,
449                    access,
450                    numa_node,
451                )
452            }
453            _ => {
454                self.inner.mapping.map_file(
455                    range.start() as usize,
456                    range.len() as usize,
457                    mappable,
458                    file_offset,
459                    writable,
460                )
461            }
462        };
463
464        if let Err(e) = map_result {
465            return Err(MappingError::new(range, e));
466        }
467
468        // Deferred protect: lower the freshly-mapped writable view to read-only
469        // so the first write faults; `page_fault`/`resolve` then raise the
470        // touched 2 MB window back to read-write.
471        #[cfg(windows)]
472        if deferred_protect {
473            if let Err(e) = self.inner.mapping.protect(
474                range.start() as usize,
475                range.len() as usize,
476                PAGE_READONLY,
477            ) {
478                return Err(MappingError::new(range, e));
479            }
480        }
481
482        // Mark shared (file-backed) RAM as THP-eligible. This is advisory:
483        // on Linux the kernel honors it for shmem/tmpfs (memfd) mappings
484        // according to `/sys/kernel/mm/transparent_hugepage/shmem_enabled`.
485        // The kernel may accept the advice without allocating huge pages;
486        // advice failures are logged but do not fail the mapping.
487        #[cfg(target_os = "linux")]
488        if transparent_hugepages {
489            if let Err(e) = self
490                .inner
491                .mapping
492                .madvise_hugepage(range.start() as usize, range.len() as usize)
493            {
494                tracing::warn!(
495                    error = &e as &dyn std::error::Error,
496                    %range,
497                    "failed to mark shared RAM as THP eligible"
498                );
499            }
500        }
501        #[cfg(not(target_os = "linux"))]
502        let _ = transparent_hugepages;
503
504        cfg_select! {
505            target_os = "linux" => {
506                if let Some(node) = numa_node {
507                    if let Err(e) = self.inner.mapping.mbind_at(
508                        range.start() as usize,
509                        range.len() as usize,
510                        node,
511                    ) {
512                        tracing::error!(
513                            error = &e as &dyn std::error::Error,
514                            %range,
515                            node,
516                            "NUMA binding failed, using default placement"
517                        );
518                    }
519                }
520            }
521            windows => {
522                // NUMA handled by the map_view_of_file_access call above.
523                let _ = numa_node;
524            }
525            _ => {
526                assert!(numa_node.is_none(), "NUMA not supported on this platform; should have been rejected at build time");
527            }
528        }
529
530        Ok(())
531    }
532
533    /// Commits private anonymous memory for a range into the VA space.
534    ///
535    /// This replaces the reserved placeholder at `range` with committed
536    /// anonymous pages, optionally bound to a host NUMA node and marked
537    /// eligible for Transparent Huge Pages.
538    fn map_private(
539        &self,
540        params: &MappingParams,
541        deferred_protect: bool,
542    ) -> Result<(), MappingError> {
543        let &MappingParams {
544            range,
545            backing: _,
546            writable: _,
547            mapping_type: _,
548            policy:
549                MemoryPolicy {
550                    numa_node,
551                    transparent_hugepages,
552                    prefetch: _,
553                },
554        } = params;
555        let offset = range.start() as usize;
556        let len = range.len() as usize;
557
558        // On Windows, deferred-protect private RAM commits read-only so the first
559        // write faults and a full 2 MB window can be raised to read-write and
560        // materialized at once, giving the loader (and guest) large pages.
561        // Elsewhere this flag is ignored.
562        if let Err(e) = self.inner.alloc(offset, len, numa_node, deferred_protect) {
563            return Err(MappingError::new(range, e));
564        }
565
566        // Name the range so it's identifiable in /proc/{pid}/smaps.
567        self.inner
568            .mapping
569            .set_name(offset, len, "guest-ram-private");
570
571        #[cfg(target_os = "linux")]
572        if transparent_hugepages {
573            if let Err(e) = self.inner.mapping.madvise_hugepage(offset, len) {
574                tracing::warn!(
575                    error = &e as &dyn std::error::Error,
576                    %range,
577                    "failed to mark private RAM as THP eligible"
578                );
579            }
580        }
581        #[cfg(not(target_os = "linux"))]
582        let _ = transparent_hugepages;
583
584        Ok(())
585    }
586
587    fn wake_waiters(&mut self, range: MemoryRange, writable: Option<bool>) {
588        let mut waiters = self.inner.waiters.lock();
589        let waiters = waiters.as_mut().unwrap();
590
591        let mut i = 0;
592        while i < waiters.len() {
593            if let Some(success) = waiters[i].complete(range, writable) {
594                waiters.swap_remove(i).done.send(success);
595            } else {
596                i += 1;
597            }
598        }
599    }
600}
601
602#[derive(Debug, Error)]
603pub enum VaMapperError {
604    #[error("failed to communicate with the memory manager")]
605    MemoryManagerGone(#[source] RpcError),
606    #[error("failed to register mapper")]
607    Registration(#[source] RemoteError),
608    #[error("failed to reserve address space")]
609    Reserve(#[source] std::io::Error),
610}
611
612/// Error returned when a lazy mapping request cannot be fulfilled.
613#[derive(Debug, Error)]
614#[error("no mapping for {0}")]
615pub struct NoMapping(MemoryRange);
616
617impl MapperInner {
618    /// Records an established mapping in the index, replacing any stale entry
619    /// for the same range.
620    fn record_mapping(&self, range: MemoryRange, props: MappingProps) {
621        if range.is_empty() {
622            return;
623        }
624        let mut mappings = self.mappings.write();
625        mappings.remove_range(range.start()..=range.end() - 1);
626        let inserted = mappings.insert(range.start()..=range.end() - 1, props);
627        assert!(
628            inserted,
629            "mapping index range should be clear after removal"
630        );
631    }
632
633    /// Removes a mapping from the index.
634    fn remove_mapping(&self, range: MemoryRange) {
635        if range.is_empty() {
636            return;
637        }
638        self.mappings
639            .write()
640            .remove_range(range.start()..=range.end() - 1);
641    }
642
643    /// Request that the mapping manager send mappings for the given range.
644    ///
645    /// Registers a waiter, sends `SendMappings` (fire-and-forget), and
646    /// awaits the waiter oneshot. The mapping manager will send `MapLazy`
647    /// or `NoMapping` messages to the mapper task, which wakes the waiter.
648    async fn request_mapping(
649        &self,
650        id: MapperId,
651        range: MemoryRange,
652        writable: bool,
653    ) -> Result<(), NoMapping> {
654        let (send, recv) = mesh::oneshot();
655        self.waiters
656            .lock()
657            .as_mut()
658            .ok_or(NoMapping(range))?
659            .push(MapWaiter {
660                range,
661                writable,
662                done: send,
663            });
664
665        tracing::debug!(%range, "waiting for mappings");
666        self.req_send.send(MappingRequest::SendMappings(id, range));
667        match recv.await {
668            Ok(true) => Ok(()),
669            Ok(false) | Err(_) => Err(NoMapping(range)),
670        }
671    }
672
673    /// Commits private anonymous memory for a range, optionally bound to a
674    /// specific host NUMA node.
675    ///
676    /// This replaces the placeholder at the given offset with committed
677    /// anonymous memory.
678    ///
679    /// When `deferred_protect` is set (Windows soft large pages), the memory is
680    /// committed read-only instead of read-write, so the first write faults and
681    /// [`VaMapper`] can upgrade a full 2 MB window to read-write at once. This
682    /// has no effect on other platforms.
683    ///
684    /// Caution: on Linux, if NUMA binding fails, the allocation itself has
685    /// still succeeded — the returned error does not imply the memory is
686    /// unmapped.
687    fn alloc(
688        &self,
689        offset: usize,
690        len: usize,
691        numa_node: Option<u32>,
692        deferred_protect: bool,
693    ) -> Result<(), std::io::Error> {
694        cfg_select! {
695            windows => {
696                // Deferred protect (soft large pages): commit read-only so the
697                // first write faults and the 2 MB window can be raised +
698                // materialized as a large page; otherwise commit read-write.
699                let protect = if deferred_protect {
700                    PAGE_READONLY
701                } else {
702                    PAGE_READWRITE
703                };
704                self.mapping.virtual_alloc(offset, len, protect, numa_node)
705            }
706            target_os = "linux" => {
707                let _ = deferred_protect;
708                self.mapping.alloc(offset, len)?;
709                if let Some(node) = numa_node {
710                    self.mapping.mbind_at(offset, len, node)?;
711                }
712                Ok(())
713            }
714            _ => {
715                let _ = deferred_protect;
716                assert!(numa_node.is_none(), "NUMA not supported on this platform; should have been rejected at build time");
717                self.mapping.alloc(offset, len)
718            }
719        }
720    }
721}
722
723impl VaMapper {
724    pub(crate) async fn new(
725        req_send: mesh::Sender<MappingRequest>,
726        len: u64,
727        remote_process: Option<RemoteProcess>,
728        minimum_alignment: Option<usize>,
729        eager: bool,
730        role: MapperRole,
731    ) -> Result<Self, VaMapperError> {
732        // Soft large pages apply only to the primary mapper, and only when the
733        // partition resolves faults; `supports_memory_fault_resolution` rides on
734        // the `Primary` variant.
735        let (primary, supports_memory_fault_resolution) = match role {
736            MapperRole::Primary {
737                supports_memory_fault_resolution,
738            } => (true, supports_memory_fault_resolution),
739            MapperRole::Secondary => (false, false),
740        };
741        let mapping = match &remote_process {
742            None => SparseMapping::new_with_minimum_alignment(
743                len as usize,
744                minimum_alignment.unwrap_or(1),
745            ),
746            Some(process) => match process {
747                #[cfg(not(windows))]
748                _ => unreachable!(),
749                #[cfg(windows)]
750                process => SparseMapping::new_remote(
751                    process.as_handle().try_clone_to_owned().unwrap().into(),
752                    None,
753                    len as usize,
754                    minimum_alignment.unwrap_or(1),
755                ),
756            },
757        }
758        .map_err(VaMapperError::Reserve)?;
759
760        // Name the VA reservation so it's identifiable in /proc/{pid}/smaps.
761        mapping.set_name(0, mapping.len(), "guest-memory");
762
763        let (send, req_recv) = mesh::channel();
764
765        let inner = Arc::new(MapperInner {
766            mapping,
767            waiters: Mutex::new(Some(Vec::new())),
768            mappings: RwLock::new(RangeMap::new()),
769            eager: AtomicBool::new(eager),
770            primary,
771            supports_memory_fault_resolution,
772            req_send,
773            host_access: OnceLock::new(),
774        });
775
776        // Spawn the mapper thread *before* the AddMapper RPC. The manager
777        // replays existing mappings to eager mappers during AddMapper, so
778        // the mapper thread must be running to respond to those RPCs.
779        //
780        // FUTURE: use a task once we resolve the block_ons in the
781        // GuestMemoryAccess implementation.
782        let thread = std::thread::Builder::new()
783            .name("mapper".to_owned())
784            .spawn({
785                let runner = MapperTask {
786                    inner: inner.clone(),
787                };
788                || block_on(runner.run(req_recv))
789            })
790            .unwrap();
791
792        let id = match inner
793            .req_send
794            .call(
795                MappingRequest::AddMapper,
796                super::manager::AddMapperParams { send, eager },
797            )
798            .await
799        {
800            Ok(Ok(id)) => id,
801            Ok(Err(e)) => {
802                // Drop inner to shut down the mapper thread (closes req_recv).
803                drop(inner);
804                let _ = thread.join();
805                return Err(VaMapperError::Registration(e));
806            }
807            Err(e) => {
808                drop(inner);
809                let _ = thread.join();
810                return Err(VaMapperError::MemoryManagerGone(e));
811            }
812        };
813
814        Ok(VaMapper {
815            inner,
816            id,
817            process: remote_process,
818            _thread: thread,
819        })
820    }
821
822    /// Returns the base pointer of the VA reservation.
823    pub fn as_ptr(&self) -> *mut u8 {
824        self.inner.mapping.as_ptr().cast()
825    }
826
827    /// Installs the callback used to recover eager-mapper faults caused by
828    /// missing host permission.
829    pub(crate) fn install_host_access(&self, host_access: Arc<dyn virt::PartitionHostAccess>) {
830        assert!(
831            self.inner
832                .host_access
833                .set(HostAccess(Arc::downgrade(&host_access)))
834                .is_ok(),
835            "host access is already installed"
836        );
837    }
838
839    /// Returns the length of the VA reservation in bytes.
840    pub fn len(&self) -> usize {
841        self.inner.mapping.len()
842    }
843
844    /// Returns true if this mapper receives mappings eagerly.
845    pub fn is_eager(&self) -> bool {
846        self.inner.eager.load(Ordering::Relaxed)
847    }
848
849    /// Returns the mapper's ID, used internally for upgrade requests.
850    pub(crate) fn mapper_id(&self) -> MapperId {
851        self.id
852    }
853
854    /// Returns the remote process, if this mapper maps into a remote process.
855    pub fn process(&self) -> Option<&RemoteProcess> {
856        self.process.as_ref()
857    }
858}
859
860/// SAFETY: the underlying VA mapping is guaranteed to be valid for the lifetime
861/// of this object.
862unsafe impl GuestMemoryAccess for VaMapper {
863    fn mapping(&self) -> Option<NonNull<u8>> {
864        // No one should be using this as a GuestMemoryAccess for remote
865        // mappings, but it's convenient to have the same type for both local
866        // and remote mappings for the sake of simplicity in
867        // `PartitionRegionMapper`.
868        assert!(self.inner.mapping.is_local());
869
870        NonNull::new(self.inner.mapping.as_ptr().cast())
871    }
872
873    fn max_address(&self) -> u64 {
874        self.inner.mapping.len() as u64
875    }
876
877    fn page_fault(
878        &self,
879        address: u64,
880        len: usize,
881        write: bool,
882        bitmap_failure: bool,
883    ) -> PageFaultAction {
884        assert!(!bitmap_failure, "bitmaps are not used");
885
886        // Soft large pages (Windows): THP-eligible ranges on the primary mapper
887        // are committed/mapped read-only, so the first *write* traps here (reads
888        // are served by the zero page and don't fault). This is the loader's
889        // path; `SoftLp::on_host_write` raises (and prefetches) the covering
890        // 2 MB window, then the write is retried. The mapping-index read lock is
891        // held across the raise; it only blocks a concurrent *writer* (a
892        // structural map/unmap), which is rare.
893        #[cfg(windows)]
894        if write {
895            let mappings = self.inner.mappings.read();
896            if let Some(&(start, end, ref props)) = mappings.get_entry(&address) {
897                if let Some(sl) = &props.soft_lp {
898                    return match sl.on_host_write(&self.inner.mapping, address, start, end) {
899                        Ok(()) => PageFaultAction::Retry,
900                        Err(err) => PageFaultAction::Fail(PageFaultError::new(
901                            GuestMemoryErrorKind::Other,
902                            err,
903                        )),
904                    };
905                }
906            }
907        }
908
909        if self.inner.eager.load(Ordering::Relaxed) {
910            // The guest-memory VA is already mapped for an eager mapper. For
911            // isolated guests, a fault can instead mean that the hypervisor has
912            // not granted userspace access to a shared page.
913            if let Some(host_access) = self
914                .inner
915                .host_access
916                .get()
917                .and_then(|host_access| host_access.0.upgrade())
918            {
919                let start = address & !(hvdef::HV_PAGE_SIZE - 1);
920                let end = address
921                    .checked_add(len as u64)
922                    .and_then(|end| end.checked_add(hvdef::HV_PAGE_SIZE - 1))
923                    .map(|end| end & !(hvdef::HV_PAGE_SIZE - 1));
924                let Some(end) = end else {
925                    return PageFaultAction::Fail(PageFaultError::new(
926                        GuestMemoryErrorKind::OutOfRange,
927                        std::io::Error::other("host-access range overflow"),
928                    ));
929                };
930                match host_access.acquire_host_access(start, end - start, write) {
931                    Ok(()) => return PageFaultAction::Retry,
932                    Err(err) => {
933                        return PageFaultAction::Fail(PageFaultError::new(
934                            GuestMemoryErrorKind::Other,
935                            std::io::Error::other(err),
936                        ));
937                    }
938                }
939            }
940
941            // Eager mapper: file-backed mappings are established proactively.
942            // If we get a page fault, the mapping was never set up or was
943            // torn down.
944            return PageFaultAction::Fail(PageFaultError::new(
945                GuestMemoryErrorKind::OutOfRange,
946                UnexpectedPageFault,
947            ));
948        }
949
950        // Lazy mapper: request the mapping on demand from the mapping manager.
951        let range = MemoryRange::bounding(address..address + len as u64);
952        if let Err(err) = block_on(self.inner.request_mapping(self.id, range, write)) {
953            return PageFaultAction::Fail(PageFaultError::new(
954                GuestMemoryErrorKind::OutOfRange,
955                err,
956            ));
957        }
958        PageFaultAction::Retry
959    }
960
961    fn sharing(&self) -> Option<GuestMemorySharing> {
962        // Private anonymous memory is committed on fault in the local process
963        // and cannot be shared to a remote DMA process, so disable DMA sharing
964        // whenever any recorded mapping is private. Derived from the mapping
965        // index rather than a static flag so it tracks the actual backings.
966        if self.inner.mappings.read().iter().any(|(_, p)| p.private) {
967            return None;
968        }
969        Some(GuestMemorySharing::new(DmaRegionProvider {
970            req_send: self.inner.req_send.clone(),
971        }))
972    }
973}
974
975impl ResolveMemoryFault for VaMapper {
976    fn resolve(
977        &self,
978        fault: MemoryRange,
979        write: bool,
980    ) -> Result<MemoryRange, GuestMemoryBackingError> {
981        if fault.end() > self.inner.mapping.len() as u64 {
982            return Err(GuestMemoryBackingError::new(
983                GuestMemoryErrorKind::OutOfRange,
984                fault.start(),
985                UnexpectedPageFault,
986            ));
987        }
988
989        // Hold the mapping-index read lock across the fault resolution. This only
990        // blocks a concurrent *writer* (a structural map/unmap), which is rare;
991        // other faulting VPs are readers and proceed in parallel. `end` is the
992        // inclusive last address of the mapping.
993        let mappings = self.inner.mappings.read();
994        let Some(&(start, end, ref props)) = mappings.get_entry(&fault.start()) else {
995            return Err(GuestMemoryBackingError::new(
996                GuestMemoryErrorKind::OutOfRange,
997                fault.start(),
998                UnexpectedPageFault,
999            ));
1000        };
1001        // The trait contract requires the resolved range to stay within the
1002        // single uniform RAM region that covers `fault.start()`. Today the only
1003        // caller faults one page at a time, so a fault never spans two mappings;
1004        // guard against a future caller passing a wider range that starts in this
1005        // mapping but extends past its end (`end` is the inclusive last address).
1006        if fault.end() > end + 1 {
1007            return Err(GuestMemoryBackingError::new(
1008                GuestMemoryErrorKind::OutOfRange,
1009                fault.start(),
1010                UnexpectedPageFault,
1011            ));
1012        }
1013        props.stats.guest_faults.increment();
1014
1015        // Soft large pages (Windows) raise the covering 2 MB window on the first
1016        // write and may resolve to the whole window; every other mapping (and
1017        // every non-Windows host) resolves to the single faulting page.
1018        match &props.soft_lp {
1019            Some(sl) => sl
1020                .resolve(&self.inner.mapping, fault, write, start, end)
1021                .map_err(|err| {
1022                    GuestMemoryBackingError::new(GuestMemoryErrorKind::Other, fault.start(), err)
1023                }),
1024            None => Ok(fault),
1025        }
1026    }
1027}
1028
1029#[cfg(test)]
1030mod tests {
1031    use sparse_mmap::SparseMapping;
1032
1033    /// Tests that private RAM pages can be allocated, written to, and read from.
1034    #[test]
1035    fn test_private_ram_alloc_write_read() {
1036        let page_size = SparseMapping::page_size();
1037        let mapping = SparseMapping::new(4 * page_size).unwrap();
1038
1039        // Allocate (commit) the first two pages.
1040        mapping.alloc(0, 2 * page_size).unwrap();
1041
1042        // Write and read through SparseMapping methods.
1043        let data = [0xABu8; 128];
1044        mapping.write_at(0, &data).unwrap();
1045
1046        let mut buf = [0u8; 128];
1047        mapping.read_at(0, &mut buf).unwrap();
1048        assert_eq!(buf, data);
1049
1050        // Verify zeros at an untouched offset within committed range.
1051        let mut zero_buf = [0xFFu8; 64];
1052        mapping.read_at(page_size, &mut zero_buf).unwrap();
1053        assert!(
1054            zero_buf.iter().all(|&b| b == 0),
1055            "untouched committed memory should be zeros"
1056        );
1057    }
1058
1059    /// Tests that commit is idempotent (committing already-committed pages is
1060    /// a no-op).
1061    #[test]
1062    fn test_private_ram_commit_idempotent() {
1063        let page_size = SparseMapping::page_size();
1064        let mapping = SparseMapping::new(4 * page_size).unwrap();
1065
1066        // Alloc then commit the same range again.
1067        mapping.alloc(0, 2 * page_size).unwrap();
1068        mapping.commit(0, 2 * page_size).unwrap();
1069        mapping.commit(0, page_size).unwrap();
1070
1071        // Write and read should work.
1072        let pattern = vec![0xEFu8; 64];
1073        mapping.write_at(0, &pattern).unwrap();
1074        let mut buf = vec![0u8; 64];
1075        mapping.read_at(0, &mut buf).unwrap();
1076        assert_eq!(buf, pattern);
1077    }
1078}