membacking/mapping_manager/va_mapper.rs
1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4//! Implements the VA mapper, which maintains a linear virtual address space for
5//! all memory mapped into a partition.
6//!
7//! VA mappers come in two modes:
8//!
9//! - **Eager**: mappings are pushed by the mapping manager when they are added
10//! and replayed when the mapper is created. Page faults on file-backed ranges
11//! fail immediately — the mapping should already be established. This is the
12//! right mode for the VP process, where hypervisors like KVM do not forward
13//! page faults back to the VMM.
14//!
15//! - **Lazy**: mappings are not pushed proactively. Instead, page faults
16//! trigger an on-demand request to the mapping manager, which finds the
17//! backing mapping and pushes it to the mapper via Rpc. This avoids the cost
18//! of notifying processes that rarely access certain mappings (e.g.,
19//! device-emulation processes with virtio-fs DAX).
20//!
21//! In both modes, private memory ranges are committed up front (Windows) or
22//! handled transparently by the kernel (Linux).
23//!
24//! On Windows, the **primary** (local) mapper's writable THP-eligible guest RAM
25//! (private *and* shared/section) additionally uses a "deferred protect" scheme
26//! for soft large pages: the range is committed/mapped read-only, and the first
27//! write fault upgrades a full 2 MB window to read-write and prefetches it (via
28//! `page_fault` for host-side writes such as the loader, or `resolve` for guest
29//! writes). Faulting a uniform 2 MB region in one operation gives the OS the
30//! opportunity to back it with a large page (which the hypervisor can map as a
31//! 2 MB SLAT entry) instead of the fragmented small pages that result from
32//! dribbled per-page writes. Non-primary (device/DMA) mappers use plain 4 KB
33//! read-write pages.
34//!
35//! When such a range is also *prefetched*, it is populated eagerly at build
36//! time instead: it stays read-write (the build-time populate cannot access a
37//! read-only mapping) and its per-window first-fault bitmap starts fully set, so
38//! `resolve` treats every window as already attempted.
39
40// UNSAFETY: Implementing the unsafe GuestMemoryAccess trait by calling unsafe
41// low level memory manipulation functions.
42#![expect(unsafe_code)]
43
44// Soft large pages are a Windows-only optimization; other targets get an
45// uninhabited stub with the same interface (`SoftLp::new` returns `None`) so the
46// fault paths compile without per-item `cfg`s.
47#[cfg_attr(not(windows), path = "va_mapper/soft_lp_stub.rs")]
48mod soft_lp;
49
50use self::soft_lp::SoftLp;
51use super::manager::DmaRegionProvider;
52use super::manager::MapperId;
53use super::manager::MapperRequest;
54use super::manager::MappingBacking;
55use super::manager::MappingError;
56use super::manager::MappingParams;
57use super::manager::MappingRequest;
58use super::manager::MemoryPolicy;
59use crate::RemoteProcess;
60use futures::executor::block_on;
61use guestmem::GuestMemoryAccess;
62use guestmem::GuestMemoryBackingError;
63use guestmem::GuestMemoryErrorKind;
64use guestmem::GuestMemorySharing;
65use guestmem::PageFaultAction;
66use guestmem::PageFaultError;
67use inspect::Inspect;
68use inspect_counters::SharedCounter;
69use memory_range::MemoryRange;
70use mesh::error::RemoteError;
71use mesh::rpc::RpcError;
72use mesh::rpc::RpcSend;
73use parking_lot::Mutex;
74use parking_lot::RwLock;
75use range_map_vec::RangeMap;
76use sparse_mmap::SparseMapping;
77use std::ptr::NonNull;
78use std::sync::Arc;
79use std::sync::OnceLock;
80use std::sync::atomic::AtomicBool;
81use std::sync::atomic::Ordering;
82use std::thread::JoinHandle;
83use thiserror::Error;
84use virt::ResolveMemoryFault;
85#[cfg(windows)]
86use windows_sys::Win32::System::Memory::PAGE_READONLY;
87#[cfg(windows)]
88use windows_sys::Win32::System::Memory::PAGE_READWRITE;
89#[cfg(windows)]
90use windows_sys::Win32::System::Memory::SECTION_MAP_READ;
91#[cfg(windows)]
92use windows_sys::Win32::System::Memory::SECTION_MAP_WRITE;
93
94#[derive(Debug, Error)]
95#[error("unexpected page fault")]
96struct UnexpectedPageFault;
97
98/// The role of a [`VaMapper`].
99///
100/// Exactly one mapper per VM is [`Primary`](Self::Primary): the loader's write
101/// target and the partition's fault resolver, and the only mapper for which soft
102/// large pages (Windows) are worthwhile, since its host backing drives the
103/// guest's SLAT. All other guest-memory access — `guest_memory()` in any
104/// process, DMA mappers, and remote partition-backing mappers — is
105/// [`Secondary`](Self::Secondary) and uses plain read-write 4 KB pages.
106///
107/// This is a role, not a location: it is set explicitly at construction rather
108/// than inferred from whether the mapping is local, so a remote primary mapper
109/// or a local secondary mapper (e.g. a device process's own local mapper) is
110/// handled correctly.
111#[derive(Debug, Copy, Clone, PartialEq, Eq)]
112pub(crate) enum MapperRole {
113 /// The single loader/partition mapper; eligible for soft large pages.
114 Primary {
115 /// Whether the partition delivers guest-memory-access faults to the
116 /// VMM. Soft large pages map a window read-only and rely on the
117 /// resulting write fault being resolved to raise it, so they are only
118 /// enabled when this is set. Carried on the `Primary` variant because it
119 /// is meaningless for a secondary mapper.
120 supports_memory_fault_resolution: bool,
121 },
122 /// Any other mapper; plain read-write 4 KB pages.
123 Secondary,
124}
125
126/// Properties recorded for each active guest-memory mapping, used to answer
127/// per-address queries (private vs. shared, soft-large-page state) without a
128/// static snapshot of the RAM layout.
129#[derive(Debug)]
130struct MappingProps {
131 /// Backed by private anonymous memory (committed up front) rather than a
132 /// shared file/section mapping.
133 private: bool,
134 /// General per-mapping fault counters, always present. See [`FaultStats`].
135 stats: FaultStats,
136 /// Soft-large-page (Windows THP) state, or `None` when the scheme does not
137 /// apply (non-primary/device mappers, read-only or non-THP ranges, and every
138 /// non-Windows host). See the [`soft_lp`] module.
139 soft_lp: Option<SoftLp>,
140}
141
142/// Per-mapping fault counters, exposed via `Inspect`.
143///
144/// Recorded for every mapping regardless of host OS, role, or backing, so
145/// general fault accounting is available even on mappings that never use soft
146/// large pages. Kept per mapping — one set of counters per backing, and thus
147/// per NUMA node — so they scale for large multi-NUMA-node VMs rather than
148/// contending on a single global counter. The counters are plain atomics
149/// (`SharedCounter`), bumped in place under the mapping-index read lock.
150#[derive(Debug, Default, Inspect)]
151struct FaultStats {
152 /// Guest memory faults resolved for this mapping.
153 guest_faults: SharedCounter,
154}
155
156/// A virtual address space mapper for guest memory.
157///
158/// Maintains a reserved VA range and maps file-backed or anonymous memory
159/// into it as directed by the mapping manager.
160pub struct VaMapper {
161 inner: Arc<MapperInner>,
162 id: MapperId,
163 process: Option<RemoteProcess>,
164 _thread: JoinHandle<()>,
165}
166
167impl std::fmt::Debug for VaMapper {
168 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
169 f.debug_struct("VaMapper")
170 .field("inner", &self.inner)
171 .field("_thread", &self._thread)
172 .finish()
173 }
174}
175
176impl Drop for VaMapper {
177 fn drop(&mut self) {
178 // Do not join the mapper thread here. The mapping manager must process
179 // this request before the mapper request channel closes, and joining in
180 // Drop could deadlock if the manager task needs the current executor to
181 // make progress. Once the manager removes its sender, the mapper thread
182 // exits naturally.
183 self.inner
184 .req_send
185 .send(MappingRequest::RemoveMapper(self.id));
186 }
187}
188
189impl Inspect for VaMapper {
190 /// Contributes each mapping's counters to the shared `mappings` node, keyed
191 /// by GPA range, so the stats sit alongside the mapping they describe (the
192 /// mapping-manager entry with the same range merges with this one). Every
193 /// mapping contributes a `faults` child (general fault accounting); only
194 /// soft-large-page mappings (the primary mapper's writable THP ranges on
195 /// Windows) additionally contribute a `soft_large_pages` child.
196 fn inspect(&self, req: inspect::Request<'_>) {
197 req.respond().field(
198 "mappings",
199 inspect::adhoc(|req| {
200 let mut resp = req.respond();
201 let mappings = self.inner.mappings.read();
202 for (range, props) in mappings.iter() {
203 let range = MemoryRange::new(*range.start()..*range.end() + 1);
204 resp.field(
205 &range.to_string(),
206 inspect::adhoc(|req| {
207 let mut resp = req.respond();
208 resp.field("faults", &props.stats);
209 if let Some(sl) = &props.soft_lp {
210 resp.field("soft_large_pages", sl);
211 }
212 }),
213 );
214 }
215 }),
216 );
217 }
218}
219
220#[derive(Debug)]
221struct MapperInner {
222 mapping: SparseMapping,
223 /// Waiters for lazy mapping requests. `None` after the mapper task exits.
224 waiters: Mutex<Option<Vec<MapWaiter>>>,
225 /// Index of active mappings recorded as they are established, keyed by GPA.
226 /// Written by the mapper task on map/unmap and read by the page-fault and
227 /// fault-resolution paths. Replaces a static snapshot of the RAM layout, so
228 /// hot-added ranges populate it like any other mapping.
229 mappings: RwLock<RangeMap<u64, MappingProps>>,
230 /// Whether this mapper receives mappings eagerly (pushed by the
231 /// mapping manager) or lazily (on demand via page faults).
232 /// Set by the mapping manager task after replay is complete.
233 ///
234 /// `Relaxed` ordering is sufficient: this flag is only read by the
235 /// page-fault handler to decide between eager-fail and lazy-request
236 /// paths. A stale `false` (lazy) is harmless — the lazy path
237 /// succeeds because the mapping is already established. The flag
238 /// is eventually updated after `SetEager` is processed.
239 eager: AtomicBool,
240 /// Whether this is the **primary** mapper — the one the partition and the
241 /// loader run against. Soft large pages (Windows) are only worthwhile here,
242 /// since this is the mapping whose host backing drives the guest's SLAT.
243 /// Secondary mappers use plain read-write 4 KB pages. See [`MapperRole`].
244 primary: bool,
245 /// Whether the partition delivers guest-memory-access faults to the VMM.
246 /// Soft large pages map a window read-only and rely on the resulting write
247 /// fault being resolved to raise it, so without fault resolution they would
248 /// wedge on the first guest write; the primary mapper only enables them when
249 /// this is set.
250 supports_memory_fault_resolution: bool,
251 req_send: mesh::Sender<MappingRequest>,
252 /// Maintains a weak reference to avoid a reference cycle with the partition.
253 host_access: OnceLock<HostAccess>,
254}
255
256#[derive(Clone)]
257struct HostAccess(std::sync::Weak<dyn virt::PartitionHostAccess>);
258
259impl std::fmt::Debug for HostAccess {
260 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
261 f.write_str("HostAccess")
262 }
263}
264
265/// A pending lazy mapping request.
266#[derive(Debug)]
267struct MapWaiter {
268 range: MemoryRange,
269 writable: bool,
270 done: mesh::OneshotSender<bool>,
271}
272
273impl MapWaiter {
274 /// Check whether the established mapping satisfies this waiter.
275 /// Returns `Some(true)` if fully satisfied, `Some(false)` if the
276 /// mapping doesn't meet requirements (e.g., read-only when write
277 /// needed), or `None` if the waiter still has remaining range.
278 fn complete(&mut self, range: MemoryRange, writable: Option<bool>) -> Option<bool> {
279 if range.contains_addr(self.range.start()) {
280 if writable.is_none() || (self.writable && writable == Some(false)) {
281 return Some(false);
282 }
283 let new_start = self.range.end().min(range.end());
284 let remaining = MemoryRange::new(new_start..self.range.end());
285 if remaining.is_empty() {
286 return Some(true);
287 }
288 tracing::debug!(%remaining, "waiting for more");
289 self.range = remaining;
290 }
291 None
292 }
293}
294
295struct MapperTask {
296 inner: Arc<MapperInner>,
297}
298
299impl MapperTask {
300 async fn run(mut self, mut req_recv: mesh::Receiver<MapperRequest>) {
301 while let Ok(req) = req_recv.recv().await {
302 match req {
303 MapperRequest::Unmap(rpc) => rpc.handle_sync(|range| {
304 tracing::debug!(%range, "invalidate received");
305 self.inner
306 .mapping
307 .unmap(range.start() as usize, range.len() as usize)
308 .expect("invalidate request should be valid");
309 self.inner.remove_mapping(range);
310 }),
311 MapperRequest::MapEager(rpc) => {
312 rpc.handle_failable_sync(|params| {
313 tracing::debug!(range = %params.range, "eager mapping received");
314 self.map(params)
315 });
316 }
317 MapperRequest::MapLazy(params) => {
318 tracing::debug!(range = %params.range, "lazy mapping received");
319 let (range, writable) = (params.range, params.writable);
320 match self.map(params) {
321 Ok(()) => self.wake_waiters(range, Some(writable)),
322 Err(e) => {
323 tracing::error!(
324 error = &e as &dyn std::error::Error,
325 %range,
326 "failed to map file for range"
327 );
328 self.wake_waiters(range, None);
329 }
330 }
331 }
332 MapperRequest::NoMapping(range) => {
333 // Wake up waiters. They'll see a failure when they try
334 // to access the VA.
335 tracing::debug!(%range, "no mapping received for range");
336 self.wake_waiters(range, None);
337 }
338 MapperRequest::SetEager(rpc) => rpc.handle_sync(|()| {
339 tracing::debug!("mapper upgraded to eager");
340 self.inner.eager.store(true, Ordering::Relaxed);
341 }),
342 }
343 }
344 // Don't allow more waiters.
345 *self.inner.waiters.lock() = None;
346 // Invalidate everything.
347 let _ = self.inner.mapping.unmap(0, self.inner.mapping.len());
348 }
349
350 /// Establishes a mapping in the VA space, dispatching on how it is backed.
351 fn map(&self, params: MappingParams) -> Result<(), MappingError> {
352 // Soft large pages apply only to writable THP-eligible RAM on the primary
353 // mapper (Windows); `SoftLp::new` returns `None` otherwise. See the
354 // `soft_lp` module. They also depend on the partition delivering write
355 // faults to raise deferred-protect windows, so don't even build one when
356 // the partition can't resolve faults.
357 let soft_lp = if self.inner.supports_memory_fault_resolution {
358 SoftLp::new(
359 params.range,
360 ¶ms.policy,
361 params.writable,
362 self.inner.primary,
363 )
364 } else {
365 None
366 };
367
368 // Deferred protect (map read-only, raise on the first write fault) drives
369 // the lazy soft-LP path; prefetched ranges are populated read-write
370 // eagerly at build time instead.
371 let deferred_protect = soft_lp.as_ref().is_some_and(SoftLp::deferred_protect);
372
373 let private = match ¶ms.backing {
374 MappingBacking::File {
375 mappable,
376 file_offset,
377 } => {
378 self.map_file(¶ms, mappable, *file_offset, deferred_protect)?;
379 false
380 }
381 MappingBacking::Private => {
382 self.map_private(¶ms, deferred_protect)?;
383 true
384 }
385 };
386 self.inner.record_mapping(
387 params.range,
388 MappingProps {
389 private,
390 stats: FaultStats::default(),
391 soft_lp,
392 },
393 );
394 Ok(())
395 }
396
397 /// Maps a file-backed region into the VA space, applying NUMA policy where
398 /// supported.
399 fn map_file(
400 &self,
401 params: &MappingParams,
402 mappable: &super::mappable::Mappable,
403 file_offset: u64,
404 deferred_protect: bool,
405 ) -> Result<(), MappingError> {
406 let &MappingParams {
407 range,
408 backing: _,
409 writable,
410 mapping_type: _,
411 policy:
412 MemoryPolicy {
413 numa_node,
414 transparent_hugepages,
415 prefetch: _,
416 },
417 } = params;
418 // A deferred-protect range is mapped read-write (so the view has write
419 // access and its pages can be raised back to read-write on the first
420 // write fault) and then immediately protected down to read-only just
421 // below. Mapping the view read-only up front instead would create a view
422 // whose pages cannot be raised to read-write later (`VirtualProtect`
423 // fails with ERROR_INVALID_PARAMETER). Deferred protect implies
424 // `writable`.
425 #[cfg(windows)]
426 let (protect, access) = (
427 if writable {
428 PAGE_READWRITE
429 } else {
430 PAGE_READONLY
431 },
432 if writable {
433 SECTION_MAP_READ | SECTION_MAP_WRITE
434 } else {
435 SECTION_MAP_READ
436 },
437 );
438 // `deferred_protect` is only consulted on Windows below; keep it live on
439 // other targets so the shared parameter doesn't warn.
440 let _ = deferred_protect;
441 let map_result = cfg_select! {
442 windows => {
443 self.inner.mapping.map_view_of_file_access(
444 range.start() as usize,
445 range.len() as usize,
446 mappable,
447 file_offset,
448 protect,
449 access,
450 numa_node,
451 )
452 }
453 _ => {
454 self.inner.mapping.map_file(
455 range.start() as usize,
456 range.len() as usize,
457 mappable,
458 file_offset,
459 writable,
460 )
461 }
462 };
463
464 if let Err(e) = map_result {
465 return Err(MappingError::new(range, e));
466 }
467
468 // Deferred protect: lower the freshly-mapped writable view to read-only
469 // so the first write faults; `page_fault`/`resolve` then raise the
470 // touched 2 MB window back to read-write.
471 #[cfg(windows)]
472 if deferred_protect {
473 if let Err(e) = self.inner.mapping.protect(
474 range.start() as usize,
475 range.len() as usize,
476 PAGE_READONLY,
477 ) {
478 return Err(MappingError::new(range, e));
479 }
480 }
481
482 // Mark shared (file-backed) RAM as THP-eligible. This is advisory:
483 // on Linux the kernel honors it for shmem/tmpfs (memfd) mappings
484 // according to `/sys/kernel/mm/transparent_hugepage/shmem_enabled`.
485 // The kernel may accept the advice without allocating huge pages;
486 // advice failures are logged but do not fail the mapping.
487 #[cfg(target_os = "linux")]
488 if transparent_hugepages {
489 if let Err(e) = self
490 .inner
491 .mapping
492 .madvise_hugepage(range.start() as usize, range.len() as usize)
493 {
494 tracing::warn!(
495 error = &e as &dyn std::error::Error,
496 %range,
497 "failed to mark shared RAM as THP eligible"
498 );
499 }
500 }
501 #[cfg(not(target_os = "linux"))]
502 let _ = transparent_hugepages;
503
504 cfg_select! {
505 target_os = "linux" => {
506 if let Some(node) = numa_node {
507 if let Err(e) = self.inner.mapping.mbind_at(
508 range.start() as usize,
509 range.len() as usize,
510 node,
511 ) {
512 tracing::error!(
513 error = &e as &dyn std::error::Error,
514 %range,
515 node,
516 "NUMA binding failed, using default placement"
517 );
518 }
519 }
520 }
521 windows => {
522 // NUMA handled by the map_view_of_file_access call above.
523 let _ = numa_node;
524 }
525 _ => {
526 assert!(numa_node.is_none(), "NUMA not supported on this platform; should have been rejected at build time");
527 }
528 }
529
530 Ok(())
531 }
532
533 /// Commits private anonymous memory for a range into the VA space.
534 ///
535 /// This replaces the reserved placeholder at `range` with committed
536 /// anonymous pages, optionally bound to a host NUMA node and marked
537 /// eligible for Transparent Huge Pages.
538 fn map_private(
539 &self,
540 params: &MappingParams,
541 deferred_protect: bool,
542 ) -> Result<(), MappingError> {
543 let &MappingParams {
544 range,
545 backing: _,
546 writable: _,
547 mapping_type: _,
548 policy:
549 MemoryPolicy {
550 numa_node,
551 transparent_hugepages,
552 prefetch: _,
553 },
554 } = params;
555 let offset = range.start() as usize;
556 let len = range.len() as usize;
557
558 // On Windows, deferred-protect private RAM commits read-only so the first
559 // write faults and a full 2 MB window can be raised to read-write and
560 // materialized at once, giving the loader (and guest) large pages.
561 // Elsewhere this flag is ignored.
562 if let Err(e) = self.inner.alloc(offset, len, numa_node, deferred_protect) {
563 return Err(MappingError::new(range, e));
564 }
565
566 // Name the range so it's identifiable in /proc/{pid}/smaps.
567 self.inner
568 .mapping
569 .set_name(offset, len, "guest-ram-private");
570
571 #[cfg(target_os = "linux")]
572 if transparent_hugepages {
573 if let Err(e) = self.inner.mapping.madvise_hugepage(offset, len) {
574 tracing::warn!(
575 error = &e as &dyn std::error::Error,
576 %range,
577 "failed to mark private RAM as THP eligible"
578 );
579 }
580 }
581 #[cfg(not(target_os = "linux"))]
582 let _ = transparent_hugepages;
583
584 Ok(())
585 }
586
587 fn wake_waiters(&mut self, range: MemoryRange, writable: Option<bool>) {
588 let mut waiters = self.inner.waiters.lock();
589 let waiters = waiters.as_mut().unwrap();
590
591 let mut i = 0;
592 while i < waiters.len() {
593 if let Some(success) = waiters[i].complete(range, writable) {
594 waiters.swap_remove(i).done.send(success);
595 } else {
596 i += 1;
597 }
598 }
599 }
600}
601
602#[derive(Debug, Error)]
603pub enum VaMapperError {
604 #[error("failed to communicate with the memory manager")]
605 MemoryManagerGone(#[source] RpcError),
606 #[error("failed to register mapper")]
607 Registration(#[source] RemoteError),
608 #[error("failed to reserve address space")]
609 Reserve(#[source] std::io::Error),
610}
611
612/// Error returned when a lazy mapping request cannot be fulfilled.
613#[derive(Debug, Error)]
614#[error("no mapping for {0}")]
615pub struct NoMapping(MemoryRange);
616
617impl MapperInner {
618 /// Records an established mapping in the index, replacing any stale entry
619 /// for the same range.
620 fn record_mapping(&self, range: MemoryRange, props: MappingProps) {
621 if range.is_empty() {
622 return;
623 }
624 let mut mappings = self.mappings.write();
625 mappings.remove_range(range.start()..=range.end() - 1);
626 let inserted = mappings.insert(range.start()..=range.end() - 1, props);
627 assert!(
628 inserted,
629 "mapping index range should be clear after removal"
630 );
631 }
632
633 /// Removes a mapping from the index.
634 fn remove_mapping(&self, range: MemoryRange) {
635 if range.is_empty() {
636 return;
637 }
638 self.mappings
639 .write()
640 .remove_range(range.start()..=range.end() - 1);
641 }
642
643 /// Request that the mapping manager send mappings for the given range.
644 ///
645 /// Registers a waiter, sends `SendMappings` (fire-and-forget), and
646 /// awaits the waiter oneshot. The mapping manager will send `MapLazy`
647 /// or `NoMapping` messages to the mapper task, which wakes the waiter.
648 async fn request_mapping(
649 &self,
650 id: MapperId,
651 range: MemoryRange,
652 writable: bool,
653 ) -> Result<(), NoMapping> {
654 let (send, recv) = mesh::oneshot();
655 self.waiters
656 .lock()
657 .as_mut()
658 .ok_or(NoMapping(range))?
659 .push(MapWaiter {
660 range,
661 writable,
662 done: send,
663 });
664
665 tracing::debug!(%range, "waiting for mappings");
666 self.req_send.send(MappingRequest::SendMappings(id, range));
667 match recv.await {
668 Ok(true) => Ok(()),
669 Ok(false) | Err(_) => Err(NoMapping(range)),
670 }
671 }
672
673 /// Commits private anonymous memory for a range, optionally bound to a
674 /// specific host NUMA node.
675 ///
676 /// This replaces the placeholder at the given offset with committed
677 /// anonymous memory.
678 ///
679 /// When `deferred_protect` is set (Windows soft large pages), the memory is
680 /// committed read-only instead of read-write, so the first write faults and
681 /// [`VaMapper`] can upgrade a full 2 MB window to read-write at once. This
682 /// has no effect on other platforms.
683 ///
684 /// Caution: on Linux, if NUMA binding fails, the allocation itself has
685 /// still succeeded — the returned error does not imply the memory is
686 /// unmapped.
687 fn alloc(
688 &self,
689 offset: usize,
690 len: usize,
691 numa_node: Option<u32>,
692 deferred_protect: bool,
693 ) -> Result<(), std::io::Error> {
694 cfg_select! {
695 windows => {
696 // Deferred protect (soft large pages): commit read-only so the
697 // first write faults and the 2 MB window can be raised +
698 // materialized as a large page; otherwise commit read-write.
699 let protect = if deferred_protect {
700 PAGE_READONLY
701 } else {
702 PAGE_READWRITE
703 };
704 self.mapping.virtual_alloc(offset, len, protect, numa_node)
705 }
706 target_os = "linux" => {
707 let _ = deferred_protect;
708 self.mapping.alloc(offset, len)?;
709 if let Some(node) = numa_node {
710 self.mapping.mbind_at(offset, len, node)?;
711 }
712 Ok(())
713 }
714 _ => {
715 let _ = deferred_protect;
716 assert!(numa_node.is_none(), "NUMA not supported on this platform; should have been rejected at build time");
717 self.mapping.alloc(offset, len)
718 }
719 }
720 }
721}
722
723impl VaMapper {
724 pub(crate) async fn new(
725 req_send: mesh::Sender<MappingRequest>,
726 len: u64,
727 remote_process: Option<RemoteProcess>,
728 minimum_alignment: Option<usize>,
729 eager: bool,
730 role: MapperRole,
731 ) -> Result<Self, VaMapperError> {
732 // Soft large pages apply only to the primary mapper, and only when the
733 // partition resolves faults; `supports_memory_fault_resolution` rides on
734 // the `Primary` variant.
735 let (primary, supports_memory_fault_resolution) = match role {
736 MapperRole::Primary {
737 supports_memory_fault_resolution,
738 } => (true, supports_memory_fault_resolution),
739 MapperRole::Secondary => (false, false),
740 };
741 let mapping = match &remote_process {
742 None => SparseMapping::new_with_minimum_alignment(
743 len as usize,
744 minimum_alignment.unwrap_or(1),
745 ),
746 Some(process) => match process {
747 #[cfg(not(windows))]
748 _ => unreachable!(),
749 #[cfg(windows)]
750 process => SparseMapping::new_remote(
751 process.as_handle().try_clone_to_owned().unwrap().into(),
752 None,
753 len as usize,
754 minimum_alignment.unwrap_or(1),
755 ),
756 },
757 }
758 .map_err(VaMapperError::Reserve)?;
759
760 // Name the VA reservation so it's identifiable in /proc/{pid}/smaps.
761 mapping.set_name(0, mapping.len(), "guest-memory");
762
763 let (send, req_recv) = mesh::channel();
764
765 let inner = Arc::new(MapperInner {
766 mapping,
767 waiters: Mutex::new(Some(Vec::new())),
768 mappings: RwLock::new(RangeMap::new()),
769 eager: AtomicBool::new(eager),
770 primary,
771 supports_memory_fault_resolution,
772 req_send,
773 host_access: OnceLock::new(),
774 });
775
776 // Spawn the mapper thread *before* the AddMapper RPC. The manager
777 // replays existing mappings to eager mappers during AddMapper, so
778 // the mapper thread must be running to respond to those RPCs.
779 //
780 // FUTURE: use a task once we resolve the block_ons in the
781 // GuestMemoryAccess implementation.
782 let thread = std::thread::Builder::new()
783 .name("mapper".to_owned())
784 .spawn({
785 let runner = MapperTask {
786 inner: inner.clone(),
787 };
788 || block_on(runner.run(req_recv))
789 })
790 .unwrap();
791
792 let id = match inner
793 .req_send
794 .call(
795 MappingRequest::AddMapper,
796 super::manager::AddMapperParams { send, eager },
797 )
798 .await
799 {
800 Ok(Ok(id)) => id,
801 Ok(Err(e)) => {
802 // Drop inner to shut down the mapper thread (closes req_recv).
803 drop(inner);
804 let _ = thread.join();
805 return Err(VaMapperError::Registration(e));
806 }
807 Err(e) => {
808 drop(inner);
809 let _ = thread.join();
810 return Err(VaMapperError::MemoryManagerGone(e));
811 }
812 };
813
814 Ok(VaMapper {
815 inner,
816 id,
817 process: remote_process,
818 _thread: thread,
819 })
820 }
821
822 /// Returns the base pointer of the VA reservation.
823 pub fn as_ptr(&self) -> *mut u8 {
824 self.inner.mapping.as_ptr().cast()
825 }
826
827 /// Installs the callback used to recover eager-mapper faults caused by
828 /// missing host permission.
829 pub(crate) fn install_host_access(&self, host_access: Arc<dyn virt::PartitionHostAccess>) {
830 assert!(
831 self.inner
832 .host_access
833 .set(HostAccess(Arc::downgrade(&host_access)))
834 .is_ok(),
835 "host access is already installed"
836 );
837 }
838
839 /// Returns the length of the VA reservation in bytes.
840 pub fn len(&self) -> usize {
841 self.inner.mapping.len()
842 }
843
844 /// Returns true if this mapper receives mappings eagerly.
845 pub fn is_eager(&self) -> bool {
846 self.inner.eager.load(Ordering::Relaxed)
847 }
848
849 /// Returns the mapper's ID, used internally for upgrade requests.
850 pub(crate) fn mapper_id(&self) -> MapperId {
851 self.id
852 }
853
854 /// Returns the remote process, if this mapper maps into a remote process.
855 pub fn process(&self) -> Option<&RemoteProcess> {
856 self.process.as_ref()
857 }
858}
859
860/// SAFETY: the underlying VA mapping is guaranteed to be valid for the lifetime
861/// of this object.
862unsafe impl GuestMemoryAccess for VaMapper {
863 fn mapping(&self) -> Option<NonNull<u8>> {
864 // No one should be using this as a GuestMemoryAccess for remote
865 // mappings, but it's convenient to have the same type for both local
866 // and remote mappings for the sake of simplicity in
867 // `PartitionRegionMapper`.
868 assert!(self.inner.mapping.is_local());
869
870 NonNull::new(self.inner.mapping.as_ptr().cast())
871 }
872
873 fn max_address(&self) -> u64 {
874 self.inner.mapping.len() as u64
875 }
876
877 fn page_fault(
878 &self,
879 address: u64,
880 len: usize,
881 write: bool,
882 bitmap_failure: bool,
883 ) -> PageFaultAction {
884 assert!(!bitmap_failure, "bitmaps are not used");
885
886 // Soft large pages (Windows): THP-eligible ranges on the primary mapper
887 // are committed/mapped read-only, so the first *write* traps here (reads
888 // are served by the zero page and don't fault). This is the loader's
889 // path; `SoftLp::on_host_write` raises (and prefetches) the covering
890 // 2 MB window, then the write is retried. The mapping-index read lock is
891 // held across the raise; it only blocks a concurrent *writer* (a
892 // structural map/unmap), which is rare.
893 #[cfg(windows)]
894 if write {
895 let mappings = self.inner.mappings.read();
896 if let Some(&(start, end, ref props)) = mappings.get_entry(&address) {
897 if let Some(sl) = &props.soft_lp {
898 return match sl.on_host_write(&self.inner.mapping, address, start, end) {
899 Ok(()) => PageFaultAction::Retry,
900 Err(err) => PageFaultAction::Fail(PageFaultError::new(
901 GuestMemoryErrorKind::Other,
902 err,
903 )),
904 };
905 }
906 }
907 }
908
909 if self.inner.eager.load(Ordering::Relaxed) {
910 // The guest-memory VA is already mapped for an eager mapper. For
911 // isolated guests, a fault can instead mean that the hypervisor has
912 // not granted userspace access to a shared page.
913 if let Some(host_access) = self
914 .inner
915 .host_access
916 .get()
917 .and_then(|host_access| host_access.0.upgrade())
918 {
919 let start = address & !(hvdef::HV_PAGE_SIZE - 1);
920 let end = address
921 .checked_add(len as u64)
922 .and_then(|end| end.checked_add(hvdef::HV_PAGE_SIZE - 1))
923 .map(|end| end & !(hvdef::HV_PAGE_SIZE - 1));
924 let Some(end) = end else {
925 return PageFaultAction::Fail(PageFaultError::new(
926 GuestMemoryErrorKind::OutOfRange,
927 std::io::Error::other("host-access range overflow"),
928 ));
929 };
930 match host_access.acquire_host_access(start, end - start, write) {
931 Ok(()) => return PageFaultAction::Retry,
932 Err(err) => {
933 return PageFaultAction::Fail(PageFaultError::new(
934 GuestMemoryErrorKind::Other,
935 std::io::Error::other(err),
936 ));
937 }
938 }
939 }
940
941 // Eager mapper: file-backed mappings are established proactively.
942 // If we get a page fault, the mapping was never set up or was
943 // torn down.
944 return PageFaultAction::Fail(PageFaultError::new(
945 GuestMemoryErrorKind::OutOfRange,
946 UnexpectedPageFault,
947 ));
948 }
949
950 // Lazy mapper: request the mapping on demand from the mapping manager.
951 let range = MemoryRange::bounding(address..address + len as u64);
952 if let Err(err) = block_on(self.inner.request_mapping(self.id, range, write)) {
953 return PageFaultAction::Fail(PageFaultError::new(
954 GuestMemoryErrorKind::OutOfRange,
955 err,
956 ));
957 }
958 PageFaultAction::Retry
959 }
960
961 fn sharing(&self) -> Option<GuestMemorySharing> {
962 // Private anonymous memory is committed on fault in the local process
963 // and cannot be shared to a remote DMA process, so disable DMA sharing
964 // whenever any recorded mapping is private. Derived from the mapping
965 // index rather than a static flag so it tracks the actual backings.
966 if self.inner.mappings.read().iter().any(|(_, p)| p.private) {
967 return None;
968 }
969 Some(GuestMemorySharing::new(DmaRegionProvider {
970 req_send: self.inner.req_send.clone(),
971 }))
972 }
973}
974
975impl ResolveMemoryFault for VaMapper {
976 fn resolve(
977 &self,
978 fault: MemoryRange,
979 write: bool,
980 ) -> Result<MemoryRange, GuestMemoryBackingError> {
981 if fault.end() > self.inner.mapping.len() as u64 {
982 return Err(GuestMemoryBackingError::new(
983 GuestMemoryErrorKind::OutOfRange,
984 fault.start(),
985 UnexpectedPageFault,
986 ));
987 }
988
989 // Hold the mapping-index read lock across the fault resolution. This only
990 // blocks a concurrent *writer* (a structural map/unmap), which is rare;
991 // other faulting VPs are readers and proceed in parallel. `end` is the
992 // inclusive last address of the mapping.
993 let mappings = self.inner.mappings.read();
994 let Some(&(start, end, ref props)) = mappings.get_entry(&fault.start()) else {
995 return Err(GuestMemoryBackingError::new(
996 GuestMemoryErrorKind::OutOfRange,
997 fault.start(),
998 UnexpectedPageFault,
999 ));
1000 };
1001 // The trait contract requires the resolved range to stay within the
1002 // single uniform RAM region that covers `fault.start()`. Today the only
1003 // caller faults one page at a time, so a fault never spans two mappings;
1004 // guard against a future caller passing a wider range that starts in this
1005 // mapping but extends past its end (`end` is the inclusive last address).
1006 if fault.end() > end + 1 {
1007 return Err(GuestMemoryBackingError::new(
1008 GuestMemoryErrorKind::OutOfRange,
1009 fault.start(),
1010 UnexpectedPageFault,
1011 ));
1012 }
1013 props.stats.guest_faults.increment();
1014
1015 // Soft large pages (Windows) raise the covering 2 MB window on the first
1016 // write and may resolve to the whole window; every other mapping (and
1017 // every non-Windows host) resolves to the single faulting page.
1018 match &props.soft_lp {
1019 Some(sl) => sl
1020 .resolve(&self.inner.mapping, fault, write, start, end)
1021 .map_err(|err| {
1022 GuestMemoryBackingError::new(GuestMemoryErrorKind::Other, fault.start(), err)
1023 }),
1024 None => Ok(fault),
1025 }
1026 }
1027}
1028
1029#[cfg(test)]
1030mod tests {
1031 use sparse_mmap::SparseMapping;
1032
1033 /// Tests that private RAM pages can be allocated, written to, and read from.
1034 #[test]
1035 fn test_private_ram_alloc_write_read() {
1036 let page_size = SparseMapping::page_size();
1037 let mapping = SparseMapping::new(4 * page_size).unwrap();
1038
1039 // Allocate (commit) the first two pages.
1040 mapping.alloc(0, 2 * page_size).unwrap();
1041
1042 // Write and read through SparseMapping methods.
1043 let data = [0xABu8; 128];
1044 mapping.write_at(0, &data).unwrap();
1045
1046 let mut buf = [0u8; 128];
1047 mapping.read_at(0, &mut buf).unwrap();
1048 assert_eq!(buf, data);
1049
1050 // Verify zeros at an untouched offset within committed range.
1051 let mut zero_buf = [0xFFu8; 64];
1052 mapping.read_at(page_size, &mut zero_buf).unwrap();
1053 assert!(
1054 zero_buf.iter().all(|&b| b == 0),
1055 "untouched committed memory should be zeros"
1056 );
1057 }
1058
1059 /// Tests that commit is idempotent (committing already-committed pages is
1060 /// a no-op).
1061 #[test]
1062 fn test_private_ram_commit_idempotent() {
1063 let page_size = SparseMapping::page_size();
1064 let mapping = SparseMapping::new(4 * page_size).unwrap();
1065
1066 // Alloc then commit the same range again.
1067 mapping.alloc(0, 2 * page_size).unwrap();
1068 mapping.commit(0, 2 * page_size).unwrap();
1069 mapping.commit(0, page_size).unwrap();
1070
1071 // Write and read should work.
1072 let pattern = vec![0xEFu8; 64];
1073 mapping.write_at(0, &pattern).unwrap();
1074 let mut buf = vec![0u8; 64];
1075 mapping.read_at(0, &mut buf).unwrap();
1076 assert_eq!(buf, pattern);
1077 }
1078}