Skip to main content

openhcl_boot/
main.rs

1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4//! Bare-metal OpenHCL boot loader that prepares VTL2 before Linux starts.
5//!
6//! This measured payload validates its imported regions and host parameters,
7//! establishes the VTL2 address space and processor state, constructs Linux
8//! boot parameters and a device tree, and initializes the sidecar kernel when
9//! configured. It then transfers control to the OpenHCL Linux kernel.
10
11// See build.rs.
12#![cfg_attr(minimal_rt, no_std, no_main)]
13// UNSAFETY: Interacting with low level hardware and bootloader primitives.
14#![expect(unsafe_code)]
15// Allow the allocator api when compiling with `RUSTFLAGS="--cfg nightly"`. This
16// is used for some miri tests for testing the bump allocator.
17//
18// Do not use a normal feature, as that shows errors with rust-analyzer since
19// most people are using stable and enable all features. We could remove this
20// once the allocator_api feature is stable.
21#![cfg_attr(nightly, feature(allocator_api))]
22
23mod arch;
24mod boot_logger;
25mod cmdline;
26mod dt;
27mod host_params;
28mod hypercall;
29mod memory;
30mod rt;
31mod sidecar;
32mod single_threaded;
33
34use crate::arch::setup_vtl2_memory;
35use crate::arch::setup_vtl2_vp;
36#[cfg(target_arch = "x86_64")]
37use crate::arch::tdx::get_tdx_tsc_reftime;
38use crate::arch::verify_imported_regions_hash;
39use crate::boot_logger::boot_logger_memory_init;
40use crate::boot_logger::boot_logger_runtime_init;
41use crate::hypercall::hvcall;
42use crate::memory::AddressSpaceManager;
43use crate::single_threaded::OffStackRef;
44use crate::single_threaded::off_stack;
45use arrayvec::ArrayString;
46use arrayvec::ArrayVec;
47use cmdline::BootCommandLineOptions;
48use core::fmt::Write;
49use dt::BootTimes;
50use dt::write_dt;
51use host_fdt_parser::ComInfo;
52use host_params::COMMAND_LINE_SIZE;
53use host_params::PartitionInfo;
54use host_params::shim_params::IsolationType;
55use host_params::shim_params::ShimParams;
56use hvdef::Vtl;
57use loader_defs::linux::SETUP_DTB;
58use loader_defs::linux::setup_data;
59use loader_defs::shim::ShimParamsRaw;
60use memory_range::RangeWalkResult;
61use memory_range::walk_ranges;
62use minimal_rt::enlightened_panic::enable_enlightened_panic;
63use sidecar::SidecarConfig;
64use sidecar_defs::SidecarOutput;
65use sidecar_defs::SidecarParams;
66use zerocopy::FromBytes;
67use zerocopy::FromZeros;
68use zerocopy::Immutable;
69use zerocopy::IntoBytes;
70use zerocopy::KnownLayout;
71
72#[derive(Debug)]
73struct CommandLineTooLong;
74
75impl From<core::fmt::Error> for CommandLineTooLong {
76    fn from(_: core::fmt::Error) -> Self {
77        Self
78    }
79}
80
81struct BuildKernelCommandLineParams<'a> {
82    params: &'a ShimParams,
83    cmdline: &'a mut ArrayString<COMMAND_LINE_SIZE>,
84    partition_info: &'a PartitionInfo,
85    can_trust_host: bool,
86    is_confidential_debug: bool,
87    sidecar: Option<&'a SidecarConfig<'a>>,
88    vtl2_pool_supported: bool,
89}
90
91/// Read and setup the underhill kernel command line into the specified buffer.
92fn build_kernel_command_line(
93    fn_params: BuildKernelCommandLineParams<'_>,
94) -> Result<(), CommandLineTooLong> {
95    let BuildKernelCommandLineParams {
96        params,
97        cmdline,
98        partition_info,
99        can_trust_host,
100        is_confidential_debug,
101        sidecar,
102        vtl2_pool_supported,
103    } = fn_params;
104
105    // For reference:
106    // https://www.kernel.org/doc/html/v5.15/admin-guide/kernel-parameters.html
107    const KERNEL_PARAMETERS: &[&str] = &[
108        // If a console is specified, then write everything to it.
109        "loglevel=8",
110        // Use a fixed 128KB log buffer by default.
111        "log_buf_len=128K",
112        // Enable time output on console for ohcldiag-dev.
113        "printk.time=1",
114        // Enable facility and level output on console for ohcldiag-dev.
115        "console_msg_format=syslog",
116        // Set uio parameter to configure vmbus ring buffer behavior.
117        "uio_hv_generic.no_mask=1",
118        // RELIABILITY: Dump anonymous pages and ELF headers only. Skip over
119        // huge pages and the shared pages.
120        "coredump_filter=0x33",
121        // PERF: No processor frequency governing.
122        "cpufreq.off=1",
123        // PERF: Disable the CPU idle time management entirely. It does not
124        // prevent the idle loop from running on idle CPUs, but it prevents
125        // the CPU idle time governors and drivers from being invoked.
126        "cpuidle.off=1",
127        // PERF: No perf checks for crypto algorithms to boot faster.
128        // Would have to evaluate the perf wins on the crypto manager vs
129        // delaying the boot up.
130        "cryptomgr.notests",
131        // PERF: Idle threads use HLT on x64 if there is no work.
132        // Believed to be a compromise between waking up the processor
133        // and the power consumption.
134        "idle=halt",
135        // WORKAROUND: Avoid init calls that assume presence of CMOS (Simple
136        // Boot Flag) or allocate the real-mode trampoline for APs.
137        "initcall_blacklist=init_real_mode,sbf_init",
138        // CONFIG-STATIC, PERF: Static loops-per-jiffy value to save time on boot.
139        "lpj=3000000",
140        // PERF: No broken timer check to boot faster.
141        "no_timer_check",
142        // CONFIG-STATIC, PERF: Using xsave makes VTL transitions being
143        // much slower. The xsave state is shared between VTLs, and we don't
144        // context switch it in the kernel when leaving/entering VTL2.
145        // Removing this will lead to corrupting register state and the
146        // undefined behaviour.
147        "noxsave",
148        // RELIABILITY: Panic on MCEs and faults in the kernel.
149        "oops=panic",
150        // RELIABILITY: Don't panic on kernel warnings.
151        "panic_on_warn=0",
152        // PERF, RELIABILITY: Don't print detailed information about the failing
153        // processes (memory maps, threads).
154        "panic_print=0",
155        // RELIABILITY: Reboot immediately on panic, no timeout.
156        "panic=-1",
157        // RELIABILITY: Don't print processor context information on a fatal
158        // signal. Our crash dump collection infrastructure seems reliable, and
159        // this information doesn't seem useful without a dump anyways.
160        // Additionally it may push important logs off the end of the kmsg
161        // page logged by the host.
162        //"print_fatal_signals=0",
163        // RELIABILITY: Unlimited logging to /dev/kmsg from userspace.
164        "printk.devkmsg=on",
165        // RELIABILITY: Reboot using a triple fault as the fastest method.
166        // That is also the method used for compatibility with earlier versions
167        // of the Microsoft HCL.
168        "reboot=t",
169        // CONFIG-STATIC: Type of the root file system.
170        "rootfstype=tmpfs",
171        // PERF: Deactivate kcompactd kernel thread, otherwise it will queue a
172        // scheduler timer periodically, which introduces jitters for VTL0.
173        "sysctl.vm.compaction_proactiveness=0",
174        // PERF: No TSC stability check when booting up to boot faster,
175        // also no validation during runtime.
176        "tsc=reliable",
177        // RELIABILITY: Panic on receiving an NMI.
178        "unknown_nmi_panic=1",
179        // Use vfio for MANA devices.
180        "vfio_pci.ids=1414:00ba",
181        // WORKAROUND: Enable no-IOMMU mode. This mode provides no device isolation,
182        // and no DMA translation.
183        "vfio.enable_unsafe_noiommu_mode=1",
184        // Specify the init path.
185        "rdinit=/underhill-init",
186        // Default to user-mode NVMe driver.
187        "OPENHCL_NVME_VFIO=1",
188        // The next three items reduce the memory overhead of the storvsc driver.
189        // Since it is only used for DVD, performance is not critical.
190        "hv_storvsc.storvsc_vcpus_per_sub_channel=2048",
191        // Fix number of hardware queues at 2.
192        "hv_storvsc.storvsc_max_hw_queues=2",
193        // Reduce the ring buffer size to 32K.
194        "hv_storvsc.storvsc_ringbuffer_size=0x8000",
195        // Disable eager mimalloc commit to prevent core dumps from being overly large
196        "MIMALLOC_ARENA_EAGER_COMMIT=0",
197        // Disable acpi runtime support. Unused in underhill, but some support
198        // is compiled in for the kernel (ie TDX mailbox protocol).
199        "acpi=off",
200    ];
201
202    const X86_KERNEL_PARAMETERS: &[&str] = &[
203        // Disable all attempts to use an IOMMU, including swiotlb.
204        "iommu=off",
205        // Don't probe for a PCI bus. PCI devices currently come from VPCI. When
206        // this changes, we will explicitly enumerate a PCI bus via devicetree.
207        "pci=off",
208    ];
209
210    const AARCH64_KERNEL_PARAMETERS: &[&str] = &[];
211
212    for p in KERNEL_PARAMETERS {
213        write!(cmdline, "{p} ")?;
214    }
215
216    let arch_parameters = if cfg!(target_arch = "x86_64") {
217        X86_KERNEL_PARAMETERS
218    } else {
219        AARCH64_KERNEL_PARAMETERS
220    };
221    for p in arch_parameters {
222        write!(cmdline, "{p} ")?;
223    }
224
225    const HARDWARE_ISOLATED_KERNEL_PARAMETERS: &[&str] = &[
226        // Even with iommu=off, the SWIOTLB is still allocated on AARCH64
227        // (iommu=off ignored entirely), and CVMs (memory encryption forces it
228        // on). Set it to a single area in 8MB. The first parameter controls the
229        // area size in slabs (2KB per slab), the second controls the number of
230        // areas (default is # of CPUs).
231        //
232        // This is set to 8MB on hardware isolated VMs since there are some
233        // scenarios, such as provisioning over DVD, which require a larger size
234        // since the buffer is being used.
235        "swiotlb=4096,1",
236    ];
237
238    const NON_HARDWARE_ISOLATED_KERNEL_PARAMETERS: &[&str] = &[
239        // Even with iommu=off, the SWIOTLB is still allocated on AARCH64
240        // (iommu=off ignored entirely). Set it to the minimum, saving ~63 MiB.
241        // The first parameter controls the area size, the second controls the
242        // number of areas (default is # of CPUs). Set them both to the minimum.
243        "swiotlb=1,1",
244    ];
245
246    if params.isolation_type.is_hardware_isolated() {
247        for p in HARDWARE_ISOLATED_KERNEL_PARAMETERS {
248            write!(cmdline, "{p} ")?;
249        }
250    } else {
251        for p in NON_HARDWARE_ISOLATED_KERNEL_PARAMETERS {
252            write!(cmdline, "{p} ")?;
253        }
254    }
255
256    // Enable the com3 console by default if it's available and we're not
257    // isolated, or if we are isolated but also have debugging enabled.
258    //
259    // Otherwise, set the console to ttynull so the kernel does not default to
260    // com1. This is overridden by any user customizations in the static or
261    // dynamic command line, as this console argument provided by the bootloader
262    // comes first.
263    write!(cmdline, "console=")?;
264    match (&partition_info.com3_serial, can_trust_host) {
265        (ComInfo::Ns16550 { current_speed, .. }, true) => {
266            write!(cmdline, "ttyS2,{current_speed} ")?
267        }
268        (ComInfo::Pl011 { current_speed, .. }, true) => {
269            write!(cmdline, "ttyAMA0,{current_speed} ")?
270        }
271        _ => write!(cmdline, "ttynull ")?,
272    }
273
274    if params.isolation_type != IsolationType::None {
275        write!(
276            cmdline,
277            "{}=1 ",
278            underhill_confidentiality::OPENHCL_CONFIDENTIAL_ENV_VAR_NAME
279        )?;
280    }
281
282    if is_confidential_debug {
283        write!(
284            cmdline,
285            "{}=1 ",
286            underhill_confidentiality::OPENHCL_CONFIDENTIAL_DEBUG_ENV_VAR_NAME
287        )?;
288    }
289
290    // Generate the NVMe keep alive command line which should look something
291    // like: OPENHCL_NVME_KEEP_ALIVE=disabled,host,privatepool
292    // TODO: Move from command line to device tree when stabilized.
293    write!(cmdline, "OPENHCL_NVME_KEEP_ALIVE=")?;
294
295    if partition_info.boot_options.disable_nvme_keep_alive {
296        write!(cmdline, "disabled,")?;
297    }
298
299    if partition_info.nvme_keepalive {
300        write!(cmdline, "host,")?;
301    } else {
302        write!(cmdline, "nohost,")?;
303    }
304
305    if vtl2_pool_supported {
306        write!(cmdline, "privatepool ")?;
307    } else {
308        write!(cmdline, "noprivatepool ")?;
309    }
310
311    if let Some(sidecar) = sidecar {
312        write!(cmdline, "{} ", sidecar.kernel_command_line())?;
313    }
314
315    if !cmdline.contains("hv_vmbus.message_connection_id") {
316        // HACK: Set the vmbus connection id via kernel commandline if we haven't
317        // gotten one from elsewhere.
318        //
319        // This code will be removed when the kernel supports setting connection id
320        // via device tree.
321        write!(
322            cmdline,
323            "hv_vmbus.message_connection_id=0x{:x} ",
324            partition_info.vmbus_vtl2.connection_id
325        )?;
326    }
327
328    // Prepend the computed parameters to the original command line.
329    cmdline.write_str(&partition_info.cmdline)?;
330
331    Ok(())
332}
333
334// The Linux kernel requires that the FDT fit within a single 256KB mapping, as
335// that is the maximum size the kernel can use during its early boot processes.
336// We also want our FDT to be as large as possible to support as many vCPUs as
337// possible. We set it to 256KB, but it must also be page-aligned, as leaving it
338// unaligned runs the possibility of it taking up 1 too many pages, resulting in
339// a 260KB mapping, which will fail.
340const FDT_SIZE: usize = 256 * 1024;
341
342#[repr(C, align(4096))]
343#[derive(FromBytes, IntoBytes, Immutable, KnownLayout)]
344struct Fdt {
345    header: setup_data,
346    data: [u8; FDT_SIZE - size_of::<setup_data>()],
347}
348
349/// Raw shim parameters are provided via a relative offset from the base of
350/// where the shim is loaded. Return a ShimParams structure based on the raw
351/// offset based RawShimParams.
352fn shim_parameters(shim_params_raw_offset: isize) -> ShimParams {
353    unsafe extern "C" {
354        static __ehdr_start: u8;
355    }
356
357    let shim_base = core::ptr::addr_of!(__ehdr_start) as usize;
358
359    // SAFETY: The host is required to relocate everything by the same bias, so
360    //         the shim parameters should be at the build time specified offset
361    //         from the base address of the image.
362    let raw_shim_params = unsafe {
363        &*(shim_base.wrapping_add_signed(shim_params_raw_offset) as *const ShimParamsRaw)
364    };
365
366    ShimParams::new(shim_base as u64, raw_shim_params)
367}
368
369#[cfg_attr(not(target_arch = "x86_64"), expect(dead_code))]
370mod x86_boot {
371    use crate::PageAlign;
372    use crate::memory::AddressSpaceManager;
373    use crate::single_threaded::OffStackRef;
374    use crate::single_threaded::off_stack;
375    use crate::zeroed;
376    use core::mem::size_of;
377    use core::ops::Range;
378    use core::ptr;
379    use loader_defs::linux::E820_RAM;
380    use loader_defs::linux::E820_RESERVED;
381    use loader_defs::linux::SETUP_E820_EXT;
382    use loader_defs::linux::boot_params;
383    use loader_defs::linux::e820entry;
384    use loader_defs::linux::setup_data;
385    use loader_defs::shim::MemoryVtlType;
386    use memory_range::MemoryRange;
387    use zerocopy::FromZeros;
388    use zerocopy::Immutable;
389    use zerocopy::KnownLayout;
390
391    #[repr(C)]
392    #[derive(FromZeros, Immutable, KnownLayout)]
393    pub struct E820Ext {
394        pub header: setup_data,
395        pub entries: [e820entry; 512],
396    }
397
398    fn add_e820_entry(
399        entry: Option<&mut e820entry>,
400        range: MemoryRange,
401        typ: u32,
402    ) -> Result<(), BuildE820MapError> {
403        *entry.ok_or(BuildE820MapError::OutOfE820Entries)? = e820entry {
404            addr: range.start().into(),
405            size: range.len().into(),
406            typ: typ.into(),
407        };
408        Ok(())
409    }
410
411    #[derive(Debug)]
412    pub enum BuildE820MapError {
413        /// Out of e820 entries.
414        OutOfE820Entries,
415    }
416
417    /// Build the e820 map for the kernel representing usable VTL2 ram.
418    pub fn build_e820_map(
419        boot_params: &mut boot_params,
420        ext: &mut E820Ext,
421        address_space: &AddressSpaceManager,
422    ) -> Result<bool, BuildE820MapError> {
423        boot_params.e820_entries = 0;
424        let mut entries = boot_params
425            .e820_map
426            .iter_mut()
427            .chain(ext.entries.iter_mut());
428
429        let mut n = 0;
430        for (range, typ) in address_space.vtl2_ranges() {
431            match typ {
432                MemoryVtlType::VTL2_RAM => {
433                    add_e820_entry(entries.next(), range, E820_RAM)?;
434                    n += 1;
435                }
436                MemoryVtlType::VTL2_CONFIG
437                | MemoryVtlType::VTL2_SIDECAR_IMAGE
438                | MemoryVtlType::VTL2_SIDECAR_NODE
439                | MemoryVtlType::VTL2_RESERVED
440                | MemoryVtlType::VTL2_GPA_POOL
441                | MemoryVtlType::VTL2_TDX_PAGE_TABLES
442                | MemoryVtlType::VTL2_BOOTSHIM_LOG_BUFFER
443                | MemoryVtlType::VTL2_PERSISTED_STATE_HEADER
444                | MemoryVtlType::VTL2_PERSISTED_STATE_PROTOBUF => {
445                    add_e820_entry(entries.next(), range, E820_RESERVED)?;
446                    n += 1;
447                }
448
449                _ => {
450                    panic!("unexpected vtl2 ram type {typ:?} for range {range:#?}");
451                }
452            }
453        }
454
455        let base = n.min(boot_params.e820_map.len());
456        boot_params.e820_entries = base as u8;
457
458        if base < n {
459            ext.header.len = ((n - base) * size_of::<e820entry>()) as u32;
460            Ok(true)
461        } else {
462            Ok(false)
463        }
464    }
465
466    pub fn build_boot_params(
467        address_space: &AddressSpaceManager,
468        initrd: Range<u64>,
469        cmdline: &str,
470        setup_data_head: *const setup_data,
471        setup_data_tail: &mut &mut setup_data,
472    ) -> OffStackRef<'static, PageAlign<boot_params>> {
473        let mut boot_params_storage = off_stack!(PageAlign<boot_params>, zeroed());
474        let boot_params = &mut boot_params_storage.0;
475        boot_params.hdr.type_of_loader = 0xff; // Unknown loader type
476
477        // HACK: A kernel change just in the Underhill kernel tree has a workaround
478        // to disable probe_roms and reserve_bios_regions when X86_SUBARCH_LGUEST
479        // (1) is set by the bootloader. This stops the kernel from reading VTL0
480        // memory during kernel boot, which can have catastrophic consequences
481        // during a servicing operation when VTL0 has written values to memory, or
482        // unaccepted page accesses in an isolated partition.
483        //
484        // This is only intended as a stopgap until a suitable upstreamable kernel
485        // patch is made.
486        boot_params.hdr.hardware_subarch = 1.into();
487
488        boot_params.hdr.ramdisk_image = (initrd.start as u32).into();
489        boot_params.ext_ramdisk_image = (initrd.start >> 32) as u32;
490        let initrd_len = initrd.end - initrd.start;
491        boot_params.hdr.ramdisk_size = (initrd_len as u32).into();
492        boot_params.ext_ramdisk_size = (initrd_len >> 32) as u32;
493
494        let e820_ext = OffStackRef::leak(off_stack!(E820Ext, zeroed()));
495
496        let used_ext = build_e820_map(boot_params, e820_ext, address_space)
497            .expect("building e820 map must succeed");
498
499        if used_ext {
500            e820_ext.header.ty = SETUP_E820_EXT;
501            setup_data_tail.next = ptr::from_ref(&e820_ext.header) as u64;
502            *setup_data_tail = &mut e820_ext.header;
503        }
504
505        let cmd_line_addr = cmdline.as_ptr() as u64;
506        boot_params.hdr.cmd_line_ptr = (cmd_line_addr as u32).into();
507        boot_params.ext_cmd_line_ptr = (cmd_line_addr >> 32) as u32;
508
509        boot_params.hdr.setup_data = (setup_data_head as u64).into();
510
511        boot_params_storage
512    }
513}
514
515/// Build the cc_blob containing the location of different parameters associated with SEV.
516#[cfg(target_arch = "x86_64")]
517fn build_cc_blob_sev_info(
518    cc_blob: &mut loader_defs::linux::cc_blob_sev_info,
519    shim_params: &ShimParams,
520) {
521    // TODO SNP: Currently only the first CPUID page is passed through.
522    // Consider changing this.
523    cc_blob.magic = loader_defs::linux::CC_BLOB_SEV_INFO_MAGIC;
524    cc_blob.version = 0;
525    cc_blob._reserved = 0;
526    cc_blob.secrets_phys = shim_params.secrets_start();
527    cc_blob.secrets_len = hvdef::HV_PAGE_SIZE as u32;
528    cc_blob._rsvd1 = 0;
529    cc_blob.cpuid_phys = shim_params.cpuid_start();
530    cc_blob.cpuid_len = hvdef::HV_PAGE_SIZE as u32;
531    cc_blob._rsvd2 = 0;
532}
533
534#[repr(C, align(4096))]
535#[derive(FromZeros, Immutable, KnownLayout)]
536struct PageAlign<T>(T);
537
538const fn zeroed<T: FromZeros>() -> T {
539    // SAFETY: `T` implements `FromZeros`, so this is a safe initialization of `T`.
540    unsafe { core::mem::MaybeUninit::<T>::zeroed().assume_init() }
541}
542
543fn get_ref_time(isolation: IsolationType) -> Option<u64> {
544    match isolation {
545        #[cfg(target_arch = "x86_64")]
546        IsolationType::Tdx => get_tdx_tsc_reftime(),
547        #[cfg(target_arch = "x86_64")]
548        IsolationType::Snp => None,
549        _ => Some(minimal_rt::reftime::reference_time()),
550    }
551}
552
553/// Dump diagnostics when initrd CRC32 does not match the build-time value.
554///
555/// Contents:
556///
557/// - `base` / `size` / `expected` (build-time) / `got` (first read)
558///   CRCs.
559/// - `got2` — a second CRC re-computed immediately from the same virtual
560///   range. If `got2 != got`, initrd memory is not being read consistently
561///   (typical symptom of stale/mismatched cache lines after the SNP shared
562///   -> private transition, rather than data being wrong in memory).
563/// - `head` / `tail` — first and last 16 bytes as hex, to fingerprint what
564///   is actually in memory.
565/// - `eighths` — CRC32 of eight roughly-equal slices of the initrd. This
566///   is a coarse "which region diverges" locator that is cheap to compute
567///   and stable across boots, so it can be compared with a known-good
568///   build without needing a per-page dump (which would blow the log
569///   budget for real-sized initrds).
570//
571// SNP TODO: temporary diagnostic; remove once the SNP initrd CRC mismatch
572// is root-caused.
573fn build_initrd_crc_diagnostic(p: &ShimParams, first_computed_crc: u32) -> ArrayString<384> {
574    let initrd_bytes = p.initrd();
575
576    // A second read from the same VA. If this differs from the first read,
577    // the initrd memory is not being read consistently, which typically
578    // indicates stale/mismatched cache lines rather than actual data
579    // corruption.
580    let second_computed_crc = crc32fast::hash(initrd_bytes);
581
582    // First 16 and last 16 bytes, as fixed-size arrays so we can rely on
583    // Debug's `{:02x?}` slice formatting.
584    let mut head = [0u8; 16];
585    let head_len = head.len().min(initrd_bytes.len());
586    head[..head_len].copy_from_slice(&initrd_bytes[..head_len]);
587
588    let mut tail = [0u8; 16];
589    let tail_len = tail.len().min(initrd_bytes.len());
590    if tail_len > 0 {
591        let start = initrd_bytes.len() - tail_len;
592        tail[..tail_len].copy_from_slice(&initrd_bytes[start..]);
593    }
594
595    // Split the initrd into (up to) 8 roughly-equal slices and CRC each.
596    // Bytes past the aligned slices go into the last chunk.
597    let mut eighths = [0u32; 8];
598    let n = initrd_bytes.len();
599    if n > 0 {
600        let step = n.div_ceil(8);
601        for (i, e) in eighths.iter_mut().enumerate() {
602            let start = i * step;
603            if start >= n {
604                break;
605            }
606            let end = ((i + 1) * step).min(n);
607            *e = crc32fast::hash(&initrd_bytes[start..end]);
608        }
609    }
610
611    let mut buf = ArrayString::<384>::new();
612    let _ = write!(
613        &mut buf,
614        "initrd crc mismatch: iso={:?} base={:#x} size={:#x} \
615         exp={:#x} got={:#x} got2={:#x} head={:02x?} tail={:02x?} \
616         eighths=[{:#x},{:#x},{:#x},{:#x},{:#x},{:#x},{:#x},{:#x}]",
617        p.isolation_type,
618        p.initrd_base,
619        p.initrd_size,
620        p.initrd_crc,
621        first_computed_crc,
622        second_computed_crc,
623        &head[..head_len],
624        &tail[..tail_len],
625        eighths[0],
626        eighths[1],
627        eighths[2],
628        eighths[3],
629        eighths[4],
630        eighths[5],
631        eighths[6],
632        eighths[7],
633    );
634    buf
635}
636
637fn shim_main(shim_params_raw_offset: isize) -> ! {
638    let p = shim_parameters(shim_params_raw_offset);
639    if p.isolation_type == IsolationType::None {
640        enable_enlightened_panic();
641    }
642
643    #[cfg(feature = "cvm_boot_log")]
644    arch::initialize_serial_io(&p);
645
646    // Enable the in-memory log.
647    boot_logger_memory_init(p.log_buffer);
648
649    // Enable global log crate.
650    log::set_logger(&boot_logger::BOOT_LOGGER).unwrap();
651    // TODO: allow overriding filter at runtime
652    log::set_max_level(log::LevelFilter::Info);
653
654    let boot_reftime = get_ref_time(p.isolation_type);
655
656    // The support code for the fast hypercalls does not set
657    // the Guest ID if it is not set yet as opposed to the slow
658    // hypercall code path where that is done automatically.
659    // Thus the fast hypercalls will fail as the the Guest ID has
660    // to be set first hence initialize hypercall support
661    // explicitly.
662    if !p.isolation_type.is_hardware_isolated() {
663        hvcall().initialize();
664    }
665
666    let mut static_options = BootCommandLineOptions::new();
667    if let Some(cmdline) = p.command_line().command_line() {
668        static_options.parse(cmdline);
669    }
670
671    let static_confidential_debug = static_options.confidential_debug;
672    let can_trust_host = p.isolation_type == IsolationType::None || static_confidential_debug;
673
674    let mut dt_storage = off_stack!(PartitionInfo, PartitionInfo::new());
675    let address_space = OffStackRef::leak(off_stack!(
676        AddressSpaceManager,
677        AddressSpaceManager::new_const()
678    ));
679    let partition_info = match PartitionInfo::read_from_dt(
680        &p,
681        &mut dt_storage,
682        address_space,
683        static_options,
684        can_trust_host,
685    ) {
686        Ok(val) => val,
687        Err(e) => panic!("unable to read device tree params {:?}", e),
688    };
689
690    // Enable logging ASAP. This is fine even when isolated, as we don't have
691    // any access to secrets in the boot shim.
692    boot_logger_runtime_init(p.isolation_type, partition_info.com3_serial.clone());
693    log::info!("openhcl_boot: logging enabled");
694    log::info!("serial configuration: {:#x?}", partition_info.com3_serial);
695
696    // Confidential debug will show up in boot_options only if included in the
697    // static command line, or if can_trust_host is true (so the dynamic command
698    // line has been parsed).
699    let is_confidential_debug =
700        static_confidential_debug || partition_info.boot_options.confidential_debug;
701
702    // Fill out the non-devicetree derived parts of PartitionInfo.
703    if !p.isolation_type.is_hardware_isolated()
704        && hvcall().vtl() == Vtl::Vtl2
705        && hvdef::HvRegisterVsmCapabilities::from(
706            hvcall()
707                .get_register(hvdef::HvAllArchRegisterName::VsmCapabilities.into())
708                .expect("failed to query vsm capabilities")
709                .as_u64(),
710        )
711        .vtl0_alias_map_available()
712    {
713        // If the vtl0 alias map was not provided in the devicetree, attempt to
714        // derive it from the architectural physical address bits.
715        //
716        // The value in the ID_AA64MMFR0_EL1 register used to determine the
717        // physical address bits can only represent multiples of 4. As a result,
718        // the Surface Pro X (and systems with similar CPUs) cannot properly
719        // report their address width of 39 bits. This causes the calculated
720        // alias map to be incorrect, which results in panics when trying to
721        // read memory and getting invalid data.
722        if partition_info.vtl0_alias_map.is_none() {
723            partition_info.vtl0_alias_map =
724                Some(1 << (arch::physical_address_bits(p.isolation_type) - 1));
725        }
726    } else {
727        // Ignore any devicetree-provided alias map if the conditions above
728        // aren't met.
729        partition_info.vtl0_alias_map = None;
730    }
731
732    // Rebind partition_info as no longer mutable.
733    let partition_info: &PartitionInfo = partition_info;
734
735    if partition_info.cpus.is_empty() {
736        panic!("no cpus");
737    }
738
739    validate_vp_hw_ids(partition_info);
740
741    setup_vtl2_memory(&p, partition_info, address_space);
742    setup_vtl2_vp(partition_info);
743
744    verify_imported_regions_hash(&p);
745
746    let mut sidecar_params = off_stack!(PageAlign<SidecarParams>, zeroed());
747    let mut sidecar_output = off_stack!(PageAlign<SidecarOutput>, zeroed());
748    let sidecar = sidecar::start_sidecar(
749        &p,
750        partition_info,
751        address_space,
752        &mut sidecar_params.0,
753        &mut sidecar_output.0,
754    );
755
756    // Rebind address_space as no longer mutable.
757    let address_space: &AddressSpaceManager = address_space;
758
759    let mut cmdline = off_stack!(ArrayString<COMMAND_LINE_SIZE>, ArrayString::new_const());
760    build_kernel_command_line(BuildKernelCommandLineParams {
761        params: &p,
762        cmdline: &mut cmdline,
763        partition_info,
764        can_trust_host,
765        is_confidential_debug,
766        sidecar: sidecar.as_ref(),
767        vtl2_pool_supported: address_space.has_vtl2_pool(),
768    })
769    .unwrap();
770
771    let mut fdt = off_stack!(Fdt, zeroed());
772    fdt.header.len = fdt.data.len() as u32;
773    fdt.header.ty = SETUP_DTB;
774
775    #[cfg(target_arch = "x86_64")]
776    let mut setup_data_tail = &mut fdt.header;
777    #[cfg(target_arch = "x86_64")]
778    let setup_data_head = core::ptr::from_ref(setup_data_tail);
779
780    #[cfg(target_arch = "x86_64")]
781    if p.isolation_type == IsolationType::Snp {
782        let cc_blob = OffStackRef::leak(off_stack!(loader_defs::linux::cc_blob_sev_info, zeroed()));
783        build_cc_blob_sev_info(cc_blob, &p);
784
785        let cc_data = OffStackRef::leak(off_stack!(loader_defs::linux::cc_setup_data, zeroed()));
786        cc_data.header.len = size_of::<loader_defs::linux::cc_setup_data>() as u32;
787        cc_data.header.ty = loader_defs::linux::SETUP_CC_BLOB;
788        cc_data.cc_blob_address = core::ptr::from_ref(&*cc_blob) as u32;
789
790        // Chain in the setup data.
791        setup_data_tail.next = core::ptr::from_ref(&*cc_data) as u64;
792        setup_data_tail = &mut cc_data.header;
793    }
794
795    let initrd = p.initrd_base..p.initrd_base + p.initrd_size;
796
797    // Validate the initrd crc matches what was put at file generation time.
798    let computed_crc = crc32fast::hash(p.initrd());
799    if computed_crc != p.initrd_crc && is_confidential_debug {
800        let diag = build_initrd_crc_diagnostic(&p, computed_crc);
801        log::error!("{}", diag.as_str());
802        panic!("{}", diag.as_str());
803    }
804    assert_eq!(
805        computed_crc, p.initrd_crc,
806        "computed initrd crc does not match build time calculated crc"
807    );
808
809    #[cfg(target_arch = "x86_64")]
810    let boot_params = x86_boot::build_boot_params(
811        address_space,
812        initrd.clone(),
813        &cmdline,
814        setup_data_head,
815        &mut setup_data_tail,
816    );
817
818    // Compute the ending boot time. This has to be before writing to device
819    // tree, so this is as late as we can do it.
820
821    let boot_times = boot_reftime.map(|start| BootTimes {
822        start,
823        end: get_ref_time(p.isolation_type).unwrap_or(0),
824    });
825
826    // Validate that no imported regions that are pending are not part of vtl2
827    // ram.
828    for (range, result) in walk_ranges(
829        partition_info.vtl2_ram.iter().map(|r| (r.range, ())),
830        p.imported_regions(),
831    ) {
832        match result {
833            RangeWalkResult::Neither | RangeWalkResult::Left(_) | RangeWalkResult::Both(_, _) => {}
834            RangeWalkResult::Right(accepted) => {
835                // Ranges that are not a part of VTL2 ram must have been
836                // preaccepted, as usermode expect that to be the case.
837                assert!(
838                    accepted,
839                    "range {:#x?} not in vtl2 ram was not preaccepted at launch",
840                    range
841                );
842            }
843        }
844    }
845
846    write_dt(
847        &mut fdt.data,
848        partition_info,
849        address_space,
850        p.imported_regions().map(|r| {
851            // Discard if the range was previously pending - the bootloader has
852            // accepted all pending ranges.
853            //
854            // NOTE: No VTL0 memory today is marked as pending. The check above
855            // validates that, and this code may need to change if this becomes
856            // no longer true.
857            r.0
858        }),
859        initrd,
860        &cmdline,
861        sidecar.as_ref(),
862        boot_times,
863        p.isolation_type,
864    )
865    .unwrap();
866
867    rt::verify_stack_cookie();
868
869    log::info!("uninitializing hypercalls");
870    #[cfg(not(feature = "cvm_boot_log"))]
871    log::info!("about to jump to kernel");
872
873    hvcall().uninitialize();
874
875    #[cfg(feature = "cvm_boot_log")]
876    {
877        log::info!("uninitializing serial io");
878        log::info!("about to jump to kernel");
879        arch::uninitialize_serial_io(&p);
880    }
881
882    cfg_if::cfg_if! {
883        if #[cfg(target_arch = "x86_64")] {
884            // SAFETY: the parameter blob is trusted.
885            let kernel_entry: extern "C" fn(u64, &loader_defs::linux::boot_params) -> ! =
886                unsafe { core::mem::transmute(p.kernel_entry_address) };
887            kernel_entry(0, &boot_params.0)
888        } else if #[cfg(target_arch = "aarch64")] {
889            // SAFETY: the parameter blob is trusted.
890            let kernel_entry: extern "C" fn(fdt_data: *const u8, mbz0: u64, mbz1: u64, mbz2: u64) -> ! =
891                unsafe { core::mem::transmute(p.kernel_entry_address) };
892            // Disable MMU for kernel boot without EFI, as required by the boot protocol.
893            // Flush (and invalidate) the caches, as that is required for disabling MMU.
894            // SAFETY: Just changing a bit in the register and then jumping to the kernel.
895            unsafe {
896                core::arch::asm!(
897                    "
898                    mrs     {0}, sctlr_el1
899                    bic     {0}, {0}, #0x1
900                    msr     sctlr_el1, {0}
901                    tlbi    vmalle1
902                    dsb     sy
903                    isb     sy",
904                    lateout(reg) _,
905                );
906            }
907            kernel_entry(fdt.data.as_ptr(), 0, 0, 0)
908        } else {
909            panic!("unsupported arch")
910        }
911    }
912}
913
914/// Ensure that mshv VP indexes for the CPUs listed in the partition info
915/// correspond to the N in the cpu@N devicetree node name. OpenVMM assumes that
916/// this will be the case.
917fn validate_vp_hw_ids(partition_info: &PartitionInfo) {
918    use host_params::MAX_CPU_COUNT;
919    use hypercall::HwId;
920
921    if partition_info.isolation.is_hardware_isolated() {
922        // TODO TDX SNP: we don't have a GHCB/GHCI page set up to communicate
923        // with the hypervisor here, so we can't easily perform the check. Since
924        // there is no security impact to this check, we can skip it for now; if
925        // the VM fails to boot, then this is due to a host contract violation.
926        //
927        // For TDX, we could use ENUM TOPOLOGY to validate that the TD VCPU
928        // indexes correspond to the APIC IDs in the right order. I am not
929        // certain if there are places where we depend on this mapping today.
930        return;
931    }
932
933    if hvcall().vtl() != Vtl::Vtl2 {
934        // If we're not using guest VSM, then the guest won't communicate
935        // directly with the hypervisor, so we can choose the VP indexes
936        // ourselves.
937        return;
938    }
939
940    // Ensure the host and hypervisor agree on VP index ordering.
941
942    let mut hw_ids = off_stack!(ArrayVec<HwId, MAX_CPU_COUNT>, ArrayVec::new_const());
943    hw_ids.clear();
944    hw_ids.extend(partition_info.cpus.iter().map(|c| c.reg as _));
945    let mut vp_indexes = off_stack!(ArrayVec<u32, MAX_CPU_COUNT>, ArrayVec::new_const());
946    vp_indexes.clear();
947    if let Err(err) = hvcall().get_vp_index_from_hw_id(&hw_ids, &mut vp_indexes) {
948        panic!(
949            "failed to get VP index for hardware ID {:#x}: {}",
950            hw_ids[vp_indexes.len().min(hw_ids.len() - 1)],
951            err
952        );
953    }
954    if let Some((i, &vp_index)) = vp_indexes
955        .iter()
956        .enumerate()
957        .find(|&(i, vp_index)| i as u32 != *vp_index)
958    {
959        panic!(
960            "CPU hardware ID {:#x} does not correspond to VP index {}",
961            hw_ids[i], vp_index
962        );
963    }
964}
965
966// See build.rs. See `mod rt` for the actual bootstrap code required to invoke
967// shim_main.
968#[cfg(not(minimal_rt))]
969fn main() {
970    unimplemented!("build with MINIMAL_RT_BUILD to produce a working boot loader");
971}
972
973#[cfg(test)]
974mod test {
975    use super::x86_boot::E820Ext;
976    use super::x86_boot::build_e820_map;
977    use crate::cmdline::BootCommandLineOptions;
978    use crate::dt::write_dt;
979    use crate::host_params::MAX_CPU_COUNT;
980    use crate::host_params::PartitionInfo;
981    use crate::host_params::shim_params::IsolationType;
982    use crate::memory::AddressSpaceManager;
983    use crate::memory::AddressSpaceManagerBuilder;
984    use arrayvec::ArrayString;
985    use arrayvec::ArrayVec;
986    use core::ops::Range;
987    use host_fdt_parser::ComInfo;
988    use host_fdt_parser::CpuEntry;
989    use host_fdt_parser::MemoryEntry;
990    use host_fdt_parser::VmbusInfo;
991    use igvm_defs::MemoryMapEntryType;
992    use loader_defs::linux::E820_RAM;
993    use loader_defs::linux::E820_RESERVED;
994    use loader_defs::linux::boot_params;
995    use loader_defs::linux::e820entry;
996    use memory_range::MemoryRange;
997    use memory_range::subtract_ranges;
998    use sidecar_defs::PerCpuState;
999    use zerocopy::FromZeros;
1000
1001    const HIGH_MMIO_GAP_END: u64 = 0x1000000000; //  64 GiB
1002    const VMBUS_MMIO_GAP_SIZE: u64 = 0x10000000; // 256 MiB
1003    const HIGH_MMIO_GAP_START: u64 = HIGH_MMIO_GAP_END - VMBUS_MMIO_GAP_SIZE;
1004
1005    /// Create partition info with given cpu count enabled and sequential
1006    /// apic_ids.
1007    fn new_partition_info(cpu_count: usize) -> PartitionInfo {
1008        let mut cpus: ArrayVec<CpuEntry, MAX_CPU_COUNT> = ArrayVec::new();
1009
1010        for id in 0..(cpu_count as u64) {
1011            cpus.push(CpuEntry { reg: id, vnode: 0 });
1012        }
1013
1014        let mut mmio = ArrayVec::new();
1015        mmio.push(
1016            MemoryRange::try_new(HIGH_MMIO_GAP_START..HIGH_MMIO_GAP_END).expect("valid range"),
1017        );
1018
1019        PartitionInfo {
1020            vtl2_ram: ArrayVec::new(),
1021            partition_ram: ArrayVec::new(),
1022            isolation: IsolationType::None,
1023            bsp_reg: cpus[0].reg as u32,
1024            cpus,
1025            sidecar_cpu_overrides: PerCpuState {
1026                per_cpu_state_specified: false,
1027                sidecar_starts_cpu: [true; sidecar_defs::NUM_CPUS_SUPPORTED_FOR_PER_CPU_STATE],
1028            },
1029            cmdline: ArrayString::new(),
1030            vmbus_vtl2: VmbusInfo {
1031                mmio,
1032                connection_id: 0,
1033            },
1034            vmbus_vtl0: VmbusInfo {
1035                mmio: ArrayVec::new(),
1036                connection_id: 0,
1037            },
1038            com3_serial: ComInfo::None,
1039            gic: None,
1040            pmu_gsiv: None,
1041            memory_allocation_mode: host_fdt_parser::MemoryAllocationMode::Host,
1042            entropy: None,
1043            vtl0_alias_map: None,
1044            nvme_keepalive: false,
1045            boot_options: BootCommandLineOptions::new(),
1046        }
1047    }
1048
1049    // ensure we can boot with a _lot_ of vcpus
1050    #[test]
1051    #[cfg_attr(
1052        target_arch = "aarch64",
1053        ignore = "TODO: investigate why this doesn't always work on ARM"
1054    )]
1055    fn fdt_cpu_scaling() {
1056        const MAX_CPUS: usize = 2048;
1057
1058        let mut buf = [0; 0x40000];
1059        write_dt(
1060            &mut buf,
1061            &new_partition_info(MAX_CPUS),
1062            &AddressSpaceManager::new_const(),
1063            [],
1064            0..0,
1065            &ArrayString::from("test").unwrap_or_default(),
1066            None,
1067            None,
1068            IsolationType::None,
1069        )
1070        .unwrap();
1071    }
1072
1073    // Must match the DeviceTree blob generated with the standard tooling
1074    // to ensure being compliant to the standards (or, at least, compatibility
1075    // with a widely used implementation).
1076    // For details on regenerating the test content, see `fdt_dtc_decompile`
1077    // below.
1078    #[test]
1079    #[ignore = "TODO: temporarily broken"]
1080    fn fdt_dtc_check_content() {
1081        const MAX_CPUS: usize = 2;
1082        const BUF_SIZE: usize = 0x1000;
1083
1084        // Rust cannot infer the type.
1085        let dtb_data_spans: [(usize, &[u8]); 2] = [
1086            (
1087                /* Span starts at offset */ 0,
1088                b"\xd0\x0d\xfe\xed\x00\x00\x10\x00\x00\x00\x04\x38\x00\x00\x00\x38\
1089                \x00\x00\x00\x28\x00\x00\x00\x11\x00\x00\x00\x10\x00\x00\x00\x00\
1090                \x00\x00\x00\x4a\x00\x00\x01\x6c\x00\x00\x00\x00\x00\x00\x00\x00\
1091                \x00\x00\x00\x00\x00\x00\x00\x00\x23\x61\x64\x64\x72\x65\x73\x73\
1092                \x2d\x63\x65\x6c\x6c\x73\x00\x23\x73\x69\x7a\x65\x2d\x63\x65\x6c\
1093                \x6c\x73\x00\x6d\x6f\x64\x65\x6c\x00\x72\x65\x67\x00\x64\x65\x76\
1094                \x69\x63\x65\x5f\x74\x79\x70\x65\x00\x73\x74\x61\x74\x75\x73\x00\
1095                \x63\x6f\x6d\x70\x61\x74\x69\x62\x6c\x65\x00\x72\x61\x6e\x67\x65\
1096                \x73",
1097            ),
1098            (
1099                /* Span starts at offset */ 0x430,
1100                b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x01\x00\x00\x00\x00\
1101                \x00\x00\x00\x03\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x02\
1102                \x00\x00\x00\x03\x00\x00\x00\x04\x00\x00\x00\x0f\x00\x00\x00\x00\
1103                \x00\x00\x00\x03\x00\x00\x00\x0f\x00\x00\x00\x1b\x6d\x73\x66\x74\
1104                \x2c\x75\x6e\x64\x65\x72\x68\x69\x6c\x6c\x00\x00\x00\x00\x00\x01\
1105                \x63\x70\x75\x73\x00\x00\x00\x00\x00\x00\x00\x03\x00\x00\x00\x04\
1106                \x00\x00\x00\x00\x00\x00\x00\x01\x00\x00\x00\x03\x00\x00\x00\x04\
1107                \x00\x00\x00\x0f\x00\x00\x00\x00\x00\x00\x00\x01\x63\x70\x75\x40\
1108                \x30\x00\x00\x00\x00\x00\x00\x03\x00\x00\x00\x04\x00\x00\x00\x25\
1109                \x63\x70\x75\x00\x00\x00\x00\x03\x00\x00\x00\x04\x00\x00\x00\x21\
1110                \x00\x00\x00\x00\x00\x00\x00\x03\x00\x00\x00\x05\x00\x00\x00\x31\
1111                \x6f\x6b\x61\x79\x00\x00\x00\x00\x00\x00\x00\x02\x00\x00\x00\x01\
1112                \x63\x70\x75\x40\x31\x00\x00\x00\x00\x00\x00\x03\x00\x00\x00\x04\
1113                \x00\x00\x00\x25\x63\x70\x75\x00\x00\x00\x00\x03\x00\x00\x00\x04\
1114                \x00\x00\x00\x21\x00\x00\x00\x01\x00\x00\x00\x03\x00\x00\x00\x05\
1115                \x00\x00\x00\x31\x6f\x6b\x61\x79\x00\x00\x00\x00\x00\x00\x00\x02\
1116                \x00\x00\x00\x02\x00\x00\x00\x01\x76\x6d\x62\x75\x73\x00\x00\x00\
1117                \x00\x00\x00\x03\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x02\
1118                \x00\x00\x00\x03\x00\x00\x00\x04\x00\x00\x00\x0f\x00\x00\x00\x01\
1119                \x00\x00\x00\x03\x00\x00\x00\x0b\x00\x00\x00\x38\x6d\x73\x66\x74\
1120                \x2c\x76\x6d\x62\x75\x73\x00\x00\x00\x00\x00\x03\x00\x00\x00\x14\
1121                \x00\x00\x00\x43\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x0f\
1122                \xf0\x00\x00\x00\x10\x00\x00\x00\x00\x00\x00\x02\x00\x00\x00\x02\
1123                \x00\x00\x00\x09",
1124            ),
1125        ];
1126
1127        let mut sample_buf = [0u8; BUF_SIZE];
1128        for (span_start, bytes) in dtb_data_spans {
1129            sample_buf[span_start..span_start + bytes.len()].copy_from_slice(bytes);
1130        }
1131
1132        let mut buf = [0u8; BUF_SIZE];
1133        write_dt(
1134            &mut buf,
1135            &new_partition_info(MAX_CPUS),
1136            &AddressSpaceManager::new_const(),
1137            [],
1138            0..0,
1139            &ArrayString::from("test").unwrap_or_default(),
1140            None,
1141            None,
1142            IsolationType::None,
1143        )
1144        .unwrap();
1145
1146        assert!(sample_buf == buf);
1147    }
1148
1149    // This test should be manually enabled when need to regenerate
1150    // the sample content above and validate spec compliance with `dtc`.
1151    // Before running the test, please install the DeviceTree compiler:
1152    // ```shell
1153    // sudo apt-get update && sudo apt-get install device-tree-compiler
1154    // ```
1155    #[test]
1156    #[ignore = "enabling the test requires installing additional software, \
1157                and developers will experience a break."]
1158    fn fdt_dtc_decompile() {
1159        const MAX_CPUS: usize = 2048;
1160
1161        let mut buf = [0; 0x40000];
1162        write_dt(
1163            &mut buf,
1164            &new_partition_info(MAX_CPUS),
1165            &AddressSpaceManager::new_const(),
1166            [],
1167            0..0,
1168            &ArrayString::from("test").unwrap_or_default(),
1169            None,
1170            None,
1171            IsolationType::None,
1172        )
1173        .unwrap();
1174
1175        let input_dtb_file_name = "openhcl_boot.dtb";
1176        let output_dts_file_name = "openhcl_boot.dts";
1177        std::fs::write(input_dtb_file_name, buf).unwrap();
1178        let success = std::process::Command::new("dtc")
1179            .args([input_dtb_file_name, "-I", "dtb", "-o", output_dts_file_name])
1180            .status()
1181            .unwrap()
1182            .success();
1183        assert!(success);
1184    }
1185
1186    fn new_address_space_manager(
1187        ram: &[MemoryRange],
1188        bootshim_used: MemoryRange,
1189        persisted_range: MemoryRange,
1190        parameter_range: MemoryRange,
1191        reclaim: Option<MemoryRange>,
1192    ) -> AddressSpaceManager {
1193        let ram = ram
1194            .iter()
1195            .cloned()
1196            .map(|range| MemoryEntry {
1197                range,
1198                mem_type: MemoryMapEntryType::VTL2_PROTECTABLE,
1199                vnode: 0,
1200            })
1201            .collect::<Vec<_>>();
1202        let mut address_space = AddressSpaceManager::new_const();
1203        AddressSpaceManagerBuilder::new(
1204            &mut address_space,
1205            &ram,
1206            bootshim_used,
1207            persisted_range,
1208            subtract_ranges([parameter_range], reclaim),
1209        )
1210        .init()
1211        .unwrap();
1212        address_space
1213    }
1214
1215    fn check_e820(boot_params: &boot_params, ext: &E820Ext, expected: &[(Range<u64>, u32)]) {
1216        let actual = boot_params.e820_map[..boot_params.e820_entries as usize]
1217            .iter()
1218            .chain(
1219                ext.entries
1220                    .iter()
1221                    .take((ext.header.len as usize) / size_of::<e820entry>()),
1222            );
1223
1224        assert_eq!(actual.clone().count(), expected.len());
1225
1226        for (actual, (expected_range, expected_type)) in actual.zip(expected.iter()) {
1227            let addr: u64 = actual.addr.into();
1228            let size: u64 = actual.size.into();
1229            let typ: u32 = actual.typ.into();
1230            assert_eq!(addr, expected_range.start);
1231            assert_eq!(size, expected_range.end - expected_range.start);
1232            assert_eq!(typ, *expected_type);
1233        }
1234    }
1235
1236    const PAGE_SIZE: u64 = 0x1000;
1237    const ONE_MB: u64 = 0x10_0000;
1238
1239    #[test]
1240    fn test_e820_basic() {
1241        // memmap with no param reclaim
1242        let mut boot_params: boot_params = FromZeros::new_zeroed();
1243        let mut ext = FromZeros::new_zeroed();
1244        let bootshim_used = MemoryRange::try_new(ONE_MB..3 * ONE_MB).unwrap();
1245        let persisted_header_end = ONE_MB + PAGE_SIZE;
1246        let persisted_end = ONE_MB + 4 * PAGE_SIZE;
1247        let persisted_state = MemoryRange::try_new(ONE_MB..persisted_end).unwrap();
1248        let parameter_range = MemoryRange::try_new(2 * ONE_MB..3 * ONE_MB).unwrap();
1249        let address_space = new_address_space_manager(
1250            &[MemoryRange::new(ONE_MB..4 * ONE_MB)],
1251            bootshim_used,
1252            persisted_state,
1253            parameter_range,
1254            None,
1255        );
1256
1257        assert!(build_e820_map(&mut boot_params, &mut ext, &address_space).is_ok());
1258
1259        check_e820(
1260            &boot_params,
1261            &ext,
1262            &[
1263                (ONE_MB..(persisted_header_end), E820_RESERVED),
1264                (persisted_header_end..persisted_end, E820_RESERVED),
1265                (persisted_end..2 * ONE_MB, E820_RAM),
1266                (2 * ONE_MB..3 * ONE_MB, E820_RESERVED),
1267                (3 * ONE_MB..4 * ONE_MB, E820_RAM),
1268            ],
1269        );
1270
1271        // memmap with reclaim
1272        let mut boot_params: boot_params = FromZeros::new_zeroed();
1273        let mut ext = FromZeros::new_zeroed();
1274        let bootshim_used = MemoryRange::try_new(ONE_MB..5 * ONE_MB).unwrap();
1275        let persisted_header_end = ONE_MB + PAGE_SIZE;
1276        let persisted_end = ONE_MB + 4 * PAGE_SIZE;
1277        let persisted_state = MemoryRange::try_new(ONE_MB..persisted_end).unwrap();
1278        let parameter_range = MemoryRange::try_new(2 * ONE_MB..5 * ONE_MB).unwrap();
1279        let reclaim = MemoryRange::try_new(3 * ONE_MB..4 * ONE_MB).unwrap();
1280        let address_space = new_address_space_manager(
1281            &[MemoryRange::new(ONE_MB..6 * ONE_MB)],
1282            bootshim_used,
1283            persisted_state,
1284            parameter_range,
1285            Some(reclaim),
1286        );
1287
1288        assert!(build_e820_map(&mut boot_params, &mut ext, &address_space).is_ok());
1289
1290        check_e820(
1291            &boot_params,
1292            &ext,
1293            &[
1294                (ONE_MB..(persisted_header_end), E820_RESERVED),
1295                (persisted_header_end..persisted_end, E820_RESERVED),
1296                (persisted_end..2 * ONE_MB, E820_RAM),
1297                (2 * ONE_MB..3 * ONE_MB, E820_RESERVED),
1298                (3 * ONE_MB..4 * ONE_MB, E820_RAM),
1299                (4 * ONE_MB..5 * ONE_MB, E820_RESERVED),
1300                (5 * ONE_MB..6 * ONE_MB, E820_RAM),
1301            ],
1302        );
1303
1304        // two mem ranges
1305        let mut boot_params: boot_params = FromZeros::new_zeroed();
1306        let mut ext = FromZeros::new_zeroed();
1307        let bootshim_used = MemoryRange::try_new(ONE_MB..5 * ONE_MB).unwrap();
1308        let persisted_header_end = ONE_MB + PAGE_SIZE;
1309        let persisted_end = ONE_MB + 4 * PAGE_SIZE;
1310        let persisted_state = MemoryRange::try_new(ONE_MB..persisted_end).unwrap();
1311        let parameter_range = MemoryRange::try_new(2 * ONE_MB..5 * ONE_MB).unwrap();
1312        let reclaim = MemoryRange::try_new(3 * ONE_MB..4 * ONE_MB).unwrap();
1313        let address_space = new_address_space_manager(
1314            &[
1315                MemoryRange::new(ONE_MB..4 * ONE_MB),
1316                MemoryRange::new(4 * ONE_MB..10 * ONE_MB),
1317            ],
1318            bootshim_used,
1319            persisted_state,
1320            parameter_range,
1321            Some(reclaim),
1322        );
1323
1324        assert!(build_e820_map(&mut boot_params, &mut ext, &address_space).is_ok());
1325
1326        check_e820(
1327            &boot_params,
1328            &ext,
1329            &[
1330                (ONE_MB..(persisted_header_end), E820_RESERVED),
1331                (persisted_header_end..persisted_end, E820_RESERVED),
1332                (persisted_end..2 * ONE_MB, E820_RAM),
1333                (2 * ONE_MB..3 * ONE_MB, E820_RESERVED),
1334                (3 * ONE_MB..4 * ONE_MB, E820_RAM),
1335                (4 * ONE_MB..5 * ONE_MB, E820_RESERVED),
1336                (5 * ONE_MB..10 * ONE_MB, E820_RAM),
1337            ],
1338        );
1339
1340        // memmap in 1 mb chunks
1341        let mut boot_params: boot_params = FromZeros::new_zeroed();
1342        let mut ext = FromZeros::new_zeroed();
1343        let bootshim_used = MemoryRange::try_new(ONE_MB..5 * ONE_MB).unwrap();
1344        let persisted_header_end = ONE_MB + PAGE_SIZE;
1345        let persisted_end = ONE_MB + 4 * PAGE_SIZE;
1346        let persisted_state = MemoryRange::try_new(ONE_MB..persisted_end).unwrap();
1347        let parameter_range = MemoryRange::try_new(2 * ONE_MB..5 * ONE_MB).unwrap();
1348        let reclaim = MemoryRange::try_new(3 * ONE_MB..4 * ONE_MB).unwrap();
1349        let address_space = new_address_space_manager(
1350            &[
1351                MemoryRange::new(ONE_MB..2 * ONE_MB),
1352                MemoryRange::new(2 * ONE_MB..3 * ONE_MB),
1353                MemoryRange::new(3 * ONE_MB..4 * ONE_MB),
1354                MemoryRange::new(4 * ONE_MB..5 * ONE_MB),
1355                MemoryRange::new(5 * ONE_MB..6 * ONE_MB),
1356                MemoryRange::new(6 * ONE_MB..7 * ONE_MB),
1357                MemoryRange::new(7 * ONE_MB..8 * ONE_MB),
1358            ],
1359            bootshim_used,
1360            persisted_state,
1361            parameter_range,
1362            Some(reclaim),
1363        );
1364
1365        assert!(build_e820_map(&mut boot_params, &mut ext, &address_space).is_ok());
1366
1367        check_e820(
1368            &boot_params,
1369            &ext,
1370            &[
1371                (ONE_MB..(persisted_header_end), E820_RESERVED),
1372                (persisted_header_end..persisted_end, E820_RESERVED),
1373                (persisted_end..2 * ONE_MB, E820_RAM),
1374                (2 * ONE_MB..3 * ONE_MB, E820_RESERVED),
1375                (3 * ONE_MB..4 * ONE_MB, E820_RAM),
1376                (4 * ONE_MB..5 * ONE_MB, E820_RESERVED),
1377                (5 * ONE_MB..8 * ONE_MB, E820_RAM),
1378            ],
1379        );
1380    }
1381
1382    // test e820 with spillover into ext
1383    #[test]
1384    fn test_e820_huge() {
1385        use crate::memory::AllocationPolicy;
1386        use crate::memory::AllocationType;
1387
1388        // Create 64 RAM ranges, then allocate 256 ranges to test spillover
1389        // boot_params.e820_map has E820_MAX_ENTRIES_ZEROPAGE (128) entries
1390        const E820_MAX_ENTRIES_ZEROPAGE: usize = 128;
1391        const RAM_RANGES: usize = 64;
1392        const TOTAL_ALLOCATIONS: usize = 256;
1393
1394        // Create 64 large RAM ranges (64MB each = 64 * 1MB pages per range)
1395        let mut ranges = Vec::new();
1396        for i in 0..RAM_RANGES {
1397            let start = (i as u64) * 64 * ONE_MB;
1398            let end = start + 64 * ONE_MB;
1399            ranges.push(MemoryRange::new(start..end));
1400        }
1401
1402        let bootshim_used = MemoryRange::try_new(0..ONE_MB * 2).unwrap();
1403        let persisted_range = MemoryRange::try_new(0..ONE_MB).unwrap();
1404        let parameter_range = MemoryRange::try_new(ONE_MB..2 * ONE_MB).unwrap();
1405
1406        let mut address_space = {
1407            let ram = ranges
1408                .iter()
1409                .cloned()
1410                .map(|range| MemoryEntry {
1411                    range,
1412                    mem_type: MemoryMapEntryType::VTL2_PROTECTABLE,
1413                    vnode: 0,
1414                })
1415                .collect::<Vec<_>>();
1416            let mut address_space = AddressSpaceManager::new_const();
1417            AddressSpaceManagerBuilder::new(
1418                &mut address_space,
1419                &ram,
1420                bootshim_used,
1421                persisted_range,
1422                core::iter::once(parameter_range),
1423            )
1424            .init()
1425            .unwrap();
1426            address_space
1427        };
1428
1429        for i in 0..TOTAL_ALLOCATIONS {
1430            // Intersperse sidecar node allocations with gpa pool allocations,
1431            // as otherwise the address space manager will collapse adjacent
1432            // ranges of the same type.
1433            let _allocated = address_space
1434                .allocate(
1435                    None,
1436                    ONE_MB,
1437                    if i % 2 == 0 {
1438                        AllocationType::GpaPool
1439                    } else {
1440                        AllocationType::SidecarNode
1441                    },
1442                    AllocationPolicy::LowMemory,
1443                )
1444                .expect("should be able to allocate sidecar node");
1445        }
1446
1447        let mut boot_params: boot_params = FromZeros::new_zeroed();
1448        let mut ext = FromZeros::new_zeroed();
1449        let total_ranges = address_space.vtl2_ranges().count();
1450
1451        let used_ext = build_e820_map(&mut boot_params, &mut ext, &address_space).unwrap();
1452
1453        // Verify that we used the extension
1454        assert!(used_ext, "should use extension when there are many ranges");
1455
1456        // Verify the standard e820_map is full
1457        assert_eq!(boot_params.e820_entries, E820_MAX_ENTRIES_ZEROPAGE as u8);
1458
1459        // Verify the extension has the overflow entries
1460        let ext_entries = (ext.header.len as usize) / size_of::<e820entry>();
1461        assert_eq!(ext_entries, total_ranges - E820_MAX_ENTRIES_ZEROPAGE);
1462
1463        // Verify we have the expected number of total ranges
1464        let total_e820_entries = boot_params.e820_entries as usize + ext_entries;
1465        assert_eq!(total_e820_entries, total_ranges);
1466    }
1467}