Skip to main content

openvmm_entry/
lib.rs

1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4//! This module implements the interactive control process and the entry point
5//! for the worker process.
6
7#![expect(missing_docs)]
8#![forbid(unsafe_code)]
9
10mod cli_args;
11mod crash_dump;
12mod kvp;
13mod meshworker;
14mod pidfile;
15mod repl;
16mod serial_io;
17mod storage_builder;
18mod tracing_init;
19mod ttrpc;
20mod vm_controller;
21
22// `pub` so that the missing_docs warning fires for options without
23// documentation.
24pub use cli_args::Options;
25use console_relay::ConsoleLaunchOptions;
26
27use crate::cli_args::SecureBootTemplateCli;
28use anyhow::Context;
29use anyhow::bail;
30use chipset_resources::battery::HostBatteryUpdate;
31use cli_args::DiskCliKind;
32use cli_args::EfiDiagnosticsLogLevelCli;
33use cli_args::EndpointConfigCli;
34use cli_args::IgvmPersonalityCli;
35use cli_args::NicConfigCli;
36use cli_args::ProvisionVmgs;
37use cli_args::SerialConfigCli;
38use cli_args::TpmVersionCli;
39use cli_args::UefiConsoleModeCli;
40use cli_args::VirtioBusCli;
41use cli_args::VmgsCli;
42use crash_dump::spawn_dump_handler;
43use cxl_spec::test::CxlTestDeviceHandle;
44use disk_backend_resources::DelayDiskHandle;
45use disk_backend_resources::DiskLayerDescription;
46use disk_backend_resources::layer::DiskLayerHandle;
47use disk_backend_resources::layer::RamDiskLayerHandle;
48use disk_backend_resources::layer::SqliteAutoCacheDiskLayerHandle;
49use disk_backend_resources::layer::SqliteDiskLayerHandle;
50use floppy_resources::FloppyDiskConfig;
51use framebuffer::FRAMEBUFFER_SIZE;
52use framebuffer::FramebufferAccess;
53use futures::AsyncReadExt;
54use futures::AsyncWrite;
55use futures::StreamExt;
56use futures::executor::block_on;
57use futures::io::AllowStdIo;
58use gdma_resources::GdmaDeviceHandle;
59use gdma_resources::VportDefinition;
60use guid::Guid;
61use input_core::MultiplexedInputHandle;
62use inspect::InspectMut;
63use mesh::CancelContext;
64use mesh::CellUpdater;
65use mesh::rpc::RpcSend;
66use meshworker::VmmMesh;
67use net_backend_resources::mac_address::MacAddress;
68use nvme_resources::NvmeControllerRequest;
69use openvmm_defs::config::Config;
70use openvmm_defs::config::DEFAULT_PCAT_BOOT_ORDER;
71use openvmm_defs::config::DeviceVtl;
72use openvmm_defs::config::HypervisorConfig;
73use openvmm_defs::config::LateMapVtl0MemoryPolicy;
74use openvmm_defs::config::LoadMode;
75use openvmm_defs::config::MemoryConfig;
76use openvmm_defs::config::NumaDistance;
77use openvmm_defs::config::NumaNode;
78use openvmm_defs::config::NumaTopology;
79use openvmm_defs::config::PcieDeviceConfig;
80use openvmm_defs::config::PcieMmioRangeConfig;
81use openvmm_defs::config::PciePortConfig;
82use openvmm_defs::config::PcieRootComplexConfig;
83use openvmm_defs::config::PcieSwitchConfig;
84use openvmm_defs::config::ProcessorTopologyConfig;
85use openvmm_defs::config::RootComplexCxlConfig;
86use openvmm_defs::config::SerialInformation;
87use openvmm_defs::config::VirtioBus;
88use openvmm_defs::config::VmbusConfig;
89use openvmm_defs::config::VpAssignment;
90use openvmm_defs::config::VpciDeviceConfig;
91use openvmm_defs::config::Vtl2BaseAddressType;
92use openvmm_defs::config::Vtl2Config;
93use openvmm_defs::rpc::VmRpc;
94use openvmm_defs::worker::VM_WORKER;
95use openvmm_defs::worker::VmWorkerParameters;
96use openvmm_helpers::disk::OpenDiskOptions;
97use openvmm_helpers::disk::create_disk_type;
98use openvmm_helpers::disk::open_disk_type;
99use pal_async::DefaultDriver;
100use pal_async::DefaultPool;
101use pal_async::socket::PolledSocket;
102use pal_async::task::Spawn;
103use pal_async::task::Task;
104use serial_16550_resources::ComPort;
105use serial_core::resources::DisconnectedSerialBackendHandle;
106use sparse_mmap::alloc_shared_memory;
107use std::cell::RefCell;
108use std::collections::BTreeMap;
109use std::collections::HashSet;
110use std::fmt::Write as _;
111use std::io;
112#[cfg(unix)]
113use std::io::IsTerminal;
114use std::io::Write;
115use std::net::TcpListener;
116use std::path::Path;
117use std::path::PathBuf;
118use std::sync::Arc;
119use std::thread;
120use std::time::Duration;
121use storvsp_resources::ScsiControllerRequest;
122use tpm_resources::TpmDeviceHandle;
123use tpm_resources::TpmRegisterLayout;
124use tpm_resources::TpmVersion;
125use uidevices_resources::SynthKeyboardHandle;
126use uidevices_resources::SynthMouseHandle;
127use uidevices_resources::SynthVideoHandle;
128use video_core::SharedFramebufferHandle;
129use virtio_resources::VirtioPciDeviceHandle;
130use vm_manifest_builder::BaseChipsetType;
131use vm_manifest_builder::MachineArch;
132use vm_manifest_builder::VmChipsetResult;
133use vm_manifest_builder::VmManifestBuilder;
134use vm_resource::IntoResource;
135use vm_resource::Resource;
136use vm_resource::kind::DiskHandleKind;
137use vm_resource::kind::DiskLayerHandleKind;
138use vm_resource::kind::NetEndpointHandleKind;
139use vm_resource::kind::VirtioDeviceHandle;
140use vm_resource::kind::VmbusDeviceHandleKind;
141use vmbus_serial_resources::VmbusSerialDeviceHandle;
142use vmbus_serial_resources::VmbusSerialPort;
143use vmcore::non_volatile_store::resources::EphemeralNonVolatileStoreHandle;
144use vmgs_resources::GuestStateEncryptionPolicy;
145use vmgs_resources::VmgsDisk;
146use vmgs_resources::VmgsFileHandle;
147use vmgs_resources::VmgsResource;
148use vmotherboard::ChipsetDeviceHandle;
149use vnc_worker_defs::VncParameters;
150
151pub fn openvmm_main() {
152    // Save the current state of the terminal so we can restore it back to
153    // normal before exiting.
154    #[cfg(unix)]
155    let orig_termios = io::stderr().is_terminal().then(term::get_termios);
156
157    let mut pidfile_guard: Option<pidfile::Pidfile> = None;
158    let exit_code = match do_main(&mut pidfile_guard) {
159        Ok(code) => code,
160        Err(err) => {
161            eprintln!("fatal error: {:?}", err);
162            1
163        }
164    };
165
166    // Restore the terminal to its initial state.
167    #[cfg(unix)]
168    if let Some(orig_termios) = orig_termios {
169        term::set_termios(orig_termios);
170    }
171
172    // Clean up the pidfile before terminating, since
173    // pal::process::terminate skips destructors.
174    drop(pidfile_guard);
175
176    // Terminate the process immediately without graceful shutdown of DLLs or
177    // C++ destructors or anything like that. This is all unnecessary and saves
178    // time on Windows.
179    //
180    // Do flush stdout, though, since there may be buffered data.
181    let _ = io::stdout().flush();
182    pal::process::terminate(exit_code);
183}
184
185#[derive(Default)]
186struct VmResources {
187    console_in: Option<Box<dyn AsyncWrite + Send + Unpin>>,
188    /// Keeps the dedicated serial reactor alive while serial I/O objects exist.
189    serial_driver: Option<DefaultDriver>,
190    framebuffer_access: Option<FramebufferAccess>,
191    shutdown_ic: Option<mesh::Sender<hyperv_ic_resources::shutdown::ShutdownRpc>>,
192    kvp_ic: Option<mesh::Sender<hyperv_ic_resources::kvp::KvpConnectRpc>>,
193    scsi_rpc: Option<mesh::Sender<ScsiControllerRequest>>,
194    nvme_vtl2_rpc: Option<mesh::Sender<NvmeControllerRequest>>,
195    consomme_rpc: Option<mesh::Sender<net_backend_resources::consomme::ConsommeRequest>>,
196    ged_rpc: Option<mesh::Sender<get_resources::ged::GuestEmulationRequest>>,
197    vtl2_settings: Option<vtl2_settings_proto::Vtl2Settings>,
198    /// Receives dirty rectangles from the synthetic video device for the VNC worker.
199    dirty_rect_recv: Option<mesh::Receiver<Vec<video_core::DirtyRect>>>,
200    #[cfg(windows)]
201    switch_ports: Vec<vmswitch::kernel::SwitchPort>,
202}
203
204struct ConsoleState<'a> {
205    device: &'a str,
206    input: Box<dyn AsyncWrite + Unpin + Send>,
207}
208
209/// Build a flat list of switches with their parent port assignments.
210///
211/// This function converts hierarchical CLI switch definitions into a flat list
212/// where each switch specifies its parent port directly.
213fn build_switch_list(all_switches: &[cli_args::GenericPcieSwitchCli]) -> Vec<PcieSwitchConfig> {
214    all_switches
215        .iter()
216        .map(|switch_cli| PcieSwitchConfig {
217            name: switch_cli.name.clone(),
218            parent_port: switch_cli.port_name.clone(),
219            ports: (0..switch_cli.num_downstream_ports)
220                .map(|i| PciePortConfig {
221                    name: format!("{}-downstream-{}", switch_cli.name, i),
222                    devfn: None,
223                    hotplug: switch_cli.hotplug,
224                    acs_capabilities_supported: switch_cli.acs_capabilities_supported,
225                    cxl: false,
226                    pasid: switch_cli.pasid,
227                })
228                .collect(),
229        })
230        .collect()
231}
232
233fn base_chipset_type(opt: &Options) -> BaseChipsetType {
234    if let Some(igvm) = &opt.igvm {
235        match igvm.personality {
236            IgvmPersonalityCli::Openhcl => BaseChipsetType::HclHost,
237            IgvmPersonalityCli::Uefi => BaseChipsetType::HypervGen2Uefi,
238            IgvmPersonalityCli::LinuxDirect if opt.isolation.is_some() => {
239                BaseChipsetType::EnlightenedLinuxDirect
240            }
241            IgvmPersonalityCli::LinuxDirect if opt.hv => BaseChipsetType::HyperVGen2LinuxDirect,
242            IgvmPersonalityCli::LinuxDirect => BaseChipsetType::UnenlightenedLinuxDirect,
243        }
244    } else if matches!(opt.isolation, Some(cli_args::IsolationCli::Snp)) {
245        BaseChipsetType::EnlightenedLinuxDirect
246    } else if opt.pcat {
247        BaseChipsetType::HypervGen1
248    } else if opt.uefi.is_some() {
249        BaseChipsetType::HypervGen2Uefi
250    } else if opt.hv {
251        BaseChipsetType::HyperVGen2LinuxDirect
252    } else {
253        BaseChipsetType::UnenlightenedLinuxDirect
254    }
255}
256
257/// Build the loader's [`SmbiosConfig`](openvmm_defs::config::SmbiosConfig) from
258/// the parsed `--smbios` arguments.
259///
260/// Multiple `--smbios` arguments are merged (erroring on a field set twice).
261/// String overrides left unset fall through to the loader's default identity.
262/// The system UUID defaults to the all-zero GUID unless overridden with
263/// `uuid=GUID`; `uuid=random` requests a freshly generated per-VM GUID.
264fn smbios_config_from_cli(
265    args: &[cli_args::SmbiosCli],
266) -> anyhow::Result<openvmm_defs::config::SmbiosConfig> {
267    let mut merged = cli_args::SmbiosCli::default();
268    for arg in args {
269        merged.merge(arg.clone())?;
270    }
271    let cli_args::SmbiosCli {
272        bios:
273            cli_args::SmbiosBiosCli {
274                vendor: bios_vendor,
275                version: bios_version,
276                release_date: bios_release_date,
277                release: bios_release,
278            },
279        system:
280            cli_args::SmbiosSystemCli {
281                manufacturer: system_manufacturer,
282                product_name: system_product,
283                version: system_version,
284                serial_number: system_serial,
285                sku_number: system_sku,
286                family: system_family,
287                uuid: system_uuid,
288            },
289    } = merged;
290    Ok(openvmm_defs::config::SmbiosConfig {
291        bios: openvmm_defs::config::SmbiosBiosOverrides {
292            vendor: bios_vendor,
293            version: bios_version,
294            release_date: bios_release_date,
295            release: bios_release.map(|r| (r.0, r.1)),
296        },
297        system: openvmm_defs::config::SmbiosSystemOverrides {
298            manufacturer: system_manufacturer,
299            product_name: system_product,
300            version: system_version,
301            serial_number: system_serial,
302            sku_number: system_sku,
303            family: system_family,
304            uuid: match system_uuid {
305                None => Guid::ZERO,
306                Some(cli_args::SmbiosUuid::Random) => Guid::new_random(),
307                Some(cli_args::SmbiosUuid::Fixed(guid)) => guid,
308            },
309        },
310    })
311}
312
313async fn vm_config_from_command_line(
314    spawner: impl Spawn,
315    mesh: &VmmMesh,
316    opt: &Options,
317) -> anyhow::Result<(Config, VmResources)> {
318    opt.validate_isolation_options()?;
319    opt.validate_igvm_options()?;
320
321    let (_, serial_driver) = DefaultPool::spawn_on_thread("serial");
322    let uefi = opt.effective_uefi()?;
323    let default_uefi = cli_args::UefiCli::default();
324    let uefi_options = uefi.as_ref().unwrap_or(&default_uefi);
325
326    let openhcl_vtl = if opt.vtl2 {
327        DeviceVtl::Vtl2
328    } else {
329        DeviceVtl::Vtl0
330    };
331
332    let console_state: RefCell<Option<ConsoleState<'_>>> = RefCell::new(None);
333    let setup_serial = |name: &str, cli_cfg, device| -> anyhow::Result<_> {
334        Ok(match cli_cfg {
335            SerialConfigCli::Console => {
336                if let Some(console_state) = console_state.borrow().as_ref() {
337                    bail!("console already set by {}", console_state.device);
338                }
339                let (config, serial) = serial_io::anonymous_serial_pair(&serial_driver)?;
340                let (serial_read, serial_write) = AsyncReadExt::split(serial);
341                *console_state.borrow_mut() = Some(ConsoleState {
342                    device,
343                    input: Box::new(serial_write),
344                });
345                thread::Builder::new()
346                    .name(name.to_owned())
347                    .spawn(move || {
348                        let _ = block_on(futures::io::copy(
349                            serial_read,
350                            &mut AllowStdIo::new(term::raw_stdout()),
351                        ));
352                    })
353                    .unwrap();
354                Some(config)
355            }
356            SerialConfigCli::Stderr => {
357                let (config, serial) = serial_io::anonymous_serial_pair(&serial_driver)?;
358                thread::Builder::new()
359                    .name(name.to_owned())
360                    .spawn(move || {
361                        let _ = block_on(futures::io::copy(
362                            serial,
363                            &mut AllowStdIo::new(term::raw_stderr()),
364                        ));
365                    })
366                    .unwrap();
367                Some(config)
368            }
369            SerialConfigCli::File(path) => {
370                let (config, serial) = serial_io::anonymous_serial_pair(&serial_driver)?;
371                let file = fs_err::File::create(path).context("failed to create file")?;
372
373                thread::Builder::new()
374                    .name(name.to_owned())
375                    .spawn(move || {
376                        let _ = block_on(futures::io::copy(serial, &mut AllowStdIo::new(file)));
377                    })
378                    .unwrap();
379                Some(config)
380            }
381            SerialConfigCli::None => None,
382            SerialConfigCli::Pipe(path) => {
383                Some(serial_io::bind_serial(&path).context("failed to bind serial")?)
384            }
385            SerialConfigCli::Tcp(addr) => {
386                Some(serial_io::bind_tcp_serial(&addr).context("failed to bind serial")?)
387            }
388            SerialConfigCli::NewConsole(app, window_title) => {
389                let path = console_relay::random_console_path();
390                let config =
391                    serial_io::bind_serial(&path).context("failed to bind console serial")?;
392                let window_title =
393                    window_title.unwrap_or_else(|| name.to_uppercase() + " [OpenVMM]");
394
395                console_relay::launch_console(
396                    app.or_else(openvmm_terminal_app).as_deref(),
397                    &path,
398                    ConsoleLaunchOptions {
399                        window_title: Some(window_title),
400                    },
401                )
402                .context("failed to launch console")?;
403
404                Some(config)
405            }
406        })
407    };
408
409    let mut vmbus_devices = Vec::new();
410
411    let com_debugger_mode = [
412        opt.com1.as_ref().is_some_and(|c| c.debugger_mode),
413        opt.com2.as_ref().is_some_and(|c| c.debugger_mode),
414        opt.com3.as_ref().is_some_and(|c| c.debugger_mode),
415        opt.com4.as_ref().is_some_and(|c| c.debugger_mode),
416    ];
417
418    let serial0_cfg = setup_serial(
419        "com1",
420        opt.com1
421            .clone()
422            .map_or(SerialConfigCli::Console, |c| c.backend),
423        if cfg!(guest_arch = "x86_64") {
424            "ttyS0"
425        } else {
426            "ttyAMA0"
427        },
428    )?;
429    let serial1_cfg = setup_serial(
430        "com2",
431        opt.com2
432            .clone()
433            .map_or(SerialConfigCli::None, |c| c.backend),
434        if cfg!(guest_arch = "x86_64") {
435            "ttyS1"
436        } else {
437            "ttyAMA1"
438        },
439    )?;
440    let serial2_cfg = setup_serial(
441        "com3",
442        opt.com3
443            .clone()
444            .map_or(SerialConfigCli::None, |c| c.backend),
445        if cfg!(guest_arch = "x86_64") {
446            "ttyS2"
447        } else {
448            "ttyAMA2"
449        },
450    )?;
451    let serial3_cfg = setup_serial(
452        "com4",
453        opt.com4
454            .clone()
455            .map_or(SerialConfigCli::None, |c| c.backend),
456        if cfg!(guest_arch = "x86_64") {
457            "ttyS3"
458        } else {
459            "ttyAMA3"
460        },
461    )?;
462    let with_vmbus_com1_serial = if let Some(vmbus_com1_cfg) = setup_serial(
463        "vmbus_com1",
464        opt.vmbus_com1_serial
465            .clone()
466            .unwrap_or(SerialConfigCli::None),
467        "vmbus_com1",
468    )? {
469        vmbus_devices.push((
470            openhcl_vtl,
471            VmbusSerialDeviceHandle {
472                port: VmbusSerialPort::Com1,
473                backend: vmbus_com1_cfg,
474            }
475            .into_resource(),
476        ));
477        true
478    } else {
479        false
480    };
481    let with_vmbus_com2_serial = if let Some(vmbus_com2_cfg) = setup_serial(
482        "vmbus_com2",
483        opt.vmbus_com2_serial
484            .clone()
485            .unwrap_or(SerialConfigCli::None),
486        "vmbus_com2",
487    )? {
488        vmbus_devices.push((
489            openhcl_vtl,
490            VmbusSerialDeviceHandle {
491                port: VmbusSerialPort::Com2,
492                backend: vmbus_com2_cfg,
493            }
494            .into_resource(),
495        ));
496        true
497    } else {
498        false
499    };
500    let debugcon_cfg = setup_serial(
501        "debugcon",
502        opt.debugcon
503            .clone()
504            .map(|cfg| cfg.serial)
505            .unwrap_or(SerialConfigCli::None),
506        "debugcon",
507    )?;
508
509    let virtio_console_backend = if let Some(serial_cfg) = opt.virtio_console.clone() {
510        setup_serial("virtio-console", serial_cfg, "hvc0")?
511    } else {
512        None
513    };
514
515    let mut resources = VmResources::default();
516    let mut console_str = "";
517    if let Some(ConsoleState { device, input }) = console_state.into_inner() {
518        resources.console_in = Some(input);
519        console_str = device;
520    }
521
522    if opt.shared_memory {
523        tracing::warn!("--shared-memory/-M flag has no effect and will be removed");
524    }
525    if opt.deprecated_prefetch {
526        tracing::warn!("--prefetch is deprecated; use --memory prefetch=on");
527    }
528    if opt.deprecated_private_memory {
529        tracing::warn!("--private-memory is deprecated; use --memory shared=off");
530    }
531    if opt.deprecated_thp {
532        tracing::warn!("--thp is deprecated; use --memory shared=off,thp=on");
533    }
534    if opt.deprecated_memory_backing_file.is_some() {
535        tracing::warn!("--memory-backing-file is deprecated; use --memory file=<path>");
536    }
537
538    opt.validate_memory_options()?;
539
540    const MAX_PROCESSOR_COUNT: u32 = 1024;
541
542    if opt.processors == 0 || opt.processors > MAX_PROCESSOR_COUNT {
543        bail!("invalid proc count: {}", opt.processors);
544    }
545
546    // Total SCSI channel count should not exceed the processor count
547    // (at most, one channel per VP).
548    if opt.scsi_sub_channels > (MAX_PROCESSOR_COUNT - 1) as u16 {
549        bail!(
550            "invalid SCSI sub-channel count: requested {}, max {}",
551            opt.scsi_sub_channels,
552            MAX_PROCESSOR_COUNT - 1
553        );
554    }
555
556    let with_get = opt.get || (opt.vtl2 && !opt.no_get);
557
558    let mut storage = storage_builder::StorageBuilder::new(with_get.then_some(openhcl_vtl));
559
560    // Register named controllers first, so that --disk on=<name>
561    // references can be resolved.
562    for ctrl in &opt.nvme_pci {
563        let transport = match &ctrl.transport {
564            cli_args::NvmeControllerTransport::Pcie(port) => {
565                storage_builder::NvmeControllerTransport::Pcie(port.clone())
566            }
567            cli_args::NvmeControllerTransport::Vpci(guid) => {
568                let guid = guid.unwrap_or_else(|| storage_builder::deterministic_guid(&ctrl.id));
569                storage_builder::NvmeControllerTransport::Vpci(guid)
570            }
571        };
572        storage.add_nvme_controller(ctrl.id.clone(), ctrl.vtl, transport, None)?;
573    }
574
575    for ctrl in &opt.vmbus_scsi {
576        let instance_id = storage_builder::deterministic_guid(&ctrl.id);
577        storage.add_scsi_controller(ctrl.id.clone(), ctrl.vtl, instance_id, ctrl.sub_channels)?;
578    }
579
580    for ctrl in &opt.openhcl_controller {
581        let controller_type = match ctrl.controller_type {
582            cli_args::OpenhclControllerType::Scsi => storage_builder::OpenhclControllerType::Scsi,
583            cli_args::OpenhclControllerType::Nvme => storage_builder::OpenhclControllerType::Nvme,
584        };
585        let instance_id = ctrl
586            .guid
587            .unwrap_or_else(|| storage_builder::deterministic_guid(&ctrl.id));
588        storage.add_openhcl_controller(ctrl.id.clone(), controller_type, instance_id)?;
589    }
590
591    for &cli_args::DiskCli {
592        vtl,
593        ref kind,
594        read_only,
595        is_dvd,
596        underhill,
597        ref pcie_port,
598        ref serial,
599        ref controller,
600        nsid,
601        lun,
602        ref relay,
603    } in &opt.disk
604    {
605        if serial.is_some() {
606            anyhow::bail!("`serial` is only supported by `--virtio-blk`");
607        }
608        if controller.is_none() && underhill.is_none() && relay.is_none() {
609            tracing::warn!(
610                "--disk without `on` is deprecated; \
611                 use --vmbus-scsi and --disk on=<name> instead"
612            );
613        }
614
615        let relay_target = relay
616            .as_ref()
617            .map(|(name, loc)| storage_builder::RelayTarget {
618                controller: name.clone(),
619                location: *loc,
620            });
621
622        let target = if let Some(name) = controller {
623            if pcie_port.is_some() {
624                anyhow::bail!("`on` is incompatible with `pcie_port` on `--disk`");
625            }
626            storage_builder::DiskLocation::Named {
627                controller: name.clone(),
628                nsid,
629                lun,
630            }
631        } else if pcie_port.is_some() {
632            anyhow::bail!("`--disk` is incompatible with `pcie_port` without `controller`");
633        } else {
634            if opt.no_vmbus {
635                anyhow::bail!(
636                    "`--disk` without `on=` attaches to the default VMBus SCSI controller and \
637                     cannot be used with `--no-vmbus`; use `on=<name>` to attach to a named controller"
638                );
639            }
640            storage_builder::DiskLocation::Scsi(None)
641        };
642
643        storage
644            .add(
645                vtl,
646                underhill,
647                relay_target,
648                target,
649                kind,
650                is_dvd,
651                read_only,
652            )
653            .await?;
654    }
655
656    for &cli_args::IdeDiskCli {
657        ref kind,
658        read_only,
659        channel,
660        device,
661        is_dvd,
662    } in &opt.ide
663    {
664        storage
665            .add(
666                DeviceVtl::Vtl0,
667                None,
668                None,
669                storage_builder::DiskLocation::Ide(channel, device),
670                kind,
671                is_dvd,
672                read_only,
673            )
674            .await?;
675    }
676
677    if !opt.nvme.is_empty() {
678        tracing::warn!("--nvme is deprecated; use --nvme-pci and --disk on=<name> instead");
679
680        // Pre-register implicit PCIe controllers for unique port names.
681        let mut registered_ports = std::collections::BTreeSet::new();
682        for disk in &opt.nvme {
683            if let Some(port) = &disk.pcie_port {
684                if registered_ports.insert(port.clone()) {
685                    storage.add_nvme_controller(
686                        port.clone(),
687                        DeviceVtl::Vtl0,
688                        storage_builder::NvmeControllerTransport::Pcie(port.clone()),
689                        None,
690                    ).with_context(|| format!(
691                        "legacy --nvme flag conflicts with an explicit controller named '{port}'; \
692                         use --nvme-pci and --disk on=<name> instead"
693                    ))?;
694                }
695            }
696        }
697    }
698
699    for &cli_args::DiskCli {
700        vtl,
701        ref kind,
702        read_only,
703        is_dvd,
704        underhill,
705        ref pcie_port,
706        ref serial,
707        controller: _,
708        nsid: _,
709        lun: _,
710        relay: _,
711    } in &opt.nvme
712    {
713        if serial.is_some() {
714            anyhow::bail!("`serial` is only supported by `--virtio-blk`");
715        }
716        let target = if let Some(port) = pcie_port {
717            storage_builder::DiskLocation::Named {
718                controller: port.clone(),
719                nsid: None,
720                lun: None,
721            }
722        } else {
723            storage_builder::DiskLocation::Nvme(None)
724        };
725        storage
726            .add(vtl, underhill, None, target, kind, is_dvd, read_only)
727            .await?;
728    }
729
730    for &cli_args::DiskCli {
731        vtl,
732        ref kind,
733        read_only,
734        is_dvd,
735        ref underhill,
736        ref pcie_port,
737        ref serial,
738        controller: _,
739        nsid: _,
740        lun: _,
741        relay: _,
742    } in &opt.virtio_blk
743    {
744        if underhill.is_some() {
745            anyhow::bail!("underhill not supported with virtio-blk");
746        }
747        storage
748            .add(
749                vtl,
750                None,
751                None,
752                storage_builder::DiskLocation::VirtioBlk {
753                    pcie_port: pcie_port.clone(),
754                    serial: serial.clone(),
755                },
756                kind,
757                is_dvd,
758                read_only,
759            )
760            .await?;
761    }
762
763    let mut floppy_disks = Vec::new();
764    for disk in &opt.floppy {
765        let &cli_args::FloppyDiskCli {
766            ref kind,
767            read_only,
768        } = disk;
769        floppy_disks.push(FloppyDiskConfig {
770            disk_type: disk_open(kind, read_only).await?,
771            read_only,
772        });
773    }
774
775    let mut vpci_mana_nics = [(); 3].map(|()| None);
776    let mut pcie_mana_nics = BTreeMap::<String, GdmaDeviceHandle>::new();
777    let mut underhill_nics = Vec::new();
778    let mut vpci_devices = Vec::new();
779
780    let mut nic_index = 0;
781    for cli_cfg in &opt.net {
782        if cli_cfg.pcie_port.is_some() {
783            anyhow::bail!("`--net` does not support PCIe");
784        }
785        let vport = parse_endpoint(cli_cfg, &mut nic_index, &mut resources)?;
786        if cli_cfg.underhill {
787            if !opt.no_alias_map {
788                anyhow::bail!("must specify --no-alias-map to offer NICs to VTL2");
789            }
790            let mana = vpci_mana_nics[openhcl_vtl as usize].get_or_insert_with(|| {
791                let vpci_instance_id = Guid::new_random();
792                underhill_nics.push(vtl2_settings_proto::NicDeviceLegacy {
793                    instance_id: vpci_instance_id.to_string(),
794                    subordinate_instance_id: None,
795                    max_sub_channels: None,
796                });
797                (vpci_instance_id, GdmaDeviceHandle { vports: Vec::new() })
798            });
799            mana.1.vports.push(VportDefinition {
800                mac_address: vport.mac_address,
801                endpoint: vport.endpoint,
802            });
803        } else {
804            vmbus_devices.push(vport.into_netvsp_handle());
805        }
806    }
807
808    if opt.nic {
809        let nic_config = parse_endpoint(
810            &NicConfigCli {
811                vtl: DeviceVtl::Vtl0,
812                endpoint: EndpointConfigCli::Consomme {
813                    cidr: None,
814                    host_fwd: Vec::new(),
815                },
816                max_queues: None,
817                underhill: false,
818                pcie_port: None,
819            },
820            &mut nic_index,
821            &mut resources,
822        )?;
823        vmbus_devices.push(nic_config.into_netvsp_handle());
824    }
825
826    // Build initial PCIe devices list from CLI options. Storage devices
827    // (e.g., NVMe controllers on PCIe ports) are added later by storage_builder.
828    let mut pcie_devices = Vec::new();
829    for (index, cli_cfg) in opt.pcie_remote.iter().enumerate() {
830        tracing::info!(
831            port_name = %cli_cfg.port_name,
832            socket_addr = ?cli_cfg.socket_addr,
833            "instantiating PCIe remote device"
834        );
835
836        // Generate a deterministic instance ID based on index
837        const PCIE_REMOTE_BASE_INSTANCE_ID: Guid =
838            guid::guid!("28ed784d-c059-429f-9d9a-46bea02562c0");
839        let instance_id = Guid {
840            data1: index as u32,
841            ..PCIE_REMOTE_BASE_INSTANCE_ID
842        };
843
844        pcie_devices.push(PcieDeviceConfig {
845            port_name: cli_cfg.port_name.clone(),
846            resource: pcie_remote_resources::PcieRemoteHandle {
847                instance_id,
848                socket_addr: cli_cfg.socket_addr.clone(),
849                hu: cli_cfg.hu,
850                controller: cli_cfg.controller,
851            }
852            .into_resource(),
853        });
854    }
855
856    #[cfg(windows)]
857    let mut kernel_vmnics = Vec::new();
858    #[cfg(windows)]
859    for (index, switch_id) in opt.kernel_vmnic.iter().enumerate() {
860        // Pick a random MAC address.
861        let mut mac_address = [0x00, 0x15, 0x5D, 0, 0, 0];
862        getrandom::fill(&mut mac_address[3..]).expect("rng failure");
863
864        // Pick a fixed instance ID based on the index.
865        const BASE_INSTANCE_ID: Guid = guid::guid!("00000000-435d-11ee-9f59-00155d5016fc");
866        let instance_id = Guid {
867            data1: index as u32,
868            ..BASE_INSTANCE_ID
869        };
870
871        let switch_id = if switch_id == "default" {
872            None
873        } else {
874            Some(switch_id.as_str())
875        };
876        let (port_id, port) = new_switch_port(switch_id)?;
877        resources.switch_ports.push(port);
878
879        kernel_vmnics.push(openvmm_defs::config::KernelVmNicConfig {
880            instance_id,
881            mac_address: mac_address.into(),
882            switch_port_id: port_id,
883        });
884    }
885
886    for vport in &opt.mana {
887        let vport = parse_endpoint(vport, &mut nic_index, &mut resources)?;
888        let vport_array = match (vport.vtl as usize, vport.pcie_port) {
889            (vtl, None) => {
890                &mut vpci_mana_nics[vtl]
891                    .get_or_insert_with(|| {
892                        (Guid::new_random(), GdmaDeviceHandle { vports: Vec::new() })
893                    })
894                    .1
895                    .vports
896            }
897            (0, Some(pcie_port)) => {
898                &mut pcie_mana_nics
899                    .entry(pcie_port)
900                    .or_insert(GdmaDeviceHandle { vports: Vec::new() })
901                    .vports
902            }
903            _ => anyhow::bail!("PCIe NICs only supported to VTL0"),
904        };
905        vport_array.push(VportDefinition {
906            mac_address: vport.mac_address,
907            endpoint: vport.endpoint,
908        });
909    }
910
911    vpci_devices.extend(
912        vpci_mana_nics
913            .into_iter()
914            .enumerate()
915            .filter_map(|(vtl, nic)| {
916                nic.map(|(instance_id, handle)| VpciDeviceConfig {
917                    vtl: match vtl {
918                        0 => DeviceVtl::Vtl0,
919                        1 => DeviceVtl::Vtl1,
920                        2 => DeviceVtl::Vtl2,
921                        _ => unreachable!(),
922                    },
923                    instance_id,
924                    resource: handle.into_resource(),
925                    vnode: None,
926                })
927            }),
928    );
929
930    pcie_devices.extend(
931        pcie_mana_nics
932            .into_iter()
933            .map(|(pcie_port, handle)| PcieDeviceConfig {
934                port_name: pcie_port,
935                resource: handle.into_resource(),
936            }),
937    );
938
939    for cxl_test in &opt.cxl_test {
940        pcie_devices.push(PcieDeviceConfig {
941            port_name: cxl_test.pcie_port.clone(),
942            resource: CxlTestDeviceHandle {
943                hdm_size_bytes: cxl_test.hdm_size,
944            }
945            .into_resource(),
946        });
947    }
948
949    #[cfg(guest_arch = "aarch64")]
950    let arch = MachineArch::Aarch64;
951    #[cfg(guest_arch = "x86_64")]
952    let arch = MachineArch::X86_64;
953
954    #[cfg(guest_arch = "x86_64")]
955    anyhow::ensure!(
956        opt.amd_iommu.is_empty() || opt.intel_vtd.is_empty(),
957        "--amd-iommu and --intel-vtd cannot both be used in the same VM"
958    );
959
960    #[cfg(guest_arch = "x86_64")]
961    let mut amd_iommu_names: HashSet<&str> = opt.amd_iommu.iter().map(|s| s.as_str()).collect();
962    #[cfg(guest_arch = "x86_64")]
963    let mut vtd_names: HashSet<&str> = opt.intel_vtd.iter().map(|s| s.as_str()).collect();
964
965    // Map each `--smmu` entry to its root complex, rejecting duplicate `rc=`
966    // entries up front. Entries are removed as they are matched to a root
967    // complex below; any left over refer to unknown root complexes.
968    #[cfg(guest_arch = "aarch64")]
969    let mut smmu_names: std::collections::HashMap<&str, &cli_args::SmmuCli> = {
970        let mut map = std::collections::HashMap::new();
971        for s in &opt.smmu {
972            if map.insert(s.rc_name.as_str(), s).is_some() {
973                anyhow::bail!(
974                    "--smmu specified multiple times for root complex '{}'",
975                    s.rc_name
976                );
977            }
978        }
979        map
980    };
981
982    let mut pcie_root_complexes = Vec::new();
983    for (i, rc_cli) in opt.pcie_root_complex.iter().enumerate() {
984        let ports: Vec<PciePortConfig> = opt
985            .pcie_root_port
986            .iter()
987            .filter(|port_cli| port_cli.root_complex_name == rc_cli.name)
988            .map(|port_cli| PciePortConfig {
989                name: port_cli.name.clone(),
990                devfn: port_cli.devfn,
991                hotplug: port_cli.hotplug,
992                acs_capabilities_supported: port_cli.acs_capabilities_supported,
993                cxl: port_cli.cxl,
994                pasid: port_cli.pasid,
995            })
996            .collect();
997
998        const ONE_MB: u64 = 1024 * 1024;
999        // Keep all PCI windows 1MB-granular to match layout and downstream placement rules.
1000        let low_mmio_size = (rc_cli.low_mmio as u64).next_multiple_of(ONE_MB);
1001        let high_mmio_size = rc_cli
1002            .high_mmio
1003            .checked_next_multiple_of(ONE_MB)
1004            .context("high mmio rounding error")?;
1005
1006        // Count CXL-capable ports under the root bus. If the root bus has CXL root ports, it needs CHBCR.
1007        let cxl_port_count = ports.iter().filter(|port| port.cxl).count() as u64;
1008
1009        let cxl = if cxl_port_count != 0 {
1010            Some(RootComplexCxlConfig {
1011                hdm_size: rc_cli.hdm,
1012                hdm_window_restrictions: rc_cli.hdm_window_restrictions.bits(),
1013            })
1014        } else {
1015            None
1016        };
1017        pcie_root_complexes.push(PcieRootComplexConfig {
1018            index: i as u32,
1019            name: rc_cli.name.clone(),
1020            segment: rc_cli.segment,
1021            start_bus: rc_cli.start_bus,
1022            end_bus: rc_cli.end_bus,
1023            low_mmio: if let Some(base) = rc_cli.low_mmio_base {
1024                PcieMmioRangeConfig::Fixed(
1025                    memory_range::MemoryRange::try_new(base..base.wrapping_add(low_mmio_size))
1026                        .context("invalid low MMIO range")?,
1027                )
1028            } else {
1029                PcieMmioRangeConfig::Dynamic {
1030                    size: low_mmio_size,
1031                }
1032            },
1033            high_mmio: if let Some(base) = rc_cli.high_mmio_base {
1034                PcieMmioRangeConfig::Fixed(
1035                    memory_range::MemoryRange::try_new(base..base.wrapping_add(high_mmio_size))
1036                        .context("invalid high MMIO range")?,
1037                )
1038            } else {
1039                PcieMmioRangeConfig::Dynamic {
1040                    size: high_mmio_size,
1041                }
1042            },
1043            cxl,
1044            ports,
1045            #[cfg(guest_arch = "aarch64")]
1046            iommu: smmu_names.remove(rc_cli.name.as_str()).map(|s| {
1047                openvmm_defs::config::PcieIommuConfig::Smmu {
1048                    accel: s.accel,
1049                    oas: match s.oas {
1050                        cli_args::SmmuOasCli::Auto => openvmm_defs::config::SmmuOas::Auto,
1051                        cli_args::SmmuOasCli::Fixed(bits) => {
1052                            openvmm_defs::config::SmmuOas::Fixed(bits)
1053                        }
1054                    },
1055                }
1056            }),
1057            #[cfg(guest_arch = "x86_64")]
1058            iommu: if amd_iommu_names.remove(rc_cli.name.as_str()) {
1059                Some(openvmm_defs::config::PcieIommuConfig::AmdVi)
1060            } else if vtd_names.remove(rc_cli.name.as_str()) {
1061                Some(openvmm_defs::config::PcieIommuConfig::IntelVtd)
1062            } else {
1063                None
1064            },
1065            vnode: rc_cli.vnode,
1066            preserve_bars: rc_cli.preserve_bars,
1067        });
1068    }
1069
1070    #[cfg(guest_arch = "aarch64")]
1071    if let Some(name) = smmu_names.into_keys().next() {
1072        anyhow::bail!("--smmu refers to unknown root complex '{name}'");
1073    }
1074    #[cfg(guest_arch = "x86_64")]
1075    if let Some(name) = amd_iommu_names.into_iter().next() {
1076        anyhow::bail!("--amd-iommu refers to unknown root complex '{name}'");
1077    }
1078    #[cfg(guest_arch = "x86_64")]
1079    if let Some(name) = vtd_names.into_iter().next() {
1080        anyhow::bail!("--intel-vtd refers to unknown root complex '{name}'");
1081    }
1082
1083    let pcie_switches = build_switch_list(&opt.pcie_switch);
1084    let pcie_generic_initiators = opt
1085        .pcie_generic_initiator
1086        .iter()
1087        .map(|gi| openvmm_defs::config::PcieGenericInitiatorConfig {
1088            port_name: gi.port_name.clone(),
1089            node: gi.node,
1090        })
1091        .collect();
1092    #[cfg(target_os = "linux")]
1093    let vfio_pcie_devices: Vec<PcieDeviceConfig> = {
1094        use std::collections::HashMap;
1095        use vm_resource::IntoResource;
1096
1097        // Process --iommu flags: open /dev/iommu for each declared context.
1098        let mut iommu_map: HashMap<String, std::fs::File> = HashMap::new();
1099        for iommu_cli in &opt.iommu {
1100            anyhow::ensure!(
1101                !iommu_map.contains_key(&iommu_cli.id),
1102                "duplicate --iommu id={}",
1103                iommu_cli.id
1104            );
1105            let file = std::fs::OpenOptions::new()
1106                .read(true)
1107                .write(true)
1108                .open("/dev/iommu")
1109                .context("failed to open /dev/iommu (is iommufd available?)")?;
1110            iommu_map.insert(iommu_cli.id.clone(), file);
1111        }
1112
1113        opt.vfio
1114            .iter()
1115            .map(|cli_cfg| {
1116                let sysfs_path = Path::new("/sys/bus/pci/devices").join(&cli_cfg.pci_id);
1117
1118                if let Some(iommu_id) = &cli_cfg.iommu {
1119                    // cdev + iommufd path
1120                    let iommufd = iommu_map.get(iommu_id).with_context(|| {
1121                        format!(
1122                            "--vfio device {} references iommu={iommu_id}, \
1123                             but no --iommu id={iommu_id} was specified",
1124                            cli_cfg.pci_id
1125                        )
1126                    })?;
1127                    // Clone the iommufd fd so the per-iommu manager can own it.
1128                    // The first device for a given iommu ID uses the cloned fd
1129                    // to create the IoasManager; subsequent devices reuse the
1130                    // existing manager and the cloned fd is dropped.
1131                    let iommufd = iommufd.try_clone().with_context(|| {
1132                        format!("failed to dup iommufd fd for iommu={iommu_id}")
1133                    })?;
1134
1135                    // Open the cdev device node.
1136                    let vfio_dev_dir = sysfs_path.join("vfio-dev");
1137                    let entry = std::fs::read_dir(&vfio_dev_dir)
1138                        .with_context(|| {
1139                            format!(
1140                                "failed to read {}: is {} bound to vfio-pci?",
1141                                vfio_dev_dir.display(),
1142                                cli_cfg.pci_id
1143                            )
1144                        })?
1145                        .next()
1146                        .context("no vfio-dev entry found")?
1147                        .context("failed to read vfio-dev entry")?;
1148                    let dev_path = Path::new("/dev/vfio/devices").join(entry.file_name());
1149                    let cdev = std::fs::OpenOptions::new()
1150                        .read(true)
1151                        .write(true)
1152                        .open(&dev_path)
1153                        .with_context(|| format!("failed to open {}", dev_path.display()))?;
1154
1155                    Ok(PcieDeviceConfig {
1156                        port_name: cli_cfg.port_name.clone(),
1157                        resource: vfio_assigned_device_resources::VfioCdevDeviceHandle {
1158                            pci_id: cli_cfg.pci_id.clone(),
1159                            cdev,
1160                            iommufd,
1161                            iommu_id: iommu_id.clone(),
1162                            bar_addresses: cli_cfg.bar_addresses,
1163                        }
1164                        .into_resource(),
1165                    })
1166                } else {
1167                    // Legacy group/container path
1168                    let iommu_group_link = std::fs::read_link(sysfs_path.join("iommu_group"))
1169                        .with_context(|| {
1170                            format!("failed to read IOMMU group for {}", cli_cfg.pci_id)
1171                        })?;
1172                    let group_id: u64 = iommu_group_link
1173                        .file_name()
1174                        .and_then(|s| s.to_str())
1175                        .context("invalid iommu_group symlink")?
1176                        .parse()
1177                        .context("failed to parse IOMMU group ID")?;
1178                    let group = std::fs::OpenOptions::new()
1179                        .read(true)
1180                        .write(true)
1181                        .open(format!("/dev/vfio/{group_id}"))
1182                        .with_context(|| format!("failed to open /dev/vfio/{group_id}"))?;
1183
1184                    Ok(PcieDeviceConfig {
1185                        port_name: cli_cfg.port_name.clone(),
1186                        resource: vfio_assigned_device_resources::VfioDeviceHandle {
1187                            pci_id: cli_cfg.pci_id.clone(),
1188                            group,
1189                            bar_addresses: cli_cfg.bar_addresses,
1190                        }
1191                        .into_resource(),
1192                    })
1193                }
1194            })
1195            .collect::<anyhow::Result<Vec<_>>>()?
1196    };
1197
1198    #[cfg(windows)]
1199    let vpci_resources: Vec<_> = opt
1200        .device
1201        .iter()
1202        .map(|path| -> anyhow::Result<_> {
1203            Ok(virt_whp::device::DeviceHandle(
1204                whp::VpciResource::new(
1205                    None,
1206                    Default::default(),
1207                    &whp::VpciResourceDescriptor::Sriov(path, 0, 0),
1208                )
1209                .with_context(|| format!("opening PCI device {}", path))?,
1210            ))
1211        })
1212        .collect::<Result<_, _>>()?;
1213
1214    // Create a vmbusproxy handle if needed by any devices.
1215    #[cfg(windows)]
1216    let vmbusproxy_handle = if !kernel_vmnics.is_empty() {
1217        Some(vmbus_proxy::ProxyHandle::new().context("failed to open vmbusproxy handle")?)
1218    } else {
1219        None
1220    };
1221
1222    let framebuffer = if opt.gfx || opt.vtl2_gfx || opt.vnc.vnc || opt.pcat {
1223        let vram = alloc_shared_memory(FRAMEBUFFER_SIZE, "vram")?;
1224        let (fb, fba) =
1225            framebuffer::framebuffer(vram, FRAMEBUFFER_SIZE, 0).context("creating framebuffer")?;
1226        resources.framebuffer_access = Some(fba);
1227        Some(fb)
1228    } else {
1229        None
1230    };
1231
1232    let load_mode;
1233    let with_hv;
1234
1235    let any_serial_configured = serial0_cfg.is_some()
1236        || serial1_cfg.is_some()
1237        || serial2_cfg.is_some()
1238        || serial3_cfg.is_some();
1239
1240    let has_com3 = serial2_cfg.is_some();
1241
1242    let mut chipset = VmManifestBuilder::new(base_chipset_type(opt), arch);
1243
1244    if framebuffer.is_some() {
1245        chipset = chipset.with_framebuffer();
1246    }
1247    if opt.guest_watchdog {
1248        chipset = chipset.with_guest_watchdog();
1249    }
1250    if any_serial_configured {
1251        chipset = chipset.with_serial([serial0_cfg, serial1_cfg, serial2_cfg, serial3_cfg]);
1252    }
1253    chipset = chipset.with_serial_debugger_mode(com_debugger_mode);
1254    if opt.battery {
1255        let (tx, rx) = mesh::channel();
1256        tx.send(HostBatteryUpdate::default_present());
1257        chipset = chipset.with_battery(rx);
1258    }
1259    if opt.no_vmbus {
1260        chipset = chipset.without_vmbus();
1261    }
1262    if let Some(cfg) = &opt.debugcon {
1263        chipset = chipset.with_debugcon(
1264            debugcon_cfg.unwrap_or_else(|| DisconnectedSerialBackendHandle.into_resource()),
1265            cfg.port,
1266        );
1267    }
1268
1269    let (base_template, custom_uefi_json) = {
1270        #[cfg(guest_arch = "aarch64")]
1271        use firmware_uefi_resources::aarch64_secure_boot_templates as secure_boot_templates;
1272        #[cfg(guest_arch = "x86_64")]
1273        use firmware_uefi_resources::x64_secure_boot_templates as secure_boot_templates;
1274        let base_template = opt.secure_boot_template.map(|template| match template {
1275            SecureBootTemplateCli::Windows => secure_boot_templates::microsoft_windows(),
1276            SecureBootTemplateCli::UefiCa => secure_boot_templates::microsoft_uefi_ca(),
1277        });
1278
1279        // TODO: fallback to VMGS read if no command line flag was given
1280
1281        let custom_uefi_json = match &opt.custom_uefi_json {
1282            Some(file) => Some(
1283                fs_err::read(file)
1284                    .context("opening custom uefi json file")?
1285                    .into(),
1286            ),
1287            None => None,
1288        };
1289
1290        (base_template, custom_uefi_json)
1291    };
1292
1293    if (uefi.is_some() && opt.igvm.is_none())
1294        || opt
1295            .igvm
1296            .as_ref()
1297            .is_some_and(|igvm| igvm.personality == IgvmPersonalityCli::Uefi)
1298    {
1299        let log_level = match uefi_options.diagnostics.unwrap_or_default() {
1300            EfiDiagnosticsLogLevelCli::Default => firmware_uefi_resources::LogLevel::make_default(),
1301            EfiDiagnosticsLogLevelCli::Info => firmware_uefi_resources::LogLevel::make_info(),
1302            EfiDiagnosticsLogLevelCli::Full => firmware_uefi_resources::LogLevel::make_full(),
1303        };
1304        let nvram_storage = if opt.vmgs.is_some() {
1305            VmgsFileHandle::new(vmgs_format::FileId::BIOS_NVRAM, true).into_resource()
1306        } else {
1307            EphemeralNonVolatileStoreHandle.into_resource()
1308        };
1309        chipset = chipset.with_uefi(vm_manifest_builder::UefiManifest::new(
1310            arch,
1311            base_template,
1312            custom_uefi_json,
1313            opt.secure_boot,
1314            log_level,
1315            None,
1316            nvram_storage,
1317            None,
1318        ));
1319    }
1320
1321    // Build the SMBIOS config once, up front, so that UEFI and Linux direct
1322    // boot share a single source for the VM's BIOS GUID / system UUID. The TPM
1323    // also keys off this GUID.
1324    let smbios = Box::new(smbios_config_from_cli(&opt.smbios)?);
1325    let bios_guid = smbios.system.uuid;
1326
1327    // Capture the SMBIOS config for the OpenHCL/GED path before `smbios` is
1328    // potentially moved into a non-VTL2 LoadMode below. The GED forwards only
1329    // the system identity to the paravisor and fails closed on BIOS overrides
1330    // it cannot honor, so it is delivered as the shared `SmbiosConfig`.
1331    let ged_smbios = (*smbios).clone();
1332
1333    let layout_config = chipset.layout_config();
1334    let VmChipsetResult {
1335        chipset,
1336        mut chipset_devices,
1337        pci_chipset_devices,
1338        isa_dma_controller,
1339        capabilities,
1340    } = chipset
1341        .build()
1342        .context("failed to build chipset configuration")?;
1343
1344    let tpm_version = opt.tpm.map(|cli_ver| match cli_ver {
1345        TpmVersionCli::V138 => TpmVersion::V138,
1346        TpmVersionCli::V185 => TpmVersion::V185,
1347    });
1348
1349    if opt.restore_snapshot.is_some() {
1350        // Snapshot restore: skip firmware loading entirely. Device state and
1351        // memory come from the snapshot directory.
1352        load_mode = LoadMode::None;
1353        with_hv = true;
1354    } else if let Some(igvm) = &opt.igvm {
1355        let cli_args::UefiCli {
1356            firmware,
1357            debug: _,
1358            enable_memory_protections: _,
1359            force_dma_bounce: _,
1360            force_firmware_version,
1361            disable_frontpage: _,
1362            console: _,
1363            diagnostics: _,
1364            default_boot_always_attempt: _,
1365        } = uefi_options;
1366
1367        anyhow::ensure!(
1368            firmware.is_none(),
1369            "--uefi firmware is not supported with --igvm"
1370        );
1371        anyhow::ensure!(
1372            !force_firmware_version,
1373            "--uefi force_firmware_version is not supported with --igvm"
1374        );
1375        let file = fs_err::File::open(&igvm.firmware)
1376            .context("failed to open igvm file")?
1377            .into();
1378        let cmdline = opt.cmdline.join(" ");
1379        with_hv = match igvm.personality {
1380            IgvmPersonalityCli::Openhcl | IgvmPersonalityCli::Uefi => true,
1381            IgvmPersonalityCli::LinuxDirect => opt.hv,
1382        };
1383
1384        load_mode = LoadMode::Igvm {
1385            file,
1386            cmdline,
1387            vtl2_base_address: if opt.vtl2 {
1388                opt.igvm_vtl2_relocation_type
1389            } else {
1390                Vtl2BaseAddressType::File
1391            },
1392            com_serial: has_com3.then(|| SerialInformation {
1393                io_port: ComPort::Com3.io_port(),
1394                irq: ComPort::Com3.irq().into(),
1395            }),
1396        };
1397
1398        // An IGVM launch carries no SMBIOS field of its own; the identity is
1399        // only delivered over the GET/GED channel, which is absent here. Reject
1400        // overrides that would otherwise be silently dropped.
1401        let smbios_requested = !opt.smbios.is_empty();
1402        let smbios_delivered_via_get = with_get && with_hv;
1403        if smbios_requested && !smbios_delivered_via_get {
1404            anyhow::bail!(
1405                "--smbios is not supported for IGVM launches without an OpenHCL GET channel"
1406            );
1407        }
1408    } else if opt.pcat {
1409        // Emit a nice error early instead of complaining about missing firmware.
1410        if arch != MachineArch::X86_64 {
1411            anyhow::bail!("pcat not supported on this architecture");
1412        }
1413        with_hv = true;
1414
1415        let firmware = openvmm_pcat_locator::find_pcat_bios(opt.pcat_firmware.as_deref())?;
1416        load_mode = LoadMode::Pcat {
1417            firmware,
1418            boot_order: opt
1419                .pcat_boot_order
1420                .map(|x| x.0)
1421                .unwrap_or(DEFAULT_PCAT_BOOT_ORDER),
1422            hibernation_enabled: opt.hibernation,
1423            smbios,
1424        };
1425    } else if let Some(uefi_options) = &uefi {
1426        use openvmm_defs::config::UefiConsoleMode;
1427
1428        let cli_args::UefiCli {
1429            firmware,
1430            debug,
1431            enable_memory_protections,
1432            force_dma_bounce,
1433            force_firmware_version,
1434            disable_frontpage,
1435            console,
1436            diagnostics: _,
1437            default_boot_always_attempt,
1438        } = uefi_options;
1439
1440        if opt.no_hv && cfg!(guest_arch = "x86_64") {
1441            anyhow::bail!("--no-hv is not supported on x86_64");
1442        }
1443
1444        with_hv = !opt.no_hv;
1445
1446        let default_firmware = cli_args::default_uefi_firmware();
1447        let firmware = fs_err::File::open(
1448            firmware
1449                .as_ref()
1450                .or(default_firmware.as_ref())
1451                .context("must provide uefi firmware when booting with uefi")?,
1452        )
1453        .context("failed to open uefi firmware")?;
1454
1455        // TODO: It would be better to default memory protections to on, but currently Linux does not boot via UEFI due to what
1456        //       appears to be a GRUB memory protection fault. Memory protections are therefore only enabled if configured.
1457        load_mode = LoadMode::Uefi {
1458            firmware: firmware.into(),
1459            enable_debugging: *debug,
1460            enable_memory_protections: *enable_memory_protections,
1461            disable_frontpage: *disable_frontpage,
1462            tpm_version,
1463            enable_battery: opt.battery,
1464            enable_serial: any_serial_configured,
1465            enable_vpci_boot: false,
1466            uefi_console_mode: console.map(|m| match m {
1467                UefiConsoleModeCli::Default => UefiConsoleMode::Default,
1468                UefiConsoleModeCli::Com1 => UefiConsoleMode::Com1,
1469                UefiConsoleModeCli::Com2 => UefiConsoleMode::Com2,
1470                UefiConsoleModeCli::None => UefiConsoleMode::None,
1471            }),
1472            default_boot_always_attempt: *default_boot_always_attempt,
1473            smbios,
1474            enable_vmbus: !opt.no_vmbus,
1475            force_dma_bounce: *force_dma_bounce,
1476            enable_hv: !opt.no_hv,
1477            hibernation_enabled: opt.hibernation,
1478            force_firmware_version: *force_firmware_version,
1479        };
1480    } else {
1481        // Linux Direct
1482        let mut cmdline = "panic=-1 debug".to_string();
1483
1484        with_hv = opt.hv;
1485        if with_hv && opt.pcie_root_complex.is_empty() {
1486            cmdline += " pci=off";
1487        }
1488
1489        if !console_str.is_empty() {
1490            let _ = write!(&mut cmdline, " console={}", console_str);
1491        }
1492
1493        if opt.gfx {
1494            cmdline += " console=tty";
1495        }
1496        for extra in &opt.cmdline {
1497            let _ = write!(&mut cmdline, " {}", extra);
1498        }
1499
1500        let kernel = fs_err::File::open(
1501            (opt.kernel.0)
1502                .as_ref()
1503                .context("must provide kernel when booting with linux direct")?,
1504        )
1505        .context("failed to open kernel")?;
1506        let initrd = (opt.initrd.0)
1507            .as_ref()
1508            .map(fs_err::File::open)
1509            .transpose()
1510            .context("failed to open initrd")?;
1511
1512        load_mode = LoadMode::Linux {
1513            kernel: kernel.into(),
1514            initrd: initrd.map(Into::into),
1515            cmdline,
1516            enable_serial: any_serial_configured,
1517            isolation: if matches!(opt.isolation, Some(cli_args::IsolationCli::Snp)) {
1518                openvmm_defs::config::LinuxIsolationConfig::Snp {
1519                    restricted_injection: opt.snp_restricted_injection,
1520                }
1521            } else {
1522                openvmm_defs::config::LinuxIsolationConfig::None
1523            },
1524            boot_mode: if opt.device_tree {
1525                openvmm_defs::config::LinuxDirectBootMode::DeviceTree
1526            } else {
1527                openvmm_defs::config::LinuxDirectBootMode::Acpi
1528            },
1529            smbios,
1530        };
1531    }
1532
1533    let mut vmgs = Some(if let Some(VmgsCli { kind, provision }) = &opt.vmgs {
1534        let disk = VmgsDisk {
1535            disk: disk_open(kind, false)
1536                .await
1537                .context("failed to open vmgs disk")?,
1538            encryption_policy: if opt.test_gsp_by_id {
1539                GuestStateEncryptionPolicy::GspById(true)
1540            } else {
1541                GuestStateEncryptionPolicy::None(true)
1542            },
1543        };
1544        match provision {
1545            ProvisionVmgs::OnEmpty => VmgsResource::Disk(disk),
1546            ProvisionVmgs::OnFailure => VmgsResource::ReprovisionOnFailure(disk),
1547            ProvisionVmgs::True => VmgsResource::Reprovision(disk),
1548        }
1549    } else {
1550        VmgsResource::Ephemeral
1551    });
1552
1553    if with_get && with_hv {
1554        let has_vtl0_nvme = storage.has_vtl0_nvme();
1555        let vtl2_settings = vtl2_settings_proto::Vtl2Settings {
1556            version: vtl2_settings_proto::vtl2_settings_base::Version::V1.into(),
1557            fixed: Some(Default::default()),
1558            dynamic: Some(vtl2_settings_proto::Vtl2SettingsDynamic {
1559                storage_controllers: storage.build_openhcl_settings(opt.vmbus_redirect),
1560                nic_devices: underhill_nics,
1561            }),
1562            namespace_settings: Vec::default(),
1563        };
1564
1565        // Cache the VTL2 settings for later modification via the interactive console.
1566        resources.vtl2_settings = Some(vtl2_settings.clone());
1567
1568        let (send, guest_request_recv) = mesh::channel();
1569        resources.ged_rpc = Some(send);
1570
1571        let vmgs = vmgs.take().unwrap();
1572
1573        vmbus_devices.extend([
1574            (
1575                openhcl_vtl,
1576                get_resources::gel::GuestEmulationLogHandle.into_resource(),
1577            ),
1578            (
1579                openhcl_vtl,
1580                get_resources::ged::GuestEmulationDeviceHandle {
1581                    firmware: if opt.pcat {
1582                        get_resources::ged::GuestFirmwareConfig::Pcat {
1583                            boot_order: opt
1584                                .pcat_boot_order
1585                                .map_or(DEFAULT_PCAT_BOOT_ORDER, |x| x.0)
1586                                .map(|x| match x {
1587                                    openvmm_defs::config::PcatBootDevice::Floppy => {
1588                                        get_resources::ged::PcatBootDevice::Floppy
1589                                    }
1590                                    openvmm_defs::config::PcatBootDevice::HardDrive => {
1591                                        get_resources::ged::PcatBootDevice::HardDrive
1592                                    }
1593                                    openvmm_defs::config::PcatBootDevice::Optical => {
1594                                        get_resources::ged::PcatBootDevice::Optical
1595                                    }
1596                                    openvmm_defs::config::PcatBootDevice::Network => {
1597                                        get_resources::ged::PcatBootDevice::Network
1598                                    }
1599                                }),
1600                        }
1601                    } else {
1602                        use get_resources::ged::UefiConsoleMode;
1603
1604                        get_resources::ged::GuestFirmwareConfig::Uefi {
1605                            enable_vpci_boot: has_vtl0_nvme,
1606                            firmware_debug: uefi_options.debug,
1607                            enable_memory_protections: uefi_options.enable_memory_protections,
1608                            disable_frontpage: uefi_options.disable_frontpage,
1609                            console_mode: match uefi_options.console.unwrap_or(UefiConsoleModeCli::Default) {
1610                                UefiConsoleModeCli::Default => UefiConsoleMode::Default,
1611                                UefiConsoleModeCli::Com1 => UefiConsoleMode::COM1,
1612                                UefiConsoleModeCli::Com2 => UefiConsoleMode::COM2,
1613                                UefiConsoleModeCli::None => UefiConsoleMode::None,
1614                            },
1615                            default_boot_always_attempt: uefi_options.default_boot_always_attempt,
1616                        }
1617                    },
1618                    com1: with_vmbus_com1_serial,
1619                    com2: with_vmbus_com2_serial,
1620                    serial_tx_only: opt.serial_tx_only,
1621                    vtl2_settings: Some(prost::Message::encode_to_vec(&vtl2_settings)),
1622                    vmbus_redirection: opt.vmbus_redirect,
1623                    vmgs,
1624                    framebuffer: opt
1625                        .vtl2_gfx
1626                        .then(|| SharedFramebufferHandle.into_resource()),
1627                    guest_request_recv,
1628                    tpm_version: tpm_version.map(|v| match v {
1629                        TpmVersion::V138 => get_resources::ged::GedTpmVersion::V138,
1630                        TpmVersion::V185 => get_resources::ged::GedTpmVersion::V185,
1631                    }),
1632                    firmware_event_send: None,
1633                    ipmi_sel_event_send: None,
1634                    secure_boot_enabled: opt.secure_boot,
1635                    secure_boot_template: match opt.secure_boot_template {
1636                        Some(SecureBootTemplateCli::Windows) => {
1637                            get_resources::ged::GuestSecureBootTemplateType::MicrosoftWindows
1638                        },
1639                        Some(SecureBootTemplateCli::UefiCa) => {
1640                            get_resources::ged::GuestSecureBootTemplateType::MicrosoftUefiCertificateAuthority
1641                        }
1642                        None => {
1643                            get_resources::ged::GuestSecureBootTemplateType::None
1644                        },
1645                    },
1646                    enable_battery: opt.battery,
1647                    enable_ipmi: false,
1648                    enable_hibernation: opt.hibernation,
1649                    no_persistent_secrets: true,
1650                    igvm_attest_test_config: None,
1651                    test_gsp_by_id: opt.test_gsp_by_id,
1652                    efi_diagnostics_log_level: {
1653                        match uefi_options.diagnostics.unwrap_or_default() {
1654                            EfiDiagnosticsLogLevelCli::Default => get_resources::ged::EfiDiagnosticsLogLevelType::Default,
1655                            EfiDiagnosticsLogLevelCli::Info => get_resources::ged::EfiDiagnosticsLogLevelType::Info,
1656                            EfiDiagnosticsLogLevelCli::Full => get_resources::ged::EfiDiagnosticsLogLevelType::Full,
1657                        }
1658                    },
1659                    force_dma_bounce_enabled: uefi_options.force_dma_bounce,
1660                    smbios: ged_smbios,
1661                }
1662                .into_resource(),
1663            ),
1664        ]);
1665    }
1666
1667    if let Some(tpm_version) = tpm_version
1668        && !opt.vtl2
1669    {
1670        let register_layout = if cfg!(guest_arch = "x86_64") {
1671            TpmRegisterLayout::IoPort
1672        } else {
1673            TpmRegisterLayout::Mmio
1674        };
1675
1676        let (ppi_store, nvram_store) = if opt.vmgs.is_some() {
1677            (
1678                VmgsFileHandle::new(vmgs_format::FileId::TPM_PPI, true).into_resource(),
1679                VmgsFileHandle::new(tpm_vmgs::tpm_nvram_file_id(tpm_version), true).into_resource(),
1680            )
1681        } else {
1682            (
1683                EphemeralNonVolatileStoreHandle.into_resource(),
1684                EphemeralNonVolatileStoreHandle.into_resource(),
1685            )
1686        };
1687
1688        chipset_devices.push(ChipsetDeviceHandle {
1689            name: "tpm".to_string(),
1690            resource: chipset_device_worker_defs::RemoteChipsetDeviceHandle {
1691                device: TpmDeviceHandle {
1692                    version: tpm_version,
1693                    ppi_store,
1694                    nvram_store,
1695                    nvram_size: None,
1696                    refresh_tpm_seeds: false,
1697                    ak_cert_type: tpm_resources::TpmAkCertTypeResource::None,
1698                    register_layout,
1699                    guest_secret_key: None,
1700                    logger: None,
1701                    is_confidential_vm: false,
1702                    bios_guid,
1703                }
1704                .into_resource(),
1705                worker_host: mesh.make_host("tpm", None).await?,
1706            }
1707            .into_resource(),
1708        });
1709    }
1710
1711    let vga_firmware = if opt.pcat {
1712        Some(openvmm_pcat_locator::find_svga_bios(
1713            opt.vga_firmware.as_deref(),
1714        )?)
1715    } else {
1716        None
1717    };
1718
1719    if opt.gfx {
1720        // Channel for the video device to report dirty rectangles to the VNC worker.
1721        let (dirt_send, dirt_recv) = mesh::channel();
1722        resources.dirty_rect_recv = Some(dirt_recv);
1723
1724        vmbus_devices.extend([
1725            (
1726                DeviceVtl::Vtl0,
1727                SynthVideoHandle {
1728                    framebuffer: SharedFramebufferHandle.into_resource(),
1729                    dirt_send: Some(dirt_send),
1730                }
1731                .into_resource(),
1732            ),
1733            (
1734                DeviceVtl::Vtl0,
1735                SynthKeyboardHandle {
1736                    source: MultiplexedInputHandle {
1737                        // Save 0 for PS/2
1738                        elevation: 1,
1739                    }
1740                    .into_resource(),
1741                }
1742                .into_resource(),
1743            ),
1744            (
1745                DeviceVtl::Vtl0,
1746                SynthMouseHandle {
1747                    source: MultiplexedInputHandle {
1748                        // Save 0 for PS/2
1749                        elevation: 1,
1750                    }
1751                    .into_resource(),
1752                }
1753                .into_resource(),
1754            ),
1755        ]);
1756    }
1757
1758    let vsock_listener = |path: Option<&str>| -> anyhow::Result<_> {
1759        if let Some(path) = path {
1760            cleanup_socket(path.as_ref());
1761            let listener = unix_socket::UnixListener::bind(path)
1762                .with_context(|| format!("failed to bind to hybrid vsock path: {}", path))?;
1763            Ok(Some(listener))
1764        } else {
1765            Ok(None)
1766        }
1767    };
1768
1769    let vtl0_vsock_listener = vsock_listener(opt.vmbus_vsock_path.as_deref())?;
1770    let vtl2_vsock_listener = vsock_listener(opt.vmbus_vtl2_vsock_path.as_deref())?;
1771
1772    if let Some(path) = &opt.openhcl_dump_path {
1773        let (resource, task) = spawn_dump_handler(&spawner, path.clone(), None);
1774        task.detach();
1775        vmbus_devices.push((openhcl_vtl, resource));
1776    }
1777
1778    #[cfg(guest_arch = "aarch64")]
1779    let topology_arch = openvmm_defs::config::ArchTopologyConfig::Aarch64(
1780        openvmm_defs::config::Aarch64TopologyConfig {
1781            // TODO: allow this to be configured from the command line
1782            gic_config: None,
1783            pmu_gsiv: openvmm_defs::config::PmuGsivConfig::Platform,
1784            gic_msi: match opt.gic_msi {
1785                cli_args::GicMsiCli::Auto => openvmm_defs::config::GicMsiConfig::Auto,
1786                cli_args::GicMsiCli::Its => openvmm_defs::config::GicMsiConfig::Its,
1787                cli_args::GicMsiCli::V2m => {
1788                    openvmm_defs::config::GicMsiConfig::V2m { spi_count: None }
1789                }
1790            },
1791        },
1792    );
1793    #[cfg(guest_arch = "x86_64")]
1794    let topology_arch =
1795        openvmm_defs::config::ArchTopologyConfig::X86(openvmm_defs::config::X86TopologyConfig {
1796            apic_id_offset: opt.apic_id_offset,
1797            x2apic: opt.x2apic,
1798        });
1799
1800    let with_isolation = if let Some(isolation) = &opt.isolation {
1801        match isolation {
1802            cli_args::IsolationCli::Vbs => {
1803                // TODO: For now, VBS isolation is only supported with VTL2.
1804                if !opt.vtl2 {
1805                    anyhow::bail!("VBS isolation is only currently supported with vtl2");
1806                }
1807
1808                // TODO: Alias map support is not yet implemented with isolation.
1809                if !opt.no_alias_map {
1810                    anyhow::bail!("alias map not supported with isolation");
1811                }
1812
1813                Some(openvmm_defs::config::IsolationType::Vbs)
1814            }
1815            cli_args::IsolationCli::Snp => Some(openvmm_defs::config::IsolationType::Snp {
1816                // SNP host data is currently only configurable via TTRPC.
1817                host_data: None,
1818            }),
1819        }
1820    } else {
1821        None
1822    };
1823
1824    if with_hv && !opt.no_vmbus {
1825        let (shutdown_send, shutdown_recv) = mesh::channel();
1826        resources.shutdown_ic = Some(shutdown_send);
1827        let (kvp_send, kvp_recv) = mesh::channel();
1828        resources.kvp_ic = Some(kvp_send);
1829        vmbus_devices.extend(
1830            [
1831                hyperv_ic_resources::shutdown::ShutdownIcHandle {
1832                    recv: shutdown_recv,
1833                }
1834                .into_resource(),
1835                hyperv_ic_resources::kvp::KvpIcHandle { recv: kvp_recv }.into_resource(),
1836                hyperv_ic_resources::timesync::TimesyncIcHandle.into_resource(),
1837            ]
1838            .map(|r| (DeviceVtl::Vtl0, r)),
1839        );
1840    }
1841
1842    if let Some(hive_path) = &opt.imc {
1843        let file = fs_err::File::open(hive_path).context("failed to open imc hive")?;
1844        vmbus_devices.push((
1845            DeviceVtl::Vtl0,
1846            vmbfs_resources::VmbfsImcDeviceHandle { file: file.into() }.into_resource(),
1847        ));
1848    }
1849
1850    let mut virtio_devices = Vec::new();
1851    let mut add_virtio_device =
1852        |bus, resource: Resource<VirtioDeviceHandle>, pcie_devices: &mut Vec<_>| match bus {
1853            VirtioBusCli::Auto => {
1854                // Use VPCI when possible (currently only on Windows and macOS due
1855                // to KVM backend limitations).
1856                if with_hv && (cfg!(windows) || cfg!(target_os = "macos")) {
1857                    vpci_devices.push(VpciDeviceConfig {
1858                        vtl: DeviceVtl::Vtl0,
1859                        instance_id: Guid::new_random(),
1860                        resource: VirtioPciDeviceHandle(resource).into_resource(),
1861                        vnode: None,
1862                    });
1863                } else {
1864                    virtio_devices.push((VirtioBus::Pci, resource));
1865                }
1866            }
1867            VirtioBusCli::Mmio => virtio_devices.push((VirtioBus::Mmio, resource)),
1868            VirtioBusCli::Pci => virtio_devices.push((VirtioBus::Pci, resource)),
1869            VirtioBusCli::Pcie(port_name) => pcie_devices.push(PcieDeviceConfig {
1870                port_name,
1871                resource: VirtioPciDeviceHandle(resource).into_resource(),
1872            }),
1873            VirtioBusCli::Vpci => vpci_devices.push(VpciDeviceConfig {
1874                vtl: DeviceVtl::Vtl0,
1875                instance_id: Guid::new_random(),
1876                resource: VirtioPciDeviceHandle(resource).into_resource(),
1877                vnode: None,
1878            }),
1879        };
1880
1881    for cli_cfg in &opt.virtio_net {
1882        if cli_cfg.underhill {
1883            anyhow::bail!("use --net uh:[...] to add underhill NICs")
1884        }
1885        let vport = parse_endpoint(cli_cfg, &mut nic_index, &mut resources)?;
1886        let resource = virtio_resources::net::VirtioNetHandle {
1887            max_queues: vport.max_queues,
1888            mac_address: vport.mac_address,
1889            endpoint: vport.endpoint,
1890        }
1891        .into_resource();
1892        if let Some(pcie_port) = &cli_cfg.pcie_port {
1893            pcie_devices.push(PcieDeviceConfig {
1894                port_name: pcie_port.clone(),
1895                resource: VirtioPciDeviceHandle(resource).into_resource(),
1896            });
1897        } else {
1898            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1899        }
1900    }
1901
1902    for args in &opt.virtio_fs {
1903        let resource: Resource<VirtioDeviceHandle> = virtio_resources::fs::VirtioFsHandle {
1904            tag: args.tag.clone(),
1905            fs: virtio_resources::fs::VirtioFsBackend::HostFs {
1906                root_path: args.path.clone(),
1907                mount_options: args.options.clone(),
1908            },
1909        }
1910        .into_resource();
1911        if let Some(pcie_port) = &args.pcie_port {
1912            pcie_devices.push(PcieDeviceConfig {
1913                port_name: pcie_port.clone(),
1914                resource: VirtioPciDeviceHandle(resource).into_resource(),
1915            });
1916        } else {
1917            add_virtio_device(opt.virtio_fs_bus.clone(), resource, &mut pcie_devices);
1918        }
1919    }
1920
1921    for args in &opt.virtio_fs_shmem {
1922        let resource: Resource<VirtioDeviceHandle> = virtio_resources::fs::VirtioFsHandle {
1923            tag: args.tag.clone(),
1924            fs: virtio_resources::fs::VirtioFsBackend::SectionFs {
1925                root_path: args.path.clone(),
1926            },
1927        }
1928        .into_resource();
1929        if let Some(pcie_port) = &args.pcie_port {
1930            pcie_devices.push(PcieDeviceConfig {
1931                port_name: pcie_port.clone(),
1932                resource: VirtioPciDeviceHandle(resource).into_resource(),
1933            });
1934        } else {
1935            add_virtio_device(opt.virtio_fs_bus.clone(), resource, &mut pcie_devices);
1936        }
1937    }
1938
1939    for args in &opt.virtio_9p {
1940        let resource: Resource<VirtioDeviceHandle> = virtio_resources::p9::VirtioPlan9Handle {
1941            tag: args.tag.clone(),
1942            root_path: args.path.clone(),
1943            debug: opt.virtio_9p_debug,
1944        }
1945        .into_resource();
1946        if let Some(pcie_port) = &args.pcie_port {
1947            pcie_devices.push(PcieDeviceConfig {
1948                port_name: pcie_port.clone(),
1949                resource: VirtioPciDeviceHandle(resource).into_resource(),
1950            });
1951        } else {
1952            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1953        }
1954    }
1955
1956    if let Some(pmem_args) = &opt.virtio_pmem {
1957        let resource: Resource<VirtioDeviceHandle> = virtio_resources::pmem::VirtioPmemHandle {
1958            path: pmem_args.path.clone(),
1959        }
1960        .into_resource();
1961        if let Some(pcie_port) = &pmem_args.pcie_port {
1962            pcie_devices.push(PcieDeviceConfig {
1963                port_name: pcie_port.clone(),
1964                resource: VirtioPciDeviceHandle(resource).into_resource(),
1965            });
1966        } else {
1967            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1968        }
1969    }
1970
1971    if opt.virtio_rng {
1972        let resource: Resource<VirtioDeviceHandle> =
1973            virtio_resources::rng::VirtioRngHandle.into_resource();
1974        if let Some(pcie_port) = &opt.virtio_rng_pcie_port {
1975            pcie_devices.push(PcieDeviceConfig {
1976                port_name: pcie_port.clone(),
1977                resource: VirtioPciDeviceHandle(resource).into_resource(),
1978            });
1979        } else {
1980            add_virtio_device(opt.virtio_rng_bus.clone(), resource, &mut pcie_devices);
1981        }
1982    }
1983
1984    if let Some(backend) = virtio_console_backend {
1985        let resource: Resource<VirtioDeviceHandle> =
1986            virtio_resources::console::VirtioConsoleHandle { backend }.into_resource();
1987        if let Some(pcie_port) = &opt.virtio_console_pcie_port {
1988            pcie_devices.push(PcieDeviceConfig {
1989                port_name: pcie_port.clone(),
1990                resource: VirtioPciDeviceHandle(resource).into_resource(),
1991            });
1992        } else {
1993            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1994        }
1995    }
1996
1997    // Handle --vhost-user arguments.
1998    #[cfg(target_os = "linux")]
1999    for vhost_cli in &opt.vhost_user {
2000        let stream =
2001            unix_socket::UnixStream::connect(&vhost_cli.socket_path).with_context(|| {
2002                format!(
2003                    "failed to connect to vhost-user socket: {}",
2004                    vhost_cli.socket_path
2005                )
2006            })?;
2007
2008        use crate::cli_args::VhostUserDeviceTypeCli;
2009        let resource: Resource<VirtioDeviceHandle> = match vhost_cli.device_type {
2010            VhostUserDeviceTypeCli::Fs {
2011                ref tag,
2012                num_queues,
2013                queue_size,
2014            } => virtio_resources::vhost_user::VhostUserFsHandle {
2015                socket: stream.into(),
2016                tag: tag.clone(),
2017                num_queues,
2018                queue_size,
2019            }
2020            .into_resource(),
2021            VhostUserDeviceTypeCli::Blk {
2022                num_queues,
2023                queue_size,
2024            } => virtio_resources::vhost_user::VhostUserBlkHandle {
2025                socket: stream.into(),
2026                num_queues,
2027                queue_size,
2028            }
2029            .into_resource(),
2030            VhostUserDeviceTypeCli::Other {
2031                device_id,
2032                ref queue_sizes,
2033            } => virtio_resources::vhost_user::VhostUserGenericHandle {
2034                socket: stream.into(),
2035                device_id,
2036                queue_sizes: queue_sizes.clone(),
2037            }
2038            .into_resource(),
2039        };
2040        if let Some(pcie_port) = &vhost_cli.pcie_port {
2041            pcie_devices.push(PcieDeviceConfig {
2042                port_name: pcie_port.clone(),
2043                resource: VirtioPciDeviceHandle(resource).into_resource(),
2044            });
2045        } else {
2046            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
2047        }
2048    }
2049
2050    let virtio_vsock_bus = opt.virtio_vsock_bus.clone().unwrap_or(VirtioBusCli::Auto);
2051
2052    if let Some(vsock_path) = &opt.virtio_vsock_path {
2053        let listener = vsock_listener(Some(vsock_path))?.unwrap();
2054        let resource: Resource<VirtioDeviceHandle> = virtio_resources::vsock::VirtioVsockHandle {
2055            // The guest CID does not matter since the UDS relay does not use it. It just needs
2056            // to be some non-reserved value for the guest to use.
2057            guest_cid: 0x3,
2058            base_path: vsock_path.clone(),
2059            listener,
2060        }
2061        .into_resource();
2062        add_virtio_device(virtio_vsock_bus.clone(), resource, &mut pcie_devices);
2063    }
2064
2065    #[cfg(target_os = "linux")]
2066    if let Some(guest_cid) = opt.virtio_vsock_vhost_cid {
2067        let vhost = std::fs::OpenOptions::new()
2068            .read(true)
2069            .write(true)
2070            .open("/dev/vhost-vsock")
2071            .context("failed to open /dev/vhost-vsock")?
2072            .into();
2073        let resource =
2074            virtio_resources::vsock::VirtioVsockVhostHandle { vhost, guest_cid }.into_resource();
2075        add_virtio_device(virtio_vsock_bus, resource, &mut pcie_devices);
2076    }
2077
2078    #[cfg(target_os = "linux")]
2079    pcie_devices.extend(vfio_pcie_devices);
2080
2081    let mut cfg = Config {
2082        chipset,
2083        load_mode,
2084        floppy_disks,
2085        pcie_root_complexes,
2086        pcie_ecam_below_4gb: opt.pcie_ecam_below_4gb,
2087        #[cfg(target_os = "linux")]
2088        pcie_devices,
2089        #[cfg(not(target_os = "linux"))]
2090        pcie_devices,
2091        pcie_switches,
2092        pcie_generic_initiators,
2093        vpci_devices,
2094        ide_disks: Vec::new(),
2095        numa: {
2096            if let Some(ref nodes) = opt.numa {
2097                // --numa mode: each --numa flag defines a node.
2098                NumaTopology {
2099                    nodes: nodes
2100                        .iter()
2101                        .map(|n| {
2102                            let vps = match &n.vps {
2103                                Some(vps) if vps.0.is_empty() => VpAssignment::Empty,
2104                                Some(vps) => {
2105                                    VpAssignment::Explicit(vps.expand_below(opt.processors)?)
2106                                }
2107                                None => VpAssignment::FromTopology,
2108                            };
2109                            Ok(NumaNode {
2110                                mem: Some(MemoryConfig {
2111                                    mem_size: n
2112                                        .memory
2113                                        .size
2114                                        .expect("NUMA memory size was validated")
2115                                        .0,
2116                                    prefetch_memory: n.memory.prefetch,
2117                                    private_memory: n.memory.shared == Some(false),
2118                                    transparent_hugepages: n
2119                                        .memory
2120                                        .transparent_hugepages
2121                                        .unwrap_or(!n.memory.hugepages),
2122                                    hugepages: n.memory.hugepages,
2123                                    hugepage_size: n.memory.hugepage_size.map(|m| m.0),
2124                                    host_numa_node: n.host_numa_node,
2125                                }),
2126                                vps,
2127                            })
2128                        })
2129                        .collect::<anyhow::Result<Vec<_>>>()?,
2130                    distances: opt
2131                        .numa_distance
2132                        .as_deref()
2133                        .unwrap_or(&[])
2134                        .iter()
2135                        .map(|d| NumaDistance {
2136                            src: d.src,
2137                            dst: d.dst,
2138                            distance: d.distance,
2139                        })
2140                        .collect(),
2141                }
2142            } else {
2143                // Single-node default from --memory.
2144                NumaTopology {
2145                    nodes: vec![NumaNode {
2146                        mem: Some(MemoryConfig {
2147                            mem_size: opt.memory_size(),
2148                            prefetch_memory: opt.prefetch_memory(),
2149                            private_memory: opt.private_memory(),
2150                            transparent_hugepages: opt.transparent_hugepages(),
2151                            hugepages: opt.memory.hugepages,
2152                            hugepage_size: opt.memory.hugepage_size.map(|m| m.0),
2153                            host_numa_node: None,
2154                        }),
2155                        vps: VpAssignment::FromTopology,
2156                    }],
2157                    distances: vec![],
2158                }
2159            }
2160        },
2161        processor_topology: ProcessorTopologyConfig {
2162            proc_count: opt.processors,
2163            vps_per_socket: opt.vps_per_socket,
2164            enable_smt: match opt.smt {
2165                cli_args::SmtConfigCli::Auto => None,
2166                cli_args::SmtConfigCli::Force => Some(true),
2167                cli_args::SmtConfigCli::Off => Some(false),
2168            },
2169            arch: Some(topology_arch),
2170        },
2171        hypervisor: HypervisorConfig {
2172            with_hv,
2173            with_vtl2: opt.vtl2.then_some(Vtl2Config {
2174                vtl0_alias_map: !opt.no_alias_map,
2175                late_map_vtl0_memory: match opt.late_map_vtl0_policy {
2176                    cli_args::Vtl0LateMapPolicyCli::Off => None,
2177                    cli_args::Vtl0LateMapPolicyCli::Log => Some(LateMapVtl0MemoryPolicy::Log),
2178                    cli_args::Vtl0LateMapPolicyCli::Halt => Some(LateMapVtl0MemoryPolicy::Halt),
2179                    cli_args::Vtl0LateMapPolicyCli::Exception => {
2180                        Some(LateMapVtl0MemoryPolicy::InjectException)
2181                    }
2182                },
2183            }),
2184            with_isolation,
2185            nested_virt: opt.nested_virt,
2186        },
2187        #[cfg(windows)]
2188        kernel_vmnics,
2189        input: mesh::Receiver::new(),
2190        framebuffer,
2191        vga_firmware,
2192        vtl2_gfx: opt.vtl2_gfx,
2193        virtio_devices,
2194        vmbus: (with_hv && !opt.no_vmbus).then_some(VmbusConfig {
2195            vsock_listener: vtl0_vsock_listener,
2196            vsock_path: opt.vmbus_vsock_path.clone(),
2197            vtl2_redirect: opt.vmbus_redirect,
2198            vmbus_max_version: opt.vmbus_max_version,
2199            #[cfg(windows)]
2200            vmbusproxy_handle,
2201        }),
2202        vtl2_vmbus: (with_hv && opt.vtl2).then_some(VmbusConfig {
2203            vsock_listener: vtl2_vsock_listener,
2204            vsock_path: opt.vmbus_vtl2_vsock_path.clone(),
2205            ..Default::default()
2206        }),
2207        vmbus_devices,
2208        chipset_devices,
2209        pci_chipset_devices,
2210        isa_dma_controller,
2211        chipset_capabilities: capabilities,
2212        layout: layout_config,
2213        #[cfg(windows)]
2214        vpci_resources,
2215        vmgs,
2216        firmware_event_send: None,
2217        debugger_rpc: None,
2218        rtc_delta_milliseconds: 0,
2219    };
2220
2221    storage.build_config(&mut cfg, &mut resources, opt.scsi_sub_channels)?;
2222    let mut pcie_port_names = HashSet::new();
2223    for device in &cfg.pcie_devices {
2224        anyhow::ensure!(
2225            pcie_port_names.insert(&device.port_name),
2226            "multiple devices use PCIe port '{}'",
2227            device.port_name
2228        );
2229    }
2230    resources.serial_driver = Some(serial_driver);
2231    validate_snp_config(&cfg)?;
2232    Ok((cfg, resources))
2233}
2234
2235fn validate_snp_config(cfg: &Config) -> anyhow::Result<()> {
2236    if !matches!(
2237        cfg.hypervisor.with_isolation,
2238        Some(openvmm_defs::config::IsolationType::Snp { .. })
2239    ) {
2240        return Ok(());
2241    }
2242
2243    if !matches!(
2244        cfg.load_mode,
2245        LoadMode::Linux { .. } | LoadMode::Igvm { .. }
2246    ) {
2247        anyhow::bail!("SNP isolation currently only supports Linux direct or IGVM boot");
2248    }
2249    if cfg.hypervisor.with_vtl2.is_some() {
2250        anyhow::bail!("SNP isolation currently does not support VTL2");
2251    }
2252    if cfg.vmbus.is_some() || cfg.vtl2_vmbus.is_some() || !cfg.vmbus_devices.is_empty() {
2253        anyhow::bail!("SNP isolation currently does not support VMBus devices");
2254    }
2255
2256    let only_supported_chipset_devices = cfg.chipset_devices.iter().all(|device| {
2257        matches!(
2258            device.resource.id(),
2259            "serial_16550"
2260                | "pic"
2261                | "pit"
2262                | "generic-ioapic"
2263                | "hyperv_power_management"
2264                | "missing-dev"
2265        )
2266    });
2267    let only_virtio_pcie_devices = cfg
2268        .pcie_devices
2269        .iter()
2270        .all(|device| device.resource.id() == "virtio");
2271    if !cfg.floppy_disks.is_empty()
2272        || !cfg.ide_disks.is_empty()
2273        || !cfg.virtio_devices.is_empty()
2274        || !only_virtio_pcie_devices
2275        || !cfg.vpci_devices.is_empty()
2276        || !only_supported_chipset_devices
2277        || !cfg.pci_chipset_devices.is_empty()
2278    {
2279        anyhow::bail!("SNP isolation currently only supports virtio devices");
2280    }
2281    if cfg.framebuffer.is_some() || cfg.vga_firmware.is_some() || cfg.debugger_rpc.is_some() {
2282        anyhow::bail!("SNP isolation currently does not support this VM configuration");
2283    }
2284
2285    Ok(())
2286}
2287
2288/// Gets the terminal to use for externally launched console windows.
2289pub(crate) fn openvmm_terminal_app() -> Option<PathBuf> {
2290    std::env::var_os("OPENVMM_TERM")
2291        .or_else(|| std::env::var_os("HVLITE_TERM"))
2292        .map(Into::into)
2293}
2294
2295// Tries to remove `path` if it is confirmed to be a Unix socket.
2296fn cleanup_socket(path: &Path) {
2297    #[cfg(windows)]
2298    let is_socket = pal::windows::fs::is_unix_socket(path).unwrap_or(false);
2299    #[cfg(not(windows))]
2300    let is_socket = path
2301        .metadata()
2302        .is_ok_and(|meta| std::os::unix::fs::FileTypeExt::is_socket(&meta.file_type()));
2303
2304    if is_socket {
2305        let _ = std::fs::remove_file(path);
2306    }
2307}
2308
2309#[cfg(windows)]
2310fn new_switch_port(
2311    switch_id: Option<&str>,
2312) -> anyhow::Result<(
2313    openvmm_defs::config::SwitchPortId,
2314    vmswitch::kernel::SwitchPort,
2315)> {
2316    let id = vmswitch::kernel::SwitchPortId {
2317        switch: match switch_id {
2318            Some(s) => s.parse().context("invalid switch id")?,
2319            None => vmswitch::hcn::DEFAULT_SWITCH,
2320        },
2321        port: Guid::new_random(),
2322    };
2323    let _ = vmswitch::hcn::Network::open(&id.switch)
2324        .with_context(|| format!("could not find switch {}", id.switch))?;
2325
2326    let port = vmswitch::kernel::SwitchPort::new(&id).context("failed to create switch port")?;
2327
2328    let id = openvmm_defs::config::SwitchPortId {
2329        switch: id.switch,
2330        port: id.port,
2331    };
2332    Ok((id, port))
2333}
2334
2335fn parse_endpoint(
2336    cli_cfg: &NicConfigCli,
2337    index: &mut usize,
2338    resources: &mut VmResources,
2339) -> anyhow::Result<NicConfig> {
2340    let _ = resources;
2341    let endpoint = match &cli_cfg.endpoint {
2342        EndpointConfigCli::Consomme { cidr, host_fwd } => {
2343            let ports = host_fwd
2344                .iter()
2345                .map(|fwd| {
2346                    use net_backend_resources::consomme::HostPortProtocol;
2347                    net_backend_resources::consomme::HostPortConfig {
2348                        protocol: match fwd.protocol {
2349                            cli_args::HostPortProtocolCli::Tcp => HostPortProtocol::Tcp,
2350                            cli_args::HostPortProtocolCli::Udp => HostPortProtocol::Udp,
2351                        },
2352                        host_address: fwd
2353                            .host_address
2354                            .map(net_backend_resources::consomme::HostIpAddress::from),
2355                        host_port: net_backend_resources::consomme::HostPort::Fixed(fwd.host_port),
2356                        guest_port: fwd.guest_port,
2357                    }
2358                })
2359                .collect();
2360            // Only wire the bind/unbind RPC channel to the first consomme
2361            // endpoint. Additional consomme NICs work normally but cannot be
2362            // targeted by runtime bind/unbind commands.
2363            let recv = if resources.consomme_rpc.is_none() {
2364                let (send, recv) = mesh::channel();
2365                resources.consomme_rpc = Some(send);
2366                Some(recv)
2367            } else {
2368                None
2369            };
2370            net_backend_resources::consomme::ConsommeHandle {
2371                cidr: cidr.clone(),
2372                ports,
2373                recv,
2374            }
2375            .into_resource()
2376        }
2377        EndpointConfigCli::None => net_backend_resources::null::NullHandle.into_resource(),
2378        EndpointConfigCli::Dio { id } => {
2379            #[cfg(windows)]
2380            {
2381                let (port_id, port) = new_switch_port(id.as_deref())?;
2382                resources.switch_ports.push(port);
2383                net_backend_resources::dio::WindowsDirectIoHandle {
2384                    switch_port_id: net_backend_resources::dio::SwitchPortId {
2385                        switch: port_id.switch,
2386                        port: port_id.port,
2387                    },
2388                }
2389                .into_resource()
2390            }
2391
2392            #[cfg(not(windows))]
2393            {
2394                let _ = id;
2395                bail!("cannot use dio on non-windows platforms")
2396            }
2397        }
2398        EndpointConfigCli::Tap { name } => {
2399            #[cfg(target_os = "linux")]
2400            {
2401                let fd = net_tap::tap::open_tap(name)
2402                    .with_context(|| format!("failed to open TAP device '{name}'"))?;
2403                net_backend_resources::tap::TapHandle { fd }.into_resource()
2404            }
2405
2406            #[cfg(not(target_os = "linux"))]
2407            {
2408                let _ = name;
2409                bail!("TAP backend is only supported on Linux")
2410            }
2411        }
2412    };
2413
2414    // Pick a random MAC address.
2415    let mut mac_address = [0x00, 0x15, 0x5D, 0, 0, 0];
2416    getrandom::fill(&mut mac_address[3..]).expect("rng failure");
2417
2418    // Pick a fixed instance ID based on the index.
2419    const BASE_INSTANCE_ID: Guid = guid::guid!("00000000-da43-11ed-936a-00155d6db52f");
2420    let instance_id = Guid {
2421        data1: *index as u32,
2422        ..BASE_INSTANCE_ID
2423    };
2424    *index += 1;
2425
2426    Ok(NicConfig {
2427        vtl: cli_cfg.vtl,
2428        instance_id,
2429        endpoint,
2430        mac_address: mac_address.into(),
2431        max_queues: cli_cfg.max_queues,
2432        pcie_port: cli_cfg.pcie_port.clone(),
2433    })
2434}
2435
2436#[derive(Debug)]
2437struct NicConfig {
2438    vtl: DeviceVtl,
2439    instance_id: Guid,
2440    mac_address: MacAddress,
2441    endpoint: Resource<NetEndpointHandleKind>,
2442    max_queues: Option<u16>,
2443    pcie_port: Option<String>,
2444}
2445
2446impl NicConfig {
2447    fn into_netvsp_handle(self) -> (DeviceVtl, Resource<VmbusDeviceHandleKind>) {
2448        (
2449            self.vtl,
2450            netvsp_resources::NetvspHandle {
2451                instance_id: self.instance_id,
2452                mac_address: self.mac_address,
2453                endpoint: self.endpoint,
2454                max_queues: self.max_queues,
2455            }
2456            .into_resource(),
2457        )
2458    }
2459}
2460
2461enum LayerOrDisk {
2462    Layer(DiskLayerDescription),
2463    Disk(Resource<DiskHandleKind>),
2464}
2465
2466async fn disk_open(
2467    disk_cli: &DiskCliKind,
2468    read_only: bool,
2469) -> anyhow::Result<Resource<DiskHandleKind>> {
2470    let mut layers = Vec::new();
2471    disk_open_inner(disk_cli, read_only, &mut layers).await?;
2472    if layers.len() == 1 && matches!(layers[0], LayerOrDisk::Disk(_)) {
2473        let LayerOrDisk::Disk(disk) = layers.pop().unwrap() else {
2474            unreachable!()
2475        };
2476        Ok(disk)
2477    } else {
2478        Ok(Resource::new(disk_backend_resources::LayeredDiskHandle {
2479            layers: layers
2480                .into_iter()
2481                .map(|layer| match layer {
2482                    LayerOrDisk::Layer(layer) => layer,
2483                    LayerOrDisk::Disk(disk) => DiskLayerDescription {
2484                        layer: DiskLayerHandle(disk).into_resource(),
2485                        read_cache: false,
2486                        write_through: false,
2487                    },
2488                })
2489                .collect(),
2490        }))
2491    }
2492}
2493
2494fn disk_open_inner<'a>(
2495    disk_cli: &'a DiskCliKind,
2496    read_only: bool,
2497    layers: &'a mut Vec<LayerOrDisk>,
2498) -> futures::future::BoxFuture<'a, anyhow::Result<()>> {
2499    Box::pin(async move {
2500        fn layer<T: IntoResource<DiskLayerHandleKind>>(layer: T) -> LayerOrDisk {
2501            LayerOrDisk::Layer(layer.into_resource().into())
2502        }
2503        fn disk<T: IntoResource<DiskHandleKind>>(disk: T) -> LayerOrDisk {
2504            LayerOrDisk::Disk(disk.into_resource())
2505        }
2506        match disk_cli {
2507            &DiskCliKind::Memory(len) => {
2508                layers.push(layer(RamDiskLayerHandle {
2509                    len: Some(len),
2510                    sector_size: None,
2511                }));
2512            }
2513            DiskCliKind::File {
2514                path,
2515                create_with_len,
2516                direct,
2517            } => layers.push(LayerOrDisk::Disk(if let Some(size) = create_with_len {
2518                create_disk_type(
2519                    path,
2520                    *size,
2521                    OpenDiskOptions {
2522                        read_only: false,
2523                        direct: *direct,
2524                    },
2525                )
2526                .with_context(|| format!("failed to create {}", path.display()))?
2527            } else {
2528                open_disk_type(
2529                    path,
2530                    OpenDiskOptions {
2531                        read_only,
2532                        direct: *direct,
2533                    },
2534                )
2535                .await
2536                .with_context(|| format!("failed to open {}", path.display()))?
2537            })),
2538            DiskCliKind::Blob { kind, url } => {
2539                layers.push(disk(disk_backend_resources::BlobDiskHandle {
2540                    url: url.to_owned(),
2541                    format: match kind {
2542                        cli_args::BlobKind::Flat => disk_backend_resources::BlobDiskFormat::Flat,
2543                        cli_args::BlobKind::Vhd1 => {
2544                            disk_backend_resources::BlobDiskFormat::FixedVhd1
2545                        }
2546                    },
2547                }))
2548            }
2549            DiskCliKind::MemoryDiff(inner) => {
2550                layers.push(layer(RamDiskLayerHandle {
2551                    len: None,
2552                    sector_size: None,
2553                }));
2554                disk_open_inner(inner, true, layers).await?;
2555            }
2556            DiskCliKind::PersistentReservationsWrapper(inner) => {
2557                layers.push(disk(disk_backend_resources::DiskWithReservationsHandle(
2558                    disk_open(inner, read_only).await?,
2559                )))
2560            }
2561            DiskCliKind::DelayDiskWrapper {
2562                delay_ms,
2563                disk: inner,
2564            } => layers.push(disk(DelayDiskHandle {
2565                delay: CellUpdater::new(Duration::from_millis(*delay_ms)).cell(),
2566                disk: disk_open(inner, read_only).await?,
2567            })),
2568            DiskCliKind::Crypt {
2569                disk: inner,
2570                cipher,
2571                key_file,
2572            } => layers.push(disk(disk_crypt_resources::DiskCryptHandle {
2573                disk: disk_open(inner, read_only).await?,
2574                cipher: match cipher {
2575                    cli_args::DiskCipher::XtsAes256 => disk_crypt_resources::Cipher::XtsAes256,
2576                },
2577                key: fs_err::read(key_file).context("failed to read key file")?,
2578            })),
2579            DiskCliKind::Sqlite {
2580                path,
2581                create_with_len,
2582            } => {
2583                // FUTURE: this code should be responsible for opening
2584                // file-handle(s) itself, and passing them into sqlite via a custom
2585                // vfs. For now though - simply check if the file exists or not, and
2586                // perform early validation of filesystem-level create options.
2587                match (create_with_len.is_some(), path.exists()) {
2588                    (true, true) => anyhow::bail!(
2589                        "cannot create new sqlite disk at {} - file already exists",
2590                        path.display()
2591                    ),
2592                    (false, false) => anyhow::bail!(
2593                        "cannot open sqlite disk at {} - file not found",
2594                        path.display()
2595                    ),
2596                    _ => {}
2597                }
2598
2599                layers.push(layer(SqliteDiskLayerHandle {
2600                    dbhd_path: path.display().to_string(),
2601                    format_dbhd: create_with_len.map(|len| {
2602                        disk_backend_resources::layer::SqliteDiskLayerFormatParams {
2603                            logically_read_only: false,
2604                            len: Some(len),
2605                        }
2606                    }),
2607                }));
2608            }
2609            DiskCliKind::SqliteDiff { path, create, disk } => {
2610                // FUTURE: this code should be responsible for opening
2611                // file-handle(s) itself, and passing them into sqlite via a custom
2612                // vfs. For now though - simply check if the file exists or not, and
2613                // perform early validation of filesystem-level create options.
2614                match (create, path.exists()) {
2615                    (true, true) => anyhow::bail!(
2616                        "cannot create new sqlite disk at {} - file already exists",
2617                        path.display()
2618                    ),
2619                    (false, false) => anyhow::bail!(
2620                        "cannot open sqlite disk at {} - file not found",
2621                        path.display()
2622                    ),
2623                    _ => {}
2624                }
2625
2626                layers.push(layer(SqliteDiskLayerHandle {
2627                    dbhd_path: path.display().to_string(),
2628                    format_dbhd: create.then_some(
2629                        disk_backend_resources::layer::SqliteDiskLayerFormatParams {
2630                            logically_read_only: false,
2631                            len: None,
2632                        },
2633                    ),
2634                }));
2635                disk_open_inner(disk, true, layers).await?;
2636            }
2637            DiskCliKind::AutoCacheSqlite {
2638                cache_path,
2639                key,
2640                disk,
2641            } => {
2642                layers.push(LayerOrDisk::Layer(DiskLayerDescription {
2643                    read_cache: true,
2644                    write_through: false,
2645                    layer: SqliteAutoCacheDiskLayerHandle {
2646                        cache_path: cache_path.clone(),
2647                        cache_key: key.clone(),
2648                    }
2649                    .into_resource(),
2650                }));
2651                disk_open_inner(disk, read_only, layers).await?;
2652            }
2653        }
2654        Ok(())
2655    })
2656}
2657
2658/// Get the system page size.
2659pub(crate) fn system_page_size() -> u32 {
2660    sparse_mmap::SparseMapping::page_size() as u32
2661}
2662
2663/// The guest architecture string, derived from the compile-time `guest_arch` cfg.
2664pub(crate) const GUEST_ARCH: &str = if cfg!(guest_arch = "x86_64") {
2665    "x86_64"
2666} else {
2667    "aarch64"
2668};
2669
2670/// Open a snapshot directory and validate it against the current VM config.
2671/// Returns the shared memory fd (from memory.bin) and the saved device state.
2672fn prepare_snapshot_restore(
2673    snapshot_dir: &Path,
2674    opt: &Options,
2675) -> anyhow::Result<(
2676    openvmm_defs::worker::SharedMemoryFd,
2677    mesh::payload::message::ProtobufMessage,
2678)> {
2679    let (manifest, state_bytes) = openvmm_helpers::snapshot::read_snapshot(snapshot_dir)?;
2680
2681    // Validate manifest against current VM config.
2682    openvmm_helpers::snapshot::validate_manifest(
2683        &manifest,
2684        GUEST_ARCH,
2685        opt.memory_size(),
2686        opt.processors,
2687        system_page_size(),
2688    )?;
2689
2690    // Open memory.bin (existing file, no create, no resize).
2691    let memory_file = fs_err::OpenOptions::new()
2692        .read(true)
2693        .write(true)
2694        .open(snapshot_dir.join("memory.bin"))?;
2695
2696    // Validate file size matches expected memory size.
2697    let file_size = memory_file.metadata()?.len();
2698    if file_size != manifest.memory_size_bytes {
2699        anyhow::bail!(
2700            "memory.bin size ({file_size} bytes) doesn't match manifest ({} bytes)",
2701            manifest.memory_size_bytes,
2702        );
2703    }
2704
2705    let shared_memory_fd =
2706        openvmm_helpers::shared_memory::file_to_shared_memory_fd(memory_file.into())?;
2707
2708    // Reconstruct ProtobufMessage from the saved state bytes.
2709    // The save side wrote mesh::payload::encode(ProtobufMessage), so we decode
2710    // back to ProtobufMessage.
2711    let state_msg: mesh::payload::message::ProtobufMessage = mesh::payload::decode(&state_bytes)
2712        .context("failed to decode saved state from snapshot")?;
2713
2714    Ok((shared_memory_fd, state_msg))
2715}
2716
2717fn do_main(pidfile_guard: &mut Option<pidfile::Pidfile>) -> anyhow::Result<i32> {
2718    #[cfg(windows)]
2719    pal::windows::disable_hard_error_dialog();
2720
2721    tracing_init::enable_tracing()?;
2722
2723    // Try to run as a worker host.
2724    // On success the worker runs to completion and then exits the process (does
2725    // not return). Any worker host setup errors are return and bubbled up.
2726    meshworker::run_vmm_mesh_host()?;
2727
2728    let opt = cli_args::parse_options();
2729
2730    // Print the version number. This comes after argument parsing to not interfere
2731    // with --version and --help.
2732    tracing::info!(version = openvmm_build_info::get().version());
2733
2734    if let Some(path) = &opt.write_saved_state_proto {
2735        mesh::payload::protofile::DescriptorWriter::new(vmcore::save_restore::saved_state_roots())
2736            .write_to_path(path)
2737            .context("failed to write protobuf descriptors")?;
2738        return Ok(0);
2739    }
2740
2741    if let Some(ref path) = opt.pidfile {
2742        *pidfile_guard = Some(pidfile::Pidfile::new(path).context("failed to create pidfile")?);
2743    }
2744
2745    if let Some(path) = opt.relay_console_path {
2746        let console_title = opt.relay_console_title.unwrap_or_default();
2747        return console_relay::relay_console(&path, console_title.as_str()).map(|()| 0);
2748    }
2749
2750    #[cfg(any(feature = "grpc", feature = "ttrpc"))]
2751    {
2752        let rpc = opt
2753            .rpc
2754            .as_ref()
2755            .map(|rpc| {
2756                let transport = match rpc.transport {
2757                    cli_args::RpcTransportCli::Auto => ttrpc::RpcTransport::Auto,
2758                    cli_args::RpcTransportCli::Ttrpc => ttrpc::RpcTransport::Ttrpc,
2759                    cli_args::RpcTransportCli::Grpc => ttrpc::RpcTransport::Grpc,
2760                };
2761                (rpc.path.as_path(), transport)
2762            })
2763            .or_else(|| {
2764                opt.ttrpc
2765                    .as_deref()
2766                    .map(|p| (p, ttrpc::RpcTransport::Ttrpc))
2767            })
2768            .or_else(|| opt.grpc.as_deref().map(|p| (p, ttrpc::RpcTransport::Grpc)));
2769
2770        if let Some((path, transport)) = rpc {
2771            return block_on(async {
2772                let _ = std::fs::remove_file(path);
2773                let listener =
2774                    unix_socket::UnixListener::bind(path).context("failed to bind to socket")?;
2775
2776                // This is a local launch
2777                let mut handle =
2778                    mesh_worker::launch_local_worker::<ttrpc::TtrpcWorker>(ttrpc::Parameters {
2779                        listener,
2780                        transport,
2781                    })
2782                    .await?;
2783
2784                tracing::info!(%transport, path = %path.display(), "listening");
2785
2786                // Signal the parent process that the server is ready.
2787                pal::close_stdout().context("failed to close stdout")?;
2788
2789                handle.join().await?;
2790
2791                Ok(0)
2792            });
2793        }
2794    }
2795
2796    DefaultPool::run_with(async |driver| run_control(&driver, opt).await)
2797}
2798
2799fn new_hvsock_service_id(port: u32) -> Guid {
2800    // This GUID is an embedding of the AF_VSOCK port into an
2801    // AF_HYPERV service ID.
2802    Guid {
2803        data1: port,
2804        .."00000000-facb-11e6-bd58-64006a7986d3".parse().unwrap()
2805    }
2806}
2807
2808async fn run_control(driver: &DefaultDriver, opt: Options) -> anyhow::Result<i32> {
2809    let mut mesh = Some(VmmMesh::new(&driver, opt.single_process)?);
2810    let result = run_control_inner(driver, &mut mesh, opt).await;
2811    // If setup failed before the mesh was handed to the controller, shut it
2812    // down so the child host process exits cleanly without noisy logs.
2813    if let Some(mesh) = mesh {
2814        mesh.shutdown().await;
2815    }
2816    result
2817}
2818
2819async fn run_control_inner(
2820    driver: &DefaultDriver,
2821    mesh_slot: &mut Option<VmmMesh>,
2822    opt: Options,
2823) -> anyhow::Result<i32> {
2824    let mesh = mesh_slot.as_ref().unwrap();
2825    let (mut vm_config, mut resources) = vm_config_from_command_line(driver, mesh, &opt).await?;
2826
2827    let mut vnc_worker = None;
2828    if opt.gfx || opt.vnc.vnc {
2829        // Parse the listen address. Try as a full SocketAddr (host:port) first;
2830        // fall back to a bare IP, using the configured port.
2831        let addr: std::net::SocketAddr = if let Ok(sa) =
2832            opt.vnc.vnc_listen.parse::<std::net::SocketAddr>()
2833        {
2834            sa
2835        } else {
2836            let ip: std::net::IpAddr = opt.vnc.vnc_listen.parse().with_context(|| {
2837                format!(
2838                    "invalid VNC listen address: {} (expected IP address or socket address like [::1]:5900)",
2839                    opt.vnc.vnc_listen
2840                )
2841            })?;
2842            std::net::SocketAddr::new(ip, opt.vnc.vnc_port)
2843        };
2844
2845        let socket = socket2::Socket::new(
2846            if addr.is_ipv6() {
2847                socket2::Domain::IPV6
2848            } else {
2849                socket2::Domain::IPV4
2850            },
2851            socket2::Type::STREAM,
2852            None,
2853        )
2854        .with_context(|| format!("creating VNC socket for {}", addr))?;
2855
2856        if addr.is_ipv6() {
2857            if let Err(e) = socket.set_only_v6(false) {
2858                tracing::warn!(
2859                    error = %e,
2860                    "failed to enable dual-stack on IPv6 VNC socket, IPv4 clients may not be able to connect"
2861                );
2862            }
2863        }
2864        socket.set_reuse_address(true)?;
2865        socket
2866            .bind(&addr.into())
2867            .with_context(|| format!("binding VNC socket to {}", addr))?;
2868        socket
2869            .listen(128)
2870            .with_context(|| format!("listening on VNC socket {}", addr))?;
2871        let listener: TcpListener = socket.into();
2872
2873        if !addr.ip().is_loopback() {
2874            tracing::warn!(
2875                address = %addr,
2876                "VNC server listening on non-localhost address without authentication"
2877            );
2878        }
2879
2880        let input_send = vm_config.input.sender();
2881        let framebuffer = resources
2882            .framebuffer_access
2883            .take()
2884            .expect("synth video enabled");
2885
2886        let vnc_host = mesh
2887            .make_host("vnc", None)
2888            .await
2889            .context("spawning vnc process failed")?;
2890
2891        vnc_worker = Some(
2892            vnc_host
2893                .launch_worker(
2894                    vnc_worker_defs::VNC_WORKER_TCP,
2895                    VncParameters {
2896                        listener,
2897                        framebuffer,
2898                        input_send,
2899                        dirty_recv: resources.dirty_rect_recv.take(),
2900                        max_clients: opt.vnc.vnc_max_clients,
2901                        evict_oldest: opt.vnc.vnc_evict_oldest,
2902                    },
2903                )
2904                .await?,
2905        )
2906    }
2907
2908    // spin up the debug worker
2909    let gdb_worker = if let Some(port) = opt.gdb {
2910        let listener = TcpListener::bind(format!("127.0.0.1:{}", port))
2911            .with_context(|| format!("binding to gdb port {}", port))?;
2912
2913        let (req_tx, req_rx) = mesh::channel();
2914        vm_config.debugger_rpc = Some(req_rx);
2915
2916        let gdb_host = mesh
2917            .make_host("gdb", None)
2918            .await
2919            .context("spawning gdbstub process failed")?;
2920
2921        Some(
2922            gdb_host
2923                .launch_worker(
2924                    debug_worker_defs::DEBUGGER_WORKER,
2925                    debug_worker_defs::DebuggerParameters {
2926                        listener,
2927                        req_chan: req_tx,
2928                        vp_count: vm_config.processor_topology.proc_count,
2929                        target_arch: if cfg!(guest_arch = "x86_64") {
2930                            debug_worker_defs::TargetArch::X86_64
2931                        } else {
2932                            debug_worker_defs::TargetArch::Aarch64
2933                        },
2934                    },
2935                )
2936                .await
2937                .context("failed to launch gdbstub worker")?,
2938        )
2939    } else {
2940        None
2941    };
2942
2943    // spin up the VM
2944    let (vm_rpc, rpc_recv) = mesh::channel();
2945    let (notify_send, notify_recv) = mesh::channel();
2946    let vm_worker = {
2947        let vm_host = mesh.make_host("vm", opt.log_file.clone()).await?;
2948
2949        let (shared_memory, saved_state) = if let Some(snapshot_dir) = &opt.restore_snapshot {
2950            let (fd, state_msg) = prepare_snapshot_restore(snapshot_dir, &opt)?;
2951            (Some(fd), Some(state_msg))
2952        } else {
2953            let shared_memory = opt
2954                .memory_backing_file()
2955                .map(|path| {
2956                    openvmm_helpers::shared_memory::open_memory_backing_file(
2957                        path,
2958                        opt.memory_size(),
2959                    )
2960                })
2961                .transpose()?;
2962            (shared_memory, None)
2963        };
2964
2965        let params = VmWorkerParameters {
2966            hypervisor: match &opt.hypervisor {
2967                Some(name) => openvmm_helpers::hypervisor::hypervisor_resource(name)?,
2968                None => openvmm_helpers::hypervisor::choose_hypervisor()?,
2969            },
2970            cfg: vm_config,
2971            saved_state,
2972            shared_memory,
2973            rpc: rpc_recv,
2974            notify: notify_send,
2975        };
2976        vm_host
2977            .launch_worker(VM_WORKER, params)
2978            .await
2979            .context("failed to launch vm worker")?
2980    };
2981
2982    if opt.restore_snapshot.is_some() {
2983        tracing::info!("restoring VM from snapshot");
2984    }
2985
2986    if !opt.paused {
2987        vm_rpc.call(VmRpc::Resume, ()).await?;
2988    }
2989
2990    let paravisor_diag = Arc::new(diag_client::DiagClient::from_dialer(
2991        driver.clone(),
2992        DiagDialer {
2993            driver: driver.clone(),
2994            vm_rpc: vm_rpc.clone(),
2995            openhcl_vtl: if opt.vtl2 {
2996                DeviceVtl::Vtl2
2997            } else {
2998                DeviceVtl::Vtl0
2999            },
3000        },
3001    ));
3002
3003    let diag_inspector = DiagInspector::new(driver.clone(), paravisor_diag.clone());
3004
3005    // Create channels between the REPL and VmController.
3006    let (vm_controller_send, vm_controller_recv) = mesh::channel();
3007    let (vm_controller_event_send, vm_controller_event_recv) = mesh::channel();
3008
3009    let has_vtl2 = resources.vtl2_settings.is_some();
3010    let serial_driver = resources
3011        .serial_driver
3012        .take()
3013        .expect("serial driver must outlive serial resources");
3014
3015    // Build the VmController with exclusive resources.
3016    let controller = vm_controller::VmController {
3017        mesh: mesh_slot.take().unwrap(),
3018        vm_worker,
3019        vnc_worker,
3020        gdb_worker,
3021        diag_inspector: Some(diag_inspector),
3022        vtl2_settings: resources.vtl2_settings,
3023        ged_rpc: resources.ged_rpc.clone(),
3024        vm_rpc: vm_rpc.clone(),
3025        paravisor_diag: Some(paravisor_diag),
3026        igvm_path: opt.igvm.as_ref().map(|igvm| igvm.firmware.clone()),
3027        memory_backing_file: opt.memory_backing_file().cloned(),
3028        memory: opt.memory_size(),
3029        processors: opt.processors,
3030        log_file: opt.log_file.clone(),
3031        crash_dump_path: opt.crash_dump_path.clone(),
3032        guest_power_actions: vm_controller::GuestPowerActions {
3033            shutdown: opt.guest_shutdown_action,
3034            reset: opt.guest_reset_action,
3035            crash: opt.guest_crash_action,
3036            watchdog: opt.guest_watchdog_action,
3037        },
3038    };
3039
3040    // Spawn the VmController as a task.
3041    let controller_task = driver.spawn(
3042        "vm-controller",
3043        controller.run(vm_controller_recv, vm_controller_event_send, notify_recv),
3044    );
3045
3046    // Run the REPL with shareable resources.
3047    let repl_result = repl::run_repl(
3048        driver,
3049        repl::ReplResources {
3050            vm_rpc,
3051            vm_controller: vm_controller_send,
3052            vm_controller_events: vm_controller_event_recv,
3053            scsi_rpc: resources.scsi_rpc,
3054            nvme_vtl2_rpc: resources.nvme_vtl2_rpc,
3055            consomme_rpc: resources.consomme_rpc,
3056            shutdown_ic: resources.shutdown_ic,
3057            kvp_ic: resources.kvp_ic,
3058            console_in: resources.console_in,
3059            has_vtl2,
3060        },
3061    )
3062    .await;
3063
3064    // Wait for the controller task to finish (it stops the VM worker and
3065    // shuts down the mesh).
3066    controller_task.await;
3067    drop(serial_driver);
3068
3069    // run_repl returns the exit status: the code the guest drove via an opt-in
3070    // exit (VmControllerEvent::ExitRequested), or 0 when the VM stopped normally.
3071    repl_result
3072}
3073
3074struct DiagDialer {
3075    driver: DefaultDriver,
3076    vm_rpc: mesh::Sender<VmRpc>,
3077    openhcl_vtl: DeviceVtl,
3078}
3079
3080impl mesh_rpc::client::Dial for DiagDialer {
3081    type Stream = PolledSocket<unix_socket::UnixStream>;
3082
3083    async fn dial(&mut self) -> io::Result<Self::Stream> {
3084        let service_id = new_hvsock_service_id(1);
3085        let socket = self
3086            .vm_rpc
3087            .call_failable(
3088                VmRpc::ConnectHvsock,
3089                (
3090                    CancelContext::new().with_timeout(Duration::from_secs(2)),
3091                    service_id,
3092                    self.openhcl_vtl,
3093                ),
3094            )
3095            .await
3096            .map_err(io::Error::other)?;
3097
3098        PolledSocket::new(&self.driver, socket)
3099    }
3100}
3101
3102/// An object that implements [`InspectMut`] by sending an inspect request over
3103/// TTRPC to the guest (typically the paravisor running in VTL2), then stitching
3104/// the response back into the inspect tree.
3105///
3106/// This also caches the TTRPC connection to the guest so that only the first
3107/// inspect request has to wait for the connection to be established.
3108pub(crate) struct DiagInspector(DiagInspectorInner);
3109
3110enum DiagInspectorInner {
3111    NotStarted(DefaultDriver, Arc<diag_client::DiagClient>),
3112    Started {
3113        send: mesh::Sender<inspect::Deferred>,
3114        _task: Task<()>,
3115    },
3116    Invalid,
3117}
3118
3119impl DiagInspector {
3120    pub fn new(driver: DefaultDriver, diag_client: Arc<diag_client::DiagClient>) -> Self {
3121        Self(DiagInspectorInner::NotStarted(driver, diag_client))
3122    }
3123
3124    fn start(&mut self) -> &mesh::Sender<inspect::Deferred> {
3125        loop {
3126            match self.0 {
3127                DiagInspectorInner::NotStarted { .. } => {
3128                    let DiagInspectorInner::NotStarted(driver, client) =
3129                        std::mem::replace(&mut self.0, DiagInspectorInner::Invalid)
3130                    else {
3131                        unreachable!()
3132                    };
3133                    let (send, recv) = mesh::channel();
3134                    let task = driver.clone().spawn("diag-inspect", async move {
3135                        Self::run(&client, recv).await
3136                    });
3137
3138                    self.0 = DiagInspectorInner::Started { send, _task: task };
3139                }
3140                DiagInspectorInner::Started { ref send, .. } => break send,
3141                DiagInspectorInner::Invalid => unreachable!(),
3142            }
3143        }
3144    }
3145
3146    async fn run(
3147        diag_client: &diag_client::DiagClient,
3148        mut recv: mesh::Receiver<inspect::Deferred>,
3149    ) {
3150        while let Some(deferred) = recv.next().await {
3151            let info = deferred.external_request();
3152            let result = match info.request_type {
3153                inspect::ExternalRequestType::Inspect { depth } => {
3154                    if depth == 0 {
3155                        Ok(inspect::Node::Unevaluated)
3156                    } else {
3157                        // TODO: Support taking timeouts from the command line
3158                        diag_client
3159                            .inspect(info.path, Some(depth - 1), Some(Duration::from_secs(1)))
3160                            .await
3161                    }
3162                }
3163                inspect::ExternalRequestType::Update { value } => {
3164                    (diag_client.update(info.path, value).await).map(inspect::Node::Value)
3165                }
3166            };
3167            deferred.complete_external(
3168                result.unwrap_or_else(|err| {
3169                    inspect::Node::Failed(inspect::Error::Mesh(format!("{err:#}")))
3170                }),
3171                inspect::SensitivityLevel::Unspecified,
3172            )
3173        }
3174    }
3175}
3176
3177impl InspectMut for DiagInspector {
3178    fn inspect_mut(&mut self, req: inspect::Request<'_>) {
3179        self.start().send(req.defer());
3180    }
3181}
3182
3183#[cfg(test)]
3184mod tests {
3185    use super::*;
3186    use clap::Parser;
3187    use std::fs::File;
3188    use test_with_tracing::test;
3189
3190    #[test]
3191    fn maps_igvm_personalities_to_chipsets() {
3192        for (args, expected) in [
3193            (
3194                vec!["openvmm", "--igvm", "firmware=guest.igvm,personality=uefi"],
3195                BaseChipsetType::HypervGen2Uefi,
3196            ),
3197            (
3198                vec![
3199                    "openvmm",
3200                    "--igvm",
3201                    "firmware=guest.igvm,personality=linux-direct",
3202                ],
3203                BaseChipsetType::UnenlightenedLinuxDirect,
3204            ),
3205            (
3206                vec![
3207                    "openvmm",
3208                    "--igvm",
3209                    "firmware=guest.igvm,personality=linux-direct",
3210                    "--hv",
3211                ],
3212                BaseChipsetType::HyperVGen2LinuxDirect,
3213            ),
3214            (
3215                vec![
3216                    "openvmm",
3217                    "--igvm",
3218                    "firmware=guest.igvm,personality=linux-direct",
3219                    "--isolation",
3220                    "snp",
3221                ],
3222                BaseChipsetType::EnlightenedLinuxDirect,
3223            ),
3224            (
3225                vec![
3226                    "openvmm",
3227                    "--igvm",
3228                    "firmware=guest.igvm,personality=openhcl",
3229                    "--hv",
3230                    "--vtl2",
3231                ],
3232                BaseChipsetType::HclHost,
3233            ),
3234        ] {
3235            let opt = Options::try_parse_from(args).unwrap();
3236            opt.validate_igvm_options().unwrap();
3237            assert!(
3238                std::mem::discriminant(&base_chipset_type(&opt))
3239                    == std::mem::discriminant(&expected)
3240            );
3241        }
3242    }
3243
3244    #[test]
3245    fn builds_openhcl_igvm_config_without_host_uefi() {
3246        DefaultPool::run_with(async |driver| {
3247            let temp_dir = tempfile::tempdir().unwrap();
3248            let igvm_path = temp_dir.path().join("guest.igvm");
3249            File::create(&igvm_path).unwrap();
3250            let igvm_options = format!("firmware={},personality=openhcl", igvm_path.display());
3251            let mesh = VmmMesh::new(&driver, true).unwrap();
3252
3253            for (extra, expected_alias, expected_policy) in [
3254                (vec![], true, Some(LateMapVtl0MemoryPolicy::Halt)),
3255                (
3256                    vec!["--uefi", "console=com1,disable_frontpage"],
3257                    true,
3258                    Some(LateMapVtl0MemoryPolicy::Halt),
3259                ),
3260                (
3261                    vec!["--net", "uh:consomme", "--no-alias-map"],
3262                    false,
3263                    Some(LateMapVtl0MemoryPolicy::Halt),
3264                ),
3265                (
3266                    vec!["--isolation", "vbs", "--no-alias-map"],
3267                    false,
3268                    Some(LateMapVtl0MemoryPolicy::Halt),
3269                ),
3270                (
3271                    vec!["--no-alias-map", "--late-map-vtl0-policy", "exception"],
3272                    false,
3273                    Some(LateMapVtl0MemoryPolicy::InjectException),
3274                ),
3275                (
3276                    vec!["--late-map-vtl0-policy", "log"],
3277                    true,
3278                    Some(LateMapVtl0MemoryPolicy::Log),
3279                ),
3280                (
3281                    vec!["--igvm-vtl2-relocation-type", "vtl2=filesize"],
3282                    true,
3283                    Some(LateMapVtl0MemoryPolicy::Halt),
3284                ),
3285                (
3286                    vec![
3287                        "--igvm-vtl2-relocation-type",
3288                        "vtl2=filesize",
3289                        "--late-map-vtl0-policy",
3290                        "off",
3291                    ],
3292                    true,
3293                    None,
3294                ),
3295            ] {
3296                let mut args = vec![
3297                    "openvmm",
3298                    "--igvm",
3299                    &igvm_options,
3300                    "--hv",
3301                    "--vtl2",
3302                    "--single-process",
3303                ];
3304                args.extend(extra);
3305                let opt = Options::try_parse_from(args).unwrap();
3306                let (config, resources) = vm_config_from_command_line(driver.clone(), &mesh, &opt)
3307                    .await
3308                    .unwrap();
3309
3310                assert!(matches!(config.load_mode, LoadMode::Igvm { .. }));
3311                assert!(config.hypervisor.with_hv);
3312                let vtl2 = config.hypervisor.with_vtl2.as_ref().unwrap();
3313                assert_eq!(vtl2.vtl0_alias_map, expected_alias);
3314                assert_eq!(vtl2.late_map_vtl0_memory, expected_policy);
3315                assert!(config.vmbus.is_some());
3316                assert!(config.vtl2_vmbus.is_some());
3317                assert!(resources.ged_rpc.is_some());
3318                for id in ["ged", "gel"] {
3319                    assert!(
3320                        config.vmbus_devices.iter().any(|(vtl, resource)| {
3321                            *vtl == DeviceVtl::Vtl2 && resource.id() == id
3322                        })
3323                    );
3324                }
3325                assert!(
3326                    config
3327                        .chipset_devices
3328                        .iter()
3329                        .all(|device| { device.resource.id() != "hyperv_firmware_uefi" })
3330                );
3331            }
3332            for (extra, expected_error) in [
3333                (
3334                    ["--net", "uh:consomme"],
3335                    "must specify --no-alias-map to offer NICs to VTL2",
3336                ),
3337                (
3338                    ["--isolation", "vbs"],
3339                    "alias map not supported with isolation",
3340                ),
3341            ] {
3342                let mut args = vec![
3343                    "openvmm",
3344                    "--igvm",
3345                    &igvm_options,
3346                    "--hv",
3347                    "--vtl2",
3348                    "--single-process",
3349                ];
3350                args.extend(extra);
3351                let opt = Options::try_parse_from(args).unwrap();
3352                let err = vm_config_from_command_line(driver.clone(), &mesh, &opt)
3353                    .await
3354                    .err()
3355                    .unwrap();
3356                assert_eq!(err.to_string(), expected_error);
3357            }
3358            mesh.shutdown().await;
3359        });
3360    }
3361
3362    #[test]
3363    fn maps_virtio_vsock_to_named_pcie_port() {
3364        DefaultPool::run_with(async |driver| {
3365            let temp_dir = tempfile::tempdir().unwrap();
3366            let kernel_path = temp_dir.path().join("kernel");
3367            File::create(&kernel_path).unwrap();
3368            let initrd_path = temp_dir.path().join("initrd");
3369            File::create(&initrd_path).unwrap();
3370            let socket_path = temp_dir.path().join("vsock");
3371            let opt = Options::try_parse_from([
3372                "openvmm",
3373                "--kernel",
3374                kernel_path.to_str().unwrap(),
3375                "--initrd",
3376                initrd_path.to_str().unwrap(),
3377                "--virtio-vsock-path",
3378                socket_path.to_str().unwrap(),
3379                "--virtio-vsock-bus",
3380                "pcie:custom",
3381                "--single-process",
3382            ])
3383            .unwrap();
3384            let mesh = VmmMesh::new(&driver, true).unwrap();
3385
3386            let (config, _resources) = vm_config_from_command_line(driver, &mesh, &opt)
3387                .await
3388                .unwrap();
3389
3390            assert_eq!(config.pcie_devices.len(), 1);
3391            assert_eq!(config.pcie_devices[0].port_name, "custom");
3392            mesh.shutdown().await;
3393        });
3394    }
3395
3396    #[test]
3397    fn maps_virtio_fs_and_rng_to_named_pcie_ports() {
3398        DefaultPool::run_with(async |driver| {
3399            let temp_dir = tempfile::tempdir().unwrap();
3400            let kernel_path = temp_dir.path().join("kernel");
3401            File::create(&kernel_path).unwrap();
3402            let initrd_path = temp_dir.path().join("initrd");
3403            File::create(&initrd_path).unwrap();
3404            let root_path = temp_dir.path().to_str().unwrap();
3405            let opt = Options::try_parse_from([
3406                "openvmm",
3407                "--kernel",
3408                kernel_path.to_str().unwrap(),
3409                "--initrd",
3410                initrd_path.to_str().unwrap(),
3411                "--virtio-fs",
3412                &format!("fs,{root_path}"),
3413                "--virtio-fs-bus",
3414                "pcie:fs",
3415                "--virtio-rng",
3416                "--virtio-rng-bus",
3417                "pcie:rng",
3418                "--single-process",
3419            ])
3420            .unwrap();
3421            let mesh = VmmMesh::new(&driver, true).unwrap();
3422
3423            let (config, _resources) = vm_config_from_command_line(driver, &mesh, &opt)
3424                .await
3425                .unwrap();
3426
3427            let port_names: Vec<_> = config
3428                .pcie_devices
3429                .iter()
3430                .map(|device| device.port_name.as_str())
3431                .collect();
3432            assert_eq!(port_names, ["fs", "rng"]);
3433            mesh.shutdown().await;
3434        });
3435    }
3436
3437    #[test]
3438    fn rejects_duplicate_pcie_port_assignments() {
3439        DefaultPool::run_with(async |driver| {
3440            let temp_dir = tempfile::tempdir().unwrap();
3441            let kernel_path = temp_dir.path().join("kernel");
3442            File::create(&kernel_path).unwrap();
3443            let initrd_path = temp_dir.path().join("initrd");
3444            File::create(&initrd_path).unwrap();
3445            let root_path = temp_dir.path().to_str().unwrap();
3446            let opt = Options::try_parse_from([
3447                "openvmm",
3448                "--kernel",
3449                kernel_path.to_str().unwrap(),
3450                "--initrd",
3451                initrd_path.to_str().unwrap(),
3452                "--virtio-fs",
3453                &format!("pcie_port=custom:fs,{root_path}"),
3454                "--virtio-rng",
3455                "--virtio-rng-bus",
3456                "pcie:custom",
3457                "--single-process",
3458            ])
3459            .unwrap();
3460            let mesh = VmmMesh::new(&driver, true).unwrap();
3461
3462            let error = vm_config_from_command_line(driver, &mesh, &opt)
3463                .await
3464                .err()
3465                .unwrap();
3466
3467            assert_eq!(error.to_string(), "multiple devices use PCIe port 'custom'");
3468            mesh.shutdown().await;
3469        });
3470    }
3471
3472    #[test]
3473    fn rejects_duplicate_pcie_port_assignment_from_storage() {
3474        DefaultPool::run_with(async |driver| {
3475            let temp_dir = tempfile::tempdir().unwrap();
3476            let kernel_path = temp_dir.path().join("kernel");
3477            File::create(&kernel_path).unwrap();
3478            let initrd_path = temp_dir.path().join("initrd");
3479            File::create(&initrd_path).unwrap();
3480            let opt = Options::try_parse_from([
3481                "openvmm",
3482                "--kernel",
3483                kernel_path.to_str().unwrap(),
3484                "--initrd",
3485                initrd_path.to_str().unwrap(),
3486                "--nvme-pci",
3487                "id=nvme0,pcie_port=custom",
3488                "--virtio-rng",
3489                "--virtio-rng-bus",
3490                "pcie:custom",
3491                "--single-process",
3492            ])
3493            .unwrap();
3494            let mesh = VmmMesh::new(&driver, true).unwrap();
3495
3496            let error = vm_config_from_command_line(driver, &mesh, &opt)
3497                .await
3498                .err()
3499                .unwrap();
3500
3501            assert_eq!(error.to_string(), "multiple devices use PCIe port 'custom'");
3502            mesh.shutdown().await;
3503        });
3504    }
3505}