Skip to main content

openvmm_entry/
lib.rs

1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4//! This module implements the interactive control process and the entry point
5//! for the worker process.
6
7#![expect(missing_docs)]
8#![forbid(unsafe_code)]
9
10mod cli_args;
11mod crash_dump;
12mod kvp;
13mod meshworker;
14mod pidfile;
15mod repl;
16mod serial_io;
17mod storage_builder;
18mod tracing_init;
19mod ttrpc;
20mod vm_controller;
21
22// `pub` so that the missing_docs warning fires for options without
23// documentation.
24pub use cli_args::Options;
25use console_relay::ConsoleLaunchOptions;
26
27use crate::cli_args::SecureBootTemplateCli;
28use anyhow::Context;
29use anyhow::bail;
30use chipset_resources::battery::HostBatteryUpdate;
31use cli_args::DiskCliKind;
32use cli_args::EfiDiagnosticsLogLevelCli;
33use cli_args::EndpointConfigCli;
34use cli_args::IgvmPersonalityCli;
35use cli_args::NicConfigCli;
36use cli_args::ProvisionVmgs;
37use cli_args::SerialConfigCli;
38use cli_args::TpmVersionCli;
39use cli_args::UefiConsoleModeCli;
40use cli_args::VirtioBusCli;
41use cli_args::VmgsCli;
42use crash_dump::spawn_dump_handler;
43use cxl_spec::test::CxlTestDeviceHandle;
44use disk_backend_resources::DelayDiskHandle;
45use disk_backend_resources::DiskLayerDescription;
46use disk_backend_resources::layer::DiskLayerHandle;
47use disk_backend_resources::layer::RamDiskLayerHandle;
48use disk_backend_resources::layer::SqliteAutoCacheDiskLayerHandle;
49use disk_backend_resources::layer::SqliteDiskLayerHandle;
50use floppy_resources::FloppyDiskConfig;
51use framebuffer::FRAMEBUFFER_SIZE;
52use framebuffer::FramebufferAccess;
53use futures::AsyncReadExt;
54use futures::AsyncWrite;
55use futures::StreamExt;
56use futures::executor::block_on;
57use futures::io::AllowStdIo;
58use gdma_resources::GdmaDeviceHandle;
59use gdma_resources::VportDefinition;
60use guid::Guid;
61use input_core::MultiplexedInputHandle;
62use inspect::InspectMut;
63use mesh::CancelContext;
64use mesh::CellUpdater;
65use mesh::rpc::RpcSend;
66use meshworker::VmmMesh;
67use net_backend_resources::mac_address::MacAddress;
68use nvme_resources::NvmeControllerRequest;
69use openvmm_defs::config::Config;
70use openvmm_defs::config::DEFAULT_PCAT_BOOT_ORDER;
71use openvmm_defs::config::DeviceVtl;
72use openvmm_defs::config::HypervisorConfig;
73use openvmm_defs::config::LateMapVtl0MemoryPolicy;
74use openvmm_defs::config::LoadMode;
75use openvmm_defs::config::MemoryConfig;
76use openvmm_defs::config::NumaDistance;
77use openvmm_defs::config::NumaNode;
78use openvmm_defs::config::NumaTopology;
79use openvmm_defs::config::PcieDeviceConfig;
80use openvmm_defs::config::PcieMmioRangeConfig;
81use openvmm_defs::config::PciePortConfig;
82use openvmm_defs::config::PcieRootComplexConfig;
83use openvmm_defs::config::PcieSwitchConfig;
84use openvmm_defs::config::ProcessorTopologyConfig;
85use openvmm_defs::config::RootComplexCxlConfig;
86use openvmm_defs::config::SerialInformation;
87use openvmm_defs::config::VirtioBus;
88use openvmm_defs::config::VmbusConfig;
89use openvmm_defs::config::VpAssignment;
90use openvmm_defs::config::VpciDeviceConfig;
91use openvmm_defs::config::Vtl2BaseAddressType;
92use openvmm_defs::config::Vtl2Config;
93use openvmm_defs::rpc::VmRpc;
94use openvmm_defs::worker::VM_WORKER;
95use openvmm_defs::worker::VmWorkerParameters;
96use openvmm_helpers::disk::OpenDiskOptions;
97use openvmm_helpers::disk::create_disk_type;
98use openvmm_helpers::disk::open_disk_type;
99use pal_async::DefaultDriver;
100use pal_async::DefaultPool;
101use pal_async::socket::PolledSocket;
102use pal_async::task::Spawn;
103use pal_async::task::Task;
104use serial_16550_resources::ComPort;
105use serial_core::resources::DisconnectedSerialBackendHandle;
106use sparse_mmap::alloc_shared_memory;
107use std::cell::RefCell;
108use std::collections::BTreeMap;
109use std::collections::HashSet;
110use std::fmt::Write as _;
111use std::io;
112#[cfg(unix)]
113use std::io::IsTerminal;
114use std::io::Write;
115use std::net::TcpListener;
116use std::path::Path;
117use std::path::PathBuf;
118use std::sync::Arc;
119use std::thread;
120use std::time::Duration;
121use storvsp_resources::ScsiControllerRequest;
122use tpm_resources::TpmDeviceHandle;
123use tpm_resources::TpmRegisterLayout;
124use tpm_resources::TpmVersion;
125use uidevices_resources::SynthKeyboardHandle;
126use uidevices_resources::SynthMouseHandle;
127use uidevices_resources::SynthVideoHandle;
128use video_core::SharedFramebufferHandle;
129use virtio_resources::VirtioPciDeviceHandle;
130use vm_manifest_builder::BaseChipsetType;
131use vm_manifest_builder::MachineArch;
132use vm_manifest_builder::VmChipsetResult;
133use vm_manifest_builder::VmManifestBuilder;
134use vm_resource::IntoResource;
135use vm_resource::Resource;
136use vm_resource::kind::DiskHandleKind;
137use vm_resource::kind::DiskLayerHandleKind;
138use vm_resource::kind::NetEndpointHandleKind;
139use vm_resource::kind::VirtioDeviceHandle;
140use vm_resource::kind::VmbusDeviceHandleKind;
141use vmbus_serial_resources::VmbusSerialDeviceHandle;
142use vmbus_serial_resources::VmbusSerialPort;
143use vmcore::non_volatile_store::resources::EphemeralNonVolatileStoreHandle;
144use vmgs_resources::GuestStateEncryptionPolicy;
145use vmgs_resources::VmgsDisk;
146use vmgs_resources::VmgsFileHandle;
147use vmgs_resources::VmgsResource;
148use vmotherboard::ChipsetDeviceHandle;
149use vnc_worker_defs::VncParameters;
150
151pub fn openvmm_main() {
152    // Save the current state of the terminal so we can restore it back to
153    // normal before exiting.
154    #[cfg(unix)]
155    let orig_termios = io::stderr().is_terminal().then(term::get_termios);
156
157    let mut pidfile_guard: Option<pidfile::Pidfile> = None;
158    let exit_code = match do_main(&mut pidfile_guard) {
159        Ok(code) => code,
160        Err(err) => {
161            eprintln!("fatal error: {:?}", err);
162            1
163        }
164    };
165
166    // Restore the terminal to its initial state.
167    #[cfg(unix)]
168    if let Some(orig_termios) = orig_termios {
169        term::set_termios(orig_termios);
170    }
171
172    // Clean up the pidfile before terminating, since
173    // pal::process::terminate skips destructors.
174    drop(pidfile_guard);
175
176    // Terminate the process immediately without graceful shutdown of DLLs or
177    // C++ destructors or anything like that. This is all unnecessary and saves
178    // time on Windows.
179    //
180    // Do flush stdout, though, since there may be buffered data.
181    let _ = io::stdout().flush();
182    pal::process::terminate(exit_code);
183}
184
185#[derive(Default)]
186struct VmResources {
187    console_in: Option<Box<dyn AsyncWrite + Send + Unpin>>,
188    /// Keeps the dedicated serial reactor alive while serial I/O objects exist.
189    serial_driver: Option<DefaultDriver>,
190    framebuffer_access: Option<FramebufferAccess>,
191    shutdown_ic: Option<mesh::Sender<hyperv_ic_resources::shutdown::ShutdownRpc>>,
192    kvp_ic: Option<mesh::Sender<hyperv_ic_resources::kvp::KvpConnectRpc>>,
193    scsi_rpc: Option<mesh::Sender<ScsiControllerRequest>>,
194    nvme_vtl2_rpc: Option<mesh::Sender<NvmeControllerRequest>>,
195    consomme_rpc: Option<mesh::Sender<net_backend_resources::consomme::ConsommeRequest>>,
196    ged_rpc: Option<mesh::Sender<get_resources::ged::GuestEmulationRequest>>,
197    vtl2_settings: Option<vtl2_settings_proto::Vtl2Settings>,
198    /// Receives dirty rectangles from the synthetic video device for the VNC worker.
199    dirty_rect_recv: Option<mesh::Receiver<Vec<video_core::DirtyRect>>>,
200    #[cfg(windows)]
201    switch_ports: Vec<vmswitch::kernel::SwitchPort>,
202}
203
204struct ConsoleState<'a> {
205    device: &'a str,
206    input: Box<dyn AsyncWrite + Unpin + Send>,
207}
208
209/// Build a flat list of switches with their parent port assignments.
210///
211/// This function converts hierarchical CLI switch definitions into a flat list
212/// where each switch specifies its parent port directly.
213fn build_switch_list(all_switches: &[cli_args::GenericPcieSwitchCli]) -> Vec<PcieSwitchConfig> {
214    all_switches
215        .iter()
216        .map(|switch_cli| PcieSwitchConfig {
217            name: switch_cli.name.clone(),
218            parent_port: switch_cli.port_name.clone(),
219            ports: (0..switch_cli.num_downstream_ports)
220                .map(|i| PciePortConfig {
221                    name: format!("{}-downstream-{}", switch_cli.name, i),
222                    devfn: None,
223                    hotplug: switch_cli.hotplug,
224                    acs_capabilities_supported: switch_cli.acs_capabilities_supported,
225                    cxl: false,
226                    pasid: switch_cli.pasid,
227                })
228                .collect(),
229        })
230        .collect()
231}
232
233fn base_chipset_type(opt: &Options) -> BaseChipsetType {
234    if opt.igvm.is_some() {
235        match opt.igvm_personality {
236            None => BaseChipsetType::HclHost,
237            Some(IgvmPersonalityCli::Uefi) => BaseChipsetType::HypervGen2Uefi,
238            Some(IgvmPersonalityCli::LinuxDirect)
239                if matches!(opt.isolation, Some(cli_args::IsolationCli::Snp)) =>
240            {
241                BaseChipsetType::EnlightenedLinuxDirect
242            }
243            Some(IgvmPersonalityCli::LinuxDirect) if opt.hv => {
244                BaseChipsetType::HyperVGen2LinuxDirect
245            }
246            Some(IgvmPersonalityCli::LinuxDirect) => BaseChipsetType::UnenlightenedLinuxDirect,
247        }
248    } else if matches!(opt.isolation, Some(cli_args::IsolationCli::Snp)) {
249        BaseChipsetType::EnlightenedLinuxDirect
250    } else if opt.pcat {
251        BaseChipsetType::HypervGen1
252    } else if opt.uefi.is_some() {
253        BaseChipsetType::HypervGen2Uefi
254    } else if opt.hv {
255        BaseChipsetType::HyperVGen2LinuxDirect
256    } else {
257        BaseChipsetType::UnenlightenedLinuxDirect
258    }
259}
260
261/// Build the loader's [`SmbiosConfig`](openvmm_defs::config::SmbiosConfig) from
262/// the parsed `--smbios` arguments.
263///
264/// Multiple `--smbios` arguments are merged (erroring on a field set twice).
265/// String overrides left unset fall through to the loader's default identity.
266/// The system UUID defaults to the all-zero GUID unless overridden with
267/// `uuid=GUID`; `uuid=random` requests a freshly generated per-VM GUID.
268fn smbios_config_from_cli(
269    args: &[cli_args::SmbiosCli],
270) -> anyhow::Result<openvmm_defs::config::SmbiosConfig> {
271    let mut merged = cli_args::SmbiosCli::default();
272    for arg in args {
273        merged.merge(arg.clone())?;
274    }
275    let cli_args::SmbiosCli {
276        bios:
277            cli_args::SmbiosBiosCli {
278                vendor: bios_vendor,
279                version: bios_version,
280                release_date: bios_release_date,
281                release: bios_release,
282            },
283        system:
284            cli_args::SmbiosSystemCli {
285                manufacturer: system_manufacturer,
286                product_name: system_product,
287                version: system_version,
288                serial_number: system_serial,
289                sku_number: system_sku,
290                family: system_family,
291                uuid: system_uuid,
292            },
293    } = merged;
294    Ok(openvmm_defs::config::SmbiosConfig {
295        bios: openvmm_defs::config::SmbiosBiosOverrides {
296            vendor: bios_vendor,
297            version: bios_version,
298            release_date: bios_release_date,
299            release: bios_release.map(|r| (r.0, r.1)),
300        },
301        system: openvmm_defs::config::SmbiosSystemOverrides {
302            manufacturer: system_manufacturer,
303            product_name: system_product,
304            version: system_version,
305            serial_number: system_serial,
306            sku_number: system_sku,
307            family: system_family,
308            uuid: match system_uuid {
309                None => Guid::ZERO,
310                Some(cli_args::SmbiosUuid::Random) => Guid::new_random(),
311                Some(cli_args::SmbiosUuid::Fixed(guid)) => guid,
312            },
313        },
314    })
315}
316
317async fn vm_config_from_command_line(
318    spawner: impl Spawn,
319    mesh: &VmmMesh,
320    opt: &Options,
321) -> anyhow::Result<(Config, VmResources)> {
322    opt.validate_isolation_options()?;
323    opt.validate_igvm_options()?;
324
325    let (_, serial_driver) = DefaultPool::spawn_on_thread("serial");
326    let uefi = opt.effective_uefi()?;
327    let default_uefi = cli_args::UefiCli::default();
328    let uefi_options = uefi.as_ref().unwrap_or(&default_uefi);
329
330    let openhcl_vtl = if opt.vtl2 {
331        DeviceVtl::Vtl2
332    } else {
333        DeviceVtl::Vtl0
334    };
335
336    let console_state: RefCell<Option<ConsoleState<'_>>> = RefCell::new(None);
337    let setup_serial = |name: &str, cli_cfg, device| -> anyhow::Result<_> {
338        Ok(match cli_cfg {
339            SerialConfigCli::Console => {
340                if let Some(console_state) = console_state.borrow().as_ref() {
341                    bail!("console already set by {}", console_state.device);
342                }
343                let (config, serial) = serial_io::anonymous_serial_pair(&serial_driver)?;
344                let (serial_read, serial_write) = AsyncReadExt::split(serial);
345                *console_state.borrow_mut() = Some(ConsoleState {
346                    device,
347                    input: Box::new(serial_write),
348                });
349                thread::Builder::new()
350                    .name(name.to_owned())
351                    .spawn(move || {
352                        let _ = block_on(futures::io::copy(
353                            serial_read,
354                            &mut AllowStdIo::new(term::raw_stdout()),
355                        ));
356                    })
357                    .unwrap();
358                Some(config)
359            }
360            SerialConfigCli::Stderr => {
361                let (config, serial) = serial_io::anonymous_serial_pair(&serial_driver)?;
362                thread::Builder::new()
363                    .name(name.to_owned())
364                    .spawn(move || {
365                        let _ = block_on(futures::io::copy(
366                            serial,
367                            &mut AllowStdIo::new(term::raw_stderr()),
368                        ));
369                    })
370                    .unwrap();
371                Some(config)
372            }
373            SerialConfigCli::File(path) => {
374                let (config, serial) = serial_io::anonymous_serial_pair(&serial_driver)?;
375                let file = fs_err::File::create(path).context("failed to create file")?;
376
377                thread::Builder::new()
378                    .name(name.to_owned())
379                    .spawn(move || {
380                        let _ = block_on(futures::io::copy(serial, &mut AllowStdIo::new(file)));
381                    })
382                    .unwrap();
383                Some(config)
384            }
385            SerialConfigCli::None => None,
386            SerialConfigCli::Pipe(path) => {
387                Some(serial_io::bind_serial(&path).context("failed to bind serial")?)
388            }
389            SerialConfigCli::Tcp(addr) => {
390                Some(serial_io::bind_tcp_serial(&addr).context("failed to bind serial")?)
391            }
392            SerialConfigCli::NewConsole(app, window_title) => {
393                let path = console_relay::random_console_path();
394                let config =
395                    serial_io::bind_serial(&path).context("failed to bind console serial")?;
396                let window_title =
397                    window_title.unwrap_or_else(|| name.to_uppercase() + " [OpenVMM]");
398
399                console_relay::launch_console(
400                    app.or_else(openvmm_terminal_app).as_deref(),
401                    &path,
402                    ConsoleLaunchOptions {
403                        window_title: Some(window_title),
404                    },
405                )
406                .context("failed to launch console")?;
407
408                Some(config)
409            }
410        })
411    };
412
413    let mut vmbus_devices = Vec::new();
414
415    let com_debugger_mode = [
416        opt.com1.as_ref().is_some_and(|c| c.debugger_mode),
417        opt.com2.as_ref().is_some_and(|c| c.debugger_mode),
418        opt.com3.as_ref().is_some_and(|c| c.debugger_mode),
419        opt.com4.as_ref().is_some_and(|c| c.debugger_mode),
420    ];
421
422    let serial0_cfg = setup_serial(
423        "com1",
424        opt.com1
425            .clone()
426            .map_or(SerialConfigCli::Console, |c| c.backend),
427        if cfg!(guest_arch = "x86_64") {
428            "ttyS0"
429        } else {
430            "ttyAMA0"
431        },
432    )?;
433    let serial1_cfg = setup_serial(
434        "com2",
435        opt.com2
436            .clone()
437            .map_or(SerialConfigCli::None, |c| c.backend),
438        if cfg!(guest_arch = "x86_64") {
439            "ttyS1"
440        } else {
441            "ttyAMA1"
442        },
443    )?;
444    let serial2_cfg = setup_serial(
445        "com3",
446        opt.com3
447            .clone()
448            .map_or(SerialConfigCli::None, |c| c.backend),
449        if cfg!(guest_arch = "x86_64") {
450            "ttyS2"
451        } else {
452            "ttyAMA2"
453        },
454    )?;
455    let serial3_cfg = setup_serial(
456        "com4",
457        opt.com4
458            .clone()
459            .map_or(SerialConfigCli::None, |c| c.backend),
460        if cfg!(guest_arch = "x86_64") {
461            "ttyS3"
462        } else {
463            "ttyAMA3"
464        },
465    )?;
466    let with_vmbus_com1_serial = if let Some(vmbus_com1_cfg) = setup_serial(
467        "vmbus_com1",
468        opt.vmbus_com1_serial
469            .clone()
470            .unwrap_or(SerialConfigCli::None),
471        "vmbus_com1",
472    )? {
473        vmbus_devices.push((
474            openhcl_vtl,
475            VmbusSerialDeviceHandle {
476                port: VmbusSerialPort::Com1,
477                backend: vmbus_com1_cfg,
478            }
479            .into_resource(),
480        ));
481        true
482    } else {
483        false
484    };
485    let with_vmbus_com2_serial = if let Some(vmbus_com2_cfg) = setup_serial(
486        "vmbus_com2",
487        opt.vmbus_com2_serial
488            .clone()
489            .unwrap_or(SerialConfigCli::None),
490        "vmbus_com2",
491    )? {
492        vmbus_devices.push((
493            openhcl_vtl,
494            VmbusSerialDeviceHandle {
495                port: VmbusSerialPort::Com2,
496                backend: vmbus_com2_cfg,
497            }
498            .into_resource(),
499        ));
500        true
501    } else {
502        false
503    };
504    let debugcon_cfg = setup_serial(
505        "debugcon",
506        opt.debugcon
507            .clone()
508            .map(|cfg| cfg.serial)
509            .unwrap_or(SerialConfigCli::None),
510        "debugcon",
511    )?;
512
513    let virtio_console_backend = if let Some(serial_cfg) = opt.virtio_console.clone() {
514        setup_serial("virtio-console", serial_cfg, "hvc0")?
515    } else {
516        None
517    };
518
519    let mut resources = VmResources::default();
520    let mut console_str = "";
521    if let Some(ConsoleState { device, input }) = console_state.into_inner() {
522        resources.console_in = Some(input);
523        console_str = device;
524    }
525
526    if opt.shared_memory {
527        tracing::warn!("--shared-memory/-M flag has no effect and will be removed");
528    }
529    if opt.deprecated_prefetch {
530        tracing::warn!("--prefetch is deprecated; use --memory prefetch=on");
531    }
532    if opt.deprecated_private_memory {
533        tracing::warn!("--private-memory is deprecated; use --memory shared=off");
534    }
535    if opt.deprecated_thp {
536        tracing::warn!("--thp is deprecated; use --memory shared=off,thp=on");
537    }
538    if opt.deprecated_memory_backing_file.is_some() {
539        tracing::warn!("--memory-backing-file is deprecated; use --memory file=<path>");
540    }
541
542    opt.validate_memory_options()?;
543
544    const MAX_PROCESSOR_COUNT: u32 = 1024;
545
546    if opt.processors == 0 || opt.processors > MAX_PROCESSOR_COUNT {
547        bail!("invalid proc count: {}", opt.processors);
548    }
549
550    // Total SCSI channel count should not exceed the processor count
551    // (at most, one channel per VP).
552    if opt.scsi_sub_channels > (MAX_PROCESSOR_COUNT - 1) as u16 {
553        bail!(
554            "invalid SCSI sub-channel count: requested {}, max {}",
555            opt.scsi_sub_channels,
556            MAX_PROCESSOR_COUNT - 1
557        );
558    }
559
560    let with_get = opt.get || (opt.vtl2 && !opt.no_get);
561
562    let mut storage = storage_builder::StorageBuilder::new(with_get.then_some(openhcl_vtl));
563
564    // Register named controllers first, so that --disk on=<name>
565    // references can be resolved.
566    for ctrl in &opt.nvme_pci {
567        let transport = match &ctrl.transport {
568            cli_args::NvmeControllerTransport::Pcie(port) => {
569                storage_builder::NvmeControllerTransport::Pcie(port.clone())
570            }
571            cli_args::NvmeControllerTransport::Vpci(guid) => {
572                let guid = guid.unwrap_or_else(|| storage_builder::deterministic_guid(&ctrl.id));
573                storage_builder::NvmeControllerTransport::Vpci(guid)
574            }
575        };
576        storage.add_nvme_controller(ctrl.id.clone(), ctrl.vtl, transport, None)?;
577    }
578
579    for ctrl in &opt.vmbus_scsi {
580        let instance_id = storage_builder::deterministic_guid(&ctrl.id);
581        storage.add_scsi_controller(ctrl.id.clone(), ctrl.vtl, instance_id, ctrl.sub_channels)?;
582    }
583
584    for ctrl in &opt.openhcl_controller {
585        let controller_type = match ctrl.controller_type {
586            cli_args::OpenhclControllerType::Scsi => storage_builder::OpenhclControllerType::Scsi,
587            cli_args::OpenhclControllerType::Nvme => storage_builder::OpenhclControllerType::Nvme,
588        };
589        let instance_id = ctrl
590            .guid
591            .unwrap_or_else(|| storage_builder::deterministic_guid(&ctrl.id));
592        storage.add_openhcl_controller(ctrl.id.clone(), controller_type, instance_id)?;
593    }
594
595    for &cli_args::DiskCli {
596        vtl,
597        ref kind,
598        read_only,
599        is_dvd,
600        underhill,
601        ref pcie_port,
602        ref serial,
603        ref controller,
604        nsid,
605        lun,
606        ref relay,
607    } in &opt.disk
608    {
609        if serial.is_some() {
610            anyhow::bail!("`serial` is only supported by `--virtio-blk`");
611        }
612        if controller.is_none() && underhill.is_none() && relay.is_none() {
613            tracing::warn!(
614                "--disk without `on` is deprecated; \
615                 use --vmbus-scsi and --disk on=<name> instead"
616            );
617        }
618
619        let relay_target = relay
620            .as_ref()
621            .map(|(name, loc)| storage_builder::RelayTarget {
622                controller: name.clone(),
623                location: *loc,
624            });
625
626        let target = if let Some(name) = controller {
627            if pcie_port.is_some() {
628                anyhow::bail!("`on` is incompatible with `pcie_port` on `--disk`");
629            }
630            storage_builder::DiskLocation::Named {
631                controller: name.clone(),
632                nsid,
633                lun,
634            }
635        } else if pcie_port.is_some() {
636            anyhow::bail!("`--disk` is incompatible with `pcie_port` without `controller`");
637        } else {
638            if opt.no_vmbus {
639                anyhow::bail!(
640                    "`--disk` without `on=` attaches to the default VMBus SCSI controller and \
641                     cannot be used with `--no-vmbus`; use `on=<name>` to attach to a named controller"
642                );
643            }
644            storage_builder::DiskLocation::Scsi(None)
645        };
646
647        storage
648            .add(
649                vtl,
650                underhill,
651                relay_target,
652                target,
653                kind,
654                is_dvd,
655                read_only,
656            )
657            .await?;
658    }
659
660    for &cli_args::IdeDiskCli {
661        ref kind,
662        read_only,
663        channel,
664        device,
665        is_dvd,
666    } in &opt.ide
667    {
668        storage
669            .add(
670                DeviceVtl::Vtl0,
671                None,
672                None,
673                storage_builder::DiskLocation::Ide(channel, device),
674                kind,
675                is_dvd,
676                read_only,
677            )
678            .await?;
679    }
680
681    if !opt.nvme.is_empty() {
682        tracing::warn!("--nvme is deprecated; use --nvme-pci and --disk on=<name> instead");
683
684        // Pre-register implicit PCIe controllers for unique port names.
685        let mut registered_ports = std::collections::BTreeSet::new();
686        for disk in &opt.nvme {
687            if let Some(port) = &disk.pcie_port {
688                if registered_ports.insert(port.clone()) {
689                    storage.add_nvme_controller(
690                        port.clone(),
691                        DeviceVtl::Vtl0,
692                        storage_builder::NvmeControllerTransport::Pcie(port.clone()),
693                        None,
694                    ).with_context(|| format!(
695                        "legacy --nvme flag conflicts with an explicit controller named '{port}'; \
696                         use --nvme-pci and --disk on=<name> instead"
697                    ))?;
698                }
699            }
700        }
701    }
702
703    for &cli_args::DiskCli {
704        vtl,
705        ref kind,
706        read_only,
707        is_dvd,
708        underhill,
709        ref pcie_port,
710        ref serial,
711        controller: _,
712        nsid: _,
713        lun: _,
714        relay: _,
715    } in &opt.nvme
716    {
717        if serial.is_some() {
718            anyhow::bail!("`serial` is only supported by `--virtio-blk`");
719        }
720        let target = if let Some(port) = pcie_port {
721            storage_builder::DiskLocation::Named {
722                controller: port.clone(),
723                nsid: None,
724                lun: None,
725            }
726        } else {
727            storage_builder::DiskLocation::Nvme(None)
728        };
729        storage
730            .add(vtl, underhill, None, target, kind, is_dvd, read_only)
731            .await?;
732    }
733
734    for &cli_args::DiskCli {
735        vtl,
736        ref kind,
737        read_only,
738        is_dvd,
739        ref underhill,
740        ref pcie_port,
741        ref serial,
742        controller: _,
743        nsid: _,
744        lun: _,
745        relay: _,
746    } in &opt.virtio_blk
747    {
748        if underhill.is_some() {
749            anyhow::bail!("underhill not supported with virtio-blk");
750        }
751        storage
752            .add(
753                vtl,
754                None,
755                None,
756                storage_builder::DiskLocation::VirtioBlk {
757                    pcie_port: pcie_port.clone(),
758                    serial: serial.clone(),
759                },
760                kind,
761                is_dvd,
762                read_only,
763            )
764            .await?;
765    }
766
767    let mut floppy_disks = Vec::new();
768    for disk in &opt.floppy {
769        let &cli_args::FloppyDiskCli {
770            ref kind,
771            read_only,
772        } = disk;
773        floppy_disks.push(FloppyDiskConfig {
774            disk_type: disk_open(kind, read_only).await?,
775            read_only,
776        });
777    }
778
779    let mut vpci_mana_nics = [(); 3].map(|()| None);
780    let mut pcie_mana_nics = BTreeMap::<String, GdmaDeviceHandle>::new();
781    let mut underhill_nics = Vec::new();
782    let mut vpci_devices = Vec::new();
783
784    let mut nic_index = 0;
785    for cli_cfg in &opt.net {
786        if cli_cfg.pcie_port.is_some() {
787            anyhow::bail!("`--net` does not support PCIe");
788        }
789        let vport = parse_endpoint(cli_cfg, &mut nic_index, &mut resources)?;
790        if cli_cfg.underhill {
791            if !opt.no_alias_map {
792                anyhow::bail!("must specify --no-alias-map to offer NICs to VTL2");
793            }
794            let mana = vpci_mana_nics[openhcl_vtl as usize].get_or_insert_with(|| {
795                let vpci_instance_id = Guid::new_random();
796                underhill_nics.push(vtl2_settings_proto::NicDeviceLegacy {
797                    instance_id: vpci_instance_id.to_string(),
798                    subordinate_instance_id: None,
799                    max_sub_channels: None,
800                });
801                (vpci_instance_id, GdmaDeviceHandle { vports: Vec::new() })
802            });
803            mana.1.vports.push(VportDefinition {
804                mac_address: vport.mac_address,
805                endpoint: vport.endpoint,
806            });
807        } else {
808            vmbus_devices.push(vport.into_netvsp_handle());
809        }
810    }
811
812    if opt.nic {
813        let nic_config = parse_endpoint(
814            &NicConfigCli {
815                vtl: DeviceVtl::Vtl0,
816                endpoint: EndpointConfigCli::Consomme {
817                    cidr: None,
818                    host_fwd: Vec::new(),
819                },
820                max_queues: None,
821                underhill: false,
822                pcie_port: None,
823            },
824            &mut nic_index,
825            &mut resources,
826        )?;
827        vmbus_devices.push(nic_config.into_netvsp_handle());
828    }
829
830    // Build initial PCIe devices list from CLI options. Storage devices
831    // (e.g., NVMe controllers on PCIe ports) are added later by storage_builder.
832    let mut pcie_devices = Vec::new();
833    for (index, cli_cfg) in opt.pcie_remote.iter().enumerate() {
834        tracing::info!(
835            port_name = %cli_cfg.port_name,
836            socket_addr = ?cli_cfg.socket_addr,
837            "instantiating PCIe remote device"
838        );
839
840        // Generate a deterministic instance ID based on index
841        const PCIE_REMOTE_BASE_INSTANCE_ID: Guid =
842            guid::guid!("28ed784d-c059-429f-9d9a-46bea02562c0");
843        let instance_id = Guid {
844            data1: index as u32,
845            ..PCIE_REMOTE_BASE_INSTANCE_ID
846        };
847
848        pcie_devices.push(PcieDeviceConfig {
849            port_name: cli_cfg.port_name.clone(),
850            resource: pcie_remote_resources::PcieRemoteHandle {
851                instance_id,
852                socket_addr: cli_cfg.socket_addr.clone(),
853                hu: cli_cfg.hu,
854                controller: cli_cfg.controller,
855            }
856            .into_resource(),
857        });
858    }
859
860    #[cfg(windows)]
861    let mut kernel_vmnics = Vec::new();
862    #[cfg(windows)]
863    for (index, switch_id) in opt.kernel_vmnic.iter().enumerate() {
864        // Pick a random MAC address.
865        let mut mac_address = [0x00, 0x15, 0x5D, 0, 0, 0];
866        getrandom::fill(&mut mac_address[3..]).expect("rng failure");
867
868        // Pick a fixed instance ID based on the index.
869        const BASE_INSTANCE_ID: Guid = guid::guid!("00000000-435d-11ee-9f59-00155d5016fc");
870        let instance_id = Guid {
871            data1: index as u32,
872            ..BASE_INSTANCE_ID
873        };
874
875        let switch_id = if switch_id == "default" {
876            None
877        } else {
878            Some(switch_id.as_str())
879        };
880        let (port_id, port) = new_switch_port(switch_id)?;
881        resources.switch_ports.push(port);
882
883        kernel_vmnics.push(openvmm_defs::config::KernelVmNicConfig {
884            instance_id,
885            mac_address: mac_address.into(),
886            switch_port_id: port_id,
887        });
888    }
889
890    for vport in &opt.mana {
891        let vport = parse_endpoint(vport, &mut nic_index, &mut resources)?;
892        let vport_array = match (vport.vtl as usize, vport.pcie_port) {
893            (vtl, None) => {
894                &mut vpci_mana_nics[vtl]
895                    .get_or_insert_with(|| {
896                        (Guid::new_random(), GdmaDeviceHandle { vports: Vec::new() })
897                    })
898                    .1
899                    .vports
900            }
901            (0, Some(pcie_port)) => {
902                &mut pcie_mana_nics
903                    .entry(pcie_port)
904                    .or_insert(GdmaDeviceHandle { vports: Vec::new() })
905                    .vports
906            }
907            _ => anyhow::bail!("PCIe NICs only supported to VTL0"),
908        };
909        vport_array.push(VportDefinition {
910            mac_address: vport.mac_address,
911            endpoint: vport.endpoint,
912        });
913    }
914
915    vpci_devices.extend(
916        vpci_mana_nics
917            .into_iter()
918            .enumerate()
919            .filter_map(|(vtl, nic)| {
920                nic.map(|(instance_id, handle)| VpciDeviceConfig {
921                    vtl: match vtl {
922                        0 => DeviceVtl::Vtl0,
923                        1 => DeviceVtl::Vtl1,
924                        2 => DeviceVtl::Vtl2,
925                        _ => unreachable!(),
926                    },
927                    instance_id,
928                    resource: handle.into_resource(),
929                    vnode: None,
930                })
931            }),
932    );
933
934    pcie_devices.extend(
935        pcie_mana_nics
936            .into_iter()
937            .map(|(pcie_port, handle)| PcieDeviceConfig {
938                port_name: pcie_port,
939                resource: handle.into_resource(),
940            }),
941    );
942
943    for cxl_test in &opt.cxl_test {
944        pcie_devices.push(PcieDeviceConfig {
945            port_name: cxl_test.pcie_port.clone(),
946            resource: CxlTestDeviceHandle {
947                hdm_size_bytes: cxl_test.hdm_size,
948            }
949            .into_resource(),
950        });
951    }
952
953    #[cfg(guest_arch = "aarch64")]
954    let arch = MachineArch::Aarch64;
955    #[cfg(guest_arch = "x86_64")]
956    let arch = MachineArch::X86_64;
957
958    #[cfg(guest_arch = "x86_64")]
959    anyhow::ensure!(
960        opt.amd_iommu.is_empty() || opt.intel_vtd.is_empty(),
961        "--amd-iommu and --intel-vtd cannot both be used in the same VM"
962    );
963
964    #[cfg(guest_arch = "x86_64")]
965    let mut amd_iommu_names: HashSet<&str> = opt.amd_iommu.iter().map(|s| s.as_str()).collect();
966    #[cfg(guest_arch = "x86_64")]
967    let mut vtd_names: HashSet<&str> = opt.intel_vtd.iter().map(|s| s.as_str()).collect();
968
969    // Map each `--smmu` entry to its root complex, rejecting duplicate `rc=`
970    // entries up front. Entries are removed as they are matched to a root
971    // complex below; any left over refer to unknown root complexes.
972    #[cfg(guest_arch = "aarch64")]
973    let mut smmu_names: std::collections::HashMap<&str, &cli_args::SmmuCli> = {
974        let mut map = std::collections::HashMap::new();
975        for s in &opt.smmu {
976            if map.insert(s.rc_name.as_str(), s).is_some() {
977                anyhow::bail!(
978                    "--smmu specified multiple times for root complex '{}'",
979                    s.rc_name
980                );
981            }
982        }
983        map
984    };
985
986    let mut pcie_root_complexes = Vec::new();
987    for (i, rc_cli) in opt.pcie_root_complex.iter().enumerate() {
988        let ports: Vec<PciePortConfig> = opt
989            .pcie_root_port
990            .iter()
991            .filter(|port_cli| port_cli.root_complex_name == rc_cli.name)
992            .map(|port_cli| PciePortConfig {
993                name: port_cli.name.clone(),
994                devfn: port_cli.devfn,
995                hotplug: port_cli.hotplug,
996                acs_capabilities_supported: port_cli.acs_capabilities_supported,
997                cxl: port_cli.cxl,
998                pasid: port_cli.pasid,
999            })
1000            .collect();
1001
1002        const ONE_MB: u64 = 1024 * 1024;
1003        // Keep all PCI windows 1MB-granular to match layout and downstream placement rules.
1004        let low_mmio_size = (rc_cli.low_mmio as u64).next_multiple_of(ONE_MB);
1005        let high_mmio_size = rc_cli
1006            .high_mmio
1007            .checked_next_multiple_of(ONE_MB)
1008            .context("high mmio rounding error")?;
1009
1010        // Count CXL-capable ports under the root bus. If the root bus has CXL root ports, it needs CHBCR.
1011        let cxl_port_count = ports.iter().filter(|port| port.cxl).count() as u64;
1012
1013        let cxl = if cxl_port_count != 0 {
1014            Some(RootComplexCxlConfig {
1015                hdm_size: rc_cli.hdm,
1016                hdm_window_restrictions: rc_cli.hdm_window_restrictions.bits(),
1017            })
1018        } else {
1019            None
1020        };
1021        pcie_root_complexes.push(PcieRootComplexConfig {
1022            index: i as u32,
1023            name: rc_cli.name.clone(),
1024            segment: rc_cli.segment,
1025            start_bus: rc_cli.start_bus,
1026            end_bus: rc_cli.end_bus,
1027            low_mmio: if let Some(base) = rc_cli.low_mmio_base {
1028                PcieMmioRangeConfig::Fixed(
1029                    memory_range::MemoryRange::try_new(base..base.wrapping_add(low_mmio_size))
1030                        .context("invalid low MMIO range")?,
1031                )
1032            } else {
1033                PcieMmioRangeConfig::Dynamic {
1034                    size: low_mmio_size,
1035                }
1036            },
1037            high_mmio: if let Some(base) = rc_cli.high_mmio_base {
1038                PcieMmioRangeConfig::Fixed(
1039                    memory_range::MemoryRange::try_new(base..base.wrapping_add(high_mmio_size))
1040                        .context("invalid high MMIO range")?,
1041                )
1042            } else {
1043                PcieMmioRangeConfig::Dynamic {
1044                    size: high_mmio_size,
1045                }
1046            },
1047            cxl,
1048            ports,
1049            #[cfg(guest_arch = "aarch64")]
1050            iommu: smmu_names.remove(rc_cli.name.as_str()).map(|s| {
1051                openvmm_defs::config::PcieIommuConfig::Smmu {
1052                    accel: s.accel,
1053                    oas: match s.oas {
1054                        cli_args::SmmuOasCli::Auto => openvmm_defs::config::SmmuOas::Auto,
1055                        cli_args::SmmuOasCli::Fixed(bits) => {
1056                            openvmm_defs::config::SmmuOas::Fixed(bits)
1057                        }
1058                    },
1059                }
1060            }),
1061            #[cfg(guest_arch = "x86_64")]
1062            iommu: if amd_iommu_names.remove(rc_cli.name.as_str()) {
1063                Some(openvmm_defs::config::PcieIommuConfig::AmdVi)
1064            } else if vtd_names.remove(rc_cli.name.as_str()) {
1065                Some(openvmm_defs::config::PcieIommuConfig::IntelVtd)
1066            } else {
1067                None
1068            },
1069            vnode: rc_cli.vnode,
1070            preserve_bars: rc_cli.preserve_bars,
1071        });
1072    }
1073
1074    #[cfg(guest_arch = "aarch64")]
1075    if let Some(name) = smmu_names.into_keys().next() {
1076        anyhow::bail!("--smmu refers to unknown root complex '{name}'");
1077    }
1078    #[cfg(guest_arch = "x86_64")]
1079    if let Some(name) = amd_iommu_names.into_iter().next() {
1080        anyhow::bail!("--amd-iommu refers to unknown root complex '{name}'");
1081    }
1082    #[cfg(guest_arch = "x86_64")]
1083    if let Some(name) = vtd_names.into_iter().next() {
1084        anyhow::bail!("--intel-vtd refers to unknown root complex '{name}'");
1085    }
1086
1087    let pcie_switches = build_switch_list(&opt.pcie_switch);
1088    let pcie_generic_initiators = opt
1089        .pcie_generic_initiator
1090        .iter()
1091        .map(|gi| openvmm_defs::config::PcieGenericInitiatorConfig {
1092            port_name: gi.port_name.clone(),
1093            node: gi.node,
1094        })
1095        .collect();
1096    #[cfg(target_os = "linux")]
1097    let vfio_pcie_devices: Vec<PcieDeviceConfig> = {
1098        use std::collections::HashMap;
1099        use vm_resource::IntoResource;
1100
1101        // Process --iommu flags: open /dev/iommu for each declared context.
1102        let mut iommu_map: HashMap<String, std::fs::File> = HashMap::new();
1103        for iommu_cli in &opt.iommu {
1104            anyhow::ensure!(
1105                !iommu_map.contains_key(&iommu_cli.id),
1106                "duplicate --iommu id={}",
1107                iommu_cli.id
1108            );
1109            let file = std::fs::OpenOptions::new()
1110                .read(true)
1111                .write(true)
1112                .open("/dev/iommu")
1113                .context("failed to open /dev/iommu (is iommufd available?)")?;
1114            iommu_map.insert(iommu_cli.id.clone(), file);
1115        }
1116
1117        opt.vfio
1118            .iter()
1119            .map(|cli_cfg| {
1120                let sysfs_path = Path::new("/sys/bus/pci/devices").join(&cli_cfg.pci_id);
1121
1122                if let Some(iommu_id) = &cli_cfg.iommu {
1123                    // cdev + iommufd path
1124                    let iommufd = iommu_map.get(iommu_id).with_context(|| {
1125                        format!(
1126                            "--vfio device {} references iommu={iommu_id}, \
1127                             but no --iommu id={iommu_id} was specified",
1128                            cli_cfg.pci_id
1129                        )
1130                    })?;
1131                    // Clone the iommufd fd so the per-iommu manager can own it.
1132                    // The first device for a given iommu ID uses the cloned fd
1133                    // to create the IoasManager; subsequent devices reuse the
1134                    // existing manager and the cloned fd is dropped.
1135                    let iommufd = iommufd.try_clone().with_context(|| {
1136                        format!("failed to dup iommufd fd for iommu={iommu_id}")
1137                    })?;
1138
1139                    // Open the cdev device node.
1140                    let vfio_dev_dir = sysfs_path.join("vfio-dev");
1141                    let entry = std::fs::read_dir(&vfio_dev_dir)
1142                        .with_context(|| {
1143                            format!(
1144                                "failed to read {}: is {} bound to vfio-pci?",
1145                                vfio_dev_dir.display(),
1146                                cli_cfg.pci_id
1147                            )
1148                        })?
1149                        .next()
1150                        .context("no vfio-dev entry found")?
1151                        .context("failed to read vfio-dev entry")?;
1152                    let dev_path = Path::new("/dev/vfio/devices").join(entry.file_name());
1153                    let cdev = std::fs::OpenOptions::new()
1154                        .read(true)
1155                        .write(true)
1156                        .open(&dev_path)
1157                        .with_context(|| format!("failed to open {}", dev_path.display()))?;
1158
1159                    Ok(PcieDeviceConfig {
1160                        port_name: cli_cfg.port_name.clone(),
1161                        resource: vfio_assigned_device_resources::VfioCdevDeviceHandle {
1162                            pci_id: cli_cfg.pci_id.clone(),
1163                            cdev,
1164                            iommufd,
1165                            iommu_id: iommu_id.clone(),
1166                            bar_addresses: cli_cfg.bar_addresses,
1167                        }
1168                        .into_resource(),
1169                    })
1170                } else {
1171                    // Legacy group/container path
1172                    let iommu_group_link = std::fs::read_link(sysfs_path.join("iommu_group"))
1173                        .with_context(|| {
1174                            format!("failed to read IOMMU group for {}", cli_cfg.pci_id)
1175                        })?;
1176                    let group_id: u64 = iommu_group_link
1177                        .file_name()
1178                        .and_then(|s| s.to_str())
1179                        .context("invalid iommu_group symlink")?
1180                        .parse()
1181                        .context("failed to parse IOMMU group ID")?;
1182                    let group = std::fs::OpenOptions::new()
1183                        .read(true)
1184                        .write(true)
1185                        .open(format!("/dev/vfio/{group_id}"))
1186                        .with_context(|| format!("failed to open /dev/vfio/{group_id}"))?;
1187
1188                    Ok(PcieDeviceConfig {
1189                        port_name: cli_cfg.port_name.clone(),
1190                        resource: vfio_assigned_device_resources::VfioDeviceHandle {
1191                            pci_id: cli_cfg.pci_id.clone(),
1192                            group,
1193                            bar_addresses: cli_cfg.bar_addresses,
1194                        }
1195                        .into_resource(),
1196                    })
1197                }
1198            })
1199            .collect::<anyhow::Result<Vec<_>>>()?
1200    };
1201
1202    #[cfg(windows)]
1203    let vpci_resources: Vec<_> = opt
1204        .device
1205        .iter()
1206        .map(|path| -> anyhow::Result<_> {
1207            Ok(virt_whp::device::DeviceHandle(
1208                whp::VpciResource::new(
1209                    None,
1210                    Default::default(),
1211                    &whp::VpciResourceDescriptor::Sriov(path, 0, 0),
1212                )
1213                .with_context(|| format!("opening PCI device {}", path))?,
1214            ))
1215        })
1216        .collect::<Result<_, _>>()?;
1217
1218    // Create a vmbusproxy handle if needed by any devices.
1219    #[cfg(windows)]
1220    let vmbusproxy_handle = if !kernel_vmnics.is_empty() {
1221        Some(vmbus_proxy::ProxyHandle::new().context("failed to open vmbusproxy handle")?)
1222    } else {
1223        None
1224    };
1225
1226    let framebuffer = if opt.gfx || opt.vtl2_gfx || opt.vnc.vnc || opt.pcat {
1227        let vram = alloc_shared_memory(FRAMEBUFFER_SIZE, "vram")?;
1228        let (fb, fba) =
1229            framebuffer::framebuffer(vram, FRAMEBUFFER_SIZE, 0).context("creating framebuffer")?;
1230        resources.framebuffer_access = Some(fba);
1231        Some(fb)
1232    } else {
1233        None
1234    };
1235
1236    let load_mode;
1237    let with_hv;
1238
1239    let any_serial_configured = serial0_cfg.is_some()
1240        || serial1_cfg.is_some()
1241        || serial2_cfg.is_some()
1242        || serial3_cfg.is_some();
1243
1244    let has_com3 = serial2_cfg.is_some();
1245
1246    let mut chipset = VmManifestBuilder::new(base_chipset_type(opt), arch);
1247
1248    if framebuffer.is_some() {
1249        chipset = chipset.with_framebuffer();
1250    }
1251    if opt.guest_watchdog {
1252        chipset = chipset.with_guest_watchdog();
1253    }
1254    if any_serial_configured {
1255        chipset = chipset.with_serial([serial0_cfg, serial1_cfg, serial2_cfg, serial3_cfg]);
1256    }
1257    chipset = chipset.with_serial_debugger_mode(com_debugger_mode);
1258    if opt.battery {
1259        let (tx, rx) = mesh::channel();
1260        tx.send(HostBatteryUpdate::default_present());
1261        chipset = chipset.with_battery(rx);
1262    }
1263    if opt.no_vmbus {
1264        chipset = chipset.without_vmbus();
1265    }
1266    if let Some(cfg) = &opt.debugcon {
1267        chipset = chipset.with_debugcon(
1268            debugcon_cfg.unwrap_or_else(|| DisconnectedSerialBackendHandle.into_resource()),
1269            cfg.port,
1270        );
1271    }
1272
1273    let (base_template, custom_uefi_json) = {
1274        #[cfg(guest_arch = "aarch64")]
1275        use firmware_uefi_resources::aarch64_secure_boot_templates as secure_boot_templates;
1276        #[cfg(guest_arch = "x86_64")]
1277        use firmware_uefi_resources::x64_secure_boot_templates as secure_boot_templates;
1278        let base_template = opt.secure_boot_template.map(|template| match template {
1279            SecureBootTemplateCli::Windows => secure_boot_templates::microsoft_windows(),
1280            SecureBootTemplateCli::UefiCa => secure_boot_templates::microsoft_uefi_ca(),
1281        });
1282
1283        // TODO: fallback to VMGS read if no command line flag was given
1284
1285        let custom_uefi_json = match &opt.custom_uefi_json {
1286            Some(file) => Some(
1287                fs_err::read(file)
1288                    .context("opening custom uefi json file")?
1289                    .into(),
1290            ),
1291            None => None,
1292        };
1293
1294        (base_template, custom_uefi_json)
1295    };
1296
1297    if uefi.is_some() || matches!(opt.igvm_personality, Some(IgvmPersonalityCli::Uefi)) {
1298        let log_level = match uefi_options.diagnostics.unwrap_or_default() {
1299            EfiDiagnosticsLogLevelCli::Default => firmware_uefi_resources::LogLevel::make_default(),
1300            EfiDiagnosticsLogLevelCli::Info => firmware_uefi_resources::LogLevel::make_info(),
1301            EfiDiagnosticsLogLevelCli::Full => firmware_uefi_resources::LogLevel::make_full(),
1302        };
1303        let nvram_storage = if opt.vmgs.is_some() {
1304            VmgsFileHandle::new(vmgs_format::FileId::BIOS_NVRAM, true).into_resource()
1305        } else {
1306            EphemeralNonVolatileStoreHandle.into_resource()
1307        };
1308        chipset = chipset.with_uefi(vm_manifest_builder::UefiManifest::new(
1309            arch,
1310            base_template,
1311            custom_uefi_json,
1312            opt.secure_boot,
1313            log_level,
1314            None,
1315            nvram_storage,
1316            None,
1317        ));
1318    }
1319
1320    // Build the SMBIOS config once, up front, so that UEFI and Linux direct
1321    // boot share a single source for the VM's BIOS GUID / system UUID. The TPM
1322    // also keys off this GUID.
1323    let smbios = Box::new(smbios_config_from_cli(&opt.smbios)?);
1324    let bios_guid = smbios.system.uuid;
1325
1326    // Capture the SMBIOS config for the OpenHCL/GED path before `smbios` is
1327    // potentially moved into a non-VTL2 LoadMode below. The GED forwards only
1328    // the system identity to the paravisor and fails closed on BIOS overrides
1329    // it cannot honor, so it is delivered as the shared `SmbiosConfig`.
1330    let ged_smbios = (*smbios).clone();
1331
1332    let layout_config = chipset.layout_config();
1333    let VmChipsetResult {
1334        chipset,
1335        mut chipset_devices,
1336        pci_chipset_devices,
1337        isa_dma_controller,
1338        capabilities,
1339    } = chipset
1340        .build()
1341        .context("failed to build chipset configuration")?;
1342
1343    let tpm_version = opt.tpm.map(|cli_ver| match cli_ver {
1344        TpmVersionCli::V138 => TpmVersion::V138,
1345        TpmVersionCli::V185 => TpmVersion::V185,
1346    });
1347
1348    if opt.restore_snapshot.is_some() {
1349        // Snapshot restore: skip firmware loading entirely. Device state and
1350        // memory come from the snapshot directory.
1351        load_mode = LoadMode::None;
1352        with_hv = true;
1353    } else if let Some(path) = &opt.igvm {
1354        let cli_args::UefiCli {
1355            firmware,
1356            debug: _,
1357            enable_memory_protections: _,
1358            force_dma_bounce: _,
1359            force_firmware_version,
1360            disable_frontpage: _,
1361            console: _,
1362            diagnostics: _,
1363            default_boot_always_attempt: _,
1364        } = uefi_options;
1365
1366        anyhow::ensure!(
1367            firmware.is_none(),
1368            "--uefi firmware is not supported with --igvm"
1369        );
1370        anyhow::ensure!(
1371            !force_firmware_version,
1372            "--uefi force_firmware_version is not supported with --igvm"
1373        );
1374        let file = fs_err::File::open(path)
1375            .context("failed to open igvm file")?
1376            .into();
1377        let cmdline = opt.cmdline.join(" ");
1378        with_hv = match opt.igvm_personality {
1379            None | Some(IgvmPersonalityCli::Uefi) => true,
1380            Some(IgvmPersonalityCli::LinuxDirect) => opt.hv,
1381        };
1382
1383        load_mode = LoadMode::Igvm {
1384            file,
1385            cmdline,
1386            vtl2_base_address: if opt.vtl2 {
1387                opt.igvm_vtl2_relocation_type
1388            } else {
1389                Vtl2BaseAddressType::File
1390            },
1391            com_serial: has_com3.then(|| SerialInformation {
1392                io_port: ComPort::Com3.io_port(),
1393                irq: ComPort::Com3.irq().into(),
1394            }),
1395        };
1396
1397        // An IGVM launch carries no SMBIOS field of its own; the identity is
1398        // only delivered over the GET/GED channel, which is absent here. Reject
1399        // overrides that would otherwise be silently dropped.
1400        let smbios_requested = !opt.smbios.is_empty();
1401        let smbios_delivered_via_get = with_get && with_hv;
1402        if smbios_requested && !smbios_delivered_via_get {
1403            anyhow::bail!(
1404                "--smbios is not supported for IGVM launches without an OpenHCL GET channel"
1405            );
1406        }
1407    } else if opt.pcat {
1408        // Emit a nice error early instead of complaining about missing firmware.
1409        if arch != MachineArch::X86_64 {
1410            anyhow::bail!("pcat not supported on this architecture");
1411        }
1412        with_hv = true;
1413
1414        let firmware = openvmm_pcat_locator::find_pcat_bios(opt.pcat_firmware.as_deref())?;
1415        load_mode = LoadMode::Pcat {
1416            firmware,
1417            boot_order: opt
1418                .pcat_boot_order
1419                .map(|x| x.0)
1420                .unwrap_or(DEFAULT_PCAT_BOOT_ORDER),
1421            hibernation_enabled: opt.hibernation,
1422            smbios,
1423        };
1424    } else if let Some(uefi_options) = &uefi {
1425        use openvmm_defs::config::UefiConsoleMode;
1426
1427        let cli_args::UefiCli {
1428            firmware,
1429            debug,
1430            enable_memory_protections,
1431            force_dma_bounce,
1432            force_firmware_version,
1433            disable_frontpage,
1434            console,
1435            diagnostics: _,
1436            default_boot_always_attempt,
1437        } = uefi_options;
1438
1439        if opt.no_hv && cfg!(guest_arch = "x86_64") {
1440            anyhow::bail!("--no-hv is not supported on x86_64");
1441        }
1442
1443        with_hv = !opt.no_hv;
1444
1445        let default_firmware = cli_args::default_uefi_firmware();
1446        let firmware = fs_err::File::open(
1447            firmware
1448                .as_ref()
1449                .or(default_firmware.as_ref())
1450                .context("must provide uefi firmware when booting with uefi")?,
1451        )
1452        .context("failed to open uefi firmware")?;
1453
1454        // TODO: It would be better to default memory protections to on, but currently Linux does not boot via UEFI due to what
1455        //       appears to be a GRUB memory protection fault. Memory protections are therefore only enabled if configured.
1456        load_mode = LoadMode::Uefi {
1457            firmware: firmware.into(),
1458            enable_debugging: *debug,
1459            enable_memory_protections: *enable_memory_protections,
1460            disable_frontpage: *disable_frontpage,
1461            tpm_version,
1462            enable_battery: opt.battery,
1463            enable_serial: any_serial_configured,
1464            enable_vpci_boot: false,
1465            uefi_console_mode: console.map(|m| match m {
1466                UefiConsoleModeCli::Default => UefiConsoleMode::Default,
1467                UefiConsoleModeCli::Com1 => UefiConsoleMode::Com1,
1468                UefiConsoleModeCli::Com2 => UefiConsoleMode::Com2,
1469                UefiConsoleModeCli::None => UefiConsoleMode::None,
1470            }),
1471            default_boot_always_attempt: *default_boot_always_attempt,
1472            smbios,
1473            enable_vmbus: !opt.no_vmbus,
1474            force_dma_bounce: *force_dma_bounce,
1475            enable_hv: !opt.no_hv,
1476            hibernation_enabled: opt.hibernation,
1477            force_firmware_version: *force_firmware_version,
1478        };
1479    } else {
1480        // Linux Direct
1481        let mut cmdline = "panic=-1 debug".to_string();
1482
1483        with_hv = opt.hv;
1484        if with_hv && opt.pcie_root_complex.is_empty() {
1485            cmdline += " pci=off";
1486        }
1487
1488        if !console_str.is_empty() {
1489            let _ = write!(&mut cmdline, " console={}", console_str);
1490        }
1491
1492        if opt.gfx {
1493            cmdline += " console=tty";
1494        }
1495        for extra in &opt.cmdline {
1496            let _ = write!(&mut cmdline, " {}", extra);
1497        }
1498
1499        let kernel = fs_err::File::open(
1500            (opt.kernel.0)
1501                .as_ref()
1502                .context("must provide kernel when booting with linux direct")?,
1503        )
1504        .context("failed to open kernel")?;
1505        let initrd = (opt.initrd.0)
1506            .as_ref()
1507            .map(fs_err::File::open)
1508            .transpose()
1509            .context("failed to open initrd")?;
1510
1511        load_mode = LoadMode::Linux {
1512            kernel: kernel.into(),
1513            initrd: initrd.map(Into::into),
1514            cmdline,
1515            enable_serial: any_serial_configured,
1516            isolation: if matches!(opt.isolation, Some(cli_args::IsolationCli::Snp)) {
1517                openvmm_defs::config::LinuxIsolationConfig::Snp {
1518                    restricted_injection: opt.snp_restricted_injection,
1519                }
1520            } else {
1521                openvmm_defs::config::LinuxIsolationConfig::None
1522            },
1523            boot_mode: if opt.device_tree {
1524                openvmm_defs::config::LinuxDirectBootMode::DeviceTree
1525            } else {
1526                openvmm_defs::config::LinuxDirectBootMode::Acpi
1527            },
1528            smbios,
1529        };
1530    }
1531
1532    let mut vmgs = Some(if let Some(VmgsCli { kind, provision }) = &opt.vmgs {
1533        let disk = VmgsDisk {
1534            disk: disk_open(kind, false)
1535                .await
1536                .context("failed to open vmgs disk")?,
1537            encryption_policy: if opt.test_gsp_by_id {
1538                GuestStateEncryptionPolicy::GspById(true)
1539            } else {
1540                GuestStateEncryptionPolicy::None(true)
1541            },
1542        };
1543        match provision {
1544            ProvisionVmgs::OnEmpty => VmgsResource::Disk(disk),
1545            ProvisionVmgs::OnFailure => VmgsResource::ReprovisionOnFailure(disk),
1546            ProvisionVmgs::True => VmgsResource::Reprovision(disk),
1547        }
1548    } else {
1549        VmgsResource::Ephemeral
1550    });
1551
1552    if with_get && with_hv {
1553        let has_vtl0_nvme = storage.has_vtl0_nvme();
1554        let vtl2_settings = vtl2_settings_proto::Vtl2Settings {
1555            version: vtl2_settings_proto::vtl2_settings_base::Version::V1.into(),
1556            fixed: Some(Default::default()),
1557            dynamic: Some(vtl2_settings_proto::Vtl2SettingsDynamic {
1558                storage_controllers: storage.build_openhcl_settings(opt.vmbus_redirect),
1559                nic_devices: underhill_nics,
1560            }),
1561            namespace_settings: Vec::default(),
1562        };
1563
1564        // Cache the VTL2 settings for later modification via the interactive console.
1565        resources.vtl2_settings = Some(vtl2_settings.clone());
1566
1567        let (send, guest_request_recv) = mesh::channel();
1568        resources.ged_rpc = Some(send);
1569
1570        let vmgs = vmgs.take().unwrap();
1571
1572        vmbus_devices.extend([
1573            (
1574                openhcl_vtl,
1575                get_resources::gel::GuestEmulationLogHandle.into_resource(),
1576            ),
1577            (
1578                openhcl_vtl,
1579                get_resources::ged::GuestEmulationDeviceHandle {
1580                    firmware: if opt.pcat {
1581                        get_resources::ged::GuestFirmwareConfig::Pcat {
1582                            boot_order: opt
1583                                .pcat_boot_order
1584                                .map_or(DEFAULT_PCAT_BOOT_ORDER, |x| x.0)
1585                                .map(|x| match x {
1586                                    openvmm_defs::config::PcatBootDevice::Floppy => {
1587                                        get_resources::ged::PcatBootDevice::Floppy
1588                                    }
1589                                    openvmm_defs::config::PcatBootDevice::HardDrive => {
1590                                        get_resources::ged::PcatBootDevice::HardDrive
1591                                    }
1592                                    openvmm_defs::config::PcatBootDevice::Optical => {
1593                                        get_resources::ged::PcatBootDevice::Optical
1594                                    }
1595                                    openvmm_defs::config::PcatBootDevice::Network => {
1596                                        get_resources::ged::PcatBootDevice::Network
1597                                    }
1598                                }),
1599                        }
1600                    } else {
1601                        use get_resources::ged::UefiConsoleMode;
1602
1603                        get_resources::ged::GuestFirmwareConfig::Uefi {
1604                            enable_vpci_boot: has_vtl0_nvme,
1605                            firmware_debug: uefi_options.debug,
1606                            enable_memory_protections: uefi_options.enable_memory_protections,
1607                            disable_frontpage: uefi_options.disable_frontpage,
1608                            console_mode: match uefi_options.console.unwrap_or(UefiConsoleModeCli::Default) {
1609                                UefiConsoleModeCli::Default => UefiConsoleMode::Default,
1610                                UefiConsoleModeCli::Com1 => UefiConsoleMode::COM1,
1611                                UefiConsoleModeCli::Com2 => UefiConsoleMode::COM2,
1612                                UefiConsoleModeCli::None => UefiConsoleMode::None,
1613                            },
1614                            default_boot_always_attempt: uefi_options.default_boot_always_attempt,
1615                        }
1616                    },
1617                    com1: with_vmbus_com1_serial,
1618                    com2: with_vmbus_com2_serial,
1619                    serial_tx_only: opt.serial_tx_only,
1620                    vtl2_settings: Some(prost::Message::encode_to_vec(&vtl2_settings)),
1621                    vmbus_redirection: opt.vmbus_redirect,
1622                    vmgs,
1623                    framebuffer: opt
1624                        .vtl2_gfx
1625                        .then(|| SharedFramebufferHandle.into_resource()),
1626                    guest_request_recv,
1627                    tpm_version: tpm_version.map(|v| match v {
1628                        TpmVersion::V138 => get_resources::ged::GedTpmVersion::V138,
1629                        TpmVersion::V185 => get_resources::ged::GedTpmVersion::V185,
1630                    }),
1631                    firmware_event_send: None,
1632                    ipmi_sel_event_send: None,
1633                    secure_boot_enabled: opt.secure_boot,
1634                    secure_boot_template: match opt.secure_boot_template {
1635                        Some(SecureBootTemplateCli::Windows) => {
1636                            get_resources::ged::GuestSecureBootTemplateType::MicrosoftWindows
1637                        },
1638                        Some(SecureBootTemplateCli::UefiCa) => {
1639                            get_resources::ged::GuestSecureBootTemplateType::MicrosoftUefiCertificateAuthority
1640                        }
1641                        None => {
1642                            get_resources::ged::GuestSecureBootTemplateType::None
1643                        },
1644                    },
1645                    enable_battery: opt.battery,
1646                    enable_ipmi: false,
1647                    enable_hibernation: opt.hibernation,
1648                    no_persistent_secrets: true,
1649                    igvm_attest_test_config: None,
1650                    test_gsp_by_id: opt.test_gsp_by_id,
1651                    efi_diagnostics_log_level: {
1652                        match uefi_options.diagnostics.unwrap_or_default() {
1653                            EfiDiagnosticsLogLevelCli::Default => get_resources::ged::EfiDiagnosticsLogLevelType::Default,
1654                            EfiDiagnosticsLogLevelCli::Info => get_resources::ged::EfiDiagnosticsLogLevelType::Info,
1655                            EfiDiagnosticsLogLevelCli::Full => get_resources::ged::EfiDiagnosticsLogLevelType::Full,
1656                        }
1657                    },
1658                    force_dma_bounce_enabled: uefi_options.force_dma_bounce,
1659                    smbios: ged_smbios,
1660                }
1661                .into_resource(),
1662            ),
1663        ]);
1664    }
1665
1666    if let Some(tpm_version) = tpm_version
1667        && !opt.vtl2
1668    {
1669        let register_layout = if cfg!(guest_arch = "x86_64") {
1670            TpmRegisterLayout::IoPort
1671        } else {
1672            TpmRegisterLayout::Mmio
1673        };
1674
1675        let (ppi_store, nvram_store) = if opt.vmgs.is_some() {
1676            (
1677                VmgsFileHandle::new(vmgs_format::FileId::TPM_PPI, true).into_resource(),
1678                VmgsFileHandle::new(tpm_vmgs::tpm_nvram_file_id(tpm_version), true).into_resource(),
1679            )
1680        } else {
1681            (
1682                EphemeralNonVolatileStoreHandle.into_resource(),
1683                EphemeralNonVolatileStoreHandle.into_resource(),
1684            )
1685        };
1686
1687        chipset_devices.push(ChipsetDeviceHandle {
1688            name: "tpm".to_string(),
1689            resource: chipset_device_worker_defs::RemoteChipsetDeviceHandle {
1690                device: TpmDeviceHandle {
1691                    version: tpm_version,
1692                    ppi_store,
1693                    nvram_store,
1694                    nvram_size: None,
1695                    refresh_tpm_seeds: false,
1696                    ak_cert_type: tpm_resources::TpmAkCertTypeResource::None,
1697                    register_layout,
1698                    guest_secret_key: None,
1699                    logger: None,
1700                    is_confidential_vm: false,
1701                    bios_guid,
1702                }
1703                .into_resource(),
1704                worker_host: mesh.make_host("tpm", None).await?,
1705            }
1706            .into_resource(),
1707        });
1708    }
1709
1710    let vga_firmware = if opt.pcat {
1711        Some(openvmm_pcat_locator::find_svga_bios(
1712            opt.vga_firmware.as_deref(),
1713        )?)
1714    } else {
1715        None
1716    };
1717
1718    if opt.gfx {
1719        // Channel for the video device to report dirty rectangles to the VNC worker.
1720        let (dirt_send, dirt_recv) = mesh::channel();
1721        resources.dirty_rect_recv = Some(dirt_recv);
1722
1723        vmbus_devices.extend([
1724            (
1725                DeviceVtl::Vtl0,
1726                SynthVideoHandle {
1727                    framebuffer: SharedFramebufferHandle.into_resource(),
1728                    dirt_send: Some(dirt_send),
1729                }
1730                .into_resource(),
1731            ),
1732            (
1733                DeviceVtl::Vtl0,
1734                SynthKeyboardHandle {
1735                    source: MultiplexedInputHandle {
1736                        // Save 0 for PS/2
1737                        elevation: 1,
1738                    }
1739                    .into_resource(),
1740                }
1741                .into_resource(),
1742            ),
1743            (
1744                DeviceVtl::Vtl0,
1745                SynthMouseHandle {
1746                    source: MultiplexedInputHandle {
1747                        // Save 0 for PS/2
1748                        elevation: 1,
1749                    }
1750                    .into_resource(),
1751                }
1752                .into_resource(),
1753            ),
1754        ]);
1755    }
1756
1757    let vsock_listener = |path: Option<&str>| -> anyhow::Result<_> {
1758        if let Some(path) = path {
1759            cleanup_socket(path.as_ref());
1760            let listener = unix_socket::UnixListener::bind(path)
1761                .with_context(|| format!("failed to bind to hybrid vsock path: {}", path))?;
1762            Ok(Some(listener))
1763        } else {
1764            Ok(None)
1765        }
1766    };
1767
1768    let vtl0_vsock_listener = vsock_listener(opt.vmbus_vsock_path.as_deref())?;
1769    let vtl2_vsock_listener = vsock_listener(opt.vmbus_vtl2_vsock_path.as_deref())?;
1770
1771    if let Some(path) = &opt.openhcl_dump_path {
1772        let (resource, task) = spawn_dump_handler(&spawner, path.clone(), None);
1773        task.detach();
1774        vmbus_devices.push((openhcl_vtl, resource));
1775    }
1776
1777    #[cfg(guest_arch = "aarch64")]
1778    let topology_arch = openvmm_defs::config::ArchTopologyConfig::Aarch64(
1779        openvmm_defs::config::Aarch64TopologyConfig {
1780            // TODO: allow this to be configured from the command line
1781            gic_config: None,
1782            pmu_gsiv: openvmm_defs::config::PmuGsivConfig::Platform,
1783            gic_msi: match opt.gic_msi {
1784                cli_args::GicMsiCli::Auto => openvmm_defs::config::GicMsiConfig::Auto,
1785                cli_args::GicMsiCli::Its => openvmm_defs::config::GicMsiConfig::Its,
1786                cli_args::GicMsiCli::V2m => {
1787                    openvmm_defs::config::GicMsiConfig::V2m { spi_count: None }
1788                }
1789            },
1790        },
1791    );
1792    #[cfg(guest_arch = "x86_64")]
1793    let topology_arch =
1794        openvmm_defs::config::ArchTopologyConfig::X86(openvmm_defs::config::X86TopologyConfig {
1795            apic_id_offset: opt.apic_id_offset,
1796            x2apic: opt.x2apic,
1797        });
1798
1799    let with_isolation = if let Some(isolation) = &opt.isolation {
1800        match isolation {
1801            cli_args::IsolationCli::Vbs => {
1802                // TODO: For now, VBS isolation is only supported with VTL2.
1803                if !opt.vtl2 {
1804                    anyhow::bail!("VBS isolation is only currently supported with vtl2");
1805                }
1806
1807                // TODO: Alias map support is not yet implemented with isolation.
1808                if !opt.no_alias_map {
1809                    anyhow::bail!("alias map not supported with isolation");
1810                }
1811
1812                Some(openvmm_defs::config::IsolationType::Vbs)
1813            }
1814            cli_args::IsolationCli::Snp => Some(openvmm_defs::config::IsolationType::Snp),
1815        }
1816    } else {
1817        None
1818    };
1819
1820    if with_hv && !opt.no_vmbus {
1821        let (shutdown_send, shutdown_recv) = mesh::channel();
1822        resources.shutdown_ic = Some(shutdown_send);
1823        let (kvp_send, kvp_recv) = mesh::channel();
1824        resources.kvp_ic = Some(kvp_send);
1825        vmbus_devices.extend(
1826            [
1827                hyperv_ic_resources::shutdown::ShutdownIcHandle {
1828                    recv: shutdown_recv,
1829                }
1830                .into_resource(),
1831                hyperv_ic_resources::kvp::KvpIcHandle { recv: kvp_recv }.into_resource(),
1832                hyperv_ic_resources::timesync::TimesyncIcHandle.into_resource(),
1833            ]
1834            .map(|r| (DeviceVtl::Vtl0, r)),
1835        );
1836    }
1837
1838    if let Some(hive_path) = &opt.imc {
1839        let file = fs_err::File::open(hive_path).context("failed to open imc hive")?;
1840        vmbus_devices.push((
1841            DeviceVtl::Vtl0,
1842            vmbfs_resources::VmbfsImcDeviceHandle { file: file.into() }.into_resource(),
1843        ));
1844    }
1845
1846    let mut virtio_devices = Vec::new();
1847    let mut add_virtio_device =
1848        |bus, resource: Resource<VirtioDeviceHandle>, pcie_devices: &mut Vec<_>| match bus {
1849            VirtioBusCli::Auto => {
1850                // Use VPCI when possible (currently only on Windows and macOS due
1851                // to KVM backend limitations).
1852                if with_hv && (cfg!(windows) || cfg!(target_os = "macos")) {
1853                    vpci_devices.push(VpciDeviceConfig {
1854                        vtl: DeviceVtl::Vtl0,
1855                        instance_id: Guid::new_random(),
1856                        resource: VirtioPciDeviceHandle(resource).into_resource(),
1857                        vnode: None,
1858                    });
1859                } else {
1860                    virtio_devices.push((VirtioBus::Pci, resource));
1861                }
1862            }
1863            VirtioBusCli::Mmio => virtio_devices.push((VirtioBus::Mmio, resource)),
1864            VirtioBusCli::Pci => virtio_devices.push((VirtioBus::Pci, resource)),
1865            VirtioBusCli::Pcie(port_name) => pcie_devices.push(PcieDeviceConfig {
1866                port_name,
1867                resource: VirtioPciDeviceHandle(resource).into_resource(),
1868            }),
1869            VirtioBusCli::Vpci => vpci_devices.push(VpciDeviceConfig {
1870                vtl: DeviceVtl::Vtl0,
1871                instance_id: Guid::new_random(),
1872                resource: VirtioPciDeviceHandle(resource).into_resource(),
1873                vnode: None,
1874            }),
1875        };
1876
1877    for cli_cfg in &opt.virtio_net {
1878        if cli_cfg.underhill {
1879            anyhow::bail!("use --net uh:[...] to add underhill NICs")
1880        }
1881        let vport = parse_endpoint(cli_cfg, &mut nic_index, &mut resources)?;
1882        let resource = virtio_resources::net::VirtioNetHandle {
1883            max_queues: vport.max_queues,
1884            mac_address: vport.mac_address,
1885            endpoint: vport.endpoint,
1886        }
1887        .into_resource();
1888        if let Some(pcie_port) = &cli_cfg.pcie_port {
1889            pcie_devices.push(PcieDeviceConfig {
1890                port_name: pcie_port.clone(),
1891                resource: VirtioPciDeviceHandle(resource).into_resource(),
1892            });
1893        } else {
1894            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1895        }
1896    }
1897
1898    for args in &opt.virtio_fs {
1899        let resource: Resource<VirtioDeviceHandle> = virtio_resources::fs::VirtioFsHandle {
1900            tag: args.tag.clone(),
1901            fs: virtio_resources::fs::VirtioFsBackend::HostFs {
1902                root_path: args.path.clone(),
1903                mount_options: args.options.clone(),
1904            },
1905        }
1906        .into_resource();
1907        if let Some(pcie_port) = &args.pcie_port {
1908            pcie_devices.push(PcieDeviceConfig {
1909                port_name: pcie_port.clone(),
1910                resource: VirtioPciDeviceHandle(resource).into_resource(),
1911            });
1912        } else {
1913            add_virtio_device(opt.virtio_fs_bus.clone(), resource, &mut pcie_devices);
1914        }
1915    }
1916
1917    for args in &opt.virtio_fs_shmem {
1918        let resource: Resource<VirtioDeviceHandle> = virtio_resources::fs::VirtioFsHandle {
1919            tag: args.tag.clone(),
1920            fs: virtio_resources::fs::VirtioFsBackend::SectionFs {
1921                root_path: args.path.clone(),
1922            },
1923        }
1924        .into_resource();
1925        if let Some(pcie_port) = &args.pcie_port {
1926            pcie_devices.push(PcieDeviceConfig {
1927                port_name: pcie_port.clone(),
1928                resource: VirtioPciDeviceHandle(resource).into_resource(),
1929            });
1930        } else {
1931            add_virtio_device(opt.virtio_fs_bus.clone(), resource, &mut pcie_devices);
1932        }
1933    }
1934
1935    for args in &opt.virtio_9p {
1936        let resource: Resource<VirtioDeviceHandle> = virtio_resources::p9::VirtioPlan9Handle {
1937            tag: args.tag.clone(),
1938            root_path: args.path.clone(),
1939            debug: opt.virtio_9p_debug,
1940        }
1941        .into_resource();
1942        if let Some(pcie_port) = &args.pcie_port {
1943            pcie_devices.push(PcieDeviceConfig {
1944                port_name: pcie_port.clone(),
1945                resource: VirtioPciDeviceHandle(resource).into_resource(),
1946            });
1947        } else {
1948            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1949        }
1950    }
1951
1952    if let Some(pmem_args) = &opt.virtio_pmem {
1953        let resource: Resource<VirtioDeviceHandle> = virtio_resources::pmem::VirtioPmemHandle {
1954            path: pmem_args.path.clone(),
1955        }
1956        .into_resource();
1957        if let Some(pcie_port) = &pmem_args.pcie_port {
1958            pcie_devices.push(PcieDeviceConfig {
1959                port_name: pcie_port.clone(),
1960                resource: VirtioPciDeviceHandle(resource).into_resource(),
1961            });
1962        } else {
1963            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1964        }
1965    }
1966
1967    if opt.virtio_rng {
1968        let resource: Resource<VirtioDeviceHandle> =
1969            virtio_resources::rng::VirtioRngHandle.into_resource();
1970        if let Some(pcie_port) = &opt.virtio_rng_pcie_port {
1971            pcie_devices.push(PcieDeviceConfig {
1972                port_name: pcie_port.clone(),
1973                resource: VirtioPciDeviceHandle(resource).into_resource(),
1974            });
1975        } else {
1976            add_virtio_device(opt.virtio_rng_bus.clone(), resource, &mut pcie_devices);
1977        }
1978    }
1979
1980    if let Some(backend) = virtio_console_backend {
1981        let resource: Resource<VirtioDeviceHandle> =
1982            virtio_resources::console::VirtioConsoleHandle { backend }.into_resource();
1983        if let Some(pcie_port) = &opt.virtio_console_pcie_port {
1984            pcie_devices.push(PcieDeviceConfig {
1985                port_name: pcie_port.clone(),
1986                resource: VirtioPciDeviceHandle(resource).into_resource(),
1987            });
1988        } else {
1989            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
1990        }
1991    }
1992
1993    // Handle --vhost-user arguments.
1994    #[cfg(target_os = "linux")]
1995    for vhost_cli in &opt.vhost_user {
1996        let stream =
1997            unix_socket::UnixStream::connect(&vhost_cli.socket_path).with_context(|| {
1998                format!(
1999                    "failed to connect to vhost-user socket: {}",
2000                    vhost_cli.socket_path
2001                )
2002            })?;
2003
2004        use crate::cli_args::VhostUserDeviceTypeCli;
2005        let resource: Resource<VirtioDeviceHandle> = match vhost_cli.device_type {
2006            VhostUserDeviceTypeCli::Fs {
2007                ref tag,
2008                num_queues,
2009                queue_size,
2010            } => virtio_resources::vhost_user::VhostUserFsHandle {
2011                socket: stream.into(),
2012                tag: tag.clone(),
2013                num_queues,
2014                queue_size,
2015            }
2016            .into_resource(),
2017            VhostUserDeviceTypeCli::Blk {
2018                num_queues,
2019                queue_size,
2020            } => virtio_resources::vhost_user::VhostUserBlkHandle {
2021                socket: stream.into(),
2022                num_queues,
2023                queue_size,
2024            }
2025            .into_resource(),
2026            VhostUserDeviceTypeCli::Other {
2027                device_id,
2028                ref queue_sizes,
2029            } => virtio_resources::vhost_user::VhostUserGenericHandle {
2030                socket: stream.into(),
2031                device_id,
2032                queue_sizes: queue_sizes.clone(),
2033            }
2034            .into_resource(),
2035        };
2036        if let Some(pcie_port) = &vhost_cli.pcie_port {
2037            pcie_devices.push(PcieDeviceConfig {
2038                port_name: pcie_port.clone(),
2039                resource: VirtioPciDeviceHandle(resource).into_resource(),
2040            });
2041        } else {
2042            add_virtio_device(VirtioBusCli::Auto, resource, &mut pcie_devices);
2043        }
2044    }
2045
2046    let virtio_vsock_bus = opt.virtio_vsock_bus.clone().unwrap_or(VirtioBusCli::Auto);
2047
2048    if let Some(vsock_path) = &opt.virtio_vsock_path {
2049        let listener = vsock_listener(Some(vsock_path))?.unwrap();
2050        let resource: Resource<VirtioDeviceHandle> = virtio_resources::vsock::VirtioVsockHandle {
2051            // The guest CID does not matter since the UDS relay does not use it. It just needs
2052            // to be some non-reserved value for the guest to use.
2053            guest_cid: 0x3,
2054            base_path: vsock_path.clone(),
2055            listener,
2056        }
2057        .into_resource();
2058        add_virtio_device(virtio_vsock_bus.clone(), resource, &mut pcie_devices);
2059    }
2060
2061    #[cfg(target_os = "linux")]
2062    if let Some(guest_cid) = opt.virtio_vsock_vhost_cid {
2063        let vhost = std::fs::OpenOptions::new()
2064            .read(true)
2065            .write(true)
2066            .open("/dev/vhost-vsock")
2067            .context("failed to open /dev/vhost-vsock")?
2068            .into();
2069        let resource =
2070            virtio_resources::vsock::VirtioVsockVhostHandle { vhost, guest_cid }.into_resource();
2071        add_virtio_device(virtio_vsock_bus, resource, &mut pcie_devices);
2072    }
2073
2074    #[cfg(target_os = "linux")]
2075    pcie_devices.extend(vfio_pcie_devices);
2076
2077    let mut cfg = Config {
2078        chipset,
2079        load_mode,
2080        floppy_disks,
2081        pcie_root_complexes,
2082        pcie_ecam_below_4gb: opt.pcie_ecam_below_4gb,
2083        #[cfg(target_os = "linux")]
2084        pcie_devices,
2085        #[cfg(not(target_os = "linux"))]
2086        pcie_devices,
2087        pcie_switches,
2088        pcie_generic_initiators,
2089        vpci_devices,
2090        ide_disks: Vec::new(),
2091        numa: {
2092            if let Some(ref nodes) = opt.numa {
2093                // --numa mode: each --numa flag defines a node.
2094                NumaTopology {
2095                    nodes: nodes
2096                        .iter()
2097                        .map(|n| {
2098                            let vps = match &n.vps {
2099                                Some(vps) if vps.0.is_empty() => VpAssignment::Empty,
2100                                Some(vps) => {
2101                                    VpAssignment::Explicit(vps.expand_below(opt.processors)?)
2102                                }
2103                                None => VpAssignment::FromTopology,
2104                            };
2105                            Ok(NumaNode {
2106                                mem: Some(MemoryConfig {
2107                                    mem_size: n
2108                                        .memory
2109                                        .size
2110                                        .expect("NUMA memory size was validated")
2111                                        .0,
2112                                    prefetch_memory: n.memory.prefetch,
2113                                    private_memory: n.memory.shared == Some(false),
2114                                    transparent_hugepages: n
2115                                        .memory
2116                                        .transparent_hugepages
2117                                        .unwrap_or(!n.memory.hugepages),
2118                                    hugepages: n.memory.hugepages,
2119                                    hugepage_size: n.memory.hugepage_size.map(|m| m.0),
2120                                    host_numa_node: n.host_numa_node,
2121                                }),
2122                                vps,
2123                            })
2124                        })
2125                        .collect::<anyhow::Result<Vec<_>>>()?,
2126                    distances: opt
2127                        .numa_distance
2128                        .as_deref()
2129                        .unwrap_or(&[])
2130                        .iter()
2131                        .map(|d| NumaDistance {
2132                            src: d.src,
2133                            dst: d.dst,
2134                            distance: d.distance,
2135                        })
2136                        .collect(),
2137                }
2138            } else {
2139                // Single-node default from --memory.
2140                NumaTopology {
2141                    nodes: vec![NumaNode {
2142                        mem: Some(MemoryConfig {
2143                            mem_size: opt.memory_size(),
2144                            prefetch_memory: opt.prefetch_memory(),
2145                            private_memory: opt.private_memory(),
2146                            transparent_hugepages: opt.transparent_hugepages(),
2147                            hugepages: opt.memory.hugepages,
2148                            hugepage_size: opt.memory.hugepage_size.map(|m| m.0),
2149                            host_numa_node: None,
2150                        }),
2151                        vps: VpAssignment::FromTopology,
2152                    }],
2153                    distances: vec![],
2154                }
2155            }
2156        },
2157        processor_topology: ProcessorTopologyConfig {
2158            proc_count: opt.processors,
2159            vps_per_socket: opt.vps_per_socket,
2160            enable_smt: match opt.smt {
2161                cli_args::SmtConfigCli::Auto => None,
2162                cli_args::SmtConfigCli::Force => Some(true),
2163                cli_args::SmtConfigCli::Off => Some(false),
2164            },
2165            arch: Some(topology_arch),
2166        },
2167        hypervisor: HypervisorConfig {
2168            with_hv,
2169            with_vtl2: opt.vtl2.then_some(Vtl2Config {
2170                vtl0_alias_map: !opt.no_alias_map,
2171                late_map_vtl0_memory: match opt.late_map_vtl0_policy {
2172                    cli_args::Vtl0LateMapPolicyCli::Off => None,
2173                    cli_args::Vtl0LateMapPolicyCli::Log => Some(LateMapVtl0MemoryPolicy::Log),
2174                    cli_args::Vtl0LateMapPolicyCli::Halt => Some(LateMapVtl0MemoryPolicy::Halt),
2175                    cli_args::Vtl0LateMapPolicyCli::Exception => {
2176                        Some(LateMapVtl0MemoryPolicy::InjectException)
2177                    }
2178                },
2179            }),
2180            with_isolation,
2181            nested_virt: opt.nested_virt,
2182        },
2183        #[cfg(windows)]
2184        kernel_vmnics,
2185        input: mesh::Receiver::new(),
2186        framebuffer,
2187        vga_firmware,
2188        vtl2_gfx: opt.vtl2_gfx,
2189        virtio_devices,
2190        vmbus: (with_hv && !opt.no_vmbus).then_some(VmbusConfig {
2191            vsock_listener: vtl0_vsock_listener,
2192            vsock_path: opt.vmbus_vsock_path.clone(),
2193            vtl2_redirect: opt.vmbus_redirect,
2194            vmbus_max_version: opt.vmbus_max_version,
2195            #[cfg(windows)]
2196            vmbusproxy_handle,
2197        }),
2198        vtl2_vmbus: (with_hv && opt.vtl2).then_some(VmbusConfig {
2199            vsock_listener: vtl2_vsock_listener,
2200            vsock_path: opt.vmbus_vtl2_vsock_path.clone(),
2201            ..Default::default()
2202        }),
2203        vmbus_devices,
2204        chipset_devices,
2205        pci_chipset_devices,
2206        isa_dma_controller,
2207        chipset_capabilities: capabilities,
2208        layout: layout_config,
2209        #[cfg(windows)]
2210        vpci_resources,
2211        vmgs,
2212        firmware_event_send: None,
2213        debugger_rpc: None,
2214        rtc_delta_milliseconds: 0,
2215    };
2216
2217    storage.build_config(&mut cfg, &mut resources, opt.scsi_sub_channels)?;
2218    let mut pcie_port_names = HashSet::new();
2219    for device in &cfg.pcie_devices {
2220        anyhow::ensure!(
2221            pcie_port_names.insert(&device.port_name),
2222            "multiple devices use PCIe port '{}'",
2223            device.port_name
2224        );
2225    }
2226    resources.serial_driver = Some(serial_driver);
2227    validate_snp_config(&cfg)?;
2228    Ok((cfg, resources))
2229}
2230
2231fn validate_snp_config(cfg: &Config) -> anyhow::Result<()> {
2232    if cfg.hypervisor.with_isolation != Some(openvmm_defs::config::IsolationType::Snp) {
2233        return Ok(());
2234    }
2235
2236    if !matches!(
2237        cfg.load_mode,
2238        LoadMode::Linux { .. } | LoadMode::Igvm { .. }
2239    ) {
2240        anyhow::bail!("SNP isolation currently only supports Linux direct or IGVM boot");
2241    }
2242    if cfg.hypervisor.with_vtl2.is_some() {
2243        anyhow::bail!("SNP isolation currently does not support VTL2");
2244    }
2245    if cfg.vmbus.is_some() || cfg.vtl2_vmbus.is_some() || !cfg.vmbus_devices.is_empty() {
2246        anyhow::bail!("SNP isolation currently does not support VMBus devices");
2247    }
2248
2249    let only_supported_chipset_devices = cfg.chipset_devices.iter().all(|device| {
2250        matches!(
2251            device.resource.id(),
2252            "serial_16550"
2253                | "pic"
2254                | "pit"
2255                | "generic-ioapic"
2256                | "hyperv_power_management"
2257                | "missing-dev"
2258        )
2259    });
2260    let only_virtio_pcie_devices = cfg
2261        .pcie_devices
2262        .iter()
2263        .all(|device| device.resource.id() == "virtio");
2264    if !cfg.floppy_disks.is_empty()
2265        || !cfg.ide_disks.is_empty()
2266        || !cfg.virtio_devices.is_empty()
2267        || !only_virtio_pcie_devices
2268        || !cfg.vpci_devices.is_empty()
2269        || !only_supported_chipset_devices
2270        || !cfg.pci_chipset_devices.is_empty()
2271    {
2272        anyhow::bail!("SNP isolation currently only supports virtio devices");
2273    }
2274    if cfg.framebuffer.is_some() || cfg.vga_firmware.is_some() || cfg.debugger_rpc.is_some() {
2275        anyhow::bail!("SNP isolation currently does not support this VM configuration");
2276    }
2277
2278    Ok(())
2279}
2280
2281/// Gets the terminal to use for externally launched console windows.
2282pub(crate) fn openvmm_terminal_app() -> Option<PathBuf> {
2283    std::env::var_os("OPENVMM_TERM")
2284        .or_else(|| std::env::var_os("HVLITE_TERM"))
2285        .map(Into::into)
2286}
2287
2288// Tries to remove `path` if it is confirmed to be a Unix socket.
2289fn cleanup_socket(path: &Path) {
2290    #[cfg(windows)]
2291    let is_socket = pal::windows::fs::is_unix_socket(path).unwrap_or(false);
2292    #[cfg(not(windows))]
2293    let is_socket = path
2294        .metadata()
2295        .is_ok_and(|meta| std::os::unix::fs::FileTypeExt::is_socket(&meta.file_type()));
2296
2297    if is_socket {
2298        let _ = std::fs::remove_file(path);
2299    }
2300}
2301
2302#[cfg(windows)]
2303fn new_switch_port(
2304    switch_id: Option<&str>,
2305) -> anyhow::Result<(
2306    openvmm_defs::config::SwitchPortId,
2307    vmswitch::kernel::SwitchPort,
2308)> {
2309    let id = vmswitch::kernel::SwitchPortId {
2310        switch: match switch_id {
2311            Some(s) => s.parse().context("invalid switch id")?,
2312            None => vmswitch::hcn::DEFAULT_SWITCH,
2313        },
2314        port: Guid::new_random(),
2315    };
2316    let _ = vmswitch::hcn::Network::open(&id.switch)
2317        .with_context(|| format!("could not find switch {}", id.switch))?;
2318
2319    let port = vmswitch::kernel::SwitchPort::new(&id).context("failed to create switch port")?;
2320
2321    let id = openvmm_defs::config::SwitchPortId {
2322        switch: id.switch,
2323        port: id.port,
2324    };
2325    Ok((id, port))
2326}
2327
2328fn parse_endpoint(
2329    cli_cfg: &NicConfigCli,
2330    index: &mut usize,
2331    resources: &mut VmResources,
2332) -> anyhow::Result<NicConfig> {
2333    let _ = resources;
2334    let endpoint = match &cli_cfg.endpoint {
2335        EndpointConfigCli::Consomme { cidr, host_fwd } => {
2336            let ports = host_fwd
2337                .iter()
2338                .map(|fwd| {
2339                    use net_backend_resources::consomme::HostPortProtocol;
2340                    net_backend_resources::consomme::HostPortConfig {
2341                        protocol: match fwd.protocol {
2342                            cli_args::HostPortProtocolCli::Tcp => HostPortProtocol::Tcp,
2343                            cli_args::HostPortProtocolCli::Udp => HostPortProtocol::Udp,
2344                        },
2345                        host_address: fwd
2346                            .host_address
2347                            .map(net_backend_resources::consomme::HostIpAddress::from),
2348                        host_port: net_backend_resources::consomme::HostPort::Fixed(fwd.host_port),
2349                        guest_port: fwd.guest_port,
2350                    }
2351                })
2352                .collect();
2353            // Only wire the bind/unbind RPC channel to the first consomme
2354            // endpoint. Additional consomme NICs work normally but cannot be
2355            // targeted by runtime bind/unbind commands.
2356            let recv = if resources.consomme_rpc.is_none() {
2357                let (send, recv) = mesh::channel();
2358                resources.consomme_rpc = Some(send);
2359                Some(recv)
2360            } else {
2361                None
2362            };
2363            net_backend_resources::consomme::ConsommeHandle {
2364                cidr: cidr.clone(),
2365                ports,
2366                recv,
2367            }
2368            .into_resource()
2369        }
2370        EndpointConfigCli::None => net_backend_resources::null::NullHandle.into_resource(),
2371        EndpointConfigCli::Dio { id } => {
2372            #[cfg(windows)]
2373            {
2374                let (port_id, port) = new_switch_port(id.as_deref())?;
2375                resources.switch_ports.push(port);
2376                net_backend_resources::dio::WindowsDirectIoHandle {
2377                    switch_port_id: net_backend_resources::dio::SwitchPortId {
2378                        switch: port_id.switch,
2379                        port: port_id.port,
2380                    },
2381                }
2382                .into_resource()
2383            }
2384
2385            #[cfg(not(windows))]
2386            {
2387                let _ = id;
2388                bail!("cannot use dio on non-windows platforms")
2389            }
2390        }
2391        EndpointConfigCli::Tap { name } => {
2392            #[cfg(target_os = "linux")]
2393            {
2394                let fd = net_tap::tap::open_tap(name)
2395                    .with_context(|| format!("failed to open TAP device '{name}'"))?;
2396                net_backend_resources::tap::TapHandle { fd }.into_resource()
2397            }
2398
2399            #[cfg(not(target_os = "linux"))]
2400            {
2401                let _ = name;
2402                bail!("TAP backend is only supported on Linux")
2403            }
2404        }
2405    };
2406
2407    // Pick a random MAC address.
2408    let mut mac_address = [0x00, 0x15, 0x5D, 0, 0, 0];
2409    getrandom::fill(&mut mac_address[3..]).expect("rng failure");
2410
2411    // Pick a fixed instance ID based on the index.
2412    const BASE_INSTANCE_ID: Guid = guid::guid!("00000000-da43-11ed-936a-00155d6db52f");
2413    let instance_id = Guid {
2414        data1: *index as u32,
2415        ..BASE_INSTANCE_ID
2416    };
2417    *index += 1;
2418
2419    Ok(NicConfig {
2420        vtl: cli_cfg.vtl,
2421        instance_id,
2422        endpoint,
2423        mac_address: mac_address.into(),
2424        max_queues: cli_cfg.max_queues,
2425        pcie_port: cli_cfg.pcie_port.clone(),
2426    })
2427}
2428
2429#[derive(Debug)]
2430struct NicConfig {
2431    vtl: DeviceVtl,
2432    instance_id: Guid,
2433    mac_address: MacAddress,
2434    endpoint: Resource<NetEndpointHandleKind>,
2435    max_queues: Option<u16>,
2436    pcie_port: Option<String>,
2437}
2438
2439impl NicConfig {
2440    fn into_netvsp_handle(self) -> (DeviceVtl, Resource<VmbusDeviceHandleKind>) {
2441        (
2442            self.vtl,
2443            netvsp_resources::NetvspHandle {
2444                instance_id: self.instance_id,
2445                mac_address: self.mac_address,
2446                endpoint: self.endpoint,
2447                max_queues: self.max_queues,
2448            }
2449            .into_resource(),
2450        )
2451    }
2452}
2453
2454enum LayerOrDisk {
2455    Layer(DiskLayerDescription),
2456    Disk(Resource<DiskHandleKind>),
2457}
2458
2459async fn disk_open(
2460    disk_cli: &DiskCliKind,
2461    read_only: bool,
2462) -> anyhow::Result<Resource<DiskHandleKind>> {
2463    let mut layers = Vec::new();
2464    disk_open_inner(disk_cli, read_only, &mut layers).await?;
2465    if layers.len() == 1 && matches!(layers[0], LayerOrDisk::Disk(_)) {
2466        let LayerOrDisk::Disk(disk) = layers.pop().unwrap() else {
2467            unreachable!()
2468        };
2469        Ok(disk)
2470    } else {
2471        Ok(Resource::new(disk_backend_resources::LayeredDiskHandle {
2472            layers: layers
2473                .into_iter()
2474                .map(|layer| match layer {
2475                    LayerOrDisk::Layer(layer) => layer,
2476                    LayerOrDisk::Disk(disk) => DiskLayerDescription {
2477                        layer: DiskLayerHandle(disk).into_resource(),
2478                        read_cache: false,
2479                        write_through: false,
2480                    },
2481                })
2482                .collect(),
2483        }))
2484    }
2485}
2486
2487fn disk_open_inner<'a>(
2488    disk_cli: &'a DiskCliKind,
2489    read_only: bool,
2490    layers: &'a mut Vec<LayerOrDisk>,
2491) -> futures::future::BoxFuture<'a, anyhow::Result<()>> {
2492    Box::pin(async move {
2493        fn layer<T: IntoResource<DiskLayerHandleKind>>(layer: T) -> LayerOrDisk {
2494            LayerOrDisk::Layer(layer.into_resource().into())
2495        }
2496        fn disk<T: IntoResource<DiskHandleKind>>(disk: T) -> LayerOrDisk {
2497            LayerOrDisk::Disk(disk.into_resource())
2498        }
2499        match disk_cli {
2500            &DiskCliKind::Memory(len) => {
2501                layers.push(layer(RamDiskLayerHandle {
2502                    len: Some(len),
2503                    sector_size: None,
2504                }));
2505            }
2506            DiskCliKind::File {
2507                path,
2508                create_with_len,
2509                direct,
2510            } => layers.push(LayerOrDisk::Disk(if let Some(size) = create_with_len {
2511                create_disk_type(
2512                    path,
2513                    *size,
2514                    OpenDiskOptions {
2515                        read_only: false,
2516                        direct: *direct,
2517                    },
2518                )
2519                .with_context(|| format!("failed to create {}", path.display()))?
2520            } else {
2521                open_disk_type(
2522                    path,
2523                    OpenDiskOptions {
2524                        read_only,
2525                        direct: *direct,
2526                    },
2527                )
2528                .await
2529                .with_context(|| format!("failed to open {}", path.display()))?
2530            })),
2531            DiskCliKind::Blob { kind, url } => {
2532                layers.push(disk(disk_backend_resources::BlobDiskHandle {
2533                    url: url.to_owned(),
2534                    format: match kind {
2535                        cli_args::BlobKind::Flat => disk_backend_resources::BlobDiskFormat::Flat,
2536                        cli_args::BlobKind::Vhd1 => {
2537                            disk_backend_resources::BlobDiskFormat::FixedVhd1
2538                        }
2539                    },
2540                }))
2541            }
2542            DiskCliKind::MemoryDiff(inner) => {
2543                layers.push(layer(RamDiskLayerHandle {
2544                    len: None,
2545                    sector_size: None,
2546                }));
2547                disk_open_inner(inner, true, layers).await?;
2548            }
2549            DiskCliKind::PersistentReservationsWrapper(inner) => {
2550                layers.push(disk(disk_backend_resources::DiskWithReservationsHandle(
2551                    disk_open(inner, read_only).await?,
2552                )))
2553            }
2554            DiskCliKind::DelayDiskWrapper {
2555                delay_ms,
2556                disk: inner,
2557            } => layers.push(disk(DelayDiskHandle {
2558                delay: CellUpdater::new(Duration::from_millis(*delay_ms)).cell(),
2559                disk: disk_open(inner, read_only).await?,
2560            })),
2561            DiskCliKind::Crypt {
2562                disk: inner,
2563                cipher,
2564                key_file,
2565            } => layers.push(disk(disk_crypt_resources::DiskCryptHandle {
2566                disk: disk_open(inner, read_only).await?,
2567                cipher: match cipher {
2568                    cli_args::DiskCipher::XtsAes256 => disk_crypt_resources::Cipher::XtsAes256,
2569                },
2570                key: fs_err::read(key_file).context("failed to read key file")?,
2571            })),
2572            DiskCliKind::Sqlite {
2573                path,
2574                create_with_len,
2575            } => {
2576                // FUTURE: this code should be responsible for opening
2577                // file-handle(s) itself, and passing them into sqlite via a custom
2578                // vfs. For now though - simply check if the file exists or not, and
2579                // perform early validation of filesystem-level create options.
2580                match (create_with_len.is_some(), path.exists()) {
2581                    (true, true) => anyhow::bail!(
2582                        "cannot create new sqlite disk at {} - file already exists",
2583                        path.display()
2584                    ),
2585                    (false, false) => anyhow::bail!(
2586                        "cannot open sqlite disk at {} - file not found",
2587                        path.display()
2588                    ),
2589                    _ => {}
2590                }
2591
2592                layers.push(layer(SqliteDiskLayerHandle {
2593                    dbhd_path: path.display().to_string(),
2594                    format_dbhd: create_with_len.map(|len| {
2595                        disk_backend_resources::layer::SqliteDiskLayerFormatParams {
2596                            logically_read_only: false,
2597                            len: Some(len),
2598                        }
2599                    }),
2600                }));
2601            }
2602            DiskCliKind::SqliteDiff { path, create, disk } => {
2603                // FUTURE: this code should be responsible for opening
2604                // file-handle(s) itself, and passing them into sqlite via a custom
2605                // vfs. For now though - simply check if the file exists or not, and
2606                // perform early validation of filesystem-level create options.
2607                match (create, path.exists()) {
2608                    (true, true) => anyhow::bail!(
2609                        "cannot create new sqlite disk at {} - file already exists",
2610                        path.display()
2611                    ),
2612                    (false, false) => anyhow::bail!(
2613                        "cannot open sqlite disk at {} - file not found",
2614                        path.display()
2615                    ),
2616                    _ => {}
2617                }
2618
2619                layers.push(layer(SqliteDiskLayerHandle {
2620                    dbhd_path: path.display().to_string(),
2621                    format_dbhd: create.then_some(
2622                        disk_backend_resources::layer::SqliteDiskLayerFormatParams {
2623                            logically_read_only: false,
2624                            len: None,
2625                        },
2626                    ),
2627                }));
2628                disk_open_inner(disk, true, layers).await?;
2629            }
2630            DiskCliKind::AutoCacheSqlite {
2631                cache_path,
2632                key,
2633                disk,
2634            } => {
2635                layers.push(LayerOrDisk::Layer(DiskLayerDescription {
2636                    read_cache: true,
2637                    write_through: false,
2638                    layer: SqliteAutoCacheDiskLayerHandle {
2639                        cache_path: cache_path.clone(),
2640                        cache_key: key.clone(),
2641                    }
2642                    .into_resource(),
2643                }));
2644                disk_open_inner(disk, read_only, layers).await?;
2645            }
2646        }
2647        Ok(())
2648    })
2649}
2650
2651/// Get the system page size.
2652pub(crate) fn system_page_size() -> u32 {
2653    sparse_mmap::SparseMapping::page_size() as u32
2654}
2655
2656/// The guest architecture string, derived from the compile-time `guest_arch` cfg.
2657pub(crate) const GUEST_ARCH: &str = if cfg!(guest_arch = "x86_64") {
2658    "x86_64"
2659} else {
2660    "aarch64"
2661};
2662
2663/// Open a snapshot directory and validate it against the current VM config.
2664/// Returns the shared memory fd (from memory.bin) and the saved device state.
2665fn prepare_snapshot_restore(
2666    snapshot_dir: &Path,
2667    opt: &Options,
2668) -> anyhow::Result<(
2669    openvmm_defs::worker::SharedMemoryFd,
2670    mesh::payload::message::ProtobufMessage,
2671)> {
2672    let (manifest, state_bytes) = openvmm_helpers::snapshot::read_snapshot(snapshot_dir)?;
2673
2674    // Validate manifest against current VM config.
2675    openvmm_helpers::snapshot::validate_manifest(
2676        &manifest,
2677        GUEST_ARCH,
2678        opt.memory_size(),
2679        opt.processors,
2680        system_page_size(),
2681    )?;
2682
2683    // Open memory.bin (existing file, no create, no resize).
2684    let memory_file = fs_err::OpenOptions::new()
2685        .read(true)
2686        .write(true)
2687        .open(snapshot_dir.join("memory.bin"))?;
2688
2689    // Validate file size matches expected memory size.
2690    let file_size = memory_file.metadata()?.len();
2691    if file_size != manifest.memory_size_bytes {
2692        anyhow::bail!(
2693            "memory.bin size ({file_size} bytes) doesn't match manifest ({} bytes)",
2694            manifest.memory_size_bytes,
2695        );
2696    }
2697
2698    let shared_memory_fd =
2699        openvmm_helpers::shared_memory::file_to_shared_memory_fd(memory_file.into())?;
2700
2701    // Reconstruct ProtobufMessage from the saved state bytes.
2702    // The save side wrote mesh::payload::encode(ProtobufMessage), so we decode
2703    // back to ProtobufMessage.
2704    let state_msg: mesh::payload::message::ProtobufMessage = mesh::payload::decode(&state_bytes)
2705        .context("failed to decode saved state from snapshot")?;
2706
2707    Ok((shared_memory_fd, state_msg))
2708}
2709
2710fn do_main(pidfile_guard: &mut Option<pidfile::Pidfile>) -> anyhow::Result<i32> {
2711    #[cfg(windows)]
2712    pal::windows::disable_hard_error_dialog();
2713
2714    tracing_init::enable_tracing()?;
2715
2716    // Try to run as a worker host.
2717    // On success the worker runs to completion and then exits the process (does
2718    // not return). Any worker host setup errors are return and bubbled up.
2719    meshworker::run_vmm_mesh_host()?;
2720
2721    let opt = cli_args::parse_options();
2722
2723    // Print the version number. This comes after argument parsing to not interfere
2724    // with --version and --help.
2725    tracing::info!(version = openvmm_build_info::get().version());
2726
2727    if let Some(path) = &opt.write_saved_state_proto {
2728        mesh::payload::protofile::DescriptorWriter::new(vmcore::save_restore::saved_state_roots())
2729            .write_to_path(path)
2730            .context("failed to write protobuf descriptors")?;
2731        return Ok(0);
2732    }
2733
2734    if let Some(ref path) = opt.pidfile {
2735        *pidfile_guard = Some(pidfile::Pidfile::new(path).context("failed to create pidfile")?);
2736    }
2737
2738    if let Some(path) = opt.relay_console_path {
2739        let console_title = opt.relay_console_title.unwrap_or_default();
2740        return console_relay::relay_console(&path, console_title.as_str()).map(|()| 0);
2741    }
2742
2743    #[cfg(any(feature = "grpc", feature = "ttrpc"))]
2744    {
2745        let rpc = opt
2746            .rpc
2747            .as_ref()
2748            .map(|rpc| {
2749                let transport = match rpc.transport {
2750                    cli_args::RpcTransportCli::Auto => ttrpc::RpcTransport::Auto,
2751                    cli_args::RpcTransportCli::Ttrpc => ttrpc::RpcTransport::Ttrpc,
2752                    cli_args::RpcTransportCli::Grpc => ttrpc::RpcTransport::Grpc,
2753                };
2754                (rpc.path.as_path(), transport)
2755            })
2756            .or_else(|| {
2757                opt.ttrpc
2758                    .as_deref()
2759                    .map(|p| (p, ttrpc::RpcTransport::Ttrpc))
2760            })
2761            .or_else(|| opt.grpc.as_deref().map(|p| (p, ttrpc::RpcTransport::Grpc)));
2762
2763        if let Some((path, transport)) = rpc {
2764            return block_on(async {
2765                let _ = std::fs::remove_file(path);
2766                let listener =
2767                    unix_socket::UnixListener::bind(path).context("failed to bind to socket")?;
2768
2769                // This is a local launch
2770                let mut handle =
2771                    mesh_worker::launch_local_worker::<ttrpc::TtrpcWorker>(ttrpc::Parameters {
2772                        listener,
2773                        transport,
2774                    })
2775                    .await?;
2776
2777                tracing::info!(%transport, path = %path.display(), "listening");
2778
2779                // Signal the parent process that the server is ready.
2780                pal::close_stdout().context("failed to close stdout")?;
2781
2782                handle.join().await?;
2783
2784                Ok(0)
2785            });
2786        }
2787    }
2788
2789    DefaultPool::run_with(async |driver| run_control(&driver, opt).await)
2790}
2791
2792fn new_hvsock_service_id(port: u32) -> Guid {
2793    // This GUID is an embedding of the AF_VSOCK port into an
2794    // AF_HYPERV service ID.
2795    Guid {
2796        data1: port,
2797        .."00000000-facb-11e6-bd58-64006a7986d3".parse().unwrap()
2798    }
2799}
2800
2801async fn run_control(driver: &DefaultDriver, opt: Options) -> anyhow::Result<i32> {
2802    let mut mesh = Some(VmmMesh::new(&driver, opt.single_process)?);
2803    let result = run_control_inner(driver, &mut mesh, opt).await;
2804    // If setup failed before the mesh was handed to the controller, shut it
2805    // down so the child host process exits cleanly without noisy logs.
2806    if let Some(mesh) = mesh {
2807        mesh.shutdown().await;
2808    }
2809    result
2810}
2811
2812async fn run_control_inner(
2813    driver: &DefaultDriver,
2814    mesh_slot: &mut Option<VmmMesh>,
2815    opt: Options,
2816) -> anyhow::Result<i32> {
2817    let mesh = mesh_slot.as_ref().unwrap();
2818    let (mut vm_config, mut resources) = vm_config_from_command_line(driver, mesh, &opt).await?;
2819
2820    let mut vnc_worker = None;
2821    if opt.gfx || opt.vnc.vnc {
2822        // Parse the listen address. Try as a full SocketAddr (host:port) first;
2823        // fall back to a bare IP, using the configured port.
2824        let addr: std::net::SocketAddr = if let Ok(sa) =
2825            opt.vnc.vnc_listen.parse::<std::net::SocketAddr>()
2826        {
2827            sa
2828        } else {
2829            let ip: std::net::IpAddr = opt.vnc.vnc_listen.parse().with_context(|| {
2830                format!(
2831                    "invalid VNC listen address: {} (expected IP address or socket address like [::1]:5900)",
2832                    opt.vnc.vnc_listen
2833                )
2834            })?;
2835            std::net::SocketAddr::new(ip, opt.vnc.vnc_port)
2836        };
2837
2838        let socket = socket2::Socket::new(
2839            if addr.is_ipv6() {
2840                socket2::Domain::IPV6
2841            } else {
2842                socket2::Domain::IPV4
2843            },
2844            socket2::Type::STREAM,
2845            None,
2846        )
2847        .with_context(|| format!("creating VNC socket for {}", addr))?;
2848
2849        if addr.is_ipv6() {
2850            if let Err(e) = socket.set_only_v6(false) {
2851                tracing::warn!(
2852                    error = %e,
2853                    "failed to enable dual-stack on IPv6 VNC socket, IPv4 clients may not be able to connect"
2854                );
2855            }
2856        }
2857        socket.set_reuse_address(true)?;
2858        socket
2859            .bind(&addr.into())
2860            .with_context(|| format!("binding VNC socket to {}", addr))?;
2861        socket
2862            .listen(128)
2863            .with_context(|| format!("listening on VNC socket {}", addr))?;
2864        let listener: TcpListener = socket.into();
2865
2866        if !addr.ip().is_loopback() {
2867            tracing::warn!(
2868                address = %addr,
2869                "VNC server listening on non-localhost address without authentication"
2870            );
2871        }
2872
2873        let input_send = vm_config.input.sender();
2874        let framebuffer = resources
2875            .framebuffer_access
2876            .take()
2877            .expect("synth video enabled");
2878
2879        let vnc_host = mesh
2880            .make_host("vnc", None)
2881            .await
2882            .context("spawning vnc process failed")?;
2883
2884        vnc_worker = Some(
2885            vnc_host
2886                .launch_worker(
2887                    vnc_worker_defs::VNC_WORKER_TCP,
2888                    VncParameters {
2889                        listener,
2890                        framebuffer,
2891                        input_send,
2892                        dirty_recv: resources.dirty_rect_recv.take(),
2893                        max_clients: opt.vnc.vnc_max_clients,
2894                        evict_oldest: opt.vnc.vnc_evict_oldest,
2895                    },
2896                )
2897                .await?,
2898        )
2899    }
2900
2901    // spin up the debug worker
2902    let gdb_worker = if let Some(port) = opt.gdb {
2903        let listener = TcpListener::bind(format!("127.0.0.1:{}", port))
2904            .with_context(|| format!("binding to gdb port {}", port))?;
2905
2906        let (req_tx, req_rx) = mesh::channel();
2907        vm_config.debugger_rpc = Some(req_rx);
2908
2909        let gdb_host = mesh
2910            .make_host("gdb", None)
2911            .await
2912            .context("spawning gdbstub process failed")?;
2913
2914        Some(
2915            gdb_host
2916                .launch_worker(
2917                    debug_worker_defs::DEBUGGER_WORKER,
2918                    debug_worker_defs::DebuggerParameters {
2919                        listener,
2920                        req_chan: req_tx,
2921                        vp_count: vm_config.processor_topology.proc_count,
2922                        target_arch: if cfg!(guest_arch = "x86_64") {
2923                            debug_worker_defs::TargetArch::X86_64
2924                        } else {
2925                            debug_worker_defs::TargetArch::Aarch64
2926                        },
2927                    },
2928                )
2929                .await
2930                .context("failed to launch gdbstub worker")?,
2931        )
2932    } else {
2933        None
2934    };
2935
2936    // spin up the VM
2937    let (vm_rpc, rpc_recv) = mesh::channel();
2938    let (notify_send, notify_recv) = mesh::channel();
2939    let vm_worker = {
2940        let vm_host = mesh.make_host("vm", opt.log_file.clone()).await?;
2941
2942        let (shared_memory, saved_state) = if let Some(snapshot_dir) = &opt.restore_snapshot {
2943            let (fd, state_msg) = prepare_snapshot_restore(snapshot_dir, &opt)?;
2944            (Some(fd), Some(state_msg))
2945        } else {
2946            let shared_memory = opt
2947                .memory_backing_file()
2948                .map(|path| {
2949                    openvmm_helpers::shared_memory::open_memory_backing_file(
2950                        path,
2951                        opt.memory_size(),
2952                    )
2953                })
2954                .transpose()?;
2955            (shared_memory, None)
2956        };
2957
2958        let params = VmWorkerParameters {
2959            hypervisor: match &opt.hypervisor {
2960                Some(name) => openvmm_helpers::hypervisor::hypervisor_resource(name)?,
2961                None => openvmm_helpers::hypervisor::choose_hypervisor()?,
2962            },
2963            cfg: vm_config,
2964            saved_state,
2965            shared_memory,
2966            rpc: rpc_recv,
2967            notify: notify_send,
2968        };
2969        vm_host
2970            .launch_worker(VM_WORKER, params)
2971            .await
2972            .context("failed to launch vm worker")?
2973    };
2974
2975    if opt.restore_snapshot.is_some() {
2976        tracing::info!("restoring VM from snapshot");
2977    }
2978
2979    if !opt.paused {
2980        vm_rpc.call(VmRpc::Resume, ()).await?;
2981    }
2982
2983    let paravisor_diag = Arc::new(diag_client::DiagClient::from_dialer(
2984        driver.clone(),
2985        DiagDialer {
2986            driver: driver.clone(),
2987            vm_rpc: vm_rpc.clone(),
2988            openhcl_vtl: if opt.vtl2 {
2989                DeviceVtl::Vtl2
2990            } else {
2991                DeviceVtl::Vtl0
2992            },
2993        },
2994    ));
2995
2996    let diag_inspector = DiagInspector::new(driver.clone(), paravisor_diag.clone());
2997
2998    // Create channels between the REPL and VmController.
2999    let (vm_controller_send, vm_controller_recv) = mesh::channel();
3000    let (vm_controller_event_send, vm_controller_event_recv) = mesh::channel();
3001
3002    let has_vtl2 = resources.vtl2_settings.is_some();
3003    let serial_driver = resources
3004        .serial_driver
3005        .take()
3006        .expect("serial driver must outlive serial resources");
3007
3008    // Build the VmController with exclusive resources.
3009    let controller = vm_controller::VmController {
3010        mesh: mesh_slot.take().unwrap(),
3011        vm_worker,
3012        vnc_worker,
3013        gdb_worker,
3014        diag_inspector: Some(diag_inspector),
3015        vtl2_settings: resources.vtl2_settings,
3016        ged_rpc: resources.ged_rpc.clone(),
3017        vm_rpc: vm_rpc.clone(),
3018        paravisor_diag: Some(paravisor_diag),
3019        igvm_path: opt.igvm.clone(),
3020        memory_backing_file: opt.memory_backing_file().cloned(),
3021        memory: opt.memory_size(),
3022        processors: opt.processors,
3023        log_file: opt.log_file.clone(),
3024        crash_dump_path: opt.crash_dump_path.clone(),
3025        guest_power_actions: vm_controller::GuestPowerActions {
3026            shutdown: opt.guest_shutdown_action,
3027            reset: opt.guest_reset_action,
3028            crash: opt.guest_crash_action,
3029            watchdog: opt.guest_watchdog_action,
3030        },
3031    };
3032
3033    // Spawn the VmController as a task.
3034    let controller_task = driver.spawn(
3035        "vm-controller",
3036        controller.run(vm_controller_recv, vm_controller_event_send, notify_recv),
3037    );
3038
3039    // Run the REPL with shareable resources.
3040    let repl_result = repl::run_repl(
3041        driver,
3042        repl::ReplResources {
3043            vm_rpc,
3044            vm_controller: vm_controller_send,
3045            vm_controller_events: vm_controller_event_recv,
3046            scsi_rpc: resources.scsi_rpc,
3047            nvme_vtl2_rpc: resources.nvme_vtl2_rpc,
3048            consomme_rpc: resources.consomme_rpc,
3049            shutdown_ic: resources.shutdown_ic,
3050            kvp_ic: resources.kvp_ic,
3051            console_in: resources.console_in,
3052            has_vtl2,
3053        },
3054    )
3055    .await;
3056
3057    // Wait for the controller task to finish (it stops the VM worker and
3058    // shuts down the mesh).
3059    controller_task.await;
3060    drop(serial_driver);
3061
3062    // run_repl returns the exit status: the code the guest drove via an opt-in
3063    // exit (VmControllerEvent::ExitRequested), or 0 when the VM stopped normally.
3064    repl_result
3065}
3066
3067struct DiagDialer {
3068    driver: DefaultDriver,
3069    vm_rpc: mesh::Sender<VmRpc>,
3070    openhcl_vtl: DeviceVtl,
3071}
3072
3073impl mesh_rpc::client::Dial for DiagDialer {
3074    type Stream = PolledSocket<unix_socket::UnixStream>;
3075
3076    async fn dial(&mut self) -> io::Result<Self::Stream> {
3077        let service_id = new_hvsock_service_id(1);
3078        let socket = self
3079            .vm_rpc
3080            .call_failable(
3081                VmRpc::ConnectHvsock,
3082                (
3083                    CancelContext::new().with_timeout(Duration::from_secs(2)),
3084                    service_id,
3085                    self.openhcl_vtl,
3086                ),
3087            )
3088            .await
3089            .map_err(io::Error::other)?;
3090
3091        PolledSocket::new(&self.driver, socket)
3092    }
3093}
3094
3095/// An object that implements [`InspectMut`] by sending an inspect request over
3096/// TTRPC to the guest (typically the paravisor running in VTL2), then stitching
3097/// the response back into the inspect tree.
3098///
3099/// This also caches the TTRPC connection to the guest so that only the first
3100/// inspect request has to wait for the connection to be established.
3101pub(crate) struct DiagInspector(DiagInspectorInner);
3102
3103enum DiagInspectorInner {
3104    NotStarted(DefaultDriver, Arc<diag_client::DiagClient>),
3105    Started {
3106        send: mesh::Sender<inspect::Deferred>,
3107        _task: Task<()>,
3108    },
3109    Invalid,
3110}
3111
3112impl DiagInspector {
3113    pub fn new(driver: DefaultDriver, diag_client: Arc<diag_client::DiagClient>) -> Self {
3114        Self(DiagInspectorInner::NotStarted(driver, diag_client))
3115    }
3116
3117    fn start(&mut self) -> &mesh::Sender<inspect::Deferred> {
3118        loop {
3119            match self.0 {
3120                DiagInspectorInner::NotStarted { .. } => {
3121                    let DiagInspectorInner::NotStarted(driver, client) =
3122                        std::mem::replace(&mut self.0, DiagInspectorInner::Invalid)
3123                    else {
3124                        unreachable!()
3125                    };
3126                    let (send, recv) = mesh::channel();
3127                    let task = driver.clone().spawn("diag-inspect", async move {
3128                        Self::run(&client, recv).await
3129                    });
3130
3131                    self.0 = DiagInspectorInner::Started { send, _task: task };
3132                }
3133                DiagInspectorInner::Started { ref send, .. } => break send,
3134                DiagInspectorInner::Invalid => unreachable!(),
3135            }
3136        }
3137    }
3138
3139    async fn run(
3140        diag_client: &diag_client::DiagClient,
3141        mut recv: mesh::Receiver<inspect::Deferred>,
3142    ) {
3143        while let Some(deferred) = recv.next().await {
3144            let info = deferred.external_request();
3145            let result = match info.request_type {
3146                inspect::ExternalRequestType::Inspect { depth } => {
3147                    if depth == 0 {
3148                        Ok(inspect::Node::Unevaluated)
3149                    } else {
3150                        // TODO: Support taking timeouts from the command line
3151                        diag_client
3152                            .inspect(info.path, Some(depth - 1), Some(Duration::from_secs(1)))
3153                            .await
3154                    }
3155                }
3156                inspect::ExternalRequestType::Update { value } => {
3157                    (diag_client.update(info.path, value).await).map(inspect::Node::Value)
3158                }
3159            };
3160            deferred.complete_external(
3161                result.unwrap_or_else(|err| {
3162                    inspect::Node::Failed(inspect::Error::Mesh(format!("{err:#}")))
3163                }),
3164                inspect::SensitivityLevel::Unspecified,
3165            )
3166        }
3167    }
3168}
3169
3170impl InspectMut for DiagInspector {
3171    fn inspect_mut(&mut self, req: inspect::Request<'_>) {
3172        self.start().send(req.defer());
3173    }
3174}
3175
3176#[cfg(test)]
3177mod tests {
3178    use super::*;
3179    use clap::Parser;
3180    use std::fs::File;
3181    use test_with_tracing::test;
3182
3183    #[test]
3184    fn maps_igvm_personalities_to_chipsets() {
3185        for (args, expected) in [
3186            (
3187                vec![
3188                    "openvmm",
3189                    "--igvm",
3190                    "guest.igvm",
3191                    "--igvm-personality",
3192                    "uefi",
3193                ],
3194                BaseChipsetType::HypervGen2Uefi,
3195            ),
3196            (
3197                vec![
3198                    "openvmm",
3199                    "--igvm",
3200                    "guest.igvm",
3201                    "--igvm-personality",
3202                    "linux-direct",
3203                ],
3204                BaseChipsetType::UnenlightenedLinuxDirect,
3205            ),
3206            (
3207                vec![
3208                    "openvmm",
3209                    "--igvm",
3210                    "guest.igvm",
3211                    "--igvm-personality",
3212                    "linux-direct",
3213                    "--hv",
3214                ],
3215                BaseChipsetType::HyperVGen2LinuxDirect,
3216            ),
3217            (
3218                vec![
3219                    "openvmm",
3220                    "--igvm",
3221                    "guest.igvm",
3222                    "--igvm-personality",
3223                    "linux-direct",
3224                    "--isolation",
3225                    "snp",
3226                ],
3227                BaseChipsetType::EnlightenedLinuxDirect,
3228            ),
3229            (
3230                vec!["openvmm", "--igvm", "guest.igvm", "--hv", "--vtl2"],
3231                BaseChipsetType::HclHost,
3232            ),
3233        ] {
3234            let opt = Options::try_parse_from(args).unwrap();
3235            assert!(
3236                std::mem::discriminant(&base_chipset_type(&opt))
3237                    == std::mem::discriminant(&expected)
3238            );
3239        }
3240    }
3241
3242    #[test]
3243    fn maps_virtio_vsock_to_named_pcie_port() {
3244        DefaultPool::run_with(async |driver| {
3245            let temp_dir = tempfile::tempdir().unwrap();
3246            let kernel_path = temp_dir.path().join("kernel");
3247            File::create(&kernel_path).unwrap();
3248            let initrd_path = temp_dir.path().join("initrd");
3249            File::create(&initrd_path).unwrap();
3250            let socket_path = temp_dir.path().join("vsock");
3251            let opt = Options::try_parse_from([
3252                "openvmm",
3253                "--kernel",
3254                kernel_path.to_str().unwrap(),
3255                "--initrd",
3256                initrd_path.to_str().unwrap(),
3257                "--virtio-vsock-path",
3258                socket_path.to_str().unwrap(),
3259                "--virtio-vsock-bus",
3260                "pcie:custom",
3261                "--single-process",
3262            ])
3263            .unwrap();
3264            let mesh = VmmMesh::new(&driver, true).unwrap();
3265
3266            let (config, _resources) = vm_config_from_command_line(driver, &mesh, &opt)
3267                .await
3268                .unwrap();
3269
3270            assert_eq!(config.pcie_devices.len(), 1);
3271            assert_eq!(config.pcie_devices[0].port_name, "custom");
3272            mesh.shutdown().await;
3273        });
3274    }
3275
3276    #[test]
3277    fn maps_virtio_fs_and_rng_to_named_pcie_ports() {
3278        DefaultPool::run_with(async |driver| {
3279            let temp_dir = tempfile::tempdir().unwrap();
3280            let kernel_path = temp_dir.path().join("kernel");
3281            File::create(&kernel_path).unwrap();
3282            let initrd_path = temp_dir.path().join("initrd");
3283            File::create(&initrd_path).unwrap();
3284            let root_path = temp_dir.path().to_str().unwrap();
3285            let opt = Options::try_parse_from([
3286                "openvmm",
3287                "--kernel",
3288                kernel_path.to_str().unwrap(),
3289                "--initrd",
3290                initrd_path.to_str().unwrap(),
3291                "--virtio-fs",
3292                &format!("fs,{root_path}"),
3293                "--virtio-fs-bus",
3294                "pcie:fs",
3295                "--virtio-rng",
3296                "--virtio-rng-bus",
3297                "pcie:rng",
3298                "--single-process",
3299            ])
3300            .unwrap();
3301            let mesh = VmmMesh::new(&driver, true).unwrap();
3302
3303            let (config, _resources) = vm_config_from_command_line(driver, &mesh, &opt)
3304                .await
3305                .unwrap();
3306
3307            let port_names: Vec<_> = config
3308                .pcie_devices
3309                .iter()
3310                .map(|device| device.port_name.as_str())
3311                .collect();
3312            assert_eq!(port_names, ["fs", "rng"]);
3313            mesh.shutdown().await;
3314        });
3315    }
3316
3317    #[test]
3318    fn rejects_duplicate_pcie_port_assignments() {
3319        DefaultPool::run_with(async |driver| {
3320            let temp_dir = tempfile::tempdir().unwrap();
3321            let kernel_path = temp_dir.path().join("kernel");
3322            File::create(&kernel_path).unwrap();
3323            let initrd_path = temp_dir.path().join("initrd");
3324            File::create(&initrd_path).unwrap();
3325            let root_path = temp_dir.path().to_str().unwrap();
3326            let opt = Options::try_parse_from([
3327                "openvmm",
3328                "--kernel",
3329                kernel_path.to_str().unwrap(),
3330                "--initrd",
3331                initrd_path.to_str().unwrap(),
3332                "--virtio-fs",
3333                &format!("pcie_port=custom:fs,{root_path}"),
3334                "--virtio-rng",
3335                "--virtio-rng-bus",
3336                "pcie:custom",
3337                "--single-process",
3338            ])
3339            .unwrap();
3340            let mesh = VmmMesh::new(&driver, true).unwrap();
3341
3342            let error = vm_config_from_command_line(driver, &mesh, &opt)
3343                .await
3344                .err()
3345                .unwrap();
3346
3347            assert_eq!(error.to_string(), "multiple devices use PCIe port 'custom'");
3348            mesh.shutdown().await;
3349        });
3350    }
3351
3352    #[test]
3353    fn rejects_duplicate_pcie_port_assignment_from_storage() {
3354        DefaultPool::run_with(async |driver| {
3355            let temp_dir = tempfile::tempdir().unwrap();
3356            let kernel_path = temp_dir.path().join("kernel");
3357            File::create(&kernel_path).unwrap();
3358            let initrd_path = temp_dir.path().join("initrd");
3359            File::create(&initrd_path).unwrap();
3360            let opt = Options::try_parse_from([
3361                "openvmm",
3362                "--kernel",
3363                kernel_path.to_str().unwrap(),
3364                "--initrd",
3365                initrd_path.to_str().unwrap(),
3366                "--nvme-pci",
3367                "id=nvme0,pcie_port=custom",
3368                "--virtio-rng",
3369                "--virtio-rng-bus",
3370                "pcie:custom",
3371                "--single-process",
3372            ])
3373            .unwrap();
3374            let mesh = VmmMesh::new(&driver, true).unwrap();
3375
3376            let error = vm_config_from_command_line(driver, &mesh, &opt)
3377                .await
3378                .err()
3379                .unwrap();
3380
3381            assert_eq!(error.to_string(), "multiple devices use PCIe port 'custom'");
3382            mesh.shutdown().await;
3383        });
3384    }
3385}