Skip to main content

petri/vm/qemu/
mod.rs

1// Copyright (c) Microsoft Corporation.
2// Licensed under the MIT License.
3
4//! QEMU backend for Petri. Currently only supports Aarch64 CPU emulation on
5//! Linux x64 hosts, but could be modified to support more qemu-system-arch
6//! configurations in the future.
7
8pub mod devices;
9
10use crate::Drive;
11use crate::Firmware;
12use crate::ModifyFn;
13use crate::NoPetriVmFramebufferAccess;
14use crate::NoPetriVmInspector;
15use crate::OpenHclServicingFlags;
16use crate::PetriHaltReason;
17use crate::PetriHaltReasonDetail;
18use crate::PetriInitrd;
19use crate::PetriVmConfig;
20use crate::PetriVmResources;
21use crate::PetriVmRuntime;
22use crate::PetriVmRuntimeConfig;
23use crate::PetriVmmBackend;
24use crate::ShutdownKind;
25use crate::VmmQuirks;
26use crate::openhcl_diag::OpenHclDiagHandler;
27use crate::vm::PetriVmProperties;
28use anyhow::Context;
29use async_trait::async_trait;
30use devices::DeviceConfig;
31use futures::lock::Mutex;
32use futures_concurrency::future::Race;
33use get_resources::ged::FirmwareEvent;
34use guid::Guid;
35use pal_async::DefaultDriver;
36use pal_async::pipe::PolledPipe;
37use pal_async::process::PolledChild;
38use pal_async::task::Spawn;
39use pal_async::task::Task;
40use pal_async::timer::PolledTimer;
41use petri_artifacts_common::tags::GuestQuirksInner;
42use petri_artifacts_common::tags::MachineArch;
43use petri_artifacts_core::ArtifactResolver;
44use petri_artifacts_core::ResolvedArtifact;
45use pipette_client::PipetteClient;
46use std::path::Path;
47use std::path::PathBuf;
48use std::process::Command;
49use std::sync::Arc;
50use std::time::Duration;
51use tempfile::TempPath;
52use vtl2_settings_proto::Vtl2Settings;
53
54/// The QEMU Petri backend
55#[derive(Debug)]
56pub struct QemuPetriBackend {
57    qemu_path: ResolvedArtifact,
58}
59
60/// QEMU-specific emulator configuration.
61#[derive(Debug, Default)]
62pub struct QemuPetriConfig {
63    share_9p: Option<PathBuf>,
64    devices: Vec<DeviceConfig>,
65}
66
67/// Resources needed at runtime for a QEMU Petri VM
68pub struct QemuPetriRuntime {
69    driver: DefaultDriver,
70    qemu_process: Arc<Mutex<PolledChild<std::process::Child>>>,
71    host_pipette_port: u16,
72    log_tasks: Vec<Task<anyhow::Result<()>>>,
73    output_dir: PathBuf,
74}
75
76#[async_trait]
77impl PetriVmmBackend for QemuPetriBackend {
78    type VmmConfig = QemuPetriConfig;
79    type VmRuntime = QemuPetriRuntime;
80    const SUPPORTS_VMBUS: bool = false;
81    const SUPPORTS_CPU_EMULATION: bool = true;
82
83    fn check_compat(_firmware: &Firmware, _arch: MachineArch) -> bool {
84        // Our QEMU bachend only supports linux X64 at this time
85        MachineArch::X86_64 == MachineArch::host() && cfg!(target_os = "linux")
86    }
87
88    fn quirks(_firmware: &Firmware) -> (GuestQuirksInner, VmmQuirks) {
89        (GuestQuirksInner::default(), VmmQuirks::default())
90    }
91
92    fn default_servicing_flags() -> OpenHclServicingFlags {
93        OpenHclServicingFlags {
94            enable_nvme_keepalive: false,
95            enable_mana_keepalive: false,
96            override_version_checks: true,
97            stop_timeout_hint_secs: None,
98        }
99    }
100
101    fn create_guest_dump_disk() -> anyhow::Result<
102        Option<(
103            Arc<TempPath>,
104            Box<dyn FnOnce() -> anyhow::Result<Box<dyn fatfs::ReadWriteSeek>>>,
105        )>,
106    > {
107        Ok(None)
108    }
109
110    fn build_custom_init_script(pipette_path: &str) -> Option<String> {
111        Some(format!(
112            "#!/bin/sh\n\
113            ip link set eth0 up\n\
114            ip addr add 10.0.2.15/24 dev eth0\n\
115            ip route add default via 10.0.2.2\n\
116            echo 'nameserver 10.0.2.3' > /etc/resolv.conf\n\
117            exec '/{}' --transport tcp\n",
118            pipette_path.replace('\'', "'\\''")
119        ))
120    }
121
122    fn new(resolver: &ArtifactResolver<'_>, arch: MachineArch) -> Self {
123        // QEMU bachend only supports linux X64 with an aarch64 guest
124        // TODO: we should have QEMU_SYSTEM_<GUEST_ARCH>_NATIVE symbols
125        // and match here based on guest arch.
126        assert_eq!(arch, MachineArch::Aarch64);
127        QemuPetriBackend {
128            qemu_path: resolver
129                .require(petri_artifacts_vmm_test::artifacts::QEMU_SYSTEM_AARCH64_LINUX_X64)
130                .erase(),
131        }
132    }
133
134    async fn run(
135        self,
136        config: PetriVmConfig,
137        modify_vmm_config: Option<ModifyFn<Self::VmmConfig>>,
138        resources: &PetriVmResources,
139        _properties: PetriVmProperties,
140    ) -> anyhow::Result<(Self::VmRuntime, PetriVmRuntimeConfig)> {
141        let PetriVmResources {
142            driver,
143            log_source,
144            prebuilt_initrd,
145        } = resources;
146
147        let mut qemu_config = QemuPetriConfig::default();
148        if let Some(f) = modify_vmm_config {
149            qemu_config = f.0(qemu_config);
150        }
151
152        let host_pipette_port = pick_free_port().context("failed to find a free port")?;
153
154        let mut cmd = build_qemu_command(
155            self.qemu_path.get(),
156            &config,
157            &qemu_config,
158            host_pipette_port,
159            prebuilt_initrd
160                .as_ref()
161                .context("QEMU requires a prebuilt initrd")?,
162        )?;
163        cmd.stdin(std::process::Stdio::null());
164        cmd.stdout(std::process::Stdio::piped());
165        cmd.stderr(std::process::Stdio::piped());
166
167        tracing::info!(?cmd, "launching qemu");
168        let mut qemu_process = cmd.spawn().context("failed to launch QEMU")?;
169        let qemu_stdout = qemu_process.stdout.take().expect("stdout should be piped");
170        let qemu_stderr = qemu_process.stderr.take().expect("stderr should be piped");
171
172        let qemu_process = PolledChild::<std::process::Child>::new(driver, qemu_process)
173            .context("failed to create PolledChild")?;
174
175        let mut log_tasks = Vec::new();
176
177        let qemu_stdout_pipe = PolledPipe::new(driver, child_pipe_to_file(qemu_stdout))
178            .context("failed to create polled pipe for qemu stdout")?;
179        // Since we pass `-serial mon:stdio` to qemu, the guest outputs to stdout.
180        let qemu_stdout_log_file = log_source.log_file("guest")?;
181        let qemu_stdout_task = driver.spawn(
182            "qemu_stdout",
183            crate::log_task(qemu_stdout_log_file, qemu_stdout_pipe, "qemu_stdout"),
184        );
185        log_tasks.push(qemu_stdout_task);
186
187        let qemu_stderr_pipe = PolledPipe::new(driver, child_pipe_to_file(qemu_stderr))
188            .context("failed to create polled pipe for qemu stderr")?;
189        // QEMU error messages are printed to stderr.
190        let qemu_stderr_log_file = log_source.log_file("qemu")?;
191        let qemu_stderr_task = driver.spawn(
192            "qemu_stderr",
193            crate::log_task(qemu_stderr_log_file, qemu_stderr_pipe, "qemu_stderr"),
194        );
195        log_tasks.push(qemu_stderr_task);
196
197        Ok((
198            QemuPetriRuntime {
199                driver: driver.clone(),
200                qemu_process: Arc::new(Mutex::new(qemu_process)),
201                host_pipette_port,
202                log_tasks,
203                output_dir: log_source.output_dir().to_owned(),
204            },
205            config
206                .firmware
207                .into_runtime_config(config.vmbus_storage_controllers),
208        ))
209    }
210}
211
212impl QemuPetriConfig {
213    /// Share a directory with the guest via 9p.
214    pub fn with_share_9p(mut self, path: impl AsRef<Path>) -> Self {
215        self.share_9p = Some(path.as_ref().to_path_buf());
216        self
217    }
218
219    /// Add a devices to the emulator configuration.
220    pub fn with_devices(mut self, devices: impl IntoIterator<Item = DeviceConfig>) -> Self {
221        self.devices.extend(devices);
222        self
223    }
224}
225
226#[async_trait]
227impl PetriVmRuntime for QemuPetriRuntime {
228    type VmInspector = NoPetriVmInspector;
229    type VmFramebufferAccess = NoPetriVmFramebufferAccess;
230
231    async fn teardown(mut self) -> anyhow::Result<()> {
232        futures::future::join_all(self.log_tasks.into_iter().map(|t| t.cancel())).await;
233        let mut qemu_process = self.qemu_process.lock().await;
234        qemu_process
235            .get_mut()
236            .kill()
237            .context("unable to kill qemu process")?;
238        Ok(())
239    }
240
241    async fn wait_for_halt(&mut self, _allow_reset: bool) -> anyhow::Result<PetriHaltReasonDetail> {
242        let status = self.qemu_process.lock().await.wait().await?;
243        Ok(PetriHaltReasonDetail {
244            reason: if status.success() {
245                PetriHaltReason::PowerOff
246            } else {
247                PetriHaltReason::Other
248            },
249            detail: format!("QEMU exited with status: {:?}", status.code()),
250        })
251    }
252
253    async fn wait_for_agent(&mut self, _set_high_vtl: bool) -> anyhow::Result<PipetteClient> {
254        let driver = self.driver.clone();
255        let output_dir = self.output_dir.clone();
256        let addr = std::net::SocketAddr::from(([127, 0, 0, 1], self.host_pipette_port));
257        let client_core = async move || {
258            tracing::debug!("connecting to pipette");
259            let socket = pal_async::socket::PolledSocket::connect_tcp(&driver, addr)
260                .await
261                .context("failed to connect to pipette")?;
262            tracing::debug!("handshaking with pipette");
263            PipetteClient::new(&driver, socket, &output_dir)
264                .await
265                .context("failed to create pipette client")
266        };
267
268        self.poll_until_exit_or_timeout(None, async move |_| match client_core().await {
269            Ok(client) => {
270                tracing::info!("completed pipette handshake");
271                Ok(Some(client))
272            }
273            Err(_err) => {
274                tracing::debug!("failed to connect to pipette server, retrying");
275                Ok(None)
276            }
277        })
278        .await
279    }
280
281    fn openhcl_diag(&self) -> Option<OpenHclDiagHandler> {
282        // QEMU doesn't support OpenHCL
283        None
284    }
285
286    async fn wait_for_boot_event(
287        &mut self,
288        _timeout: Option<Duration>,
289    ) -> anyhow::Result<Option<FirmwareEvent>> {
290        anyhow::bail!("QEMU backend only supports linux direct, which doesn't emit a boot event")
291    }
292
293    async fn wait_for_enlightened_shutdown_ready(&mut self) -> anyhow::Result<()> {
294        tracing::info!("QEMU doesn't support enlightened shutdown, immediately returning");
295        Ok(())
296    }
297
298    async fn send_enlightened_shutdown(&mut self, _kind: ShutdownKind) -> anyhow::Result<()> {
299        anyhow::bail!("QEMU doesn't support enlightened shutdown")
300    }
301
302    async fn restart_openhcl(
303        &mut self,
304        _new_openhcl: &ResolvedArtifact,
305        _flags: OpenHclServicingFlags,
306    ) -> anyhow::Result<()> {
307        anyhow::bail!("QEMU doesn't support OpenHCL")
308    }
309
310    async fn save_openhcl(
311        &mut self,
312        _new_openhcl: &ResolvedArtifact,
313        _flags: OpenHclServicingFlags,
314    ) -> anyhow::Result<()> {
315        anyhow::bail!("QEMU doesn't support OpenHCL")
316    }
317
318    async fn restore_openhcl(&mut self) -> anyhow::Result<()> {
319        anyhow::bail!("QEMU doesn't support OpenHCL")
320    }
321
322    async fn update_command_line(&mut self, _command_line: &str) -> anyhow::Result<()> {
323        todo!()
324    }
325
326    fn take_framebuffer_access(&mut self) -> Option<NoPetriVmFramebufferAccess> {
327        None
328    }
329
330    async fn reset(&mut self) -> anyhow::Result<()> {
331        todo!()
332    }
333
334    async fn get_guest_state_file(&self) -> anyhow::Result<Option<PathBuf>> {
335        todo!()
336    }
337
338    async fn set_vtl2_settings(&mut self, _settings: &Vtl2Settings) -> anyhow::Result<()> {
339        anyhow::bail!("QEMU doesn't support OpenHCL")
340    }
341
342    async fn set_vmbus_drive(
343        &mut self,
344        _drive: &Drive,
345        _controller_id: &Guid,
346        _controller_location: u32,
347    ) -> anyhow::Result<()> {
348        anyhow::bail!("QEMU doesn't support VMBus")
349    }
350}
351
352impl QemuPetriRuntime {
353    /// Poll `f` until it produces a value, failing if the QEMU process exits
354    /// first and giving up once `timeout` has elapsed.
355    async fn poll_until_exit_or_timeout<T>(
356        &mut self,
357        timeout: Option<Duration>,
358        f: impl AsyncFn(&Self) -> anyhow::Result<Option<T>>,
359    ) -> anyhow::Result<T> {
360        let mut qemu_process = self.qemu_process.lock().await;
361        (
362            async {
363                let mut timer = PolledTimer::new(&self.driver);
364                loop {
365                    if let Some(v) = f(self).await? {
366                        return Ok(v);
367                    }
368                    timer.sleep(Duration::from_secs(10)).await;
369                }
370            },
371            async {
372                let status = qemu_process.wait().await?;
373                Err(anyhow::anyhow!(
374                    "QEMU exited unexpectedly (status: {status})"
375                ))
376            },
377            async {
378                PolledTimer::new(&self.driver)
379                    .sleep(timeout.unwrap_or(Duration::from_mins(10)))
380                    .await;
381                Err(anyhow::anyhow!("Timed out waiting for operation"))
382            },
383        )
384            .race()
385            .await
386    }
387}
388
389/// Convert a child process's stdout/stderr pipe into a [`std::fs::File`] so it
390/// can be wrapped in a [`PolledPipe`]. The owned-handle type differs by
391/// platform, but the conversion is otherwise identical.
392#[cfg(unix)]
393pub fn child_pipe_to_file(pipe: impl Into<std::os::unix::io::OwnedFd>) -> std::fs::File {
394    std::fs::File::from(pipe.into())
395}
396
397/// Convert a child process's stdout/stderr pipe into a [`std::fs::File`] so it
398/// can be wrapped in a [`PolledPipe`]. The owned-handle type differs by
399/// platform, but the conversion is otherwise identical.
400#[cfg(windows)]
401pub fn child_pipe_to_file(pipe: impl Into<std::os::windows::io::OwnedHandle>) -> std::fs::File {
402    std::fs::File::from(pipe.into())
403}
404
405/// Find a free TCP port by binding to port 0 and reading the assigned port.
406fn pick_free_port() -> anyhow::Result<u16> {
407    let listener =
408        std::net::TcpListener::bind("127.0.0.1:0").context("failed to bind ephemeral port")?;
409    let port = listener
410        .local_addr()
411        .context("failed to get local addr")?
412        .port();
413    Ok(port)
414}
415
416/// First PCI device number (`addr=`) used for extra-device root ports.
417///
418/// QEMU's built-in devices use low device numbers. We start at 16 (0x10)
419/// to avoid collisions. The root port for the i-th extra device has
420/// devfn = `(EXTRA_DEVICE_ADDR_BASE + i) << 3`.
421pub const EXTRA_DEVICE_ADDR_BASE: usize = 16;
422
423/// Mount tag for the host 9P share
424pub const SHARE_9P_MOUNT_TAG: &str = "hostshare";
425
426/// Build the QEMU command line for a TCG launch.
427pub fn build_qemu_command(
428    binary: &Path,
429    config: &PetriVmConfig,
430    qemu_config: &QemuPetriConfig,
431    host_pipette_port: u16,
432    prebuilt_initrd: &PetriInitrd,
433) -> anyhow::Result<Command> {
434    let mut cmd = Command::new(binary);
435
436    cmd.arg("-machine")
437        .arg("virt,virtualization=on,iommu=smmuv3,gic-version=3");
438    cmd.arg("-cpu").arg("max");
439    // TODO: more complex memory topologies
440    cmd.arg("-m")
441        .arg((config.memory.startup_bytes / (1024 * 1024)).to_string());
442    // TODO: more complex CPU topologies
443    cmd.arg("-smp")
444        .arg(config.proc_topology.vp_count.to_string());
445    cmd.arg("-nographic");
446
447    let PetriInitrd { path, rdinit_param } = prebuilt_initrd;
448
449    match &config.firmware {
450        Firmware::LinuxDirect { kernel, .. } => {
451            cmd.arg("-kernel").arg(kernel);
452            cmd.arg("-initrd").arg(path.as_ref());
453        }
454        _ => anyhow::bail!("qemu backend only supports linux direct"),
455    };
456
457    cmd.arg("-append").arg(format!("rdinit={rdinit_param}"));
458    cmd.arg("-no-reboot");
459
460    // 9p: share the host directory into the guest
461    if let Some(share_dir) = qemu_config.share_9p.as_ref() {
462        cmd.arg("-fsdev").arg(format!(
463            "local,id=fsdev0,path={},security_model=none",
464            share_dir.display()
465        ));
466        cmd.arg("-device").arg(format!(
467            "virtio-9p-pci,fsdev=fsdev0,mount_tag={SHARE_9P_MOUNT_TAG}"
468        ));
469    }
470
471    // User-mode networking with port forwarding for pipette TCP
472    cmd.arg("-netdev").arg(format!(
473        "user,id=net0,hostfwd=tcp::{host_pipette_port}-:{guest_port}",
474        guest_port = pipette_client::PIPETTE_PORT,
475    ));
476    cmd.arg("-device")
477        .arg("virtio-net-pci,netdev=net0,romfile=");
478
479    // Console on serial (diagnostic only)
480    cmd.arg("-serial").arg("mon:stdio");
481
482    // Extra devices from the profile.
483    // Each device gets its own PCIe root port at a known PCI device number
484    // (`addr=`), so the VFIO setup code can find the bridge by its devfn
485    // in sysfs and enumerate the child behind it.
486    for (i, device) in qemu_config.devices.iter().enumerate() {
487        let rp_id = format!("hosting_rp{i}");
488        let addr = EXTRA_DEVICE_ADDR_BASE + i;
489        // Each root port needs a unique `slot` within its chassis. QEMU
490        // defaults every pcie-root-port to `chassis=0,slot=0`, so a second
491        // port collides with "Can't add chassis slot, error -16" (EBUSY).
492        // Number the slots from 1.
493        let slot = i + 1;
494        cmd.arg("-device").arg(format!(
495            "pcie-root-port,id={rp_id},addr={addr:#x},slot={slot}"
496        ));
497
498        match device {
499            DeviceConfig::VirtioBlk(cfg) => {
500                let node_name = format!("disk{i}");
501                let size_bytes = cfg.size;
502                cmd.arg("-blockdev")
503                    .arg(format!("null-co,node-name={node_name},size={size_bytes}"));
504                cmd.arg("-device")
505                    .arg(format!("virtio-blk-pci,drive={node_name},bus={rp_id},iommu_platform=on,disable-legacy=on,romfile="));
506            }
507            DeviceConfig::Edu(cfg) => {
508                // A conventional PCI endpoint with a register-programmed DMA
509                // engine. Widen `dma_mask` when provided so the engine can
510                // address high aarch64 guest physical addresses (its 28-bit
511                // default would clamp them).
512                let mut dev = format!("edu,bus={rp_id}");
513                if let Some(mask) = &cfg.dma_mask {
514                    dev.push_str(&format!(",dma_mask={mask:#x}"));
515                }
516                cmd.arg("-device").arg(dev);
517            }
518            DeviceConfig::IvshmemPlain(cfg) => {
519                // BAR2 is a prefetchable, RAM-backed memory window served by a
520                // host memory backend — a valid P2P DMA target BAR.
521                let mem_id = format!("ivshmem_mem{i}");
522                let size_bytes = cfg.size;
523                cmd.arg("-object")
524                    .arg(format!("memory-backend-ram,id={mem_id},size={size_bytes}"));
525                cmd.arg("-device")
526                    .arg(format!("ivshmem-plain,memdev={mem_id},bus={rp_id}"));
527            }
528        }
529    }
530
531    Ok(cmd)
532}