diff --git a/.github/workflows/dev-release.yaml b/.github/workflows/dev-release.yaml new file mode 100644 index 0000000000..7218188b0a --- /dev/null +++ b/.github/workflows/dev-release.yaml @@ -0,0 +1,110 @@ +name: Build and Release Dev + +on: + push: + branches: [dev] + +permissions: + contents: write + +concurrency: + group: dev-release + cancel-in-progress: true + +jobs: + build: + strategy: + fail-fast: false + matrix: + include: + - runner: ubuntu-22.04 + arch: x86_64 + - runner: ubuntu-22.04-arm + arch: aarch64 + runs-on: ${{ matrix.runner }} + steps: + - uses: actions/checkout@v7 + + - name: Install Rust toolchain + uses: dtolnay/rust-toolchain@stable + with: + toolchain: stable + + - name: Show toolchain + run: | + rustc --version --verbose + cargo --version --verbose + + - name: Build cloud-hypervisor (release) + run: cargo build --release + + - name: Stage artifacts + run: | + set -euo pipefail + mkdir -p dist + cp -v target/release/cloud-hypervisor "dist/cloud-hypervisor-${{ matrix.arch }}" + "./dist/cloud-hypervisor-${{ matrix.arch }}" --version + rustc --version --verbose | tr '\n' ';' > "dist/rustc-${{ matrix.arch }}.txt" + + - uses: actions/upload-artifact@v7 + with: + name: dist-${{ matrix.arch }} + path: dist/ + + release: + needs: build + runs-on: ubuntu-22.04 + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + + - uses: actions/download-artifact@v8 + with: + pattern: dist-* + merge-multiple: true + path: dist + + - name: Add provenance and checksums + run: | + set -euo pipefail + chmod +x dist/*-x86_64 dist/*-aarch64 + version=$(./dist/cloud-hypervisor-x86_64 --version | awk 'NR==1 {v=$2} END {print v}') + cat > dist/build-info.json < SHA256SUMS) + cat dist/SHA256SUMS + + - name: Create dev release + uses: softprops/action-gh-release@v3 + with: + tag_name: dev + target_commitish: ${{ github.sha }} + name: Dev Build + body: | + Branch: `${{ github.ref_name }}` + Commit: `${{ github.sha }}` + Workflow Run: `${{ github.run_id }}` + files: dist/* + prerelease: true + make_latest: false + overwrite_files: true + fail_on_unmatched_files: true + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + + - name: Move dev tag to current commit + run: | + git tag -f dev "${GITHUB_SHA}" + git push origin refs/tags/dev --force diff --git a/arch/src/aarch64/efi.rs b/arch/src/aarch64/efi.rs new file mode 100644 index 0000000000..d477a4d9d9 --- /dev/null +++ b/arch/src/aarch64/efi.rs @@ -0,0 +1,483 @@ +// Copyright 2026 The Cloud Hypervisor Authors +// +// SPDX-License-Identifier: Apache-2.0 + +// The aarch64 kernel reaches ACPI only through the EFI stub: it reads +// `linux,uefi-system-table` and the `linux,uefi-mmap-*` pointers out of the +// device tree's /chosen node and walks the EFI configuration table for the +// ACPI 2.0 RSDP. Nothing in that path requires real firmware to have produced +// the structures, so the VMM can synthesize them directly. Modelled on +// OpenVMM's aarch64 direct-boot loader. + +use std::result; + +use vm_memory::{Address, Bytes, GuestAddress, GuestMemoryBackend, GuestMemoryRegion}; + +use super::layout; +use crate::GuestMemoryMmap; + +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("Writing EFI handoff structures to guest memory")] + WriteEfiTables(#[source] vm_memory::GuestMemoryError), + #[error("EFI metadata at {0:#x} overflows into the ACPI region at {1:#x}")] + MetadataOverflow(u64, u64), + #[error("EFI memory map of {0} bytes overflows its {1}-byte page into the system table")] + MemoryMapOverflow(usize, u64), +} + +type Result = result::Result; + +// What the stub device tree has to publish for the EFI stub to pick the +// handoff up. +#[derive(Clone, Copy, Debug)] +pub struct EfiHandoff { + pub systab_addr: u64, + pub mmap_addr: u64, + pub mmap_size: u32, + pub mmap_desc_size: u32, + pub mmap_desc_ver: u32, +} + +const PAGE_SIZE: u64 = 0x1000; + +// "IBI SYST" +const EFI_SYSTEM_TABLE_SIGNATURE: u64 = 0x5453_5953_2049_4249; +const EFI_2_70_SYSTEM_TABLE_REVISION: u32 = (2 << 16) | 70; +const EFI_SYSTEM_TABLE_SIZE: u64 = 0x78; +const EFI_MEMORY_DESCRIPTOR_VERSION: u32 = 1; +const EFI_MEMORY_DESCRIPTOR_SIZE: u64 = 40; +const EFI_MEMORY_WB: u64 = 0x8; + +const EFI_BOOT_SERVICES_DATA: u32 = 4; +const EFI_RUNTIME_SERVICES_DATA: u32 = 6; +const EFI_CONVENTIONAL_MEMORY: u32 = 7; +const EFI_ACPI_RECLAIM_MEMORY: u32 = 9; + +const CONFIG_ENTRY_SIZE: u64 = 16 + 8; +const CONFIG_ENTRY_COUNT: u64 = 4; + +// GUIDs in EFI's mixed-endian encoding: the first three fields little-endian, +// the trailing eight bytes in order. +const ACPI_20_TABLE_GUID: [u8; 16] = [ + 0x71, 0xe8, 0x68, 0x88, 0xf1, 0xe4, 0xd3, 0x11, 0xbc, 0x22, 0x00, 0x80, 0xc7, 0x3c, 0x88, 0x81, +]; +const EFI_RT_PROPERTIES_TABLE_GUID: [u8; 16] = [ + 0x8a, 0x91, 0x66, 0xeb, 0xef, 0x7e, 0x2a, 0x40, 0x84, 0x2e, 0x93, 0x1d, 0x21, 0xc3, 0x8a, 0xe9, +]; +const SMBIOS3_TABLE_GUID: [u8; 16] = [ + 0x44, 0x15, 0xfd, 0xf2, 0x94, 0x97, 0x2c, 0x4a, 0x99, 0x2e, 0xe5, 0xbb, 0xcf, 0x20, 0xe3, 0x94, +]; +const LINUX_EFI_MEMRESERVE_TABLE_GUID: [u8; 16] = [ + 0xc6, 0xb0, 0x8e, 0x88, 0xde, 0x8e, 0xf5, 0x4f, 0xa8, 0xf0, 0x9a, 0xee, 0x5c, 0xb9, 0x77, 0xc2, +]; + +const EFI_RT_PROPERTIES_TABLE_VERSION: u16 = 1; +const EFI_RT_PROPERTIES_TABLE_SIZE: u64 = 8; + +const MEMRESERVE_HEADER_SIZE: u64 = 16; +const MEMRESERVE_ENTRY_SIZE: u64 = 16; + +fn align_up(value: u64, alignment: u64) -> u64 { + (value + alignment - 1) & !(alignment - 1) +} + +// UEFI 4.2 checksums the table header over header_size bytes with the crc32 +// field itself zeroed. Table-less so this stays dependency-free. +fn crc32(bytes: &[u8]) -> u32 { + let mut crc = !0u32; + for byte in bytes { + crc ^= *byte as u32; + for _ in 0..8 { + let mask = (crc & 1).wrapping_neg(); + crc = (crc >> 1) ^ (0xedb8_8320 & mask); + } + } + !crc +} + +fn memory_descriptor(typ: u32, physical_start: u64, pages: u64) -> [u8; 40] { + let mut out = [0u8; 40]; + out[0..4].copy_from_slice(&typ.to_le_bytes()); + out[8..16].copy_from_slice(&physical_start.to_le_bytes()); + out[24..32].copy_from_slice(&pages.to_le_bytes()); + out[32..40].copy_from_slice(&EFI_MEMORY_WB.to_le_bytes()); + out +} + +// The firmware-shaped view of guest memory the EFI stub consumes. `rsdp_addr` +// is where create_acpi_tables() already put the RSDP. +pub fn write_efi_tables( + guest_mem: &GuestMemoryMmap, + rsdp_addr: GuestAddress, +) -> Result { + let efi_base = layout::EFI_START.raw_value(); + let acpi_base = layout::ACPI_START.raw_value(); + + // Page 0 holds the memory map, which is sized last; the rest of the + // metadata starts on page 1. + let mut cursor = efi_base + PAGE_SIZE; + + let systab_addr = cursor; + cursor += EFI_SYSTEM_TABLE_SIZE; + + let config_table_addr = cursor; + cursor += CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE; + + // NUL-terminated UTF-16LE, as the system table's firmware_vendor expects. + let fw_vendor: Vec = "Cloud Hypervisor\0" + .encode_utf16() + .flat_map(|c| c.to_le_bytes()) + .collect(); + let fw_vendor_addr = cursor; + cursor += fw_vendor.len() as u64; + cursor = align_up(cursor, 8); + + let rt_props_addr = cursor; + cursor += EFI_RT_PROPERTIES_TABLE_SIZE; + + // struct linux_efi_memreserve: { i32 size; i32 count; u64 next; entry[] }, + // each entry a { u64 base; u64 size; } pair. The kernel links the pages it + // allocates later through `next`, so the root is never handed back as RAM. + let memreserve_addr = align_up(cursor, PAGE_SIZE); + let memreserve_capacity = (PAGE_SIZE - MEMRESERVE_HEADER_SIZE) / MEMRESERVE_ENTRY_SIZE; + cursor = memreserve_addr + PAGE_SIZE; + + if cursor > acpi_base { + return Err(Error::MetadataOverflow(cursor, acpi_base)); + } + + // Runtime services are never available in a synthesized handoff — saying so + // explicitly is what keeps the kernel from calling into them. + let mut rt_props = [0u8; EFI_RT_PROPERTIES_TABLE_SIZE as usize]; + rt_props[0..2].copy_from_slice(&EFI_RT_PROPERTIES_TABLE_VERSION.to_le_bytes()); + rt_props[2..4].copy_from_slice(&(EFI_RT_PROPERTIES_TABLE_SIZE as u16).to_le_bytes()); + guest_mem + .write_slice(&rt_props, GuestAddress(rt_props_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut config_entries = [0u8; (CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE) as usize]; + config_entries[0..16].copy_from_slice(&ACPI_20_TABLE_GUID); + config_entries[16..24].copy_from_slice(&rsdp_addr.raw_value().to_le_bytes()); + config_entries[24..40].copy_from_slice(&EFI_RT_PROPERTIES_TABLE_GUID); + config_entries[40..48].copy_from_slice(&rt_props_addr.to_le_bytes()); + // aarch64 Linux finds DMI only through this entry — it has no equivalent of + // the x86 anchor scan — so the tables setup_smbios() wrote are invisible + // without it. + config_entries[48..64].copy_from_slice(&SMBIOS3_TABLE_GUID); + config_entries[64..72].copy_from_slice(&layout::SMBIOS_START.raw_value().to_le_bytes()); + // Without this the GICv3 ITS cannot persist its LPI property and pending + // tables through efi_mem_reserve_persistent() and warns on every boot. + config_entries[72..88].copy_from_slice(&LINUX_EFI_MEMRESERVE_TABLE_GUID); + config_entries[88..96].copy_from_slice(&memreserve_addr.to_le_bytes()); + guest_mem + .write_slice(&config_entries, GuestAddress(config_table_addr)) + .map_err(Error::WriteEfiTables)?; + + guest_mem + .write_slice(&fw_vendor, GuestAddress(fw_vendor_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut memreserve = [0u8; MEMRESERVE_HEADER_SIZE as usize]; + memreserve[0..4].copy_from_slice(&(memreserve_capacity as i32).to_le_bytes()); + guest_mem + .write_slice(&memreserve, GuestAddress(memreserve_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut systab = [0u8; EFI_SYSTEM_TABLE_SIZE as usize]; + systab[0x00..0x08].copy_from_slice(&EFI_SYSTEM_TABLE_SIGNATURE.to_le_bytes()); + systab[0x08..0x0c].copy_from_slice(&EFI_2_70_SYSTEM_TABLE_REVISION.to_le_bytes()); + systab[0x0c..0x10].copy_from_slice(&(EFI_SYSTEM_TABLE_SIZE as u32).to_le_bytes()); + systab[0x18..0x20].copy_from_slice(&fw_vendor_addr.to_le_bytes()); + systab[0x20..0x24].copy_from_slice(&1u32.to_le_bytes()); + systab[0x68..0x70].copy_from_slice(&CONFIG_ENTRY_COUNT.to_le_bytes()); + systab[0x70..0x78].copy_from_slice(&config_table_addr.to_le_bytes()); + let checksum = crc32(&systab); + systab[0x10..0x14].copy_from_slice(&checksum.to_le_bytes()); + guest_mem + .write_slice(&systab, GuestAddress(systab_addr)) + .map_err(Error::WriteEfiTables)?; + + // The stub DT carries no /memory node, so this map is the only thing + // memblock is built from — every range the guest may touch has to appear. + let reserved_start = layout::FDT_START.raw_value(); + let smbios_base = layout::SMBIOS_START.raw_value(); + let reserved_end = smbios_base + layout::SMBIOS_MAX_SIZE; + let mut mmap = Vec::new(); + // The stub DT: the kernel unflattens it early, so it may be reclaimed after. + mmap.extend_from_slice(&memory_descriptor( + EFI_BOOT_SERVICES_DATA, + reserved_start, + (efi_base - reserved_start) / PAGE_SIZE, + )); + // The EFI structures, covering the whole carve-out so no hole is left to + // show up as an unavailable range. Runtime-services data rather than boot: + // the kernel keeps writing to the memreserve table after boot services end. + mmap.extend_from_slice(&memory_descriptor( + EFI_RUNTIME_SERVICES_DATA, + efi_base, + layout::EFI_MAX_SIZE / PAGE_SIZE, + )); + mmap.extend_from_slice(&memory_descriptor( + EFI_ACPI_RECLAIM_MEMORY, + acpi_base, + layout::ACPI_MAX_SIZE / PAGE_SIZE, + )); + // Runtime-services data rather than boot: is_usable_memory() gives + // boot-services ranges back to memblock as System RAM, and arm64 scans DMI + // from arm_dmi_init(), a core_initcall, long after the allocator is live. + mmap.extend_from_slice(&memory_descriptor( + EFI_RUNTIME_SERVICES_DATA, + smbios_base, + layout::SMBIOS_MAX_SIZE / PAGE_SIZE, + )); + for region in guest_mem.iter() { + let start = region.start_addr().raw_value(); + let end = start + region.len(); + for (from, to) in subtract_reserved(start, end, reserved_start, reserved_end) { + mmap.extend_from_slice(&memory_descriptor( + EFI_CONVENTIONAL_MEMORY, + from, + (to - from) / PAGE_SIZE, + )); + } + } + + // The map owns page 0 alone; the system table starts at page 1, so an + // oversized map would silently overwrite it. + if mmap.len() as u64 > PAGE_SIZE { + return Err(Error::MemoryMapOverflow(mmap.len(), PAGE_SIZE)); + } + guest_mem + .write_slice(&mmap, layout::EFI_START) + .map_err(Error::WriteEfiTables)?; + + Ok(EfiHandoff { + systab_addr, + mmap_addr: efi_base, + mmap_size: mmap.len() as u32, + mmap_desc_size: EFI_MEMORY_DESCRIPTOR_SIZE as u32, + mmap_desc_ver: EFI_MEMORY_DESCRIPTOR_VERSION, + }) +} + +// Overlapping descriptors make the kernel reject the whole map, so the +// firmware-owned span is punched out of every RAM region it intersects. +fn subtract_reserved( + start: u64, + end: u64, + reserved_start: u64, + reserved_end: u64, +) -> Vec<(u64, u64)> { + let mut out = Vec::new(); + if end <= reserved_start || start >= reserved_end { + out.push((start, end)); + return out; + } + if start < reserved_start { + out.push((start, reserved_start)); + } + if end > reserved_end { + out.push((reserved_end, end)); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + // RAM as arch_memory_regions() lays it out, big enough to hold the whole + // firmware carve-out plus a little usable memory above it. + fn test_mem() -> GuestMemoryMmap { + GuestMemoryMmap::from_ranges(&[( + layout::RAM_START, + (layout::KERNEL_START.raw_value() - layout::RAM_START.raw_value() + 0x10_0000) as usize, + )]) + .unwrap() + } + + fn read(mem: &GuestMemoryMmap, addr: u64, len: usize) -> Vec { + let mut out = vec![0u8; len]; + mem.read_slice(&mut out, GuestAddress(addr)).unwrap(); + out + } + + fn le64(bytes: &[u8]) -> u64 { + u64::from_le_bytes(bytes.try_into().unwrap()) + } + + fn le32(bytes: &[u8]) -> u32 { + u32::from_le_bytes(bytes.try_into().unwrap()) + } + + #[test] + fn crc32_matches_known_vector() { + assert_eq!(crc32(b"123456789"), 0xcbf4_3926); + } + + #[test] + fn reserved_span_is_punched_out_of_ram() { + // A region containing the whole reserved span splits in two. + assert_eq!( + subtract_reserved(0x4000_0000, 0x8000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x4040_0000, 0x8000_0000)] + ); + // A region entirely above it is untouched. + assert_eq!( + subtract_reserved(0x1_0000_0000, 0x2_0000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x1_0000_0000, 0x2_0000_0000)] + ); + // A region straddling the tail keeps only what is above. + assert_eq!( + subtract_reserved(0x4030_0000, 0x5000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x4040_0000, 0x5000_0000)] + ); + } + + #[test] + fn system_table_header_is_well_formed() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let systab = read(&mem, handoff.systab_addr, EFI_SYSTEM_TABLE_SIZE as usize); + + assert_eq!(le64(&systab[0x00..0x08]), EFI_SYSTEM_TABLE_SIGNATURE); + assert_eq!(le32(&systab[0x08..0x0c]), EFI_2_70_SYSTEM_TABLE_REVISION); + assert_eq!(le32(&systab[0x0c..0x10]), EFI_SYSTEM_TABLE_SIZE as u32); + assert_eq!(le64(&systab[0x68..0x70]), CONFIG_ENTRY_COUNT); + + // UEFI 4.2: the checksum covers header_size bytes with crc32 zeroed. + let stored = le32(&systab[0x10..0x14]); + let mut zeroed = systab.clone(); + zeroed[0x10..0x14].fill(0); + assert_eq!(stored, crc32(&zeroed)); + + // firmware_vendor must point at NUL-terminated UTF-16LE. + let vendor_addr = le64(&systab[0x18..0x20]); + let vendor = read(&mem, vendor_addr, 34); + let utf16: Vec = vendor + .as_chunks::<2>() + .0 + .iter() + .map(|c| u16::from_le_bytes(*c)) + .take_while(|c| *c != 0) + .collect(); + assert_eq!(String::from_utf16(&utf16).unwrap(), "Cloud Hypervisor"); + } + + #[test] + fn configuration_table_points_at_every_structure() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let systab = read(&mem, handoff.systab_addr, EFI_SYSTEM_TABLE_SIZE as usize); + let table_addr = le64(&systab[0x70..0x78]); + let entries = read( + &mem, + table_addr, + (CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE) as usize, + ); + + let found: Vec<([u8; 16], u64)> = entries + .as_chunks::<{ CONFIG_ENTRY_SIZE as usize }>() + .0 + .iter() + .map(|e| (e[0..16].try_into().unwrap(), le64(&e[16..24]))) + .collect(); + + let lookup = |guid: [u8; 16]| { + found + .iter() + .find(|(g, _)| *g == guid) + .unwrap_or_else(|| panic!("missing configuration table entry")) + .1 + }; + + assert_eq!(lookup(ACPI_20_TABLE_GUID), layout::RSDP_POINTER.raw_value()); + assert_eq!(lookup(SMBIOS3_TABLE_GUID), layout::SMBIOS_START.raw_value()); + + // RT Properties must declare that nothing is supported, otherwise the + // guest will call into runtime services this handoff does not have. + let rt_props = read( + &mem, + lookup(EFI_RT_PROPERTIES_TABLE_GUID), + EFI_RT_PROPERTIES_TABLE_SIZE as usize, + ); + assert_eq!(u16::from_le_bytes([rt_props[0], rt_props[1]]), 1); + assert_eq!(le32(&rt_props[4..8]), 0); + + // The memreserve table starts empty with room for entries. + let memreserve = read( + &mem, + lookup(LINUX_EFI_MEMRESERVE_TABLE_GUID), + MEMRESERVE_HEADER_SIZE as usize, + ); + assert!(le32(&memreserve[0..4]) > 0, "no capacity for entries"); + assert_eq!(le32(&memreserve[4..8]), 0, "count must start at zero"); + assert_eq!(le64(&memreserve[8..16]), 0, "next must be null"); + } + + #[test] + fn memory_map_covers_ram_without_overlapping() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + assert_eq!(handoff.mmap_desc_size, EFI_MEMORY_DESCRIPTOR_SIZE as u32); + assert_eq!(handoff.mmap_desc_ver, EFI_MEMORY_DESCRIPTOR_VERSION); + + let raw = read(&mem, handoff.mmap_addr, handoff.mmap_size as usize); + let mut ranges: Vec<(u64, u64, u32)> = raw + .as_chunks::<{ EFI_MEMORY_DESCRIPTOR_SIZE as usize }>() + .0 + .iter() + .map(|d| { + let start = le64(&d[8..16]); + (start, start + le64(&d[24..32]) * PAGE_SIZE, le32(&d[0..4])) + }) + .collect(); + assert!(!ranges.is_empty()); + ranges.sort_by_key(|r| r.0); + + // A hole or an overlap both make the kernel reject the map, and the + // stub tree has no /memory node to fall back on. + for pair in ranges.windows(2) { + assert_eq!(pair[0].1, pair[1].0, "gap or overlap in the memory map"); + } + assert_eq!(ranges[0].0, layout::FDT_START.raw_value()); + assert_eq!(ranges.last().unwrap().1, mem.last_addr().raw_value() + 1); + + // Everything firmware owns has to be typed as such; handing the ACPI + // tables or the EFI structures back as conventional memory would let + // the guest allocate over them. + let firmware_end = layout::SMBIOS_START.raw_value() + layout::SMBIOS_MAX_SIZE; + for (start, end, typ) in &ranges { + if *start < firmware_end { + assert_ne!(*typ, EFI_CONVENTIONAL_MEMORY, "{start:#x}..{end:#x}"); + } else { + assert_eq!(*typ, EFI_CONVENTIONAL_MEMORY, "{start:#x}..{end:#x}"); + } + } + // The EFI structures outlive boot services: the kernel keeps appending + // to the memreserve table after they end. + let efi_region = ranges + .iter() + .find(|(s, _, _)| *s == layout::EFI_START.raw_value()) + .expect("EFI region missing from the map"); + assert_eq!(efi_region.2, EFI_RUNTIME_SERVICES_DATA); + // So does SMBIOS: is_usable_memory() would give a boot-services range + // back to memblock as System RAM, and arm64 reads DMI from + // arm_dmi_init(), a core_initcall that runs once the allocator is up. + let smbios_region = ranges + .iter() + .find(|(s, _, _)| *s == layout::SMBIOS_START.raw_value()) + .expect("SMBIOS region missing from the map"); + assert_eq!(smbios_region.2, EFI_RUNTIME_SERVICES_DATA); + } + + #[test] + fn metadata_stays_below_the_acpi_region() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let acpi_base = layout::ACPI_START.raw_value(); + assert!(handoff.systab_addr < acpi_base); + assert!(handoff.mmap_addr + u64::from(handoff.mmap_size) < acpi_base); + } +} diff --git a/arch/src/aarch64/fdt.rs b/arch/src/aarch64/fdt.rs index e82e782935..03afb93771 100644 --- a/arch/src/aarch64/fdt.rs +++ b/arch/src/aarch64/fdt.rs @@ -27,6 +27,7 @@ use vm_memory::{Address, Bytes, GuestMemoryBackend, GuestMemoryError, GuestMemor use super::super::{DeviceType, GuestMemoryMmap, InitramfsConfig}; use super::cache::{CacheTopologyInfo, read_cache_topology}; +use super::efi::EfiHandoff; use super::layout::{ GIC_V2M_COMPATIBLE, GICV2M_SPI_BASE, GICV2M_SPI_NUM, IRQ_BASE, MEM_32BIT_DEVICES_SIZE, MEM_32BIT_DEVICES_START, MEM_PCI_IO_SIZE, MEM_PCI_IO_START, PCI_HIGH_BASE, @@ -71,6 +72,9 @@ const IRQ_TYPE_LEVEL_HI: u32 = 4; // System Power Down const KEY_POWER: u32 = 116; +// enum efi_secureboot_mode: 0 is "unset", which Ubuntu reports as a warning +const EFI_SECUREBOOT_MODE_DISABLED: u32 = 2; + /// Trait for devices to be added to the Flattened Device Tree. pub trait DeviceInfoForFdt { /// Returns the address where this device will be loaded. @@ -146,6 +150,48 @@ pub fn create_fdt( Ok(fdt_final) } +/// The device tree for ACPI boot: a `/chosen` node and nothing else. +/// +/// `dt_is_stub()` in the arm64 kernel treats any depth-1 node other than +/// `chosen` as proof of a "real" DTB and turns ACPI off, so describing no +/// hardware here is what lets the guest take its hardware from ACPI instead — +/// and is why `acpi=force` is not needed on the command line. +pub fn create_stub_fdt( + cmdline: &str, + initrd: &Option, + efi: &EfiHandoff, +) -> FdtWriterResult> { + let mut fdt = FdtWriter::new().unwrap(); + + let root_node = fdt.begin_node("")?; + fdt.property_u32("#address-cells", ADDRESS_CELLS)?; + fdt.property_u32("#size-cells", SIZE_CELLS)?; + + let chosen_node = fdt.begin_node("chosen")?; + fdt.property_string("bootargs", cmdline)?; + if let Some(initrd_config) = initrd { + let initrd_start = initrd_config.address.raw_value(); + fdt.property_u64("linux,initrd-start", initrd_start)?; + fdt.property_u64("linux,initrd-end", initrd_start + initrd_config.size as u64)?; + } + fdt.property_u64("linux,uefi-system-table", efi.systab_addr)?; + fdt.property_u64("linux,uefi-mmap-start", efi.mmap_addr)?; + fdt.property_u32("linux,uefi-mmap-size", efi.mmap_size)?; + fdt.property_u32("linux,uefi-mmap-desc-size", efi.mmap_desc_size)?; + fdt.property_u32("linux,uefi-mmap-desc-ver", efi.mmap_desc_ver)?; + // Ubuntu's kernel makes `linux,uefi-secure-boot` a required property in + // efi_get_fdt_params(); without it the EFI handoff is abandoned, the memory + // map is never installed, and — since this tree has no /memory node — + // memblock comes up empty and paging_init panics with "Failed to allocate + // page table page". Mainline ignores the property. + fdt.property_u32("linux,uefi-secure-boot", EFI_SECUREBOOT_MODE_DISABLED)?; + fdt.end_node(chosen_node)?; + + fdt.end_node(root_node)?; + + fdt.finish() +} + pub fn write_fdt_to_memory(fdt_final: &[u8], guest_mem: &GuestMemoryMmap) -> Result<()> { // Write FDT to memory. guest_mem @@ -1064,9 +1110,95 @@ fn print_node(node: FdtNode<'_, '_>, n_spaces: usize) { mod tests { use std::collections::BTreeMap; + use vm_memory::GuestAddress; + use super::*; use crate::NumaNode; + fn test_handoff() -> EfiHandoff { + EfiHandoff { + systab_addr: 0x4010_1000, + mmap_addr: 0x4010_0000, + mmap_size: 160, + mmap_desc_size: 40, + mmap_desc_ver: 1, + } + } + + // The whole ACPI mode rests on this: dt_is_stub() in the arm64 kernel + // reads any depth-1 node other than `chosen` as proof of a real DTB + // and turns ACPI back off, which would silently put us back on the device + // tree with no hardware described in it. + #[test] + fn stub_fdt_has_only_a_chosen_node() { + let blob = create_stub_fdt("console=ttyAMA0", &None, &test_handoff()).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let root = fdt.find_node("/").unwrap(); + let children: Vec<&str> = root.children().map(|c| c.name).collect(); + assert_eq!(children, vec!["chosen"]); + } + + #[test] + fn stub_fdt_publishes_the_efi_handoff() { + let handoff = test_handoff(); + let blob = create_stub_fdt("console=ttyAMA0 rw", &None, &handoff).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let chosen = fdt.find_node("/chosen").unwrap(); + + let u64_prop = |name: &str| -> u64 { + let p = chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")); + BigEndian::read_u64(p.value) + }; + let u32_prop = |name: &str| -> u32 { + let p = chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")); + BigEndian::read_u32(p.value) + }; + + assert_eq!(u64_prop("linux,uefi-system-table"), handoff.systab_addr); + assert_eq!(u64_prop("linux,uefi-mmap-start"), handoff.mmap_addr); + assert_eq!(u32_prop("linux,uefi-mmap-size"), handoff.mmap_size); + assert_eq!( + u32_prop("linux,uefi-mmap-desc-size"), + handoff.mmap_desc_size + ); + assert_eq!(u32_prop("linux,uefi-mmap-desc-ver"), handoff.mmap_desc_ver); + + // Ubuntu's efi_get_fdt_params() treats this as required; dropping it + // aborts the handoff and panics the guest in paging_init. + assert_eq!( + u32_prop("linux,uefi-secure-boot"), + EFI_SECUREBOOT_MODE_DISABLED + ); + } + + #[test] + fn stub_fdt_carries_the_initramfs_when_there_is_one() { + let initrd = Some(InitramfsConfig { + address: GuestAddress(0x8000_0000), + size: 0x10_0000, + }); + let blob = create_stub_fdt("", &initrd, &test_handoff()).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let chosen = fdt.find_node("/chosen").unwrap(); + let prop = |name: &str| { + BigEndian::read_u64( + chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")) + .value, + ) + }; + assert_eq!(prop("linux,initrd-start"), 0x8000_0000); + assert_eq!(prop("linux,initrd-end"), 0x8010_0000); + } + // Helper function to create a simple NumaNode for testing fn create_test_numa_node(cpus: Vec, device_id: Option) -> NumaNode { NumaNode { diff --git a/arch/src/aarch64/layout.rs b/arch/src/aarch64/layout.rs index a6858c761a..389a097737 100644 --- a/arch/src/aarch64/layout.rs +++ b/arch/src/aarch64/layout.rs @@ -116,6 +116,12 @@ pub const FDT_START: GuestAddress = RAM_START; /// documentation](https://www.kernel.org/doc/Documentation/arm64/booting.txt). pub const FDT_MAX_SIZE: u64 = 0x20_0000; +/// EFI handoff structures (system table, configuration table, memory map) for +/// ACPI boot. Carved from the upper half of the FDT reservation: the stub +/// device tree that mode uses is a few KiB against a 2 MiB window. +pub const EFI_START: GuestAddress = GuestAddress(RAM_START.0 + FDT_MAX_SIZE / 2); +pub const EFI_MAX_SIZE: u64 = FDT_MAX_SIZE / 2; + /// Put ACPI table above dtb pub const ACPI_START: GuestAddress = GuestAddress(RAM_START.0 + FDT_MAX_SIZE); const ACPI_SMBIOS_MAX_SIZE: u64 = 0x20_0000; diff --git a/arch/src/aarch64/mod.rs b/arch/src/aarch64/mod.rs index e92fa0663b..04088a76ee 100644 --- a/arch/src/aarch64/mod.rs +++ b/arch/src/aarch64/mod.rs @@ -4,6 +4,9 @@ /// Module for cache info. pub mod cache; +/// Module for the synthesized EFI handoff that gives an ACPI-booted guest its +/// RSDP without firmware. +pub mod efi; /// Module for the flattened device tree. pub mod fdt; /// Layout for this aarch64 system. @@ -56,6 +59,14 @@ pub enum Error { /// Error initializing PMU for vcpu #[error("Error initializing PMU for vcpu")] VcpuInitPmu, + + /// Failed to write the EFI handoff structures. + #[error("Failed to write the EFI handoff structures")] + SetupEfi(#[source] efi::Error), + + /// No RSDP address to hand the guest in ACPI boot mode. + #[error("No RSDP address to hand the guest in ACPI boot mode")] + MissingRsdp, } #[derive(Debug, Copy, Clone)] @@ -162,6 +173,33 @@ pub fn configure_system( Ok(()) } +/// The ACPI counterpart of [`configure_system`]: synthesize the EFI handoff the +/// arm64 EFI stub expects, then hand the guest a device tree that describes +/// nothing but where to find it. Everything else — CPUs, GIC, timer, PCI — comes +/// from the ACPI tables `create_acpi_tables()` has already written. +pub fn configure_system_acpi( + guest_mem: &GuestMemoryMmap, + cmdline: &str, + initrd: &Option, + rsdp_addr: GuestAddress, + smbios: Option<&smbios::SmbiosConfig>, +) -> super::Result<()> { + smbios::setup_smbios(guest_mem, smbios).map_err(Error::SmbiosSetup)?; + + let handoff = efi::write_efi_tables(guest_mem, rsdp_addr) + .map_err(|e| super::Error::PlatformSpecific(Error::SetupEfi(e)))?; + + let fdt_final = fdt::create_stub_fdt(cmdline, initrd, &handoff).map_err(|_| Error::SetupFdt)?; + + if log_enabled!(Level::Debug) { + fdt::print_fdt(&fdt_final); + } + + fdt::write_fdt_to_memory(&fdt_final, guest_mem).map_err(Error::WriteFdtToMemory)?; + + Ok(()) +} + /// Returns the memory address where the initramfs could be loaded. pub fn initramfs_load_addr( guest_mem: &GuestMemoryMmap, diff --git a/cloud-hypervisor/src/bin/ch-remote.rs b/cloud-hypervisor/src/bin/ch-remote.rs index 44de33bffc..5749c2c4ac 100644 --- a/cloud-hypervisor/src/bin/ch-remote.rs +++ b/cloud-hypervisor/src/bin/ch-remote.rs @@ -497,12 +497,10 @@ fn rest_api_do_command(matches: &ArgMatches, socket: &mut UnixStream) -> ApiResu .map_err(Error::HttpApiClient) } Some("snapshot") => { + let sub = matches.subcommand_matches("snapshot").unwrap(); let snapshot_config = snapshot_config( - matches - .subcommand_matches("snapshot") - .unwrap() - .get_one::("snapshot_config") - .unwrap(), + sub.get_one::("snapshot_config").unwrap(), + sub.get_flag("diff"), ); simple_api_command(socket, "PUT", "snapshot", Some(&snapshot_config)) .map_err(Error::HttpApiClient) @@ -722,12 +720,10 @@ fn dbus_api_do_command(matches: &ArgMatches, proxy: &DBusApi1ProxyBlocking<'_>) proxy.api_vm_add_vsock(&vsock_config) } Some("snapshot") => { + let sub = matches.subcommand_matches("snapshot").unwrap(); let snapshot_config = snapshot_config( - matches - .subcommand_matches("snapshot") - .unwrap() - .get_one::("snapshot_config") - .unwrap(), + sub.get_one::("snapshot_config").unwrap(), + sub.get_flag("diff"), ); proxy.api_vm_snapshot(&snapshot_config) } @@ -934,9 +930,14 @@ fn add_vsock_config(config: &str) -> Result { Ok(vsock_config) } -fn snapshot_config(url: &str) -> String { +fn snapshot_config(url: &str, diff: bool) -> String { let snapshot_config = api::VmSnapshotConfig { destination_url: String::from(url), + snapshot_type: if diff { + api::VmSnapshotType::Diff + } else { + api::VmSnapshotType::Full + }, }; serde_json::to_string(&snapshot_config).unwrap() @@ -1177,6 +1178,15 @@ fn get_cli_commands_sorted() -> Box<[Command]> { Command::new("shutdown-vmm").about("Shutdown the VMM"), Command::new("snapshot") .about("Create a snapshot from VM") + .arg( + Arg::new("diff") + .long("diff") + .action(clap::ArgAction::SetTrue) + .help( + "Write only pages dirtied since the previous snapshot; \ + the first --diff takes a full baseline and starts dirty tracking", + ), + ) .arg( Arg::new("snapshot_config") .index(1) diff --git a/cloud-hypervisor/tests/integration.rs b/cloud-hypervisor/tests/integration.rs index c2829460ca..3fefe79b73 100644 --- a/cloud-hypervisor/tests/integration.rs +++ b/cloud-hypervisor/tests/integration.rs @@ -8879,6 +8879,12 @@ mod ivshmem { ); } + #[test] + #[cfg(not(feature = "mshv"))] + fn test_snapshot_restore_diff_chain() { + snapshot_restore_common::_test_snapshot_restore_diff_chain(); + } + #[test] fn test_snapshot_restore_with_resume() { snapshot_restore_common::_test_snapshot_restore( @@ -9019,6 +9025,8 @@ mod ivshmem { } mod snapshot_restore_common { + #[cfg(not(feature = "mshv"))] + use std::fs::create_dir; use std::fs::{read_to_string, remove_dir_all}; use std::process::Command; @@ -9057,6 +9065,138 @@ mod snapshot_restore_common { )); } + fn snapshot_diff(api_socket: &str, url: &str) -> bool { + let output = Command::new(clh_command("ch-remote")) + .args([ + &format!("--api-socket={api_socket}"), + "snapshot", + "--diff", + url, + ]) + .output() + .unwrap(); + if !output.status.success() { + eprintln!( + "ch-remote snapshot --diff failed: {}", + String::from_utf8_lossy(&output.stderr) + ); + } + output.status.success() + } + + // A diff series baseline plus one delta, restored through memory_chain. + // tmpfs markers written before and after the baseline prove the delta + // rebases onto the baseline: marker A lives in the baseline pages, marker + // B only in the delta's dirty extents. + #[cfg(not(feature = "mshv"))] + pub(crate) fn _test_snapshot_restore_diff_chain() { + let disk_config = UbuntuDiskConfig::new(JAMMY_IMAGE_NAME.to_string()); + let guest = Guest::new(Box::new(disk_config)); + let kernel_path = direct_kernel_boot_path(); + + let api_socket_source = format!("{}.1", temp_api_path(&guest.tmp_dir)); + let net_params = format!( + "id=net123,tap=,mac={},ip={},mask=255.255.255.128", + guest.network.guest_mac0, guest.network.host_ip0 + ); + + let mut child = GuestCommand::new(&guest) + .args(["--api-socket", &api_socket_source]) + .args(["--cpus", "boot=1"]) + .args(["--memory", "size=1G"]) + .args(["--kernel", kernel_path.to_str().unwrap()]) + .args([ + "--disk", + format!( + "path={}", + guest.disk_config.disk(DiskType::OperatingSystem).unwrap() + ) + .as_str(), + format!( + "path={}", + guest.disk_config.disk(DiskType::CloudInit).unwrap() + ) + .as_str(), + ]) + .args(["--net", net_params.as_str()]) + .args(["--cmdline", DIRECT_KERNEL_BOOT_CMDLINE]) + .capture_output() + .spawn() + .unwrap(); + + let snapshot_dir = temp_snapshot_dir_path(&guest.tmp_dir); + let series_dir = format!("{snapshot_dir}/series"); + create_dir(&series_dir).unwrap(); + + let r = panic::catch_unwind(|| { + guest.wait_vm_boot().unwrap(); + + guest + .ssh_command("echo diffmarkerA > /dev/shm/marker_a") + .unwrap(); + + // First diff of a series writes the full baseline. + assert!(remote_command(&api_socket_source, "pause", None)); + assert!(snapshot_diff( + &api_socket_source, + format!("file://{series_dir}").as_str(), + )); + assert!(remote_command(&api_socket_source, "resume", None)); + + guest + .ssh_command("echo diffmarkerB > /dev/shm/marker_b") + .unwrap(); + + // Second diff lands in the same directory, carrying only the + // pages dirtied since the baseline. + assert!(remote_command(&api_socket_source, "pause", None)); + assert!(snapshot_diff( + &api_socket_source, + format!("file://{series_dir}").as_str(), + )); + }); + + kill_child(&mut child); + let output = child.wait_with_output().unwrap(); + handle_child_output(r, &output); + + // Restore discovers the delta files in the series directory. + let api_socket_restored = format!("{}.2", temp_api_path(&guest.tmp_dir)); + let mut child = GuestCommand::new(&guest) + .args(["--api-socket", &api_socket_restored]) + .args([ + "--restore", + format!("source_url=file://{series_dir},resume=true").as_str(), + ]) + .capture_output() + .spawn() + .unwrap(); + + let r = panic::catch_unwind(|| { + assert!(wait_until(Duration::from_secs(30), || remote_command( + &api_socket_restored, + "info", + None + ))); + + // Both markers must survive: A from the baseline pages the delta + // never dirtied, B from the delta's extents. + assert_eq!( + guest.ssh_command("cat /dev/shm/marker_a").unwrap().trim(), + "diffmarkerA" + ); + assert_eq!( + guest.ssh_command("cat /dev/shm/marker_b").unwrap().trim(), + "diffmarkerB" + ); + }); + + let _ = remove_dir_all(snapshot_dir.as_str()); + kill_child(&mut child); + let output = child.wait_with_output().unwrap(); + handle_child_output(r, &output); + } + /// Easy disambiguation between snapshot/restore variants. #[derive(Clone, Copy, Default)] pub(crate) struct SnapshotRestoreTest { diff --git a/docs/hotplug.md b/docs/hotplug.md index 14e60f954b..ef61a26dcf 100644 --- a/docs/hotplug.md +++ b/docs/hotplug.md @@ -159,7 +159,7 @@ The same API can also be used to reduce the desired RAM for a VM. It is importan Extra PCI devices can be added and removed from a running `cloud-hypervisor` instance. This is controlled by making a HTTP API request to the VMM to ask for the additional device to be added, or for the existing device to be removed. -Note: On AArch64 platform, PCI device hotplug can only be achieved using ACPI. Please refer to the [documentation](uefi.md#building-uefi-firmware-for-aarch64) for more information. +Note: On AArch64 platform, PCI device hotplug can only be achieved using ACPI. A guest booted through UEFI firmware gets ACPI from the firmware; please refer to the [documentation](uefi.md#building-uefi-firmware-for-aarch64) for more information. A direct-kernel-booted guest gets ACPI through `--platform acpi_boot=on`, which is the default: the VMM synthesizes the EFI handoff that leads the kernel to the ACPI tables and passes a device tree with only a `/chosen` node. The guest kernel needs `CONFIG_EFI` and `CONFIG_ACPI`; a kernel without them finds no memory and panics early in boot, so boot it with `--platform acpi_boot=off` to get the hardware device tree back, without PCI hotplug. To use PCI device hotplug start the VM with the HTTP server. diff --git a/docs/snapshot_restore.md b/docs/snapshot_restore.md index 4b1cdff006..bef738cfa3 100644 --- a/docs/snapshot_restore.md +++ b/docs/snapshot_restore.md @@ -60,6 +60,38 @@ be needed. `state.json` contains the virtual machine state. It is used to restore each component in the state it was left before the snapshot occurred. +## Diff snapshots + +Passing `snapshot_type=diff` (`ch-remote snapshot --diff `) writes only +the pages dirtied since the previous snapshot of the series. The first `diff` +request takes a full baseline and enables dirty-page tracking; each subsequent +one dumps the delta, so the pause cost is proportional to the amount of dirtied +memory rather than to the guest RAM size. A `full` request (or any snapshot +failure, a memory layout change, a migration, or deleting the VM) ends the +series; the next `diff` starts a new one with a fresh baseline. + +The whole series lives in one directory, and every `diff` request of a series +must target that same directory: + +```bash +ls /foo/snapshot/ +config.json memory-ranges memory-ranges.diff.1 memory-ranges.diff.2 state.json +``` + +Each `memory-ranges.diff.N` is a sparse file whose extents sit at the same +offsets as in the baseline `memory-ranges`. `config.json` and `state.json` +always describe the newest delta. A delta is written under a temporary name +and renamed into place, so an interrupted diff never invalidates a consistent +directory. A full dump into a directory removes any leftover delta files. + +Restore needs no extra parameter: point `source_url` at the directory. When +delta files are present, memory fills from the baseline and each delta's dirty +extents replay in sequence order; a gap in the sequence or a delta whose +length does not match the layout fails the restore before any page is +touched. The files must sit on a filesystem with `SEEK_DATA`/`SEEK_HOLE` +support, and a diff-snapshot series cannot be combined with +`memory_restore_mode=ondemand`. + ## Restore a Cloud Hypervisor VM Given that one has access to an existing snapshot in `/home/foo/snapshot`, diff --git a/fuzz/fuzz_targets/http_api.rs b/fuzz/fuzz_targets/http_api.rs index 56e9e55c6b..da1ee9186b 100644 --- a/fuzz/fuzz_targets/http_api.rs +++ b/fuzz/fuzz_targets/http_api.rs @@ -15,7 +15,7 @@ use vm_migration::MigratableError; use vmm::api::http::*; use vmm::api::{ ApiRequest, BalloonStatsResponse, RequestHandler, VmInfoResponse, VmReceiveMigrationData, - VmSendMigrationData, VmmPingResponse, + VmSendMigrationData, VmSnapshotConfig, VmmPingResponse, }; use vmm::config::RestoreConfig; use vmm::vm::{Error as VmError, VmState}; @@ -105,7 +105,7 @@ impl RequestHandler for StubApiRequestHandler { Ok(()) } - fn vm_snapshot(&mut self, _: &str) -> Result<(), VmError> { + fn vm_snapshot(&mut self, _: &VmSnapshotConfig) -> Result<(), VmError> { Ok(()) } diff --git a/hypervisor/src/kvm/mod.rs b/hypervisor/src/kvm/mod.rs index f48a620d5f..f7d0cb4338 100644 --- a/hypervisor/src/kvm/mod.rs +++ b/hypervisor/src/kvm/mod.rs @@ -82,9 +82,9 @@ use kvm_bindings::{ kvm_msr_entry, }; #[cfg(target_arch = "x86_64")] -use x86_64::check_required_kvm_extensions; -#[cfg(target_arch = "x86_64")] pub use x86_64::{CpuId, ExtendedControlRegisters, MsrEntries, VcpuKvmState}; +#[cfg(target_arch = "x86_64")] +use x86_64::{check_required_kvm_extensions, cpuid_has_shstk}; #[cfg(target_arch = "x86_64")] use crate::ClockData; @@ -202,6 +202,25 @@ ioctl_iow_nr!( 0xe3, kvm_bindings::kvm_device_attr ); +// kvm-ioctls only exposes KVM_{GET,SET}_ONE_REG for aarch64 and riscv64. +#[cfg(target_arch = "x86_64")] +ioctl_iow_nr!( + KVM_GET_ONE_REG, + kvm_bindings::KVMIO, + 0xab, + kvm_bindings::kvm_one_reg +); +#[cfg(target_arch = "x86_64")] +ioctl_iow_nr!( + KVM_SET_ONE_REG, + kvm_bindings::KVMIO, + 0xac, + kvm_bindings::kvm_one_reg +); +// KVM_X86_REG_KVM(KVM_REG_GUEST_SSP): the guest's current shadow stack pointer. It is +// a register, not an MSR, so KVM_GET_MSRS never carries it. +#[cfg(target_arch = "x86_64")] +const KVM_REG_GUEST_SSP: u64 = 0x2030_0003_0000_0000; #[cfg(feature = "sev_snp")] use igvm_defs::PAGE_SIZE_4K; @@ -2077,6 +2096,45 @@ impl KvmVcpu { let ret = unsafe { ioctl_with_ref(&self.fd, KVM_HAS_DEVICE_ATTR(), &attr) }; ret == 0 } + + /// The guest's current shadow stack pointer, or `None` when the guest CPUID + /// does not expose shadow stacks. A vCPU paused in user mode keeps its live SSP + /// here, not in MSR_IA32_PL3_SSP, so without it a restored thread resumes with + /// SSP=0 and faults on its next CALL/RET. + fn guest_ssp(&self, cpuid: &[CpuIdEntry]) -> cpu::Result> { + if !cpuid_has_shstk(cpuid) { + return Ok(None); + } + let mut ssp = 0u64; + let reg = kvm_bindings::kvm_one_reg { + id: KVM_REG_GUEST_SSP, + addr: &raw mut ssp as u64, + }; + // SAFETY: FFI call; `reg.addr` points to `ssp`, filled in by the kernel. + let ret = unsafe { ioctl_with_ref(&self.fd, KVM_GET_ONE_REG(), ®) }; + if ret < 0 { + return Err(cpu::HypervisorCpuError::GetRegister( + io::Error::last_os_error().into(), + )); + } + Ok(Some(ssp)) + } + + /// Restore the guest's current shadow stack pointer. + fn set_guest_ssp(&self, ssp: u64) -> cpu::Result<()> { + let reg = kvm_bindings::kvm_one_reg { + id: KVM_REG_GUEST_SSP, + addr: &raw const ssp as u64, + }; + // SAFETY: FFI call; `reg.addr` points to `ssp`, read by the kernel. + let ret = unsafe { ioctl_with_ref(&self.fd, KVM_SET_ONE_REG(), ®) }; + if ret < 0 { + return Err(cpu::HypervisorCpuError::SetRegister( + io::Error::last_os_error().into(), + )); + } + Ok(()) + } } /// Implementation of Vcpu trait for KVM @@ -3243,6 +3301,7 @@ impl cpu::Vcpu for KvmVcpu { let vcpu_events = self.get_vcpu_events()?; let tsc_khz = self.tsc_khz()?; + let guest_ssp = self.guest_ssp(&cpuid)?; Ok(VcpuKvmState { cpuid, @@ -3258,6 +3317,7 @@ impl cpu::Vcpu for KvmVcpu { tsc_khz, nested_state, hyperv_synic, + guest_ssp, } .into()) } @@ -3497,6 +3557,10 @@ impl cpu::Vcpu for KvmVcpu { } } + if let Some(ssp) = state.guest_ssp { + self.set_guest_ssp(ssp)?; + } + self.set_vcpu_events(&state.vcpu_events)?; Ok(()) diff --git a/hypervisor/src/kvm/x86_64/mod.rs b/hypervisor/src/kvm/x86_64/mod.rs index 5bc3ee89a0..c0a49ff310 100644 --- a/hypervisor/src/kvm/x86_64/mod.rs +++ b/hypervisor/src/kvm/x86_64/mod.rs @@ -87,6 +87,17 @@ pub struct VcpuKvmState { pub nested_state: Option, #[serde(default)] pub hyperv_synic: bool, + // Absent from snapshots taken before it was saved, and when the guest has no + // shadow stacks. + #[serde(default)] + pub guest_ssp: Option, +} + +/// CPUID.(EAX=7,ECX=0):ECX[7], CET shadow stack. +pub fn cpuid_has_shstk(cpuid: &[CpuIdEntry]) -> bool { + cpuid + .iter() + .any(|e| e.function == 7 && e.index == 0 && e.ecx & (1 << 7) != 0) } impl From for kvm_segment { diff --git a/vmm/src/api/mod.rs b/vmm/src/api/mod.rs index 6a3c86c2fc..83ca32134f 100644 --- a/vmm/src/api/mod.rs +++ b/vmm/src/api/mod.rs @@ -257,10 +257,25 @@ pub struct VmRemoveDeviceData { pub id: String, } +/// Type of a VM snapshot: full memory dump or dirty-pages-only delta. +#[derive(Copy, Clone, Default, Deserialize, Serialize, Debug, PartialEq, Eq)] +#[serde(rename_all = "lowercase")] +pub enum VmSnapshotType { + /// Complete guest memory dump. + #[default] + Full, + /// Only pages dirtied since the previous snapshot of the series. The + /// first diff request takes a full baseline and starts dirty tracking. + Diff, +} + #[derive(Clone, Deserialize, Serialize, Default, Debug)] pub struct VmSnapshotConfig { /// The snapshot destination URL pub destination_url: String, + /// Full dump or dirty-pages delta. + #[serde(default)] + pub snapshot_type: VmSnapshotType, } #[derive(Clone, Deserialize, Serialize, Default, Debug)] @@ -867,7 +882,7 @@ pub trait RequestHandler { fn vm_resume(&mut self) -> Result<(), VmError>; - fn vm_snapshot(&mut self, destination_url: &str) -> Result<(), VmError>; + fn vm_snapshot(&mut self, config: &VmSnapshotConfig) -> Result<(), VmError>; fn vm_restore(&mut self, restore_cfg: RestoreConfig) -> Result<(), VmError>; @@ -2027,7 +2042,7 @@ impl ApiAction for VmSnapshot { info!("API request event: VmSnapshot {config:?}"); let response = vmm - .vm_snapshot(&config.destination_url) + .vm_snapshot(&config) .map_err(ApiError::VmSnapshot) .map(|_| ApiResponsePayload::Empty); @@ -2594,4 +2609,21 @@ mod tests { .unwrap(); assert_eq!(data.effective_memory_mode(), MigrationMode::MemFDs); } + + #[test] + fn test_vm_snapshot_config_snapshot_type() { + let config: VmSnapshotConfig = + serde_json::from_str(r#"{"destination_url": "file:///foo"}"#).unwrap(); + assert_eq!(config.snapshot_type, VmSnapshotType::Full); + + let config: VmSnapshotConfig = + serde_json::from_str(r#"{"destination_url": "file:///foo", "snapshot_type": "diff"}"#) + .unwrap(); + assert_eq!(config.snapshot_type, VmSnapshotType::Diff); + + serde_json::from_str::( + r#"{"destination_url": "file:///foo", "snapshot_type": "bogus"}"#, + ) + .unwrap_err(); + } } diff --git a/vmm/src/api/openapi/cloud-hypervisor.yaml b/vmm/src/api/openapi/cloud-hypervisor.yaml index 1103c06350..a971bf570a 100644 --- a/vmm/src/api/openapi/cloud-hypervisor.yaml +++ b/vmm/src/api/openapi/cloud-hypervisor.yaml @@ -1023,6 +1023,13 @@ components: vfio_p2p_dma: type: boolean default: true + acpi_boot: + type: boolean + default: true + description: > + AArch64 only. Give a direct-kernel-booted guest ACPI through a + synthesized EFI handoff instead of a hardware device tree. + Firmware boot ignores it. MemoryZoneConfig: required: @@ -1679,6 +1686,14 @@ components: properties: destination_url: type: string + snapshot_type: + type: string + enum: [full, diff] + default: full + description: + With "diff", only pages dirtied since the previous snapshot of the + series are written; the first "diff" takes a full baseline and + starts dirty tracking. VmCoredumpData: type: object diff --git a/vmm/src/config.rs b/vmm/src/config.rs index 51293d43f6..ea26dc696f 100644 --- a/vmm/src/config.rs +++ b/vmm/src/config.rs @@ -890,6 +890,10 @@ impl PlatformConfig { oem_strings=,chassis_asset_tag=" .to_string(); + if cfg!(target_arch = "aarch64") { + syntax.push_str(",acpi_boot=on|off"); + } + if cfg!(feature = "tdx") { syntax.push_str(",tdx=on|off"); } @@ -958,6 +962,8 @@ impl PlatformConfig { .add("iommufd") .add("iommufd_fd") .add("vfio_p2p_dma"); + #[cfg(target_arch = "aarch64")] + parser.add("acpi_boot"); for field in SMBIOS_STRING_FIELDS { parser.add(field.key); } @@ -1008,6 +1014,12 @@ impl PlatformConfig { .map_err(Error::ParsePlatform)? .unwrap_or(Toggle(false)) .0; + #[cfg(target_arch = "aarch64")] + let acpi_boot = parser + .convert::("acpi_boot") + .map_err(Error::ParsePlatform)? + .unwrap_or(Toggle(true)) + .0; let mut platform_config = PlatformConfig { num_pci_segments, @@ -1029,6 +1041,8 @@ impl PlatformConfig { #[cfg(feature = "sev_snp")] sev_snp, vfio_p2p_dma, + #[cfg(target_arch = "aarch64")] + acpi_boot, }; for field in SMBIOS_STRING_FIELDS { @@ -5879,6 +5893,8 @@ id=\"{id}\",pci_segment={pci_segment},queue_sizes={queue_sizes}" tdx: false, #[cfg(feature = "sev_snp")] sev_snp: false, + #[cfg(target_arch = "aarch64")] + acpi_boot: default_platformconfig_acpi_boot(), } } diff --git a/vmm/src/lib.rs b/vmm/src/lib.rs index c0cb7d7ddf..757dae1d7b 100644 --- a/vmm/src/lib.rs +++ b/vmm/src/lib.rs @@ -49,7 +49,8 @@ use vmm_sys_util::sock_ctrl_msg::ScmSocket; use crate::api::{ ApiRequest, ApiResponse, BalloonStatsResponse, MigrationMode, RequestHandler, TimeoutStrategy, - VmInfoResponse, VmReceiveMigrationData, VmSendMigrationData, VmmPingResponse, + VmInfoResponse, VmReceiveMigrationData, VmSendMigrationData, VmSnapshotConfig, VmSnapshotType, + VmmPingResponse, }; use crate::config::{MemoryRestoreMode, RestoreConfig, VmMemoryZoneUpdateData, add_to_config}; #[cfg(all(target_arch = "x86_64", feature = "guest_debug"))] @@ -694,6 +695,14 @@ impl VmOwnership { } } +// An active diff-snapshot series: every delta lands in the directory the +// baseline was written to, numbered by `next_seq`. +struct SnapshotSeries { + destination_url: String, + layout: MemoryRangeTable, + next_seq: u32, +} + pub struct Vmm { epoll: EpollContext, exit_evt: EventFd, @@ -719,11 +728,23 @@ pub struct Vmm { console_socket_listener: Option>, no_shutdown: bool, check_migration_evt: EventFd, + // Memory layout at diff-snapshot series start; Some = dirty logging active. + snapshot_series: Option, } /// Time before aborting on the page fault connection. const FAULT_CONNECTION_ACCEPT_TIMEOUT: Duration = Duration::from_secs(30); +/// Whether two snapshot memory layouts describe the same regions, i.e. a diff +/// written against one applies at the same file offsets as the other. +fn same_memory_layout(a: &MemoryRangeTable, b: &MemoryRangeTable) -> bool { + a.regions().len() == b.regions().len() + && a.regions() + .iter() + .zip(b.regions()) + .all(|(x, y)| x.gpa == y.gpa && x.length == y.length) +} + /// Just a wrapper for the data that goes into /// [`ReceiveMigrationState::Configured`] struct ReceiveMigrationConfiguredData { @@ -953,6 +974,7 @@ impl Vmm { console_socket_listener: None, no_shutdown, check_migration_evt, + snapshot_series: None, }) } @@ -2559,20 +2581,84 @@ impl RequestHandler for Vmm { } } - fn vm_snapshot(&mut self, destination_url: &str) -> result::Result<(), VmError> { + fn vm_snapshot(&mut self, config: &VmSnapshotConfig) -> result::Result<(), VmError> { match self.vm { VmOwnership::Owned(ref mut vm) => { if vm.restoring() { return Err(VmError::VmRestoring); } + let requested_diff = config.snapshot_type == VmSnapshotType::Diff; + // A diff request continues the series if one is active; + // otherwise it takes a full baseline into the series + // directory and starts dirty logging. A full request ends + // any active series. + let effective_diff = requested_diff && self.snapshot_series.is_some(); + if requested_diff { + let layout = vm + .memory_range_table(MemoryRangePolicy::SkipPersisted) + .map_err(VmError::Snapshot)?; + if let Some(series) = &self.snapshot_series { + // The whole series lives in one directory; a delta + // elsewhere could never be discovered at restore. + if config.destination_url != series.destination_url { + return Err(VmError::Snapshot(MigratableError::Snapshot(anyhow!( + "a diff snapshot must target its series directory {}", + series.destination_url + )))); + } + if !same_memory_layout(&series.layout, &layout) { + // Offsets in the base file no longer match; a + // delta would rebase corruptly. End the series. + self.snapshot_series = None; + let _ = vm.stop_dirty_log(); + return Err(VmError::Snapshot(MigratableError::Snapshot(anyhow!( + "memory layout changed since the snapshot series \ + started; take a full snapshot" + )))); + } + } else { + vm.start_dirty_log().map_err(VmError::Snapshot)?; + self.snapshot_series = Some(SnapshotSeries { + destination_url: config.destination_url.clone(), + layout, + next_seq: 1, + }); + } + } else if self.snapshot_series.take().is_some() { + vm.stop_dirty_log().map_err(VmError::Snapshot)?; + } + + let diff_seq = self.snapshot_series.as_ref().map(|s| s.next_seq); // Drain console_info so that FDs are not reused let _ = self.console_info.take(); - vm.snapshot() + let result = vm + .snapshot() .map_err(VmError::Snapshot) .and_then(|snapshot| { - vm.send(&snapshot, destination_url) + if effective_diff { + // Harvest after device capture so pages dirtied by + // snapshot side effects land in this delta. + let table = vm.dirty_log().map_err(VmError::Snapshot)?; + vm.set_diff_snapshot_ranges(table, diff_seq.unwrap()); + } + vm.send(&snapshot, &config.destination_url) .map_err(VmError::SnapshotSend) - }) + }); + match (&result, &mut self.snapshot_series) { + (Ok(()), Some(series)) => { + if effective_diff { + series.next_seq += 1; + } + } + (Err(_), series @ Some(_)) => { + // The harvested bitmap (or the baseline dump) is + // incomplete; the next snapshot must be a full one. + *series = None; + let _ = vm.stop_dirty_log(); + } + _ => {} + } + result } VmOwnership::Migration { .. } => Err(VmError::VmMigrating), VmOwnership::None => Err(VmError::VmNotRunning), @@ -2580,6 +2666,7 @@ impl RequestHandler for Vmm { } fn vm_restore(&mut self, restore_cfg: RestoreConfig) -> result::Result<(), VmError> { + self.snapshot_series = None; match &self.vm { VmOwnership::Owned(_) => Err(VmError::VmAlreadyCreated), VmOwnership::Migration { .. } => Err(VmError::VmMigrating), @@ -2841,6 +2928,8 @@ impl RequestHandler for Vmm { } fn vm_delete(&mut self) -> result::Result<(), VmError> { + // Any diff-snapshot series dies with the VM. + self.snapshot_series = None; if self.vm_config.is_none() { return Ok(()); } @@ -3334,6 +3423,8 @@ impl RequestHandler for Vmm { &mut self, send_data_migration: VmSendMigrationData, ) -> result::Result<(), MigratableError> { + // Migration owns the dirty log; any diff-snapshot series ends here. + self.snapshot_series = None; match self.vm { VmOwnership::Owned(ref vm) => { if vm.restoring() { diff --git a/vmm/src/memory_manager.rs b/vmm/src/memory_manager.rs index b1e70b3be3..965c951985 100644 --- a/vmm/src/memory_manager.rs +++ b/vmm/src/memory_manager.rs @@ -6,7 +6,7 @@ #[cfg(all(target_arch = "x86_64", feature = "guest_debug"))] use std::collections::BTreeMap; use std::collections::{HashMap, HashSet}; -use std::fs::{File, OpenOptions}; +use std::fs::{self, File, OpenOptions}; use std::io::{self, Seek, SeekFrom}; use std::mem::{MaybeUninit, zeroed}; use std::num::NonZeroUsize; @@ -229,6 +229,9 @@ pub struct MemoryManager { thp: bool, user_provided_zones: bool, snapshot_memory_ranges: MemoryRangeTable, + // When set, Transportable::send writes only these dirty ranges, placed at + // their offsets within the full snapshot layout (diff snapshot). + diff_snapshot_ranges: Option<(MemoryRangeTable, u32)>, memory_zones: MemoryZones, log_dirty: bool, // Enable dirty logging for created RAM regions arch_mem_regions: Vec, @@ -394,6 +397,18 @@ pub enum Error { #[error("Error reading from snapshot file")] SnapshotRead(#[source] io::Error), + /// Error scanning the snapshot directory for diff files + #[error("Error scanning the snapshot directory for diff files")] + RestoreDiffScan(#[source] io::Error), + + /// The diff-file sequence has a gap + #[error("Diff snapshot {0} is missing from the snapshot directory")] + RestoreDiffGap(u32), + + /// A diff-snapshot series cannot be restored on demand + #[error("A diff-snapshot series cannot be combined with 'memory_restore_mode=ondemand'")] + RestoreDiffWithOnDemand, + // Error copying snapshot into region #[error("Error copying snapshot into region")] SnapshotCopy(#[source] GuestMemoryError), @@ -850,84 +865,15 @@ impl MemoryManager { file_path: PathBuf, saved_regions: &MemoryRangeTable, ) -> Result<(), Error> { - // Open (read only) the snapshot file. - let mut memory_file = OpenOptions::new() - .read(true) - .open(file_path) - .map_err(Error::SnapshotOpen)?; - - let guest_memory = self.guest_memory.memory(); - let mut file_cursor: u64 = 0; - - for range in saved_regions.regions() { - let end = file_cursor + range.length; - - // First call doubles as a SEEK_HOLE-support probe. On error, - // take the dense path which seeks-and-streams sequentially. - match next_data_extent(memory_file.as_fd(), file_cursor, end) { - Ok(mut next) => { - while let Some((data_off, ext_len)) = next { - debug_assert!(data_off >= file_cursor); - let in_region = data_off - .checked_sub(file_cursor) - .expect("extent precedes file_cursor"); - memory_file - .seek(SeekFrom::Start(data_off)) - .map_err(Error::SnapshotRead)?; - let mut done: u64 = 0; - while done < ext_len { - let n = guest_memory - .read_volatile_from( - GuestAddress(range.gpa + in_region + done), - &mut memory_file, - (ext_len - done) as usize, - ) - .map_err(Error::SnapshotCopy)?; - if n == 0 { - return Err(Error::SnapshotRead(io::Error::new( - io::ErrorKind::UnexpectedEof, - "read_volatile_from returned 0 inside data extent", - ))); - } - done += n as u64; - } - next = next_data_extent(memory_file.as_fd(), data_off + ext_len, end) - .map_err(Error::SnapshotRead)?; - } - } - Err(_) => { - memory_file - .seek(SeekFrom::Start(file_cursor)) - .map_err(Error::SnapshotRead)?; - let mut offset: u64 = 0; - // Manual partial-read loop preserves the workaround for - // https://github.com/rust-vmm/vm-memory/issues/174 - loop { - let bytes_read = guest_memory - .read_volatile_from( - GuestAddress(range.gpa + offset), - &mut memory_file, - (range.length - offset) as usize, - ) - .map_err(Error::SnapshotCopy)?; - if bytes_read == 0 { - return Err(Error::SnapshotRead(io::Error::new( - io::ErrorKind::UnexpectedEof, - "Memory snapshot file is shorter than the saved range", - ))); - } - offset += bytes_read as u64; - if offset == range.length { - break; - } - } - } - } - - file_cursor = end; - } + replay_memory_file(&self.guest_memory.memory(), file_path, saved_regions, true) + } - Ok(()) + fn apply_diff_snapshot( + &mut self, + file_path: PathBuf, + saved_regions: &MemoryRangeTable, + ) -> Result<(), Error> { + apply_diff_file(&self.guest_memory.memory(), file_path, saved_regions) } // Restore guest memory by mapping the snapshot file copy-on-write over @@ -2038,6 +1984,7 @@ impl MemoryManager { reserve: config.reserve, user_provided_zones, snapshot_memory_ranges: MemoryRangeTable::default(), + diff_snapshot_ranges: None, memory_zones, guest_ram_mappings: Vec::new(), uffd_handler: None, @@ -2084,20 +2031,49 @@ impl MemoryManager { )?; if !mem_snapshot.memory_ranges.is_empty() { - match memory_restore_mode { - MemoryRestoreMode::OnDemand => mm.lock().unwrap().restore_by_uffd( - &memory_file_path, - &mem_snapshot.memory_ranges, - exit_evt, - )?, - MemoryRestoreMode::CopyOnWrite => mm - .lock() - .unwrap() - .mmap_cow_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?, - MemoryRestoreMode::Copy => mm - .lock() - .unwrap() - .fill_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?, + let source_dir = url_to_path(source_url).map_err(Error::Restore)?; + let deltas = discover_diff_files(&source_dir)?; + if deltas.is_empty() { + match memory_restore_mode { + MemoryRestoreMode::OnDemand => mm.lock().unwrap().restore_by_uffd( + &memory_file_path, + &mem_snapshot.memory_ranges, + exit_evt, + )?, + MemoryRestoreMode::CopyOnWrite => { + mm.lock().unwrap().mmap_cow_saved_regions( + memory_file_path, + &mem_snapshot.memory_ranges, + )?; + } + MemoryRestoreMode::Copy => mm + .lock() + .unwrap() + .fill_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?, + } + } else { + // Diff-snapshot series: fill from the baseline (honoring the + // restore mode), then replay each delta's dirty extents in + // sequence order. Replayed writes over a copy-on-write + // baseline fault in privately, so the sharing story is + // unchanged. + if memory_restore_mode == MemoryRestoreMode::OnDemand { + return Err(Error::RestoreDiffWithOnDemand); + } + let mut mm_locked = mm.lock().unwrap(); + if memory_restore_mode == MemoryRestoreMode::CopyOnWrite { + mm_locked.mmap_cow_saved_regions( + memory_file_path, + &mem_snapshot.memory_ranges, + )?; + } else { + mm_locked + .fill_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?; + } + for delta in deltas { + mm_locked.apply_diff_snapshot(delta, &mem_snapshot.memory_ranges)?; + } + drop(mm_locked); } } @@ -3458,12 +3434,29 @@ pub struct MemoryManagerSnapshotData { next_hotplug_slot: usize, } +impl MemoryManager { + /// Restricts the next Transportable::send to the given dirty ranges, + /// written at their offsets within the full snapshot layout as delta + /// number `seq` of the series. + pub fn set_diff_snapshot_ranges(&mut self, table: MemoryRangeTable, seq: u32) { + self.diff_snapshot_ranges = Some((table, seq)); + } + + /// Whether the next Transportable::send writes a delta of a series. + pub fn diff_snapshot_active(&self) -> bool { + self.diff_snapshot_ranges.is_some() + } +} + impl Snapshottable for MemoryManager { fn id(&self) -> String { MEMORY_MANAGER_SNAPSHOT_ID.to_string() } fn snapshot(&mut self) -> result::Result { + // A stale diff table must not leak into this snapshot; the caller + // re-injects one after device capture when a diff is requested. + self.diff_snapshot_ranges = None; let memory_ranges = self.memory_range_table(MemoryRangePolicy::SkipPersisted)?; // Store locally this list of ranges as it will be used through the @@ -3492,8 +3485,29 @@ impl Transportable for MemoryManager { return Ok(()); } - let mut memory_file_path = url_to_path(destination_url)?; - memory_file_path.push(String::from(SNAPSHOT_FILENAME)); + let dest_dir = url_to_path(destination_url)?; + // A delta only carries dirty extents; its sequence number is its + // replay position within the series directory. + let diff_seq = self.diff_snapshot_ranges.as_ref().map(|(_, seq)| *seq); + let final_path = match diff_seq { + Some(seq) => dest_dir.join(format!("{SNAPSHOT_FILENAME}.diff.{seq}")), + None => dest_dir.join(SNAPSHOT_FILENAME), + }; + // A delta is written under a temporary name and renamed into place, + // so an interrupted diff never invalidates a consistent directory. A + // full dump instead removes any leftover deltas: restore discovers + // diff files automatically, and stale ones would replay over the new + // baseline. + let memory_file_path = if let Some(seq) = diff_seq { + let tmp = dest_dir.join(format!("{SNAPSHOT_FILENAME}.diff.{seq}.tmp")); + let _ = fs::remove_file(&tmp); + tmp + } else { + remove_diff_files(&dest_dir) + .context("Error removing stale diff files") + .map_err(MigratableError::MigrateSend)?; + final_path.clone() + }; let mut memory_file = OpenOptions::new() .read(true) @@ -3519,10 +3533,44 @@ impl Transportable for MemoryManager { // write path which never writes past the growing EOF. let sparse_layout = memory_file.set_len(total_len).is_ok(); + // (file offset, range) pairs: a full snapshot lays ranges out densely; + // a diff places each dirty range at its offset within that same layout + // so its extents apply onto a base snapshot without translation. + let mut write_plan: Vec<(u64, MemoryRange)> = Vec::new(); + let mut layout_cursor: u64 = 0; + for full in self.snapshot_memory_ranges.regions() { + match &self.diff_snapshot_ranges { + None => write_plan.push(( + layout_cursor, + MemoryRange { + gpa: full.gpa, + length: full.length, + }, + )), + Some((dirty, _)) => { + for d in dirty.regions() { + let start = d.gpa.max(full.gpa); + let end = (d.gpa + d.length).min(full.gpa + full.length); + if start < end { + write_plan.push(( + layout_cursor + (start - full.gpa), + MemoryRange { + gpa: start, + length: end - start, + }, + )); + } + } + } + } + layout_cursor += full.length; + } + debug_assert_eq!(layout_cursor, total_len); + let guest_memory = self.guest_memory.memory(); - let mut file_cursor: u64 = 0; - for range in self.snapshot_memory_ranges.regions() { + for (file_cursor, range) in write_plan { + let range = ⦥ let mut wrote_sparse = false; if sparse_layout && let Some(region) = guest_memory.find_region(GuestAddress(range.gpa)) @@ -3573,11 +3621,18 @@ impl Transportable for MemoryManager { } } } + } - file_cursor += range.length; + if memory_file_path != final_path { + memory_file + .sync_all() + .context("Error syncing memory snapshot delta") + .map_err(MigratableError::MigrateSend)?; + fs::rename(&memory_file_path, &final_path) + .with_context(|| format!("Error renaming {memory_file_path:?} to {final_path:?}")) + .map_err(MigratableError::MigrateSend)?; } - debug_assert_eq!(file_cursor, total_len); Ok(()) } } @@ -3747,6 +3802,168 @@ fn do_mmap_cow_saved_regions( Error::SnapshotMmap(io::Error::other("snapshot range file offset overflow")) })?; } + + Ok(()) +} + +// Reads a snapshot memory file into guest RAM, walking the file's data +// extents and skipping holes. When the filesystem cannot report extents, +// `dense_fallback` selects between streaming the whole layout (full +// snapshot) and failing (diff snapshot, where a hole must keep the content +// the pages already have). +fn replay_memory_file( + guest_memory: &GuestMemoryMmap, + file_path: PathBuf, + saved_regions: &MemoryRangeTable, + dense_fallback: bool, +) -> Result<(), Error> { + // Open (read only) the snapshot file. + let mut memory_file = OpenOptions::new() + .read(true) + .open(file_path) + .map_err(Error::SnapshotOpen)?; + + let mut file_cursor: u64 = 0; + + for range in saved_regions.regions() { + let end = file_cursor + range.length; + + // First call doubles as a SEEK_HOLE-support probe. On error, + // take the dense path which seeks-and-streams sequentially. + match next_data_extent(memory_file.as_fd(), file_cursor, end) { + Ok(mut next) => { + while let Some((data_off, ext_len)) = next { + debug_assert!(data_off >= file_cursor); + let in_region = data_off + .checked_sub(file_cursor) + .expect("extent precedes file_cursor"); + memory_file + .seek(SeekFrom::Start(data_off)) + .map_err(Error::SnapshotRead)?; + let mut done: u64 = 0; + while done < ext_len { + let n = guest_memory + .read_volatile_from( + GuestAddress(range.gpa + in_region + done), + &mut memory_file, + (ext_len - done) as usize, + ) + .map_err(Error::SnapshotCopy)?; + if n == 0 { + return Err(Error::SnapshotRead(io::Error::new( + io::ErrorKind::UnexpectedEof, + "read_volatile_from returned 0 inside data extent", + ))); + } + done += n as u64; + } + next = next_data_extent(memory_file.as_fd(), data_off + ext_len, end) + .map_err(Error::SnapshotRead)?; + } + } + Err(e) if !dense_fallback => { + return Err(Error::SnapshotRead(io::Error::new( + e.kind(), + format!("diff restore needs SEEK_DATA/SEEK_HOLE support: {e}"), + ))); + } + Err(_) => { + memory_file + .seek(SeekFrom::Start(file_cursor)) + .map_err(Error::SnapshotRead)?; + let mut offset: u64 = 0; + // Manual partial-read loop preserves the workaround for + // https://github.com/rust-vmm/vm-memory/issues/174 + loop { + let bytes_read = guest_memory + .read_volatile_from( + GuestAddress(range.gpa + offset), + &mut memory_file, + (range.length - offset) as usize, + ) + .map_err(Error::SnapshotCopy)?; + if bytes_read == 0 { + return Err(Error::SnapshotRead(io::Error::new( + io::ErrorKind::UnexpectedEof, + "Memory snapshot file is shorter than the saved range", + ))); + } + offset += bytes_read as u64; + if offset == range.length { + break; + } + } + } + } + + file_cursor = end; + } + + Ok(()) +} + +// Applies a diff snapshot's dirty extents over already-restored RAM; the +// hole-punched pages the delta never dirtied keep their baseline content. +// The length check rejects a file that is not from the restored series' +// layout before any page is touched. +fn apply_diff_file( + guest_memory: &GuestMemoryMmap, + file_path: PathBuf, + saved_regions: &MemoryRangeTable, +) -> Result<(), Error> { + let total_len: u64 = saved_regions.regions().iter().map(|r| r.length).sum(); + let file_len = fs::metadata(&file_path).map_err(Error::SnapshotOpen)?.len(); + if file_len != total_len { + return Err(Error::SnapshotOpen(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "diff file {} is {file_len} bytes, expected {total_len}: \ + not from this snapshot series?", + file_path.display() + ), + ))); + } + replay_memory_file(guest_memory, file_path, saved_regions, false) +} + +// Finds the deltas of a diff-snapshot series in `dir`, in replay order. +// Sequence numbers must be contiguous from 1; a gap fails the restore +// before any page is touched. A name whose suffix is not a number (for +// example a `.tmp` leftover from an interrupted diff) is ignored. +fn discover_diff_files(dir: &Path) -> Result, Error> { + let prefix = format!("{SNAPSHOT_FILENAME}.diff."); + let mut deltas: Vec<(u32, PathBuf)> = Vec::new(); + for entry in fs::read_dir(dir).map_err(Error::RestoreDiffScan)? { + let entry = entry.map_err(Error::RestoreDiffScan)?; + if let Some(name) = entry.file_name().to_str() + && let Some(rest) = name.strip_prefix(&prefix) + && let Ok(seq) = rest.parse::() + { + deltas.push((seq, entry.path())); + } + } + deltas.sort_unstable_by_key(|(seq, _)| *seq); + for (i, (seq, _)) in deltas.iter().enumerate() { + if *seq != i as u32 + 1 { + return Err(Error::RestoreDiffGap(i as u32 + 1)); + } + } + Ok(deltas.into_iter().map(|(_, path)| path).collect()) +} + +// Removes every delta file in `dir`, `.tmp` leftovers included. A full +// dump must not leave stale deltas behind: restore discovers diff files +// automatically and would replay them over the new baseline. +fn remove_diff_files(dir: &Path) -> io::Result<()> { + let prefix = format!("{SNAPSHOT_FILENAME}.diff."); + for entry in fs::read_dir(dir)? { + let entry = entry?; + if let Some(name) = entry.file_name().to_str() + && name.strip_prefix(&prefix).is_some() + { + fs::remove_file(entry.path())?; + } + } Ok(()) } @@ -4047,4 +4264,98 @@ mod tests { assert_eq!(gm.read_obj::(GuestAddress(0)).unwrap(), 0xcd); } } + + fn two_page_table(page: u64) -> MemoryRangeTable { + let mut table = MemoryRangeTable::default(); + table.push(MemoryRange { + gpa: 0, + length: 2 * page, + }); + table + } + + fn temp_file_path(dir: &tempfile::TempDir, name: &str) -> PathBuf { + dir.path().join(name) + } + + #[test] + fn diff_replay_overwrites_extents_and_keeps_holes() { + let page = page_size(); + let gm = GuestMemoryMmap::from_ranges(&[(GuestAddress(0), (2 * page) as usize)]).unwrap(); + let table = two_page_table(page); + let dir = tempfile::tempdir().unwrap(); + + // Full baseline: both pages 0x11. + let base_path = temp_file_path(&dir, "memory-ranges"); + let mut base = File::create(&base_path).unwrap(); + base.write_all(&vec![0x11u8; (2 * page) as usize]).unwrap(); + replay_memory_file(&gm, base_path, &table, true).unwrap(); + assert_eq!(gm.read_obj::(GuestAddress(0)).unwrap(), 0x11); + assert_eq!(gm.read_obj::(GuestAddress(page)).unwrap(), 0x11); + + // Sparse delta: only page 1 dirtied, written as 0xab. + let diff_path = temp_file_path(&dir, "memory-ranges.diff"); + let mut diff = File::create(&diff_path).unwrap(); + diff.set_len(2 * page).unwrap(); + diff.seek(SeekFrom::Start(page)).unwrap(); + diff.write_all(&vec![0xabu8; page as usize]).unwrap(); + apply_diff_file(&gm, diff_path, &table).unwrap(); + + // Page 0 sits in the delta's hole and must keep the baseline content. + assert_eq!(gm.read_obj::(GuestAddress(0)).unwrap(), 0x11); + assert_eq!(gm.read_obj::(GuestAddress(page - 1)).unwrap(), 0x11); + assert_eq!(gm.read_obj::(GuestAddress(page)).unwrap(), 0xab); + assert_eq!(gm.read_obj::(GuestAddress(2 * page - 1)).unwrap(), 0xab); + } + + #[test] + fn diff_replay_rejects_wrong_length_file() { + let page = page_size(); + let gm = GuestMemoryMmap::from_ranges(&[(GuestAddress(0), (2 * page) as usize)]).unwrap(); + let table = two_page_table(page); + let dir = tempfile::tempdir().unwrap(); + + // A file one page short cannot be a delta of this layout. + let diff_path = temp_file_path(&dir, "memory-ranges.diff.1"); + File::create(&diff_path).unwrap().set_len(page).unwrap(); + apply_diff_file(&gm, diff_path, &table).unwrap_err(); + } + + #[test] + fn diff_discovery_orders_ignores_and_detects_gaps() { + let dir = tempfile::tempdir().unwrap(); + for name in [ + "memory-ranges", + "memory-ranges.diff.2", + "memory-ranges.diff.1", + "memory-ranges.diff.10", + "config.json", + ] { + File::create(dir.path().join(name)).unwrap(); + } + // A gap (3..=9 missing) fails before any page is touched. + assert!(matches!( + discover_diff_files(dir.path()), + Err(Error::RestoreDiffGap(3)) + )); + + fs::remove_file(dir.path().join("memory-ranges.diff.10")).unwrap(); + // A .tmp leftover from an interrupted diff is ignored. + File::create(dir.path().join("memory-ranges.diff.3.tmp")).unwrap(); + let deltas = discover_diff_files(dir.path()).unwrap(); + assert_eq!( + deltas, + vec![ + dir.path().join("memory-ranges.diff.1"), + dir.path().join("memory-ranges.diff.2"), + ] + ); + } + + #[test] + fn diff_discovery_empty_without_deltas() { + let dir = tempfile::tempdir().unwrap(); + File::create(dir.path().join("memory-ranges")).unwrap(); + assert!(discover_diff_files(dir.path()).unwrap().is_empty()); + } } diff --git a/vmm/src/vm.rs b/vmm/src/vm.rs index f696635cef..7b84103ed1 100644 --- a/vmm/src/vm.rs +++ b/vmm/src/vm.rs @@ -14,7 +14,7 @@ use std::collections::{BTreeMap, BTreeSet, HashMap}; #[cfg(feature = "fw_cfg")] use std::ffi; -use std::fs::{File, OpenOptions}; +use std::fs::{self, File, OpenOptions}; use std::io::{self, Seek, SeekFrom, Write}; use std::num::Wrapping; use std::ops::Deref; @@ -1866,8 +1866,8 @@ impl Vm { #[cfg(target_arch = "aarch64")] fn configure_system( &mut self, - _rsdp_addr: Option, - _entry_addr: EntryPoint, + rsdp_addr: Option, + entry_addr: EntryPoint, ) -> Result<()> { let cmdline = Self::generate_cmdline( self.config.lock().unwrap().payload.as_ref().unwrap(), @@ -1935,13 +1935,26 @@ impl Vm { )) })?; - let smbios = self - .config - .lock() - .unwrap() - .platform - .as_ref() - .and_then(|p| p.smbios_config()); + let platform = self.config.lock().unwrap().platform.clone(); + let smbios = platform.as_ref().and_then(|p| p.smbios_config()); + + // firmware reads its memory and devices from the full device tree + let direct_boot = entry_addr.entry_addr != layout::UEFI_START; + + // init_pmu() above must run on this path too, or KVM_RUN fails EINVAL + if direct_boot && platform.as_ref().is_none_or(|p| p.acpi_boot) { + let rsdp_addr = rsdp_addr.ok_or(Error::ConfigureSystem( + arch::Error::PlatformSpecific(arch::aarch64::Error::MissingRsdp), + ))?; + return arch::aarch64::configure_system_acpi( + &mem, + cmdline.as_cstring().unwrap().to_str().unwrap(), + &initramfs_config, + rsdp_addr, + smbios.as_ref(), + ) + .map_err(Error::ConfigureSystem); + } arch::configure_system( &mem, @@ -3063,6 +3076,14 @@ impl Vm { self.memory_manager.lock().unwrap().memory_range_table(mode) } + /// Restricts the next snapshot send to the given dirty ranges (diff snapshot). + pub fn set_diff_snapshot_ranges(&mut self, table: MemoryRangeTable, seq: u32) { + self.memory_manager + .lock() + .unwrap() + .set_diff_snapshot_ranges(table, seq); + } + pub fn guest_memory(&self) -> GuestMemoryAtomic { self.memory_manager.lock().unwrap().guest_memory() } @@ -3422,55 +3443,68 @@ impl Snapshottable for Vm { } } +// Writes one snapshot file. A delta of a diff-snapshot series replaces the +// file already in the directory, through a temporary name and a rename so an +// interrupted write never invalidates the consistent copy; anything else +// must land in a fresh directory and fails if the file exists. +fn store_snapshot_file( + destination_url: &str, + name: &str, + bytes: &[u8], + replace: bool, +) -> result::Result<(), MigratableError> { + let mut final_path = url_to_path(destination_url)?; + final_path.push(name); + let write_path = if replace { + let tmp = final_path.with_extension("tmp"); + let _ = fs::remove_file(&tmp); + tmp + } else { + final_path.clone() + }; + let mut file = OpenOptions::new() + .read(true) + .write(true) + .create_new(true) + .open(&write_path) + .with_context(|| format!("Error creating snapshot file {write_path:?}")) + .map_err(MigratableError::MigrateSend)?; + file.write_all(bytes) + .with_context(|| format!("Error writing snapshot file {write_path:?}")) + .map_err(MigratableError::MigrateSend)?; + if replace { + file.sync_all() + .with_context(|| format!("Error syncing snapshot file {write_path:?}")) + .map_err(MigratableError::MigrateSend)?; + fs::rename(&write_path, &final_path) + .with_context(|| format!("Error renaming {write_path:?} to {final_path:?}")) + .map_err(MigratableError::MigrateSend)?; + } + Ok(()) +} + impl Transportable for Vm { fn send( &self, snapshot: &Snapshot, destination_url: &str, ) -> result::Result<(), MigratableError> { - let mut snapshot_config_path = url_to_path(destination_url)?; - snapshot_config_path.push(SNAPSHOT_CONFIG_FILE); - - // Create the snapshot config file - let mut snapshot_config_file = OpenOptions::new() - .read(true) - .write(true) - .create_new(true) - .open(snapshot_config_path) - .context("Error creating VM config snapshot file") - .map_err(MigratableError::MigrateSend)?; + let replace = self.memory_manager.lock().unwrap().diff_snapshot_active(); - // Serialize and write the snapshot config let vm_config = serde_json::to_string(self.config.lock().unwrap().deref()) .context("Error serializing VM config snapshot") .map_err(MigratableError::MigrateSend)?; + store_snapshot_file( + destination_url, + SNAPSHOT_CONFIG_FILE, + vm_config.as_bytes(), + replace, + )?; - snapshot_config_file - .write_all(vm_config.as_bytes()) - .context("Error writing VM config snapshot") - .map_err(MigratableError::MigrateSend)?; - - let mut snapshot_state_path = url_to_path(destination_url)?; - snapshot_state_path.push(SNAPSHOT_STATE_FILE); - - // Create the snapshot state file - let mut snapshot_state_file = OpenOptions::new() - .read(true) - .write(true) - .create_new(true) - .open(snapshot_state_path) - .context("Error creating VM state snapshot file") - .map_err(MigratableError::MigrateSend)?; - - // Serialize and write the snapshot state let vm_state = serde_json::to_vec(snapshot) .context("Error serializing VM state snapshot") .map_err(MigratableError::MigrateSend)?; - - snapshot_state_file - .write_all(&vm_state) - .context("Error writing VM state snapshot") - .map_err(MigratableError::MigrateSend)?; + store_snapshot_file(destination_url, SNAPSHOT_STATE_FILE, &vm_state, replace)?; // Tell the memory manager to also send/write its own snapshot. if let Some(memory_manager_snapshot) = snapshot.snapshots.get(MEMORY_MANAGER_SNAPSHOT_ID) { diff --git a/vmm/src/vm_config.rs b/vmm/src/vm_config.rs index f66c9d2387..74f35874cd 100644 --- a/vmm/src/vm_config.rs +++ b/vmm/src/vm_config.rs @@ -122,6 +122,11 @@ pub fn default_platformconfig_iommu_address_width_bits() -> u8 { DEFAULT_IOMMU_ADDRESS_WIDTH_BITS } +#[cfg(target_arch = "aarch64")] +pub fn default_platformconfig_acpi_boot() -> bool { + true +} + pub fn default_platformconfig_vfio_p2p_dma() -> bool { true } @@ -166,6 +171,13 @@ pub struct PlatformConfig { pub iommufd_fd: Option, #[serde(default = "default_platformconfig_vfio_p2p_dma")] pub vfio_p2p_dma: bool, + /// Hand the guest ACPI through a synthesized EFI handoff instead of a + /// hardware-describing device tree. PCI hotplug is ACPI-only, so a + /// direct-kernel-booted aarch64 guest cannot see hot-added or ejected + /// devices without this. Firmware boot ignores it. + #[cfg(target_arch = "aarch64")] + #[serde(default = "default_platformconfig_acpi_boot")] + pub acpi_boot: bool, } #[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))]