From ed830ab6d73eca071d00784f0f79b6295fdede2b Mon Sep 17 00:00:00 2001 From: CMGS Date: Thu, 23 Jul 2026 13:55:44 +0800 Subject: [PATCH 1/4] ci: add dev branch build and release workflow Build cloud-hypervisor on the dev branch for x86_64 and aarch64, then publish both static binaries to a rolling dev prerelease with checksums and build provenance. Signed-off-by: CMGS --- .github/workflows/dev-release.yaml | 110 +++++++++++++++++++++++++++++ 1 file changed, 110 insertions(+) create mode 100644 .github/workflows/dev-release.yaml diff --git a/.github/workflows/dev-release.yaml b/.github/workflows/dev-release.yaml new file mode 100644 index 0000000000..7218188b0a --- /dev/null +++ b/.github/workflows/dev-release.yaml @@ -0,0 +1,110 @@ +name: Build and Release Dev + +on: + push: + branches: [dev] + +permissions: + contents: write + +concurrency: + group: dev-release + cancel-in-progress: true + +jobs: + build: + strategy: + fail-fast: false + matrix: + include: + - runner: ubuntu-22.04 + arch: x86_64 + - runner: ubuntu-22.04-arm + arch: aarch64 + runs-on: ${{ matrix.runner }} + steps: + - uses: actions/checkout@v7 + + - name: Install Rust toolchain + uses: dtolnay/rust-toolchain@stable + with: + toolchain: stable + + - name: Show toolchain + run: | + rustc --version --verbose + cargo --version --verbose + + - name: Build cloud-hypervisor (release) + run: cargo build --release + + - name: Stage artifacts + run: | + set -euo pipefail + mkdir -p dist + cp -v target/release/cloud-hypervisor "dist/cloud-hypervisor-${{ matrix.arch }}" + "./dist/cloud-hypervisor-${{ matrix.arch }}" --version + rustc --version --verbose | tr '\n' ';' > "dist/rustc-${{ matrix.arch }}.txt" + + - uses: actions/upload-artifact@v7 + with: + name: dist-${{ matrix.arch }} + path: dist/ + + release: + needs: build + runs-on: ubuntu-22.04 + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + + - uses: actions/download-artifact@v8 + with: + pattern: dist-* + merge-multiple: true + path: dist + + - name: Add provenance and checksums + run: | + set -euo pipefail + chmod +x dist/*-x86_64 dist/*-aarch64 + version=$(./dist/cloud-hypervisor-x86_64 --version | awk 'NR==1 {v=$2} END {print v}') + cat > dist/build-info.json < SHA256SUMS) + cat dist/SHA256SUMS + + - name: Create dev release + uses: softprops/action-gh-release@v3 + with: + tag_name: dev + target_commitish: ${{ github.sha }} + name: Dev Build + body: | + Branch: `${{ github.ref_name }}` + Commit: `${{ github.sha }}` + Workflow Run: `${{ github.run_id }}` + files: dist/* + prerelease: true + make_latest: false + overwrite_files: true + fail_on_unmatched_files: true + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + + - name: Move dev tag to current commit + run: | + git tag -f dev "${GITHUB_SHA}" + git push origin refs/tags/dev --force From 256bc3edb97276cc4a17db57eea96cc254a56d85 Mon Sep 17 00:00:00 2001 From: CMGS Date: Sun, 19 Jul 2026 17:11:47 +0800 Subject: [PATCH 2/4] vmm: add diff snapshots Every snapshot currently dumps the entire guest RAM, so the pause a snapshot imposes grows with the memory size regardless of how little the guest has changed. Iterative checkpointing and warm pre-copy (take a snapshot, let the guest run, take another) repeat that full-memory cost every round. Add a diff snapshot: `vm.snapshot` accepts a `snapshot_type` of `full` (default) or `diff`, exposed as `ch-remote snapshot --diff`. The first diff of a series writes a full baseline and enables dirty-page tracking; each subsequent diff dumps only the pages dirtied since the previous one, so its pause is proportional to the dirtied memory rather than to the guest RAM size. Dirty pages are harvested after the device snapshot, so pages touched by snapshot side effects are included. The whole series lives in one directory, and every diff request of a series must target that directory. A delta is a sparse `memory-ranges.diff.N` whose extents sit at the same offsets as in the baseline `memory-ranges`, written under a temporary name and renamed into place so an interrupted diff never invalidates a consistent directory; `config.json` and `state.json` are replaced the same way and always describe the newest delta. A full dump removes leftover delta files, since restore discovers them by name. A full snapshot, a memory layout change, a migration, restore, or deleting the VM ends the series; the next diff starts a new one. Restore needs no new parameter: when `source_url` points at a directory containing delta files, memory fills from the baseline and each delta's dirty extents replay in sequence order through the same SEEK_DATA walk the eager restore already uses; device state and config come from the newest delta as before. A gap in the sequence or a delta whose length does not match the restored layout is rejected before any page is touched, a filesystem that cannot report extents fails the restore rather than zeroing undirtied pages, and a diff-snapshot series cannot be combined with `memory_restore_mode=ondemand`. Dirty tracking is enabled lazily on the first diff rather than at boot, so a VM that never takes a diff snapshot pays no runtime cost. Validated: an integration test writes tmpfs markers before and after the baseline and reads both back through a restore of the series directory; unit tests cover extent replay over holes, the layout-length check, and delta discovery (ordering, gap detection, ignoring temporary files). On a three-phase workload that dirtied one phase between snapshots, the diff pause measured ~17x shorter than the full one. Signed-off-by: CMGS --- cloud-hypervisor/src/bin/ch-remote.rs | 32 +- cloud-hypervisor/tests/integration.rs | 140 ++++++ docs/snapshot_restore.md | 32 ++ fuzz/fuzz_targets/http_api.rs | 4 +- vmm/src/api/mod.rs | 36 +- vmm/src/api/openapi/cloud-hypervisor.yaml | 8 + vmm/src/lib.rs | 101 ++++- vmm/src/memory_manager.rs | 507 +++++++++++++++++----- vmm/src/vm.rs | 93 ++-- 9 files changed, 799 insertions(+), 154 deletions(-) diff --git a/cloud-hypervisor/src/bin/ch-remote.rs b/cloud-hypervisor/src/bin/ch-remote.rs index 44de33bffc..5749c2c4ac 100644 --- a/cloud-hypervisor/src/bin/ch-remote.rs +++ b/cloud-hypervisor/src/bin/ch-remote.rs @@ -497,12 +497,10 @@ fn rest_api_do_command(matches: &ArgMatches, socket: &mut UnixStream) -> ApiResu .map_err(Error::HttpApiClient) } Some("snapshot") => { + let sub = matches.subcommand_matches("snapshot").unwrap(); let snapshot_config = snapshot_config( - matches - .subcommand_matches("snapshot") - .unwrap() - .get_one::("snapshot_config") - .unwrap(), + sub.get_one::("snapshot_config").unwrap(), + sub.get_flag("diff"), ); simple_api_command(socket, "PUT", "snapshot", Some(&snapshot_config)) .map_err(Error::HttpApiClient) @@ -722,12 +720,10 @@ fn dbus_api_do_command(matches: &ArgMatches, proxy: &DBusApi1ProxyBlocking<'_>) proxy.api_vm_add_vsock(&vsock_config) } Some("snapshot") => { + let sub = matches.subcommand_matches("snapshot").unwrap(); let snapshot_config = snapshot_config( - matches - .subcommand_matches("snapshot") - .unwrap() - .get_one::("snapshot_config") - .unwrap(), + sub.get_one::("snapshot_config").unwrap(), + sub.get_flag("diff"), ); proxy.api_vm_snapshot(&snapshot_config) } @@ -934,9 +930,14 @@ fn add_vsock_config(config: &str) -> Result { Ok(vsock_config) } -fn snapshot_config(url: &str) -> String { +fn snapshot_config(url: &str, diff: bool) -> String { let snapshot_config = api::VmSnapshotConfig { destination_url: String::from(url), + snapshot_type: if diff { + api::VmSnapshotType::Diff + } else { + api::VmSnapshotType::Full + }, }; serde_json::to_string(&snapshot_config).unwrap() @@ -1177,6 +1178,15 @@ fn get_cli_commands_sorted() -> Box<[Command]> { Command::new("shutdown-vmm").about("Shutdown the VMM"), Command::new("snapshot") .about("Create a snapshot from VM") + .arg( + Arg::new("diff") + .long("diff") + .action(clap::ArgAction::SetTrue) + .help( + "Write only pages dirtied since the previous snapshot; \ + the first --diff takes a full baseline and starts dirty tracking", + ), + ) .arg( Arg::new("snapshot_config") .index(1) diff --git a/cloud-hypervisor/tests/integration.rs b/cloud-hypervisor/tests/integration.rs index c2829460ca..3fefe79b73 100644 --- a/cloud-hypervisor/tests/integration.rs +++ b/cloud-hypervisor/tests/integration.rs @@ -8879,6 +8879,12 @@ mod ivshmem { ); } + #[test] + #[cfg(not(feature = "mshv"))] + fn test_snapshot_restore_diff_chain() { + snapshot_restore_common::_test_snapshot_restore_diff_chain(); + } + #[test] fn test_snapshot_restore_with_resume() { snapshot_restore_common::_test_snapshot_restore( @@ -9019,6 +9025,8 @@ mod ivshmem { } mod snapshot_restore_common { + #[cfg(not(feature = "mshv"))] + use std::fs::create_dir; use std::fs::{read_to_string, remove_dir_all}; use std::process::Command; @@ -9057,6 +9065,138 @@ mod snapshot_restore_common { )); } + fn snapshot_diff(api_socket: &str, url: &str) -> bool { + let output = Command::new(clh_command("ch-remote")) + .args([ + &format!("--api-socket={api_socket}"), + "snapshot", + "--diff", + url, + ]) + .output() + .unwrap(); + if !output.status.success() { + eprintln!( + "ch-remote snapshot --diff failed: {}", + String::from_utf8_lossy(&output.stderr) + ); + } + output.status.success() + } + + // A diff series baseline plus one delta, restored through memory_chain. + // tmpfs markers written before and after the baseline prove the delta + // rebases onto the baseline: marker A lives in the baseline pages, marker + // B only in the delta's dirty extents. + #[cfg(not(feature = "mshv"))] + pub(crate) fn _test_snapshot_restore_diff_chain() { + let disk_config = UbuntuDiskConfig::new(JAMMY_IMAGE_NAME.to_string()); + let guest = Guest::new(Box::new(disk_config)); + let kernel_path = direct_kernel_boot_path(); + + let api_socket_source = format!("{}.1", temp_api_path(&guest.tmp_dir)); + let net_params = format!( + "id=net123,tap=,mac={},ip={},mask=255.255.255.128", + guest.network.guest_mac0, guest.network.host_ip0 + ); + + let mut child = GuestCommand::new(&guest) + .args(["--api-socket", &api_socket_source]) + .args(["--cpus", "boot=1"]) + .args(["--memory", "size=1G"]) + .args(["--kernel", kernel_path.to_str().unwrap()]) + .args([ + "--disk", + format!( + "path={}", + guest.disk_config.disk(DiskType::OperatingSystem).unwrap() + ) + .as_str(), + format!( + "path={}", + guest.disk_config.disk(DiskType::CloudInit).unwrap() + ) + .as_str(), + ]) + .args(["--net", net_params.as_str()]) + .args(["--cmdline", DIRECT_KERNEL_BOOT_CMDLINE]) + .capture_output() + .spawn() + .unwrap(); + + let snapshot_dir = temp_snapshot_dir_path(&guest.tmp_dir); + let series_dir = format!("{snapshot_dir}/series"); + create_dir(&series_dir).unwrap(); + + let r = panic::catch_unwind(|| { + guest.wait_vm_boot().unwrap(); + + guest + .ssh_command("echo diffmarkerA > /dev/shm/marker_a") + .unwrap(); + + // First diff of a series writes the full baseline. + assert!(remote_command(&api_socket_source, "pause", None)); + assert!(snapshot_diff( + &api_socket_source, + format!("file://{series_dir}").as_str(), + )); + assert!(remote_command(&api_socket_source, "resume", None)); + + guest + .ssh_command("echo diffmarkerB > /dev/shm/marker_b") + .unwrap(); + + // Second diff lands in the same directory, carrying only the + // pages dirtied since the baseline. + assert!(remote_command(&api_socket_source, "pause", None)); + assert!(snapshot_diff( + &api_socket_source, + format!("file://{series_dir}").as_str(), + )); + }); + + kill_child(&mut child); + let output = child.wait_with_output().unwrap(); + handle_child_output(r, &output); + + // Restore discovers the delta files in the series directory. + let api_socket_restored = format!("{}.2", temp_api_path(&guest.tmp_dir)); + let mut child = GuestCommand::new(&guest) + .args(["--api-socket", &api_socket_restored]) + .args([ + "--restore", + format!("source_url=file://{series_dir},resume=true").as_str(), + ]) + .capture_output() + .spawn() + .unwrap(); + + let r = panic::catch_unwind(|| { + assert!(wait_until(Duration::from_secs(30), || remote_command( + &api_socket_restored, + "info", + None + ))); + + // Both markers must survive: A from the baseline pages the delta + // never dirtied, B from the delta's extents. + assert_eq!( + guest.ssh_command("cat /dev/shm/marker_a").unwrap().trim(), + "diffmarkerA" + ); + assert_eq!( + guest.ssh_command("cat /dev/shm/marker_b").unwrap().trim(), + "diffmarkerB" + ); + }); + + let _ = remove_dir_all(snapshot_dir.as_str()); + kill_child(&mut child); + let output = child.wait_with_output().unwrap(); + handle_child_output(r, &output); + } + /// Easy disambiguation between snapshot/restore variants. #[derive(Clone, Copy, Default)] pub(crate) struct SnapshotRestoreTest { diff --git a/docs/snapshot_restore.md b/docs/snapshot_restore.md index 4b1cdff006..bef738cfa3 100644 --- a/docs/snapshot_restore.md +++ b/docs/snapshot_restore.md @@ -60,6 +60,38 @@ be needed. `state.json` contains the virtual machine state. It is used to restore each component in the state it was left before the snapshot occurred. +## Diff snapshots + +Passing `snapshot_type=diff` (`ch-remote snapshot --diff `) writes only +the pages dirtied since the previous snapshot of the series. The first `diff` +request takes a full baseline and enables dirty-page tracking; each subsequent +one dumps the delta, so the pause cost is proportional to the amount of dirtied +memory rather than to the guest RAM size. A `full` request (or any snapshot +failure, a memory layout change, a migration, or deleting the VM) ends the +series; the next `diff` starts a new one with a fresh baseline. + +The whole series lives in one directory, and every `diff` request of a series +must target that same directory: + +```bash +ls /foo/snapshot/ +config.json memory-ranges memory-ranges.diff.1 memory-ranges.diff.2 state.json +``` + +Each `memory-ranges.diff.N` is a sparse file whose extents sit at the same +offsets as in the baseline `memory-ranges`. `config.json` and `state.json` +always describe the newest delta. A delta is written under a temporary name +and renamed into place, so an interrupted diff never invalidates a consistent +directory. A full dump into a directory removes any leftover delta files. + +Restore needs no extra parameter: point `source_url` at the directory. When +delta files are present, memory fills from the baseline and each delta's dirty +extents replay in sequence order; a gap in the sequence or a delta whose +length does not match the layout fails the restore before any page is +touched. The files must sit on a filesystem with `SEEK_DATA`/`SEEK_HOLE` +support, and a diff-snapshot series cannot be combined with +`memory_restore_mode=ondemand`. + ## Restore a Cloud Hypervisor VM Given that one has access to an existing snapshot in `/home/foo/snapshot`, diff --git a/fuzz/fuzz_targets/http_api.rs b/fuzz/fuzz_targets/http_api.rs index 56e9e55c6b..da1ee9186b 100644 --- a/fuzz/fuzz_targets/http_api.rs +++ b/fuzz/fuzz_targets/http_api.rs @@ -15,7 +15,7 @@ use vm_migration::MigratableError; use vmm::api::http::*; use vmm::api::{ ApiRequest, BalloonStatsResponse, RequestHandler, VmInfoResponse, VmReceiveMigrationData, - VmSendMigrationData, VmmPingResponse, + VmSendMigrationData, VmSnapshotConfig, VmmPingResponse, }; use vmm::config::RestoreConfig; use vmm::vm::{Error as VmError, VmState}; @@ -105,7 +105,7 @@ impl RequestHandler for StubApiRequestHandler { Ok(()) } - fn vm_snapshot(&mut self, _: &str) -> Result<(), VmError> { + fn vm_snapshot(&mut self, _: &VmSnapshotConfig) -> Result<(), VmError> { Ok(()) } diff --git a/vmm/src/api/mod.rs b/vmm/src/api/mod.rs index 6a3c86c2fc..83ca32134f 100644 --- a/vmm/src/api/mod.rs +++ b/vmm/src/api/mod.rs @@ -257,10 +257,25 @@ pub struct VmRemoveDeviceData { pub id: String, } +/// Type of a VM snapshot: full memory dump or dirty-pages-only delta. +#[derive(Copy, Clone, Default, Deserialize, Serialize, Debug, PartialEq, Eq)] +#[serde(rename_all = "lowercase")] +pub enum VmSnapshotType { + /// Complete guest memory dump. + #[default] + Full, + /// Only pages dirtied since the previous snapshot of the series. The + /// first diff request takes a full baseline and starts dirty tracking. + Diff, +} + #[derive(Clone, Deserialize, Serialize, Default, Debug)] pub struct VmSnapshotConfig { /// The snapshot destination URL pub destination_url: String, + /// Full dump or dirty-pages delta. + #[serde(default)] + pub snapshot_type: VmSnapshotType, } #[derive(Clone, Deserialize, Serialize, Default, Debug)] @@ -867,7 +882,7 @@ pub trait RequestHandler { fn vm_resume(&mut self) -> Result<(), VmError>; - fn vm_snapshot(&mut self, destination_url: &str) -> Result<(), VmError>; + fn vm_snapshot(&mut self, config: &VmSnapshotConfig) -> Result<(), VmError>; fn vm_restore(&mut self, restore_cfg: RestoreConfig) -> Result<(), VmError>; @@ -2027,7 +2042,7 @@ impl ApiAction for VmSnapshot { info!("API request event: VmSnapshot {config:?}"); let response = vmm - .vm_snapshot(&config.destination_url) + .vm_snapshot(&config) .map_err(ApiError::VmSnapshot) .map(|_| ApiResponsePayload::Empty); @@ -2594,4 +2609,21 @@ mod tests { .unwrap(); assert_eq!(data.effective_memory_mode(), MigrationMode::MemFDs); } + + #[test] + fn test_vm_snapshot_config_snapshot_type() { + let config: VmSnapshotConfig = + serde_json::from_str(r#"{"destination_url": "file:///foo"}"#).unwrap(); + assert_eq!(config.snapshot_type, VmSnapshotType::Full); + + let config: VmSnapshotConfig = + serde_json::from_str(r#"{"destination_url": "file:///foo", "snapshot_type": "diff"}"#) + .unwrap(); + assert_eq!(config.snapshot_type, VmSnapshotType::Diff); + + serde_json::from_str::( + r#"{"destination_url": "file:///foo", "snapshot_type": "bogus"}"#, + ) + .unwrap_err(); + } } diff --git a/vmm/src/api/openapi/cloud-hypervisor.yaml b/vmm/src/api/openapi/cloud-hypervisor.yaml index 1103c06350..588ce309ca 100644 --- a/vmm/src/api/openapi/cloud-hypervisor.yaml +++ b/vmm/src/api/openapi/cloud-hypervisor.yaml @@ -1679,6 +1679,14 @@ components: properties: destination_url: type: string + snapshot_type: + type: string + enum: [full, diff] + default: full + description: + With "diff", only pages dirtied since the previous snapshot of the + series are written; the first "diff" takes a full baseline and + starts dirty tracking. VmCoredumpData: type: object diff --git a/vmm/src/lib.rs b/vmm/src/lib.rs index c0cb7d7ddf..757dae1d7b 100644 --- a/vmm/src/lib.rs +++ b/vmm/src/lib.rs @@ -49,7 +49,8 @@ use vmm_sys_util::sock_ctrl_msg::ScmSocket; use crate::api::{ ApiRequest, ApiResponse, BalloonStatsResponse, MigrationMode, RequestHandler, TimeoutStrategy, - VmInfoResponse, VmReceiveMigrationData, VmSendMigrationData, VmmPingResponse, + VmInfoResponse, VmReceiveMigrationData, VmSendMigrationData, VmSnapshotConfig, VmSnapshotType, + VmmPingResponse, }; use crate::config::{MemoryRestoreMode, RestoreConfig, VmMemoryZoneUpdateData, add_to_config}; #[cfg(all(target_arch = "x86_64", feature = "guest_debug"))] @@ -694,6 +695,14 @@ impl VmOwnership { } } +// An active diff-snapshot series: every delta lands in the directory the +// baseline was written to, numbered by `next_seq`. +struct SnapshotSeries { + destination_url: String, + layout: MemoryRangeTable, + next_seq: u32, +} + pub struct Vmm { epoll: EpollContext, exit_evt: EventFd, @@ -719,11 +728,23 @@ pub struct Vmm { console_socket_listener: Option>, no_shutdown: bool, check_migration_evt: EventFd, + // Memory layout at diff-snapshot series start; Some = dirty logging active. + snapshot_series: Option, } /// Time before aborting on the page fault connection. const FAULT_CONNECTION_ACCEPT_TIMEOUT: Duration = Duration::from_secs(30); +/// Whether two snapshot memory layouts describe the same regions, i.e. a diff +/// written against one applies at the same file offsets as the other. +fn same_memory_layout(a: &MemoryRangeTable, b: &MemoryRangeTable) -> bool { + a.regions().len() == b.regions().len() + && a.regions() + .iter() + .zip(b.regions()) + .all(|(x, y)| x.gpa == y.gpa && x.length == y.length) +} + /// Just a wrapper for the data that goes into /// [`ReceiveMigrationState::Configured`] struct ReceiveMigrationConfiguredData { @@ -953,6 +974,7 @@ impl Vmm { console_socket_listener: None, no_shutdown, check_migration_evt, + snapshot_series: None, }) } @@ -2559,20 +2581,84 @@ impl RequestHandler for Vmm { } } - fn vm_snapshot(&mut self, destination_url: &str) -> result::Result<(), VmError> { + fn vm_snapshot(&mut self, config: &VmSnapshotConfig) -> result::Result<(), VmError> { match self.vm { VmOwnership::Owned(ref mut vm) => { if vm.restoring() { return Err(VmError::VmRestoring); } + let requested_diff = config.snapshot_type == VmSnapshotType::Diff; + // A diff request continues the series if one is active; + // otherwise it takes a full baseline into the series + // directory and starts dirty logging. A full request ends + // any active series. + let effective_diff = requested_diff && self.snapshot_series.is_some(); + if requested_diff { + let layout = vm + .memory_range_table(MemoryRangePolicy::SkipPersisted) + .map_err(VmError::Snapshot)?; + if let Some(series) = &self.snapshot_series { + // The whole series lives in one directory; a delta + // elsewhere could never be discovered at restore. + if config.destination_url != series.destination_url { + return Err(VmError::Snapshot(MigratableError::Snapshot(anyhow!( + "a diff snapshot must target its series directory {}", + series.destination_url + )))); + } + if !same_memory_layout(&series.layout, &layout) { + // Offsets in the base file no longer match; a + // delta would rebase corruptly. End the series. + self.snapshot_series = None; + let _ = vm.stop_dirty_log(); + return Err(VmError::Snapshot(MigratableError::Snapshot(anyhow!( + "memory layout changed since the snapshot series \ + started; take a full snapshot" + )))); + } + } else { + vm.start_dirty_log().map_err(VmError::Snapshot)?; + self.snapshot_series = Some(SnapshotSeries { + destination_url: config.destination_url.clone(), + layout, + next_seq: 1, + }); + } + } else if self.snapshot_series.take().is_some() { + vm.stop_dirty_log().map_err(VmError::Snapshot)?; + } + + let diff_seq = self.snapshot_series.as_ref().map(|s| s.next_seq); // Drain console_info so that FDs are not reused let _ = self.console_info.take(); - vm.snapshot() + let result = vm + .snapshot() .map_err(VmError::Snapshot) .and_then(|snapshot| { - vm.send(&snapshot, destination_url) + if effective_diff { + // Harvest after device capture so pages dirtied by + // snapshot side effects land in this delta. + let table = vm.dirty_log().map_err(VmError::Snapshot)?; + vm.set_diff_snapshot_ranges(table, diff_seq.unwrap()); + } + vm.send(&snapshot, &config.destination_url) .map_err(VmError::SnapshotSend) - }) + }); + match (&result, &mut self.snapshot_series) { + (Ok(()), Some(series)) => { + if effective_diff { + series.next_seq += 1; + } + } + (Err(_), series @ Some(_)) => { + // The harvested bitmap (or the baseline dump) is + // incomplete; the next snapshot must be a full one. + *series = None; + let _ = vm.stop_dirty_log(); + } + _ => {} + } + result } VmOwnership::Migration { .. } => Err(VmError::VmMigrating), VmOwnership::None => Err(VmError::VmNotRunning), @@ -2580,6 +2666,7 @@ impl RequestHandler for Vmm { } fn vm_restore(&mut self, restore_cfg: RestoreConfig) -> result::Result<(), VmError> { + self.snapshot_series = None; match &self.vm { VmOwnership::Owned(_) => Err(VmError::VmAlreadyCreated), VmOwnership::Migration { .. } => Err(VmError::VmMigrating), @@ -2841,6 +2928,8 @@ impl RequestHandler for Vmm { } fn vm_delete(&mut self) -> result::Result<(), VmError> { + // Any diff-snapshot series dies with the VM. + self.snapshot_series = None; if self.vm_config.is_none() { return Ok(()); } @@ -3334,6 +3423,8 @@ impl RequestHandler for Vmm { &mut self, send_data_migration: VmSendMigrationData, ) -> result::Result<(), MigratableError> { + // Migration owns the dirty log; any diff-snapshot series ends here. + self.snapshot_series = None; match self.vm { VmOwnership::Owned(ref vm) => { if vm.restoring() { diff --git a/vmm/src/memory_manager.rs b/vmm/src/memory_manager.rs index b1e70b3be3..965c951985 100644 --- a/vmm/src/memory_manager.rs +++ b/vmm/src/memory_manager.rs @@ -6,7 +6,7 @@ #[cfg(all(target_arch = "x86_64", feature = "guest_debug"))] use std::collections::BTreeMap; use std::collections::{HashMap, HashSet}; -use std::fs::{File, OpenOptions}; +use std::fs::{self, File, OpenOptions}; use std::io::{self, Seek, SeekFrom}; use std::mem::{MaybeUninit, zeroed}; use std::num::NonZeroUsize; @@ -229,6 +229,9 @@ pub struct MemoryManager { thp: bool, user_provided_zones: bool, snapshot_memory_ranges: MemoryRangeTable, + // When set, Transportable::send writes only these dirty ranges, placed at + // their offsets within the full snapshot layout (diff snapshot). + diff_snapshot_ranges: Option<(MemoryRangeTable, u32)>, memory_zones: MemoryZones, log_dirty: bool, // Enable dirty logging for created RAM regions arch_mem_regions: Vec, @@ -394,6 +397,18 @@ pub enum Error { #[error("Error reading from snapshot file")] SnapshotRead(#[source] io::Error), + /// Error scanning the snapshot directory for diff files + #[error("Error scanning the snapshot directory for diff files")] + RestoreDiffScan(#[source] io::Error), + + /// The diff-file sequence has a gap + #[error("Diff snapshot {0} is missing from the snapshot directory")] + RestoreDiffGap(u32), + + /// A diff-snapshot series cannot be restored on demand + #[error("A diff-snapshot series cannot be combined with 'memory_restore_mode=ondemand'")] + RestoreDiffWithOnDemand, + // Error copying snapshot into region #[error("Error copying snapshot into region")] SnapshotCopy(#[source] GuestMemoryError), @@ -850,84 +865,15 @@ impl MemoryManager { file_path: PathBuf, saved_regions: &MemoryRangeTable, ) -> Result<(), Error> { - // Open (read only) the snapshot file. - let mut memory_file = OpenOptions::new() - .read(true) - .open(file_path) - .map_err(Error::SnapshotOpen)?; - - let guest_memory = self.guest_memory.memory(); - let mut file_cursor: u64 = 0; - - for range in saved_regions.regions() { - let end = file_cursor + range.length; - - // First call doubles as a SEEK_HOLE-support probe. On error, - // take the dense path which seeks-and-streams sequentially. - match next_data_extent(memory_file.as_fd(), file_cursor, end) { - Ok(mut next) => { - while let Some((data_off, ext_len)) = next { - debug_assert!(data_off >= file_cursor); - let in_region = data_off - .checked_sub(file_cursor) - .expect("extent precedes file_cursor"); - memory_file - .seek(SeekFrom::Start(data_off)) - .map_err(Error::SnapshotRead)?; - let mut done: u64 = 0; - while done < ext_len { - let n = guest_memory - .read_volatile_from( - GuestAddress(range.gpa + in_region + done), - &mut memory_file, - (ext_len - done) as usize, - ) - .map_err(Error::SnapshotCopy)?; - if n == 0 { - return Err(Error::SnapshotRead(io::Error::new( - io::ErrorKind::UnexpectedEof, - "read_volatile_from returned 0 inside data extent", - ))); - } - done += n as u64; - } - next = next_data_extent(memory_file.as_fd(), data_off + ext_len, end) - .map_err(Error::SnapshotRead)?; - } - } - Err(_) => { - memory_file - .seek(SeekFrom::Start(file_cursor)) - .map_err(Error::SnapshotRead)?; - let mut offset: u64 = 0; - // Manual partial-read loop preserves the workaround for - // https://github.com/rust-vmm/vm-memory/issues/174 - loop { - let bytes_read = guest_memory - .read_volatile_from( - GuestAddress(range.gpa + offset), - &mut memory_file, - (range.length - offset) as usize, - ) - .map_err(Error::SnapshotCopy)?; - if bytes_read == 0 { - return Err(Error::SnapshotRead(io::Error::new( - io::ErrorKind::UnexpectedEof, - "Memory snapshot file is shorter than the saved range", - ))); - } - offset += bytes_read as u64; - if offset == range.length { - break; - } - } - } - } - - file_cursor = end; - } + replay_memory_file(&self.guest_memory.memory(), file_path, saved_regions, true) + } - Ok(()) + fn apply_diff_snapshot( + &mut self, + file_path: PathBuf, + saved_regions: &MemoryRangeTable, + ) -> Result<(), Error> { + apply_diff_file(&self.guest_memory.memory(), file_path, saved_regions) } // Restore guest memory by mapping the snapshot file copy-on-write over @@ -2038,6 +1984,7 @@ impl MemoryManager { reserve: config.reserve, user_provided_zones, snapshot_memory_ranges: MemoryRangeTable::default(), + diff_snapshot_ranges: None, memory_zones, guest_ram_mappings: Vec::new(), uffd_handler: None, @@ -2084,20 +2031,49 @@ impl MemoryManager { )?; if !mem_snapshot.memory_ranges.is_empty() { - match memory_restore_mode { - MemoryRestoreMode::OnDemand => mm.lock().unwrap().restore_by_uffd( - &memory_file_path, - &mem_snapshot.memory_ranges, - exit_evt, - )?, - MemoryRestoreMode::CopyOnWrite => mm - .lock() - .unwrap() - .mmap_cow_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?, - MemoryRestoreMode::Copy => mm - .lock() - .unwrap() - .fill_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?, + let source_dir = url_to_path(source_url).map_err(Error::Restore)?; + let deltas = discover_diff_files(&source_dir)?; + if deltas.is_empty() { + match memory_restore_mode { + MemoryRestoreMode::OnDemand => mm.lock().unwrap().restore_by_uffd( + &memory_file_path, + &mem_snapshot.memory_ranges, + exit_evt, + )?, + MemoryRestoreMode::CopyOnWrite => { + mm.lock().unwrap().mmap_cow_saved_regions( + memory_file_path, + &mem_snapshot.memory_ranges, + )?; + } + MemoryRestoreMode::Copy => mm + .lock() + .unwrap() + .fill_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?, + } + } else { + // Diff-snapshot series: fill from the baseline (honoring the + // restore mode), then replay each delta's dirty extents in + // sequence order. Replayed writes over a copy-on-write + // baseline fault in privately, so the sharing story is + // unchanged. + if memory_restore_mode == MemoryRestoreMode::OnDemand { + return Err(Error::RestoreDiffWithOnDemand); + } + let mut mm_locked = mm.lock().unwrap(); + if memory_restore_mode == MemoryRestoreMode::CopyOnWrite { + mm_locked.mmap_cow_saved_regions( + memory_file_path, + &mem_snapshot.memory_ranges, + )?; + } else { + mm_locked + .fill_saved_regions(memory_file_path, &mem_snapshot.memory_ranges)?; + } + for delta in deltas { + mm_locked.apply_diff_snapshot(delta, &mem_snapshot.memory_ranges)?; + } + drop(mm_locked); } } @@ -3458,12 +3434,29 @@ pub struct MemoryManagerSnapshotData { next_hotplug_slot: usize, } +impl MemoryManager { + /// Restricts the next Transportable::send to the given dirty ranges, + /// written at their offsets within the full snapshot layout as delta + /// number `seq` of the series. + pub fn set_diff_snapshot_ranges(&mut self, table: MemoryRangeTable, seq: u32) { + self.diff_snapshot_ranges = Some((table, seq)); + } + + /// Whether the next Transportable::send writes a delta of a series. + pub fn diff_snapshot_active(&self) -> bool { + self.diff_snapshot_ranges.is_some() + } +} + impl Snapshottable for MemoryManager { fn id(&self) -> String { MEMORY_MANAGER_SNAPSHOT_ID.to_string() } fn snapshot(&mut self) -> result::Result { + // A stale diff table must not leak into this snapshot; the caller + // re-injects one after device capture when a diff is requested. + self.diff_snapshot_ranges = None; let memory_ranges = self.memory_range_table(MemoryRangePolicy::SkipPersisted)?; // Store locally this list of ranges as it will be used through the @@ -3492,8 +3485,29 @@ impl Transportable for MemoryManager { return Ok(()); } - let mut memory_file_path = url_to_path(destination_url)?; - memory_file_path.push(String::from(SNAPSHOT_FILENAME)); + let dest_dir = url_to_path(destination_url)?; + // A delta only carries dirty extents; its sequence number is its + // replay position within the series directory. + let diff_seq = self.diff_snapshot_ranges.as_ref().map(|(_, seq)| *seq); + let final_path = match diff_seq { + Some(seq) => dest_dir.join(format!("{SNAPSHOT_FILENAME}.diff.{seq}")), + None => dest_dir.join(SNAPSHOT_FILENAME), + }; + // A delta is written under a temporary name and renamed into place, + // so an interrupted diff never invalidates a consistent directory. A + // full dump instead removes any leftover deltas: restore discovers + // diff files automatically, and stale ones would replay over the new + // baseline. + let memory_file_path = if let Some(seq) = diff_seq { + let tmp = dest_dir.join(format!("{SNAPSHOT_FILENAME}.diff.{seq}.tmp")); + let _ = fs::remove_file(&tmp); + tmp + } else { + remove_diff_files(&dest_dir) + .context("Error removing stale diff files") + .map_err(MigratableError::MigrateSend)?; + final_path.clone() + }; let mut memory_file = OpenOptions::new() .read(true) @@ -3519,10 +3533,44 @@ impl Transportable for MemoryManager { // write path which never writes past the growing EOF. let sparse_layout = memory_file.set_len(total_len).is_ok(); + // (file offset, range) pairs: a full snapshot lays ranges out densely; + // a diff places each dirty range at its offset within that same layout + // so its extents apply onto a base snapshot without translation. + let mut write_plan: Vec<(u64, MemoryRange)> = Vec::new(); + let mut layout_cursor: u64 = 0; + for full in self.snapshot_memory_ranges.regions() { + match &self.diff_snapshot_ranges { + None => write_plan.push(( + layout_cursor, + MemoryRange { + gpa: full.gpa, + length: full.length, + }, + )), + Some((dirty, _)) => { + for d in dirty.regions() { + let start = d.gpa.max(full.gpa); + let end = (d.gpa + d.length).min(full.gpa + full.length); + if start < end { + write_plan.push(( + layout_cursor + (start - full.gpa), + MemoryRange { + gpa: start, + length: end - start, + }, + )); + } + } + } + } + layout_cursor += full.length; + } + debug_assert_eq!(layout_cursor, total_len); + let guest_memory = self.guest_memory.memory(); - let mut file_cursor: u64 = 0; - for range in self.snapshot_memory_ranges.regions() { + for (file_cursor, range) in write_plan { + let range = ⦥ let mut wrote_sparse = false; if sparse_layout && let Some(region) = guest_memory.find_region(GuestAddress(range.gpa)) @@ -3573,11 +3621,18 @@ impl Transportable for MemoryManager { } } } + } - file_cursor += range.length; + if memory_file_path != final_path { + memory_file + .sync_all() + .context("Error syncing memory snapshot delta") + .map_err(MigratableError::MigrateSend)?; + fs::rename(&memory_file_path, &final_path) + .with_context(|| format!("Error renaming {memory_file_path:?} to {final_path:?}")) + .map_err(MigratableError::MigrateSend)?; } - debug_assert_eq!(file_cursor, total_len); Ok(()) } } @@ -3747,6 +3802,168 @@ fn do_mmap_cow_saved_regions( Error::SnapshotMmap(io::Error::other("snapshot range file offset overflow")) })?; } + + Ok(()) +} + +// Reads a snapshot memory file into guest RAM, walking the file's data +// extents and skipping holes. When the filesystem cannot report extents, +// `dense_fallback` selects between streaming the whole layout (full +// snapshot) and failing (diff snapshot, where a hole must keep the content +// the pages already have). +fn replay_memory_file( + guest_memory: &GuestMemoryMmap, + file_path: PathBuf, + saved_regions: &MemoryRangeTable, + dense_fallback: bool, +) -> Result<(), Error> { + // Open (read only) the snapshot file. + let mut memory_file = OpenOptions::new() + .read(true) + .open(file_path) + .map_err(Error::SnapshotOpen)?; + + let mut file_cursor: u64 = 0; + + for range in saved_regions.regions() { + let end = file_cursor + range.length; + + // First call doubles as a SEEK_HOLE-support probe. On error, + // take the dense path which seeks-and-streams sequentially. + match next_data_extent(memory_file.as_fd(), file_cursor, end) { + Ok(mut next) => { + while let Some((data_off, ext_len)) = next { + debug_assert!(data_off >= file_cursor); + let in_region = data_off + .checked_sub(file_cursor) + .expect("extent precedes file_cursor"); + memory_file + .seek(SeekFrom::Start(data_off)) + .map_err(Error::SnapshotRead)?; + let mut done: u64 = 0; + while done < ext_len { + let n = guest_memory + .read_volatile_from( + GuestAddress(range.gpa + in_region + done), + &mut memory_file, + (ext_len - done) as usize, + ) + .map_err(Error::SnapshotCopy)?; + if n == 0 { + return Err(Error::SnapshotRead(io::Error::new( + io::ErrorKind::UnexpectedEof, + "read_volatile_from returned 0 inside data extent", + ))); + } + done += n as u64; + } + next = next_data_extent(memory_file.as_fd(), data_off + ext_len, end) + .map_err(Error::SnapshotRead)?; + } + } + Err(e) if !dense_fallback => { + return Err(Error::SnapshotRead(io::Error::new( + e.kind(), + format!("diff restore needs SEEK_DATA/SEEK_HOLE support: {e}"), + ))); + } + Err(_) => { + memory_file + .seek(SeekFrom::Start(file_cursor)) + .map_err(Error::SnapshotRead)?; + let mut offset: u64 = 0; + // Manual partial-read loop preserves the workaround for + // https://github.com/rust-vmm/vm-memory/issues/174 + loop { + let bytes_read = guest_memory + .read_volatile_from( + GuestAddress(range.gpa + offset), + &mut memory_file, + (range.length - offset) as usize, + ) + .map_err(Error::SnapshotCopy)?; + if bytes_read == 0 { + return Err(Error::SnapshotRead(io::Error::new( + io::ErrorKind::UnexpectedEof, + "Memory snapshot file is shorter than the saved range", + ))); + } + offset += bytes_read as u64; + if offset == range.length { + break; + } + } + } + } + + file_cursor = end; + } + + Ok(()) +} + +// Applies a diff snapshot's dirty extents over already-restored RAM; the +// hole-punched pages the delta never dirtied keep their baseline content. +// The length check rejects a file that is not from the restored series' +// layout before any page is touched. +fn apply_diff_file( + guest_memory: &GuestMemoryMmap, + file_path: PathBuf, + saved_regions: &MemoryRangeTable, +) -> Result<(), Error> { + let total_len: u64 = saved_regions.regions().iter().map(|r| r.length).sum(); + let file_len = fs::metadata(&file_path).map_err(Error::SnapshotOpen)?.len(); + if file_len != total_len { + return Err(Error::SnapshotOpen(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "diff file {} is {file_len} bytes, expected {total_len}: \ + not from this snapshot series?", + file_path.display() + ), + ))); + } + replay_memory_file(guest_memory, file_path, saved_regions, false) +} + +// Finds the deltas of a diff-snapshot series in `dir`, in replay order. +// Sequence numbers must be contiguous from 1; a gap fails the restore +// before any page is touched. A name whose suffix is not a number (for +// example a `.tmp` leftover from an interrupted diff) is ignored. +fn discover_diff_files(dir: &Path) -> Result, Error> { + let prefix = format!("{SNAPSHOT_FILENAME}.diff."); + let mut deltas: Vec<(u32, PathBuf)> = Vec::new(); + for entry in fs::read_dir(dir).map_err(Error::RestoreDiffScan)? { + let entry = entry.map_err(Error::RestoreDiffScan)?; + if let Some(name) = entry.file_name().to_str() + && let Some(rest) = name.strip_prefix(&prefix) + && let Ok(seq) = rest.parse::() + { + deltas.push((seq, entry.path())); + } + } + deltas.sort_unstable_by_key(|(seq, _)| *seq); + for (i, (seq, _)) in deltas.iter().enumerate() { + if *seq != i as u32 + 1 { + return Err(Error::RestoreDiffGap(i as u32 + 1)); + } + } + Ok(deltas.into_iter().map(|(_, path)| path).collect()) +} + +// Removes every delta file in `dir`, `.tmp` leftovers included. A full +// dump must not leave stale deltas behind: restore discovers diff files +// automatically and would replay them over the new baseline. +fn remove_diff_files(dir: &Path) -> io::Result<()> { + let prefix = format!("{SNAPSHOT_FILENAME}.diff."); + for entry in fs::read_dir(dir)? { + let entry = entry?; + if let Some(name) = entry.file_name().to_str() + && name.strip_prefix(&prefix).is_some() + { + fs::remove_file(entry.path())?; + } + } Ok(()) } @@ -4047,4 +4264,98 @@ mod tests { assert_eq!(gm.read_obj::(GuestAddress(0)).unwrap(), 0xcd); } } + + fn two_page_table(page: u64) -> MemoryRangeTable { + let mut table = MemoryRangeTable::default(); + table.push(MemoryRange { + gpa: 0, + length: 2 * page, + }); + table + } + + fn temp_file_path(dir: &tempfile::TempDir, name: &str) -> PathBuf { + dir.path().join(name) + } + + #[test] + fn diff_replay_overwrites_extents_and_keeps_holes() { + let page = page_size(); + let gm = GuestMemoryMmap::from_ranges(&[(GuestAddress(0), (2 * page) as usize)]).unwrap(); + let table = two_page_table(page); + let dir = tempfile::tempdir().unwrap(); + + // Full baseline: both pages 0x11. + let base_path = temp_file_path(&dir, "memory-ranges"); + let mut base = File::create(&base_path).unwrap(); + base.write_all(&vec![0x11u8; (2 * page) as usize]).unwrap(); + replay_memory_file(&gm, base_path, &table, true).unwrap(); + assert_eq!(gm.read_obj::(GuestAddress(0)).unwrap(), 0x11); + assert_eq!(gm.read_obj::(GuestAddress(page)).unwrap(), 0x11); + + // Sparse delta: only page 1 dirtied, written as 0xab. + let diff_path = temp_file_path(&dir, "memory-ranges.diff"); + let mut diff = File::create(&diff_path).unwrap(); + diff.set_len(2 * page).unwrap(); + diff.seek(SeekFrom::Start(page)).unwrap(); + diff.write_all(&vec![0xabu8; page as usize]).unwrap(); + apply_diff_file(&gm, diff_path, &table).unwrap(); + + // Page 0 sits in the delta's hole and must keep the baseline content. + assert_eq!(gm.read_obj::(GuestAddress(0)).unwrap(), 0x11); + assert_eq!(gm.read_obj::(GuestAddress(page - 1)).unwrap(), 0x11); + assert_eq!(gm.read_obj::(GuestAddress(page)).unwrap(), 0xab); + assert_eq!(gm.read_obj::(GuestAddress(2 * page - 1)).unwrap(), 0xab); + } + + #[test] + fn diff_replay_rejects_wrong_length_file() { + let page = page_size(); + let gm = GuestMemoryMmap::from_ranges(&[(GuestAddress(0), (2 * page) as usize)]).unwrap(); + let table = two_page_table(page); + let dir = tempfile::tempdir().unwrap(); + + // A file one page short cannot be a delta of this layout. + let diff_path = temp_file_path(&dir, "memory-ranges.diff.1"); + File::create(&diff_path).unwrap().set_len(page).unwrap(); + apply_diff_file(&gm, diff_path, &table).unwrap_err(); + } + + #[test] + fn diff_discovery_orders_ignores_and_detects_gaps() { + let dir = tempfile::tempdir().unwrap(); + for name in [ + "memory-ranges", + "memory-ranges.diff.2", + "memory-ranges.diff.1", + "memory-ranges.diff.10", + "config.json", + ] { + File::create(dir.path().join(name)).unwrap(); + } + // A gap (3..=9 missing) fails before any page is touched. + assert!(matches!( + discover_diff_files(dir.path()), + Err(Error::RestoreDiffGap(3)) + )); + + fs::remove_file(dir.path().join("memory-ranges.diff.10")).unwrap(); + // A .tmp leftover from an interrupted diff is ignored. + File::create(dir.path().join("memory-ranges.diff.3.tmp")).unwrap(); + let deltas = discover_diff_files(dir.path()).unwrap(); + assert_eq!( + deltas, + vec![ + dir.path().join("memory-ranges.diff.1"), + dir.path().join("memory-ranges.diff.2"), + ] + ); + } + + #[test] + fn diff_discovery_empty_without_deltas() { + let dir = tempfile::tempdir().unwrap(); + File::create(dir.path().join("memory-ranges")).unwrap(); + assert!(discover_diff_files(dir.path()).unwrap().is_empty()); + } } diff --git a/vmm/src/vm.rs b/vmm/src/vm.rs index f696635cef..b5c3072867 100644 --- a/vmm/src/vm.rs +++ b/vmm/src/vm.rs @@ -14,7 +14,7 @@ use std::collections::{BTreeMap, BTreeSet, HashMap}; #[cfg(feature = "fw_cfg")] use std::ffi; -use std::fs::{File, OpenOptions}; +use std::fs::{self, File, OpenOptions}; use std::io::{self, Seek, SeekFrom, Write}; use std::num::Wrapping; use std::ops::Deref; @@ -3063,6 +3063,14 @@ impl Vm { self.memory_manager.lock().unwrap().memory_range_table(mode) } + /// Restricts the next snapshot send to the given dirty ranges (diff snapshot). + pub fn set_diff_snapshot_ranges(&mut self, table: MemoryRangeTable, seq: u32) { + self.memory_manager + .lock() + .unwrap() + .set_diff_snapshot_ranges(table, seq); + } + pub fn guest_memory(&self) -> GuestMemoryAtomic { self.memory_manager.lock().unwrap().guest_memory() } @@ -3422,55 +3430,68 @@ impl Snapshottable for Vm { } } +// Writes one snapshot file. A delta of a diff-snapshot series replaces the +// file already in the directory, through a temporary name and a rename so an +// interrupted write never invalidates the consistent copy; anything else +// must land in a fresh directory and fails if the file exists. +fn store_snapshot_file( + destination_url: &str, + name: &str, + bytes: &[u8], + replace: bool, +) -> result::Result<(), MigratableError> { + let mut final_path = url_to_path(destination_url)?; + final_path.push(name); + let write_path = if replace { + let tmp = final_path.with_extension("tmp"); + let _ = fs::remove_file(&tmp); + tmp + } else { + final_path.clone() + }; + let mut file = OpenOptions::new() + .read(true) + .write(true) + .create_new(true) + .open(&write_path) + .with_context(|| format!("Error creating snapshot file {write_path:?}")) + .map_err(MigratableError::MigrateSend)?; + file.write_all(bytes) + .with_context(|| format!("Error writing snapshot file {write_path:?}")) + .map_err(MigratableError::MigrateSend)?; + if replace { + file.sync_all() + .with_context(|| format!("Error syncing snapshot file {write_path:?}")) + .map_err(MigratableError::MigrateSend)?; + fs::rename(&write_path, &final_path) + .with_context(|| format!("Error renaming {write_path:?} to {final_path:?}")) + .map_err(MigratableError::MigrateSend)?; + } + Ok(()) +} + impl Transportable for Vm { fn send( &self, snapshot: &Snapshot, destination_url: &str, ) -> result::Result<(), MigratableError> { - let mut snapshot_config_path = url_to_path(destination_url)?; - snapshot_config_path.push(SNAPSHOT_CONFIG_FILE); - - // Create the snapshot config file - let mut snapshot_config_file = OpenOptions::new() - .read(true) - .write(true) - .create_new(true) - .open(snapshot_config_path) - .context("Error creating VM config snapshot file") - .map_err(MigratableError::MigrateSend)?; + let replace = self.memory_manager.lock().unwrap().diff_snapshot_active(); - // Serialize and write the snapshot config let vm_config = serde_json::to_string(self.config.lock().unwrap().deref()) .context("Error serializing VM config snapshot") .map_err(MigratableError::MigrateSend)?; + store_snapshot_file( + destination_url, + SNAPSHOT_CONFIG_FILE, + vm_config.as_bytes(), + replace, + )?; - snapshot_config_file - .write_all(vm_config.as_bytes()) - .context("Error writing VM config snapshot") - .map_err(MigratableError::MigrateSend)?; - - let mut snapshot_state_path = url_to_path(destination_url)?; - snapshot_state_path.push(SNAPSHOT_STATE_FILE); - - // Create the snapshot state file - let mut snapshot_state_file = OpenOptions::new() - .read(true) - .write(true) - .create_new(true) - .open(snapshot_state_path) - .context("Error creating VM state snapshot file") - .map_err(MigratableError::MigrateSend)?; - - // Serialize and write the snapshot state let vm_state = serde_json::to_vec(snapshot) .context("Error serializing VM state snapshot") .map_err(MigratableError::MigrateSend)?; - - snapshot_state_file - .write_all(&vm_state) - .context("Error writing VM state snapshot") - .map_err(MigratableError::MigrateSend)?; + store_snapshot_file(destination_url, SNAPSHOT_STATE_FILE, &vm_state, replace)?; // Tell the memory manager to also send/write its own snapshot. if let Some(memory_manager_snapshot) = snapshot.snapshots.get(MEMORY_MANAGER_SNAPSHOT_ID) { From 62259fe081269ee56743228ab3bc62a79c836c58 Mon Sep 17 00:00:00 2001 From: tonic Date: Sat, 19 Sep 2026 22:34:53 +0800 Subject: [PATCH 3/4] arch: aarch64: give direct boot ACPI via a synthesized EFI handoff MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A direct-kernel-booted aarch64 guest only ever saw the FDT, so every ACPI table create_acpi_tables() had already written went unused. PCI hotplug is ACPI-only (GED -> PHPR.PSCN -> PCNT -> DVNT/B0EJ), which left vm.add-net and vm.remove-device returning success while the guest noticed neither: a hot-added NIC never appeared, and an ejected one stayed on the bus forever along with its tap. The arm64 kernel does not need real firmware for this. It takes the RSDP from the EFI configuration table, which it finds through `linux,uefi-system-table` and the `linux,uefi-mmap-*` pointers in the device tree's /chosen node, and it does not care who produced those structures. So synthesize them: - arch/src/aarch64/efi.rs builds an EFI System Table, a configuration table and an EFI memory map. - fdt::create_stub_fdt() emits a device tree holding only /chosen. dt_is_stub() treats any other depth-1 node as proof of a real DTB and disables ACPI, so describing no hardware is what lets the guest take its hardware from ACPI — and is why acpi=force is not needed. Selected by `--platform acpi_boot=on|off`, defaulting on for aarch64. It applies to direct kernel boot only. A firmware boot keeps the full device tree whatever the option says: the firmware reads its memory and devices from that tree and publishes ACPI by itself. The stub tree also carries `linux,uefi-secure-boot = 2`: Ubuntu's kernel makes it a required property in efi_get_fdt_params(), and without it the EFI handoff is abandoned, the memory map is never installed, and — with no /memory node — memblock comes up empty and paging_init panics with "Failed to allocate page table page". The value is efi_secureboot_mode_disabled; 0 is efi_secureboot_mode_unset, which the same kernel reports as "Secure boot could not be determined (mode 0)" at warning level. Mainline ignores the property. Four configuration tables are published. ACPI 2.0 carries the RSDP. EFI RT Properties declares no runtime services, so the guest never calls into firmware that does not exist. SMBIOS3 is how aarch64 Linux locates DMI at all, which makes the tables setup_smbios() writes reachable for the first time on this path. LINUX_EFI_MEMRESERVE lets the GICv3 ITS persist its LPI property and pending tables through efi_mem_reserve_persistent(); without it the guest warns twice per boot out of irq-gic-v3-its.c. The EFI and SMBIOS ranges are both typed as runtime-services data. is_usable_memory() hands boot-services ranges back to memblock as System RAM, and both of these outlive early boot: the kernel keeps appending to the memreserve table after boot services end, and arm64 reads DMI from arm_dmi_init(), a core_initcall that runs once the allocator is already up. Verified on a c4a-highmem-96-metal host, Ubuntu 24.04 arm64 guest: - /sys/firmware/{acpi,efi,dmi} present, ACPI0013 GED bound to a GIC SPI, GICv3/ITS, PSCI, PMU, the SPCR console and DMI all taken from ACPI - zero kernel warnings, and no unavailable ranges in the memory map - vm.add-net: GED interrupt fires, the NIC appears with no guest-side rescan - vm.remove-device: the guest ejects it and it leaves CH's device tree - clone: the new MAC appears by itself and DHCPs a distinct lease, the snapshot's NIC is gone, and no restore tap is left behind - acpi_boot=off still boots the same guest through the device tree Signed-off-by: CMGS --- arch/src/aarch64/efi.rs | 483 ++++++++++++++++++++++ arch/src/aarch64/fdt.rs | 132 ++++++ arch/src/aarch64/layout.rs | 6 + arch/src/aarch64/mod.rs | 38 ++ docs/hotplug.md | 2 +- vmm/src/api/openapi/cloud-hypervisor.yaml | 7 + vmm/src/config.rs | 16 + vmm/src/vm.rs | 31 +- vmm/src/vm_config.rs | 12 + 9 files changed, 717 insertions(+), 10 deletions(-) create mode 100644 arch/src/aarch64/efi.rs diff --git a/arch/src/aarch64/efi.rs b/arch/src/aarch64/efi.rs new file mode 100644 index 0000000000..d477a4d9d9 --- /dev/null +++ b/arch/src/aarch64/efi.rs @@ -0,0 +1,483 @@ +// Copyright 2026 The Cloud Hypervisor Authors +// +// SPDX-License-Identifier: Apache-2.0 + +// The aarch64 kernel reaches ACPI only through the EFI stub: it reads +// `linux,uefi-system-table` and the `linux,uefi-mmap-*` pointers out of the +// device tree's /chosen node and walks the EFI configuration table for the +// ACPI 2.0 RSDP. Nothing in that path requires real firmware to have produced +// the structures, so the VMM can synthesize them directly. Modelled on +// OpenVMM's aarch64 direct-boot loader. + +use std::result; + +use vm_memory::{Address, Bytes, GuestAddress, GuestMemoryBackend, GuestMemoryRegion}; + +use super::layout; +use crate::GuestMemoryMmap; + +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("Writing EFI handoff structures to guest memory")] + WriteEfiTables(#[source] vm_memory::GuestMemoryError), + #[error("EFI metadata at {0:#x} overflows into the ACPI region at {1:#x}")] + MetadataOverflow(u64, u64), + #[error("EFI memory map of {0} bytes overflows its {1}-byte page into the system table")] + MemoryMapOverflow(usize, u64), +} + +type Result = result::Result; + +// What the stub device tree has to publish for the EFI stub to pick the +// handoff up. +#[derive(Clone, Copy, Debug)] +pub struct EfiHandoff { + pub systab_addr: u64, + pub mmap_addr: u64, + pub mmap_size: u32, + pub mmap_desc_size: u32, + pub mmap_desc_ver: u32, +} + +const PAGE_SIZE: u64 = 0x1000; + +// "IBI SYST" +const EFI_SYSTEM_TABLE_SIGNATURE: u64 = 0x5453_5953_2049_4249; +const EFI_2_70_SYSTEM_TABLE_REVISION: u32 = (2 << 16) | 70; +const EFI_SYSTEM_TABLE_SIZE: u64 = 0x78; +const EFI_MEMORY_DESCRIPTOR_VERSION: u32 = 1; +const EFI_MEMORY_DESCRIPTOR_SIZE: u64 = 40; +const EFI_MEMORY_WB: u64 = 0x8; + +const EFI_BOOT_SERVICES_DATA: u32 = 4; +const EFI_RUNTIME_SERVICES_DATA: u32 = 6; +const EFI_CONVENTIONAL_MEMORY: u32 = 7; +const EFI_ACPI_RECLAIM_MEMORY: u32 = 9; + +const CONFIG_ENTRY_SIZE: u64 = 16 + 8; +const CONFIG_ENTRY_COUNT: u64 = 4; + +// GUIDs in EFI's mixed-endian encoding: the first three fields little-endian, +// the trailing eight bytes in order. +const ACPI_20_TABLE_GUID: [u8; 16] = [ + 0x71, 0xe8, 0x68, 0x88, 0xf1, 0xe4, 0xd3, 0x11, 0xbc, 0x22, 0x00, 0x80, 0xc7, 0x3c, 0x88, 0x81, +]; +const EFI_RT_PROPERTIES_TABLE_GUID: [u8; 16] = [ + 0x8a, 0x91, 0x66, 0xeb, 0xef, 0x7e, 0x2a, 0x40, 0x84, 0x2e, 0x93, 0x1d, 0x21, 0xc3, 0x8a, 0xe9, +]; +const SMBIOS3_TABLE_GUID: [u8; 16] = [ + 0x44, 0x15, 0xfd, 0xf2, 0x94, 0x97, 0x2c, 0x4a, 0x99, 0x2e, 0xe5, 0xbb, 0xcf, 0x20, 0xe3, 0x94, +]; +const LINUX_EFI_MEMRESERVE_TABLE_GUID: [u8; 16] = [ + 0xc6, 0xb0, 0x8e, 0x88, 0xde, 0x8e, 0xf5, 0x4f, 0xa8, 0xf0, 0x9a, 0xee, 0x5c, 0xb9, 0x77, 0xc2, +]; + +const EFI_RT_PROPERTIES_TABLE_VERSION: u16 = 1; +const EFI_RT_PROPERTIES_TABLE_SIZE: u64 = 8; + +const MEMRESERVE_HEADER_SIZE: u64 = 16; +const MEMRESERVE_ENTRY_SIZE: u64 = 16; + +fn align_up(value: u64, alignment: u64) -> u64 { + (value + alignment - 1) & !(alignment - 1) +} + +// UEFI 4.2 checksums the table header over header_size bytes with the crc32 +// field itself zeroed. Table-less so this stays dependency-free. +fn crc32(bytes: &[u8]) -> u32 { + let mut crc = !0u32; + for byte in bytes { + crc ^= *byte as u32; + for _ in 0..8 { + let mask = (crc & 1).wrapping_neg(); + crc = (crc >> 1) ^ (0xedb8_8320 & mask); + } + } + !crc +} + +fn memory_descriptor(typ: u32, physical_start: u64, pages: u64) -> [u8; 40] { + let mut out = [0u8; 40]; + out[0..4].copy_from_slice(&typ.to_le_bytes()); + out[8..16].copy_from_slice(&physical_start.to_le_bytes()); + out[24..32].copy_from_slice(&pages.to_le_bytes()); + out[32..40].copy_from_slice(&EFI_MEMORY_WB.to_le_bytes()); + out +} + +// The firmware-shaped view of guest memory the EFI stub consumes. `rsdp_addr` +// is where create_acpi_tables() already put the RSDP. +pub fn write_efi_tables( + guest_mem: &GuestMemoryMmap, + rsdp_addr: GuestAddress, +) -> Result { + let efi_base = layout::EFI_START.raw_value(); + let acpi_base = layout::ACPI_START.raw_value(); + + // Page 0 holds the memory map, which is sized last; the rest of the + // metadata starts on page 1. + let mut cursor = efi_base + PAGE_SIZE; + + let systab_addr = cursor; + cursor += EFI_SYSTEM_TABLE_SIZE; + + let config_table_addr = cursor; + cursor += CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE; + + // NUL-terminated UTF-16LE, as the system table's firmware_vendor expects. + let fw_vendor: Vec = "Cloud Hypervisor\0" + .encode_utf16() + .flat_map(|c| c.to_le_bytes()) + .collect(); + let fw_vendor_addr = cursor; + cursor += fw_vendor.len() as u64; + cursor = align_up(cursor, 8); + + let rt_props_addr = cursor; + cursor += EFI_RT_PROPERTIES_TABLE_SIZE; + + // struct linux_efi_memreserve: { i32 size; i32 count; u64 next; entry[] }, + // each entry a { u64 base; u64 size; } pair. The kernel links the pages it + // allocates later through `next`, so the root is never handed back as RAM. + let memreserve_addr = align_up(cursor, PAGE_SIZE); + let memreserve_capacity = (PAGE_SIZE - MEMRESERVE_HEADER_SIZE) / MEMRESERVE_ENTRY_SIZE; + cursor = memreserve_addr + PAGE_SIZE; + + if cursor > acpi_base { + return Err(Error::MetadataOverflow(cursor, acpi_base)); + } + + // Runtime services are never available in a synthesized handoff — saying so + // explicitly is what keeps the kernel from calling into them. + let mut rt_props = [0u8; EFI_RT_PROPERTIES_TABLE_SIZE as usize]; + rt_props[0..2].copy_from_slice(&EFI_RT_PROPERTIES_TABLE_VERSION.to_le_bytes()); + rt_props[2..4].copy_from_slice(&(EFI_RT_PROPERTIES_TABLE_SIZE as u16).to_le_bytes()); + guest_mem + .write_slice(&rt_props, GuestAddress(rt_props_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut config_entries = [0u8; (CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE) as usize]; + config_entries[0..16].copy_from_slice(&ACPI_20_TABLE_GUID); + config_entries[16..24].copy_from_slice(&rsdp_addr.raw_value().to_le_bytes()); + config_entries[24..40].copy_from_slice(&EFI_RT_PROPERTIES_TABLE_GUID); + config_entries[40..48].copy_from_slice(&rt_props_addr.to_le_bytes()); + // aarch64 Linux finds DMI only through this entry — it has no equivalent of + // the x86 anchor scan — so the tables setup_smbios() wrote are invisible + // without it. + config_entries[48..64].copy_from_slice(&SMBIOS3_TABLE_GUID); + config_entries[64..72].copy_from_slice(&layout::SMBIOS_START.raw_value().to_le_bytes()); + // Without this the GICv3 ITS cannot persist its LPI property and pending + // tables through efi_mem_reserve_persistent() and warns on every boot. + config_entries[72..88].copy_from_slice(&LINUX_EFI_MEMRESERVE_TABLE_GUID); + config_entries[88..96].copy_from_slice(&memreserve_addr.to_le_bytes()); + guest_mem + .write_slice(&config_entries, GuestAddress(config_table_addr)) + .map_err(Error::WriteEfiTables)?; + + guest_mem + .write_slice(&fw_vendor, GuestAddress(fw_vendor_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut memreserve = [0u8; MEMRESERVE_HEADER_SIZE as usize]; + memreserve[0..4].copy_from_slice(&(memreserve_capacity as i32).to_le_bytes()); + guest_mem + .write_slice(&memreserve, GuestAddress(memreserve_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut systab = [0u8; EFI_SYSTEM_TABLE_SIZE as usize]; + systab[0x00..0x08].copy_from_slice(&EFI_SYSTEM_TABLE_SIGNATURE.to_le_bytes()); + systab[0x08..0x0c].copy_from_slice(&EFI_2_70_SYSTEM_TABLE_REVISION.to_le_bytes()); + systab[0x0c..0x10].copy_from_slice(&(EFI_SYSTEM_TABLE_SIZE as u32).to_le_bytes()); + systab[0x18..0x20].copy_from_slice(&fw_vendor_addr.to_le_bytes()); + systab[0x20..0x24].copy_from_slice(&1u32.to_le_bytes()); + systab[0x68..0x70].copy_from_slice(&CONFIG_ENTRY_COUNT.to_le_bytes()); + systab[0x70..0x78].copy_from_slice(&config_table_addr.to_le_bytes()); + let checksum = crc32(&systab); + systab[0x10..0x14].copy_from_slice(&checksum.to_le_bytes()); + guest_mem + .write_slice(&systab, GuestAddress(systab_addr)) + .map_err(Error::WriteEfiTables)?; + + // The stub DT carries no /memory node, so this map is the only thing + // memblock is built from — every range the guest may touch has to appear. + let reserved_start = layout::FDT_START.raw_value(); + let smbios_base = layout::SMBIOS_START.raw_value(); + let reserved_end = smbios_base + layout::SMBIOS_MAX_SIZE; + let mut mmap = Vec::new(); + // The stub DT: the kernel unflattens it early, so it may be reclaimed after. + mmap.extend_from_slice(&memory_descriptor( + EFI_BOOT_SERVICES_DATA, + reserved_start, + (efi_base - reserved_start) / PAGE_SIZE, + )); + // The EFI structures, covering the whole carve-out so no hole is left to + // show up as an unavailable range. Runtime-services data rather than boot: + // the kernel keeps writing to the memreserve table after boot services end. + mmap.extend_from_slice(&memory_descriptor( + EFI_RUNTIME_SERVICES_DATA, + efi_base, + layout::EFI_MAX_SIZE / PAGE_SIZE, + )); + mmap.extend_from_slice(&memory_descriptor( + EFI_ACPI_RECLAIM_MEMORY, + acpi_base, + layout::ACPI_MAX_SIZE / PAGE_SIZE, + )); + // Runtime-services data rather than boot: is_usable_memory() gives + // boot-services ranges back to memblock as System RAM, and arm64 scans DMI + // from arm_dmi_init(), a core_initcall, long after the allocator is live. + mmap.extend_from_slice(&memory_descriptor( + EFI_RUNTIME_SERVICES_DATA, + smbios_base, + layout::SMBIOS_MAX_SIZE / PAGE_SIZE, + )); + for region in guest_mem.iter() { + let start = region.start_addr().raw_value(); + let end = start + region.len(); + for (from, to) in subtract_reserved(start, end, reserved_start, reserved_end) { + mmap.extend_from_slice(&memory_descriptor( + EFI_CONVENTIONAL_MEMORY, + from, + (to - from) / PAGE_SIZE, + )); + } + } + + // The map owns page 0 alone; the system table starts at page 1, so an + // oversized map would silently overwrite it. + if mmap.len() as u64 > PAGE_SIZE { + return Err(Error::MemoryMapOverflow(mmap.len(), PAGE_SIZE)); + } + guest_mem + .write_slice(&mmap, layout::EFI_START) + .map_err(Error::WriteEfiTables)?; + + Ok(EfiHandoff { + systab_addr, + mmap_addr: efi_base, + mmap_size: mmap.len() as u32, + mmap_desc_size: EFI_MEMORY_DESCRIPTOR_SIZE as u32, + mmap_desc_ver: EFI_MEMORY_DESCRIPTOR_VERSION, + }) +} + +// Overlapping descriptors make the kernel reject the whole map, so the +// firmware-owned span is punched out of every RAM region it intersects. +fn subtract_reserved( + start: u64, + end: u64, + reserved_start: u64, + reserved_end: u64, +) -> Vec<(u64, u64)> { + let mut out = Vec::new(); + if end <= reserved_start || start >= reserved_end { + out.push((start, end)); + return out; + } + if start < reserved_start { + out.push((start, reserved_start)); + } + if end > reserved_end { + out.push((reserved_end, end)); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + // RAM as arch_memory_regions() lays it out, big enough to hold the whole + // firmware carve-out plus a little usable memory above it. + fn test_mem() -> GuestMemoryMmap { + GuestMemoryMmap::from_ranges(&[( + layout::RAM_START, + (layout::KERNEL_START.raw_value() - layout::RAM_START.raw_value() + 0x10_0000) as usize, + )]) + .unwrap() + } + + fn read(mem: &GuestMemoryMmap, addr: u64, len: usize) -> Vec { + let mut out = vec![0u8; len]; + mem.read_slice(&mut out, GuestAddress(addr)).unwrap(); + out + } + + fn le64(bytes: &[u8]) -> u64 { + u64::from_le_bytes(bytes.try_into().unwrap()) + } + + fn le32(bytes: &[u8]) -> u32 { + u32::from_le_bytes(bytes.try_into().unwrap()) + } + + #[test] + fn crc32_matches_known_vector() { + assert_eq!(crc32(b"123456789"), 0xcbf4_3926); + } + + #[test] + fn reserved_span_is_punched_out_of_ram() { + // A region containing the whole reserved span splits in two. + assert_eq!( + subtract_reserved(0x4000_0000, 0x8000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x4040_0000, 0x8000_0000)] + ); + // A region entirely above it is untouched. + assert_eq!( + subtract_reserved(0x1_0000_0000, 0x2_0000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x1_0000_0000, 0x2_0000_0000)] + ); + // A region straddling the tail keeps only what is above. + assert_eq!( + subtract_reserved(0x4030_0000, 0x5000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x4040_0000, 0x5000_0000)] + ); + } + + #[test] + fn system_table_header_is_well_formed() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let systab = read(&mem, handoff.systab_addr, EFI_SYSTEM_TABLE_SIZE as usize); + + assert_eq!(le64(&systab[0x00..0x08]), EFI_SYSTEM_TABLE_SIGNATURE); + assert_eq!(le32(&systab[0x08..0x0c]), EFI_2_70_SYSTEM_TABLE_REVISION); + assert_eq!(le32(&systab[0x0c..0x10]), EFI_SYSTEM_TABLE_SIZE as u32); + assert_eq!(le64(&systab[0x68..0x70]), CONFIG_ENTRY_COUNT); + + // UEFI 4.2: the checksum covers header_size bytes with crc32 zeroed. + let stored = le32(&systab[0x10..0x14]); + let mut zeroed = systab.clone(); + zeroed[0x10..0x14].fill(0); + assert_eq!(stored, crc32(&zeroed)); + + // firmware_vendor must point at NUL-terminated UTF-16LE. + let vendor_addr = le64(&systab[0x18..0x20]); + let vendor = read(&mem, vendor_addr, 34); + let utf16: Vec = vendor + .as_chunks::<2>() + .0 + .iter() + .map(|c| u16::from_le_bytes(*c)) + .take_while(|c| *c != 0) + .collect(); + assert_eq!(String::from_utf16(&utf16).unwrap(), "Cloud Hypervisor"); + } + + #[test] + fn configuration_table_points_at_every_structure() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let systab = read(&mem, handoff.systab_addr, EFI_SYSTEM_TABLE_SIZE as usize); + let table_addr = le64(&systab[0x70..0x78]); + let entries = read( + &mem, + table_addr, + (CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE) as usize, + ); + + let found: Vec<([u8; 16], u64)> = entries + .as_chunks::<{ CONFIG_ENTRY_SIZE as usize }>() + .0 + .iter() + .map(|e| (e[0..16].try_into().unwrap(), le64(&e[16..24]))) + .collect(); + + let lookup = |guid: [u8; 16]| { + found + .iter() + .find(|(g, _)| *g == guid) + .unwrap_or_else(|| panic!("missing configuration table entry")) + .1 + }; + + assert_eq!(lookup(ACPI_20_TABLE_GUID), layout::RSDP_POINTER.raw_value()); + assert_eq!(lookup(SMBIOS3_TABLE_GUID), layout::SMBIOS_START.raw_value()); + + // RT Properties must declare that nothing is supported, otherwise the + // guest will call into runtime services this handoff does not have. + let rt_props = read( + &mem, + lookup(EFI_RT_PROPERTIES_TABLE_GUID), + EFI_RT_PROPERTIES_TABLE_SIZE as usize, + ); + assert_eq!(u16::from_le_bytes([rt_props[0], rt_props[1]]), 1); + assert_eq!(le32(&rt_props[4..8]), 0); + + // The memreserve table starts empty with room for entries. + let memreserve = read( + &mem, + lookup(LINUX_EFI_MEMRESERVE_TABLE_GUID), + MEMRESERVE_HEADER_SIZE as usize, + ); + assert!(le32(&memreserve[0..4]) > 0, "no capacity for entries"); + assert_eq!(le32(&memreserve[4..8]), 0, "count must start at zero"); + assert_eq!(le64(&memreserve[8..16]), 0, "next must be null"); + } + + #[test] + fn memory_map_covers_ram_without_overlapping() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + assert_eq!(handoff.mmap_desc_size, EFI_MEMORY_DESCRIPTOR_SIZE as u32); + assert_eq!(handoff.mmap_desc_ver, EFI_MEMORY_DESCRIPTOR_VERSION); + + let raw = read(&mem, handoff.mmap_addr, handoff.mmap_size as usize); + let mut ranges: Vec<(u64, u64, u32)> = raw + .as_chunks::<{ EFI_MEMORY_DESCRIPTOR_SIZE as usize }>() + .0 + .iter() + .map(|d| { + let start = le64(&d[8..16]); + (start, start + le64(&d[24..32]) * PAGE_SIZE, le32(&d[0..4])) + }) + .collect(); + assert!(!ranges.is_empty()); + ranges.sort_by_key(|r| r.0); + + // A hole or an overlap both make the kernel reject the map, and the + // stub tree has no /memory node to fall back on. + for pair in ranges.windows(2) { + assert_eq!(pair[0].1, pair[1].0, "gap or overlap in the memory map"); + } + assert_eq!(ranges[0].0, layout::FDT_START.raw_value()); + assert_eq!(ranges.last().unwrap().1, mem.last_addr().raw_value() + 1); + + // Everything firmware owns has to be typed as such; handing the ACPI + // tables or the EFI structures back as conventional memory would let + // the guest allocate over them. + let firmware_end = layout::SMBIOS_START.raw_value() + layout::SMBIOS_MAX_SIZE; + for (start, end, typ) in &ranges { + if *start < firmware_end { + assert_ne!(*typ, EFI_CONVENTIONAL_MEMORY, "{start:#x}..{end:#x}"); + } else { + assert_eq!(*typ, EFI_CONVENTIONAL_MEMORY, "{start:#x}..{end:#x}"); + } + } + // The EFI structures outlive boot services: the kernel keeps appending + // to the memreserve table after they end. + let efi_region = ranges + .iter() + .find(|(s, _, _)| *s == layout::EFI_START.raw_value()) + .expect("EFI region missing from the map"); + assert_eq!(efi_region.2, EFI_RUNTIME_SERVICES_DATA); + // So does SMBIOS: is_usable_memory() would give a boot-services range + // back to memblock as System RAM, and arm64 reads DMI from + // arm_dmi_init(), a core_initcall that runs once the allocator is up. + let smbios_region = ranges + .iter() + .find(|(s, _, _)| *s == layout::SMBIOS_START.raw_value()) + .expect("SMBIOS region missing from the map"); + assert_eq!(smbios_region.2, EFI_RUNTIME_SERVICES_DATA); + } + + #[test] + fn metadata_stays_below_the_acpi_region() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let acpi_base = layout::ACPI_START.raw_value(); + assert!(handoff.systab_addr < acpi_base); + assert!(handoff.mmap_addr + u64::from(handoff.mmap_size) < acpi_base); + } +} diff --git a/arch/src/aarch64/fdt.rs b/arch/src/aarch64/fdt.rs index e82e782935..03afb93771 100644 --- a/arch/src/aarch64/fdt.rs +++ b/arch/src/aarch64/fdt.rs @@ -27,6 +27,7 @@ use vm_memory::{Address, Bytes, GuestMemoryBackend, GuestMemoryError, GuestMemor use super::super::{DeviceType, GuestMemoryMmap, InitramfsConfig}; use super::cache::{CacheTopologyInfo, read_cache_topology}; +use super::efi::EfiHandoff; use super::layout::{ GIC_V2M_COMPATIBLE, GICV2M_SPI_BASE, GICV2M_SPI_NUM, IRQ_BASE, MEM_32BIT_DEVICES_SIZE, MEM_32BIT_DEVICES_START, MEM_PCI_IO_SIZE, MEM_PCI_IO_START, PCI_HIGH_BASE, @@ -71,6 +72,9 @@ const IRQ_TYPE_LEVEL_HI: u32 = 4; // System Power Down const KEY_POWER: u32 = 116; +// enum efi_secureboot_mode: 0 is "unset", which Ubuntu reports as a warning +const EFI_SECUREBOOT_MODE_DISABLED: u32 = 2; + /// Trait for devices to be added to the Flattened Device Tree. pub trait DeviceInfoForFdt { /// Returns the address where this device will be loaded. @@ -146,6 +150,48 @@ pub fn create_fdt( Ok(fdt_final) } +/// The device tree for ACPI boot: a `/chosen` node and nothing else. +/// +/// `dt_is_stub()` in the arm64 kernel treats any depth-1 node other than +/// `chosen` as proof of a "real" DTB and turns ACPI off, so describing no +/// hardware here is what lets the guest take its hardware from ACPI instead — +/// and is why `acpi=force` is not needed on the command line. +pub fn create_stub_fdt( + cmdline: &str, + initrd: &Option, + efi: &EfiHandoff, +) -> FdtWriterResult> { + let mut fdt = FdtWriter::new().unwrap(); + + let root_node = fdt.begin_node("")?; + fdt.property_u32("#address-cells", ADDRESS_CELLS)?; + fdt.property_u32("#size-cells", SIZE_CELLS)?; + + let chosen_node = fdt.begin_node("chosen")?; + fdt.property_string("bootargs", cmdline)?; + if let Some(initrd_config) = initrd { + let initrd_start = initrd_config.address.raw_value(); + fdt.property_u64("linux,initrd-start", initrd_start)?; + fdt.property_u64("linux,initrd-end", initrd_start + initrd_config.size as u64)?; + } + fdt.property_u64("linux,uefi-system-table", efi.systab_addr)?; + fdt.property_u64("linux,uefi-mmap-start", efi.mmap_addr)?; + fdt.property_u32("linux,uefi-mmap-size", efi.mmap_size)?; + fdt.property_u32("linux,uefi-mmap-desc-size", efi.mmap_desc_size)?; + fdt.property_u32("linux,uefi-mmap-desc-ver", efi.mmap_desc_ver)?; + // Ubuntu's kernel makes `linux,uefi-secure-boot` a required property in + // efi_get_fdt_params(); without it the EFI handoff is abandoned, the memory + // map is never installed, and — since this tree has no /memory node — + // memblock comes up empty and paging_init panics with "Failed to allocate + // page table page". Mainline ignores the property. + fdt.property_u32("linux,uefi-secure-boot", EFI_SECUREBOOT_MODE_DISABLED)?; + fdt.end_node(chosen_node)?; + + fdt.end_node(root_node)?; + + fdt.finish() +} + pub fn write_fdt_to_memory(fdt_final: &[u8], guest_mem: &GuestMemoryMmap) -> Result<()> { // Write FDT to memory. guest_mem @@ -1064,9 +1110,95 @@ fn print_node(node: FdtNode<'_, '_>, n_spaces: usize) { mod tests { use std::collections::BTreeMap; + use vm_memory::GuestAddress; + use super::*; use crate::NumaNode; + fn test_handoff() -> EfiHandoff { + EfiHandoff { + systab_addr: 0x4010_1000, + mmap_addr: 0x4010_0000, + mmap_size: 160, + mmap_desc_size: 40, + mmap_desc_ver: 1, + } + } + + // The whole ACPI mode rests on this: dt_is_stub() in the arm64 kernel + // reads any depth-1 node other than `chosen` as proof of a real DTB + // and turns ACPI back off, which would silently put us back on the device + // tree with no hardware described in it. + #[test] + fn stub_fdt_has_only_a_chosen_node() { + let blob = create_stub_fdt("console=ttyAMA0", &None, &test_handoff()).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let root = fdt.find_node("/").unwrap(); + let children: Vec<&str> = root.children().map(|c| c.name).collect(); + assert_eq!(children, vec!["chosen"]); + } + + #[test] + fn stub_fdt_publishes_the_efi_handoff() { + let handoff = test_handoff(); + let blob = create_stub_fdt("console=ttyAMA0 rw", &None, &handoff).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let chosen = fdt.find_node("/chosen").unwrap(); + + let u64_prop = |name: &str| -> u64 { + let p = chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")); + BigEndian::read_u64(p.value) + }; + let u32_prop = |name: &str| -> u32 { + let p = chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")); + BigEndian::read_u32(p.value) + }; + + assert_eq!(u64_prop("linux,uefi-system-table"), handoff.systab_addr); + assert_eq!(u64_prop("linux,uefi-mmap-start"), handoff.mmap_addr); + assert_eq!(u32_prop("linux,uefi-mmap-size"), handoff.mmap_size); + assert_eq!( + u32_prop("linux,uefi-mmap-desc-size"), + handoff.mmap_desc_size + ); + assert_eq!(u32_prop("linux,uefi-mmap-desc-ver"), handoff.mmap_desc_ver); + + // Ubuntu's efi_get_fdt_params() treats this as required; dropping it + // aborts the handoff and panics the guest in paging_init. + assert_eq!( + u32_prop("linux,uefi-secure-boot"), + EFI_SECUREBOOT_MODE_DISABLED + ); + } + + #[test] + fn stub_fdt_carries_the_initramfs_when_there_is_one() { + let initrd = Some(InitramfsConfig { + address: GuestAddress(0x8000_0000), + size: 0x10_0000, + }); + let blob = create_stub_fdt("", &initrd, &test_handoff()).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let chosen = fdt.find_node("/chosen").unwrap(); + let prop = |name: &str| { + BigEndian::read_u64( + chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")) + .value, + ) + }; + assert_eq!(prop("linux,initrd-start"), 0x8000_0000); + assert_eq!(prop("linux,initrd-end"), 0x8010_0000); + } + // Helper function to create a simple NumaNode for testing fn create_test_numa_node(cpus: Vec, device_id: Option) -> NumaNode { NumaNode { diff --git a/arch/src/aarch64/layout.rs b/arch/src/aarch64/layout.rs index a6858c761a..389a097737 100644 --- a/arch/src/aarch64/layout.rs +++ b/arch/src/aarch64/layout.rs @@ -116,6 +116,12 @@ pub const FDT_START: GuestAddress = RAM_START; /// documentation](https://www.kernel.org/doc/Documentation/arm64/booting.txt). pub const FDT_MAX_SIZE: u64 = 0x20_0000; +/// EFI handoff structures (system table, configuration table, memory map) for +/// ACPI boot. Carved from the upper half of the FDT reservation: the stub +/// device tree that mode uses is a few KiB against a 2 MiB window. +pub const EFI_START: GuestAddress = GuestAddress(RAM_START.0 + FDT_MAX_SIZE / 2); +pub const EFI_MAX_SIZE: u64 = FDT_MAX_SIZE / 2; + /// Put ACPI table above dtb pub const ACPI_START: GuestAddress = GuestAddress(RAM_START.0 + FDT_MAX_SIZE); const ACPI_SMBIOS_MAX_SIZE: u64 = 0x20_0000; diff --git a/arch/src/aarch64/mod.rs b/arch/src/aarch64/mod.rs index e92fa0663b..04088a76ee 100644 --- a/arch/src/aarch64/mod.rs +++ b/arch/src/aarch64/mod.rs @@ -4,6 +4,9 @@ /// Module for cache info. pub mod cache; +/// Module for the synthesized EFI handoff that gives an ACPI-booted guest its +/// RSDP without firmware. +pub mod efi; /// Module for the flattened device tree. pub mod fdt; /// Layout for this aarch64 system. @@ -56,6 +59,14 @@ pub enum Error { /// Error initializing PMU for vcpu #[error("Error initializing PMU for vcpu")] VcpuInitPmu, + + /// Failed to write the EFI handoff structures. + #[error("Failed to write the EFI handoff structures")] + SetupEfi(#[source] efi::Error), + + /// No RSDP address to hand the guest in ACPI boot mode. + #[error("No RSDP address to hand the guest in ACPI boot mode")] + MissingRsdp, } #[derive(Debug, Copy, Clone)] @@ -162,6 +173,33 @@ pub fn configure_system( Ok(()) } +/// The ACPI counterpart of [`configure_system`]: synthesize the EFI handoff the +/// arm64 EFI stub expects, then hand the guest a device tree that describes +/// nothing but where to find it. Everything else — CPUs, GIC, timer, PCI — comes +/// from the ACPI tables `create_acpi_tables()` has already written. +pub fn configure_system_acpi( + guest_mem: &GuestMemoryMmap, + cmdline: &str, + initrd: &Option, + rsdp_addr: GuestAddress, + smbios: Option<&smbios::SmbiosConfig>, +) -> super::Result<()> { + smbios::setup_smbios(guest_mem, smbios).map_err(Error::SmbiosSetup)?; + + let handoff = efi::write_efi_tables(guest_mem, rsdp_addr) + .map_err(|e| super::Error::PlatformSpecific(Error::SetupEfi(e)))?; + + let fdt_final = fdt::create_stub_fdt(cmdline, initrd, &handoff).map_err(|_| Error::SetupFdt)?; + + if log_enabled!(Level::Debug) { + fdt::print_fdt(&fdt_final); + } + + fdt::write_fdt_to_memory(&fdt_final, guest_mem).map_err(Error::WriteFdtToMemory)?; + + Ok(()) +} + /// Returns the memory address where the initramfs could be loaded. pub fn initramfs_load_addr( guest_mem: &GuestMemoryMmap, diff --git a/docs/hotplug.md b/docs/hotplug.md index 14e60f954b..ef61a26dcf 100644 --- a/docs/hotplug.md +++ b/docs/hotplug.md @@ -159,7 +159,7 @@ The same API can also be used to reduce the desired RAM for a VM. It is importan Extra PCI devices can be added and removed from a running `cloud-hypervisor` instance. This is controlled by making a HTTP API request to the VMM to ask for the additional device to be added, or for the existing device to be removed. -Note: On AArch64 platform, PCI device hotplug can only be achieved using ACPI. Please refer to the [documentation](uefi.md#building-uefi-firmware-for-aarch64) for more information. +Note: On AArch64 platform, PCI device hotplug can only be achieved using ACPI. A guest booted through UEFI firmware gets ACPI from the firmware; please refer to the [documentation](uefi.md#building-uefi-firmware-for-aarch64) for more information. A direct-kernel-booted guest gets ACPI through `--platform acpi_boot=on`, which is the default: the VMM synthesizes the EFI handoff that leads the kernel to the ACPI tables and passes a device tree with only a `/chosen` node. The guest kernel needs `CONFIG_EFI` and `CONFIG_ACPI`; a kernel without them finds no memory and panics early in boot, so boot it with `--platform acpi_boot=off` to get the hardware device tree back, without PCI hotplug. To use PCI device hotplug start the VM with the HTTP server. diff --git a/vmm/src/api/openapi/cloud-hypervisor.yaml b/vmm/src/api/openapi/cloud-hypervisor.yaml index 588ce309ca..a971bf570a 100644 --- a/vmm/src/api/openapi/cloud-hypervisor.yaml +++ b/vmm/src/api/openapi/cloud-hypervisor.yaml @@ -1023,6 +1023,13 @@ components: vfio_p2p_dma: type: boolean default: true + acpi_boot: + type: boolean + default: true + description: > + AArch64 only. Give a direct-kernel-booted guest ACPI through a + synthesized EFI handoff instead of a hardware device tree. + Firmware boot ignores it. MemoryZoneConfig: required: diff --git a/vmm/src/config.rs b/vmm/src/config.rs index 51293d43f6..ea26dc696f 100644 --- a/vmm/src/config.rs +++ b/vmm/src/config.rs @@ -890,6 +890,10 @@ impl PlatformConfig { oem_strings=,chassis_asset_tag=" .to_string(); + if cfg!(target_arch = "aarch64") { + syntax.push_str(",acpi_boot=on|off"); + } + if cfg!(feature = "tdx") { syntax.push_str(",tdx=on|off"); } @@ -958,6 +962,8 @@ impl PlatformConfig { .add("iommufd") .add("iommufd_fd") .add("vfio_p2p_dma"); + #[cfg(target_arch = "aarch64")] + parser.add("acpi_boot"); for field in SMBIOS_STRING_FIELDS { parser.add(field.key); } @@ -1008,6 +1014,12 @@ impl PlatformConfig { .map_err(Error::ParsePlatform)? .unwrap_or(Toggle(false)) .0; + #[cfg(target_arch = "aarch64")] + let acpi_boot = parser + .convert::("acpi_boot") + .map_err(Error::ParsePlatform)? + .unwrap_or(Toggle(true)) + .0; let mut platform_config = PlatformConfig { num_pci_segments, @@ -1029,6 +1041,8 @@ impl PlatformConfig { #[cfg(feature = "sev_snp")] sev_snp, vfio_p2p_dma, + #[cfg(target_arch = "aarch64")] + acpi_boot, }; for field in SMBIOS_STRING_FIELDS { @@ -5879,6 +5893,8 @@ id=\"{id}\",pci_segment={pci_segment},queue_sizes={queue_sizes}" tdx: false, #[cfg(feature = "sev_snp")] sev_snp: false, + #[cfg(target_arch = "aarch64")] + acpi_boot: default_platformconfig_acpi_boot(), } } diff --git a/vmm/src/vm.rs b/vmm/src/vm.rs index b5c3072867..7b84103ed1 100644 --- a/vmm/src/vm.rs +++ b/vmm/src/vm.rs @@ -1866,8 +1866,8 @@ impl Vm { #[cfg(target_arch = "aarch64")] fn configure_system( &mut self, - _rsdp_addr: Option, - _entry_addr: EntryPoint, + rsdp_addr: Option, + entry_addr: EntryPoint, ) -> Result<()> { let cmdline = Self::generate_cmdline( self.config.lock().unwrap().payload.as_ref().unwrap(), @@ -1935,13 +1935,26 @@ impl Vm { )) })?; - let smbios = self - .config - .lock() - .unwrap() - .platform - .as_ref() - .and_then(|p| p.smbios_config()); + let platform = self.config.lock().unwrap().platform.clone(); + let smbios = platform.as_ref().and_then(|p| p.smbios_config()); + + // firmware reads its memory and devices from the full device tree + let direct_boot = entry_addr.entry_addr != layout::UEFI_START; + + // init_pmu() above must run on this path too, or KVM_RUN fails EINVAL + if direct_boot && platform.as_ref().is_none_or(|p| p.acpi_boot) { + let rsdp_addr = rsdp_addr.ok_or(Error::ConfigureSystem( + arch::Error::PlatformSpecific(arch::aarch64::Error::MissingRsdp), + ))?; + return arch::aarch64::configure_system_acpi( + &mem, + cmdline.as_cstring().unwrap().to_str().unwrap(), + &initramfs_config, + rsdp_addr, + smbios.as_ref(), + ) + .map_err(Error::ConfigureSystem); + } arch::configure_system( &mem, diff --git a/vmm/src/vm_config.rs b/vmm/src/vm_config.rs index f66c9d2387..74f35874cd 100644 --- a/vmm/src/vm_config.rs +++ b/vmm/src/vm_config.rs @@ -122,6 +122,11 @@ pub fn default_platformconfig_iommu_address_width_bits() -> u8 { DEFAULT_IOMMU_ADDRESS_WIDTH_BITS } +#[cfg(target_arch = "aarch64")] +pub fn default_platformconfig_acpi_boot() -> bool { + true +} + pub fn default_platformconfig_vfio_p2p_dma() -> bool { true } @@ -166,6 +171,13 @@ pub struct PlatformConfig { pub iommufd_fd: Option, #[serde(default = "default_platformconfig_vfio_p2p_dma")] pub vfio_p2p_dma: bool, + /// Hand the guest ACPI through a synthesized EFI handoff instead of a + /// hardware-describing device tree. PCI hotplug is ACPI-only, so a + /// direct-kernel-booted aarch64 guest cannot see hot-added or ejected + /// devices without this. Firmware boot ignores it. + #[cfg(target_arch = "aarch64")] + #[serde(default = "default_platformconfig_acpi_boot")] + pub acpi_boot: bool, } #[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))] From 31ac24177e8dd9fc803472f422347bdfa3ad7936 Mon Sep 17 00:00:00 2001 From: tonic Date: Wed, 23 Sep 2026 22:43:45 +0800 Subject: [PATCH 4/4] hypervisor: kvm: save and restore the guest shadow stack pointer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On a host whose KVM exposes CET shadow stacks to the guest (CPUID.(EAX=7,ECX=0):ECX[7]; e.g. Linux 7.0 on AMD Zen 3), a Windows guest bugchecks 0x50 PAGE_FAULT_IN_NONPAGED_AREA at 0xfffffffffffffff8, from nt!KePopulateContinuationContext, right after being restored from some snapshots — about one in five on our node. A vCPU paused while running a user thread with shadow stacks enabled keeps its live SSP in the SSP register, not in MSR_IA32_PL3_SSP. KVM exposes that register only as KVM_REG_GUEST_SSP through KVM_{GET,SET}_ONE_REG, which Cloud Hypervisor never saved. The thread therefore resumed with SSP=0, took a shadow-stack #PF on its next CALL/RET, and the kernel faulted again reading 0 - 8 while building the exception's continuation context. Snapshots taken while every vCPU was in the kernel or running a thread without shadow stacks were unaffected, which is why only some of them failed. Save the register in VcpuKvmState when the guest CPUID exposes shadow stacks and restore it after the MSRs, as QEMU does. The field is optional, so snapshots taken before this change still restore, with the old behaviour. Signed-off-by: tonic --- hypervisor/src/kvm/mod.rs | 68 +++++++++++++++++++++++++++++++- hypervisor/src/kvm/x86_64/mod.rs | 11 ++++++ 2 files changed, 77 insertions(+), 2 deletions(-) diff --git a/hypervisor/src/kvm/mod.rs b/hypervisor/src/kvm/mod.rs index f48a620d5f..f7d0cb4338 100644 --- a/hypervisor/src/kvm/mod.rs +++ b/hypervisor/src/kvm/mod.rs @@ -82,9 +82,9 @@ use kvm_bindings::{ kvm_msr_entry, }; #[cfg(target_arch = "x86_64")] -use x86_64::check_required_kvm_extensions; -#[cfg(target_arch = "x86_64")] pub use x86_64::{CpuId, ExtendedControlRegisters, MsrEntries, VcpuKvmState}; +#[cfg(target_arch = "x86_64")] +use x86_64::{check_required_kvm_extensions, cpuid_has_shstk}; #[cfg(target_arch = "x86_64")] use crate::ClockData; @@ -202,6 +202,25 @@ ioctl_iow_nr!( 0xe3, kvm_bindings::kvm_device_attr ); +// kvm-ioctls only exposes KVM_{GET,SET}_ONE_REG for aarch64 and riscv64. +#[cfg(target_arch = "x86_64")] +ioctl_iow_nr!( + KVM_GET_ONE_REG, + kvm_bindings::KVMIO, + 0xab, + kvm_bindings::kvm_one_reg +); +#[cfg(target_arch = "x86_64")] +ioctl_iow_nr!( + KVM_SET_ONE_REG, + kvm_bindings::KVMIO, + 0xac, + kvm_bindings::kvm_one_reg +); +// KVM_X86_REG_KVM(KVM_REG_GUEST_SSP): the guest's current shadow stack pointer. It is +// a register, not an MSR, so KVM_GET_MSRS never carries it. +#[cfg(target_arch = "x86_64")] +const KVM_REG_GUEST_SSP: u64 = 0x2030_0003_0000_0000; #[cfg(feature = "sev_snp")] use igvm_defs::PAGE_SIZE_4K; @@ -2077,6 +2096,45 @@ impl KvmVcpu { let ret = unsafe { ioctl_with_ref(&self.fd, KVM_HAS_DEVICE_ATTR(), &attr) }; ret == 0 } + + /// The guest's current shadow stack pointer, or `None` when the guest CPUID + /// does not expose shadow stacks. A vCPU paused in user mode keeps its live SSP + /// here, not in MSR_IA32_PL3_SSP, so without it a restored thread resumes with + /// SSP=0 and faults on its next CALL/RET. + fn guest_ssp(&self, cpuid: &[CpuIdEntry]) -> cpu::Result> { + if !cpuid_has_shstk(cpuid) { + return Ok(None); + } + let mut ssp = 0u64; + let reg = kvm_bindings::kvm_one_reg { + id: KVM_REG_GUEST_SSP, + addr: &raw mut ssp as u64, + }; + // SAFETY: FFI call; `reg.addr` points to `ssp`, filled in by the kernel. + let ret = unsafe { ioctl_with_ref(&self.fd, KVM_GET_ONE_REG(), ®) }; + if ret < 0 { + return Err(cpu::HypervisorCpuError::GetRegister( + io::Error::last_os_error().into(), + )); + } + Ok(Some(ssp)) + } + + /// Restore the guest's current shadow stack pointer. + fn set_guest_ssp(&self, ssp: u64) -> cpu::Result<()> { + let reg = kvm_bindings::kvm_one_reg { + id: KVM_REG_GUEST_SSP, + addr: &raw const ssp as u64, + }; + // SAFETY: FFI call; `reg.addr` points to `ssp`, read by the kernel. + let ret = unsafe { ioctl_with_ref(&self.fd, KVM_SET_ONE_REG(), ®) }; + if ret < 0 { + return Err(cpu::HypervisorCpuError::SetRegister( + io::Error::last_os_error().into(), + )); + } + Ok(()) + } } /// Implementation of Vcpu trait for KVM @@ -3243,6 +3301,7 @@ impl cpu::Vcpu for KvmVcpu { let vcpu_events = self.get_vcpu_events()?; let tsc_khz = self.tsc_khz()?; + let guest_ssp = self.guest_ssp(&cpuid)?; Ok(VcpuKvmState { cpuid, @@ -3258,6 +3317,7 @@ impl cpu::Vcpu for KvmVcpu { tsc_khz, nested_state, hyperv_synic, + guest_ssp, } .into()) } @@ -3497,6 +3557,10 @@ impl cpu::Vcpu for KvmVcpu { } } + if let Some(ssp) = state.guest_ssp { + self.set_guest_ssp(ssp)?; + } + self.set_vcpu_events(&state.vcpu_events)?; Ok(()) diff --git a/hypervisor/src/kvm/x86_64/mod.rs b/hypervisor/src/kvm/x86_64/mod.rs index 5bc3ee89a0..c0a49ff310 100644 --- a/hypervisor/src/kvm/x86_64/mod.rs +++ b/hypervisor/src/kvm/x86_64/mod.rs @@ -87,6 +87,17 @@ pub struct VcpuKvmState { pub nested_state: Option, #[serde(default)] pub hyperv_synic: bool, + // Absent from snapshots taken before it was saved, and when the guest has no + // shadow stacks. + #[serde(default)] + pub guest_ssp: Option, +} + +/// CPUID.(EAX=7,ECX=0):ECX[7], CET shadow stack. +pub fn cpuid_has_shstk(cpuid: &[CpuIdEntry]) -> bool { + cpuid + .iter() + .any(|e| e.function == 7 && e.index == 0 && e.ecx & (1 << 7) != 0) } impl From for kvm_segment {