From 8108d55f85b78fd67581ae2d087d554a31e89a3c Mon Sep 17 00:00:00 2001 From: tonic Date: Sat, 19 Sep 2026 22:34:53 +0800 Subject: [PATCH] aarch64: give direct-booted guests ACPI via a synthesized EFI handoff MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A direct-kernel-booted aarch64 guest only ever saw the FDT, so every ACPI table create_acpi_tables() had already written went unused. PCI hotplug is ACPI-only (GED -> PHPR.PSCN -> PCNT -> DVNT/B0EJ), which left vm.add-net and vm.remove-device returning success while the guest noticed neither: a hot-added NIC never appeared, and an ejected one stayed on the bus forever along with its tap. The arm64 kernel does not need real firmware for this. It takes the RSDP from the EFI configuration table, which it finds through `linux,uefi-system-table` and the `linux,uefi-mmap-*` pointers in the device tree's /chosen node, and it does not care who produced those structures. So synthesize them: - arch/src/aarch64/efi.rs builds an EFI System Table, a configuration table and an EFI memory map. - fdt::create_stub_fdt() emits a device tree holding only /chosen. dt_scan_depth1_nodes() treats any other depth-1 node as proof of a real DTB and disables ACPI, so describing no hardware is what lets the guest take its hardware from ACPI — and is why acpi=force is not needed. Selected by `--platform acpi_boot=on|off`, defaulting on for aarch64. The stub tree also carries `linux,uefi-secure-boot = 0`: Ubuntu's kernel makes it a required property in efi_get_fdt_params(), and without it the EFI handoff is abandoned, the memory map is never installed, and — with no /memory node — memblock comes up empty and paging_init panics with "Failed to allocate page table page". Mainline ignores the property. Four configuration tables are published. ACPI 2.0 carries the RSDP. EFI RT Properties declares no runtime services, so the guest never calls into firmware that does not exist. SMBIOS3 is how aarch64 Linux locates DMI at all, which makes the tables setup_smbios() writes reachable for the first time on this path. LINUX_EFI_MEMRESERVE lets the GICv3 ITS persist its LPI property and pending tables through efi_mem_reserve_persistent(); without it the guest warns twice per boot out of irq-gic-v3-its.c. Verified on a c4a-highmem-96-metal host, Ubuntu 24.04 arm64 guest: - /sys/firmware/{acpi,efi,dmi} present, ACPI0013 GED bound to a GIC SPI, GICv3/ITS, PSCI, PMU, the SPCR console and DMI all taken from ACPI - zero kernel warnings, and no unavailable ranges in the memory map - vm.add-net: GED interrupt fires, the NIC appears with no guest-side rescan - vm.remove-device: the guest ejects it and it leaves CH's device tree - clone: the new MAC appears by itself and DHCPs a distinct lease, the snapshot's NIC is gone, and no restore tap is left behind - acpi_boot=off still boots the same guest through the device tree --- arch/src/aarch64/efi.rs | 465 +++++++++++++++++++++++++++++++++++++ arch/src/aarch64/fdt.rs | 126 ++++++++++ arch/src/aarch64/layout.rs | 6 + arch/src/aarch64/mod.rs | 34 +++ vmm/src/config.rs | 16 ++ vmm/src/vm.rs | 31 ++- vmm/src/vm_config.rs | 12 + 7 files changed, 682 insertions(+), 8 deletions(-) create mode 100644 arch/src/aarch64/efi.rs diff --git a/arch/src/aarch64/efi.rs b/arch/src/aarch64/efi.rs new file mode 100644 index 0000000000..118c8d477c --- /dev/null +++ b/arch/src/aarch64/efi.rs @@ -0,0 +1,465 @@ +// Copyright 2026 The Cloud Hypervisor Authors +// +// SPDX-License-Identifier: Apache-2.0 + +// The aarch64 kernel reaches ACPI only through the EFI stub: it reads +// `linux,uefi-system-table` and the `linux,uefi-mmap-*` pointers out of the +// device tree's /chosen node and walks the EFI configuration table for the +// ACPI 2.0 RSDP. Nothing in that path requires real firmware to have produced +// the structures, so the VMM can synthesize them directly. Modelled on +// OpenVMM's aarch64 direct-boot loader. + +use std::result; + +use vm_memory::{Address, Bytes, GuestAddress, GuestMemoryBackend, GuestMemoryRegion}; + +use super::layout; +use crate::GuestMemoryMmap; + +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("Writing EFI handoff structures to guest memory")] + WriteEfiTables(#[source] vm_memory::GuestMemoryError), + #[error("EFI metadata at {0:#x} overflows into the ACPI region at {1:#x}")] + MetadataOverflow(u64, u64), +} + +type Result = result::Result; + +// What the stub device tree has to publish for the EFI stub to pick the +// handoff up. +#[derive(Clone, Copy, Debug)] +pub struct EfiHandoff { + pub systab_addr: u64, + pub mmap_addr: u64, + pub mmap_size: u32, + pub mmap_desc_size: u32, + pub mmap_desc_ver: u32, +} + +const PAGE_SIZE: u64 = 0x1000; + +// "IBI SYST" +const EFI_SYSTEM_TABLE_SIGNATURE: u64 = 0x5453_5953_2049_4249; +const EFI_2_70_SYSTEM_TABLE_REVISION: u32 = (2 << 16) | 70; +const EFI_SYSTEM_TABLE_SIZE: u64 = 0x78; +const EFI_MEMORY_DESCRIPTOR_VERSION: u32 = 1; +const EFI_MEMORY_DESCRIPTOR_SIZE: u64 = 40; +const EFI_MEMORY_WB: u64 = 0x8; + +const EFI_BOOT_SERVICES_DATA: u32 = 4; +const EFI_RUNTIME_SERVICES_DATA: u32 = 6; +const EFI_CONVENTIONAL_MEMORY: u32 = 7; +const EFI_ACPI_RECLAIM_MEMORY: u32 = 9; + +const CONFIG_ENTRY_SIZE: u64 = 16 + 8; +const CONFIG_ENTRY_COUNT: u64 = 4; + +// GUIDs in EFI's mixed-endian encoding: the first three fields little-endian, +// the trailing eight bytes in order. +const ACPI_20_TABLE_GUID: [u8; 16] = [ + 0x71, 0xe8, 0x68, 0x88, 0xf1, 0xe4, 0xd3, 0x11, 0xbc, 0x22, 0x00, 0x80, 0xc7, 0x3c, 0x88, 0x81, +]; +const EFI_RT_PROPERTIES_TABLE_GUID: [u8; 16] = [ + 0x8a, 0x91, 0x66, 0xeb, 0xef, 0x7e, 0x2a, 0x40, 0x84, 0x2e, 0x93, 0x1d, 0x21, 0xc3, 0x8a, 0xe9, +]; +const SMBIOS3_TABLE_GUID: [u8; 16] = [ + 0x44, 0x15, 0xfd, 0xf2, 0x94, 0x97, 0x2c, 0x4a, 0x99, 0x2e, 0xe5, 0xbb, 0xcf, 0x20, 0xe3, 0x94, +]; +const LINUX_EFI_MEMRESERVE_TABLE_GUID: [u8; 16] = [ + 0xc6, 0xb0, 0x8e, 0x88, 0xde, 0x8e, 0xf5, 0x4f, 0xa8, 0xf0, 0x9a, 0xee, 0x5c, 0xb9, 0x77, 0xc2, +]; + +const EFI_RT_PROPERTIES_TABLE_VERSION: u16 = 1; +const EFI_RT_PROPERTIES_TABLE_SIZE: u64 = 8; + +const MEMRESERVE_HEADER_SIZE: u64 = 16; +const MEMRESERVE_ENTRY_SIZE: u64 = 16; + +fn align_up(value: u64, alignment: u64) -> u64 { + (value + alignment - 1) & !(alignment - 1) +} + +// UEFI 4.2 checksums the table header over header_size bytes with the crc32 +// field itself zeroed. Table-less so this stays dependency-free. +fn crc32(bytes: &[u8]) -> u32 { + let mut crc = !0u32; + for byte in bytes { + crc ^= *byte as u32; + for _ in 0..8 { + let mask = (crc & 1).wrapping_neg(); + crc = (crc >> 1) ^ (0xedb8_8320 & mask); + } + } + !crc +} + +fn memory_descriptor(typ: u32, physical_start: u64, pages: u64) -> [u8; 40] { + let mut out = [0u8; 40]; + out[0..4].copy_from_slice(&typ.to_le_bytes()); + out[8..16].copy_from_slice(&physical_start.to_le_bytes()); + out[24..32].copy_from_slice(&pages.to_le_bytes()); + out[32..40].copy_from_slice(&EFI_MEMORY_WB.to_le_bytes()); + out +} + +// The firmware-shaped view of guest memory the EFI stub consumes. `rsdp_addr` +// is where create_acpi_tables() already put the RSDP. +pub fn write_efi_tables( + guest_mem: &GuestMemoryMmap, + rsdp_addr: GuestAddress, +) -> Result { + let efi_base = layout::EFI_START.raw_value(); + let acpi_base = layout::ACPI_START.raw_value(); + + // Page 0 holds the memory map, which is sized last; the rest of the + // metadata starts on page 1. + let mut cursor = efi_base + PAGE_SIZE; + + let systab_addr = cursor; + cursor += EFI_SYSTEM_TABLE_SIZE; + + let config_table_addr = cursor; + cursor += CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE; + + // NUL-terminated UTF-16LE, as the system table's firmware_vendor expects. + let fw_vendor: Vec = "Cloud Hypervisor\0" + .encode_utf16() + .flat_map(|c| c.to_le_bytes()) + .collect(); + let fw_vendor_addr = cursor; + cursor += fw_vendor.len() as u64; + cursor = align_up(cursor, 8); + + let rt_props_addr = cursor; + cursor += EFI_RT_PROPERTIES_TABLE_SIZE; + + // struct linux_efi_memreserve: { i32 size; i32 count; u64 next; entry[] }, + // each entry a { u64 base; u64 size; } pair. The kernel appends to this in + // place, so it gets its own page and is never handed back as usable memory. + let memreserve_addr = align_up(cursor, PAGE_SIZE); + let memreserve_capacity = (PAGE_SIZE - MEMRESERVE_HEADER_SIZE) / MEMRESERVE_ENTRY_SIZE; + cursor = memreserve_addr + PAGE_SIZE; + + if cursor > acpi_base { + return Err(Error::MetadataOverflow(cursor, acpi_base)); + } + + // Runtime services are never available in a synthesized handoff — saying so + // explicitly is what keeps the kernel from calling into them. + let mut rt_props = [0u8; EFI_RT_PROPERTIES_TABLE_SIZE as usize]; + rt_props[0..2].copy_from_slice(&EFI_RT_PROPERTIES_TABLE_VERSION.to_le_bytes()); + rt_props[2..4].copy_from_slice(&(EFI_RT_PROPERTIES_TABLE_SIZE as u16).to_le_bytes()); + guest_mem + .write_slice(&rt_props, GuestAddress(rt_props_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut config_entries = [0u8; (CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE) as usize]; + config_entries[0..16].copy_from_slice(&ACPI_20_TABLE_GUID); + config_entries[16..24].copy_from_slice(&rsdp_addr.raw_value().to_le_bytes()); + config_entries[24..40].copy_from_slice(&EFI_RT_PROPERTIES_TABLE_GUID); + config_entries[40..48].copy_from_slice(&rt_props_addr.to_le_bytes()); + // aarch64 Linux finds DMI only through this entry — it has no equivalent of + // the x86 anchor scan — so the tables setup_smbios() wrote are invisible + // without it. + config_entries[48..64].copy_from_slice(&SMBIOS3_TABLE_GUID); + config_entries[64..72].copy_from_slice(&layout::SMBIOS_START.raw_value().to_le_bytes()); + // Without this the GICv3 ITS cannot persist its LPI property and pending + // tables through efi_mem_reserve_persistent() and warns on every boot. + config_entries[72..88].copy_from_slice(&LINUX_EFI_MEMRESERVE_TABLE_GUID); + config_entries[88..96].copy_from_slice(&memreserve_addr.to_le_bytes()); + guest_mem + .write_slice(&config_entries, GuestAddress(config_table_addr)) + .map_err(Error::WriteEfiTables)?; + + guest_mem + .write_slice(&fw_vendor, GuestAddress(fw_vendor_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut memreserve = [0u8; MEMRESERVE_HEADER_SIZE as usize]; + memreserve[0..4].copy_from_slice(&(memreserve_capacity as i32).to_le_bytes()); + guest_mem + .write_slice(&memreserve, GuestAddress(memreserve_addr)) + .map_err(Error::WriteEfiTables)?; + + let mut systab = [0u8; EFI_SYSTEM_TABLE_SIZE as usize]; + systab[0x00..0x08].copy_from_slice(&EFI_SYSTEM_TABLE_SIGNATURE.to_le_bytes()); + systab[0x08..0x0c].copy_from_slice(&EFI_2_70_SYSTEM_TABLE_REVISION.to_le_bytes()); + systab[0x0c..0x10].copy_from_slice(&(EFI_SYSTEM_TABLE_SIZE as u32).to_le_bytes()); + systab[0x18..0x20].copy_from_slice(&fw_vendor_addr.to_le_bytes()); + systab[0x20..0x24].copy_from_slice(&1u32.to_le_bytes()); + systab[0x68..0x70].copy_from_slice(&CONFIG_ENTRY_COUNT.to_le_bytes()); + systab[0x70..0x78].copy_from_slice(&config_table_addr.to_le_bytes()); + let checksum = crc32(&systab); + systab[0x10..0x14].copy_from_slice(&checksum.to_le_bytes()); + guest_mem + .write_slice(&systab, GuestAddress(systab_addr)) + .map_err(Error::WriteEfiTables)?; + + // The stub DT carries no /memory node, so this map is the only thing + // memblock is built from — every range the guest may touch has to appear. + let reserved_start = layout::FDT_START.raw_value(); + let smbios_base = layout::SMBIOS_START.raw_value(); + let reserved_end = smbios_base + layout::SMBIOS_MAX_SIZE; + let mut mmap = Vec::new(); + // The stub DT: the kernel unflattens it early, so it may be reclaimed after. + mmap.extend_from_slice(&memory_descriptor( + EFI_BOOT_SERVICES_DATA, + reserved_start, + (efi_base - reserved_start) / PAGE_SIZE, + )); + // The EFI structures, covering the whole carve-out so no hole is left to + // show up as an unavailable range. Runtime-services data rather than boot: + // the kernel keeps writing to the memreserve table after boot services end. + mmap.extend_from_slice(&memory_descriptor( + EFI_RUNTIME_SERVICES_DATA, + efi_base, + (acpi_base - efi_base) / PAGE_SIZE, + )); + mmap.extend_from_slice(&memory_descriptor( + EFI_ACPI_RECLAIM_MEMORY, + acpi_base, + layout::ACPI_MAX_SIZE / PAGE_SIZE, + )); + mmap.extend_from_slice(&memory_descriptor( + EFI_BOOT_SERVICES_DATA, + smbios_base, + layout::SMBIOS_MAX_SIZE / PAGE_SIZE, + )); + for region in guest_mem.iter() { + let start = region.start_addr().raw_value(); + let end = start + region.len(); + for (from, to) in subtract_reserved(start, end, reserved_start, reserved_end) { + mmap.extend_from_slice(&memory_descriptor( + EFI_CONVENTIONAL_MEMORY, + from, + (to - from) / PAGE_SIZE, + )); + } + } + + guest_mem + .write_slice(&mmap, layout::EFI_START) + .map_err(Error::WriteEfiTables)?; + + Ok(EfiHandoff { + systab_addr, + mmap_addr: efi_base, + mmap_size: mmap.len() as u32, + mmap_desc_size: EFI_MEMORY_DESCRIPTOR_SIZE as u32, + mmap_desc_ver: EFI_MEMORY_DESCRIPTOR_VERSION, + }) +} + +// Overlapping descriptors make the kernel reject the whole map, so the +// firmware-owned span is punched out of every RAM region it intersects. +fn subtract_reserved( + start: u64, + end: u64, + reserved_start: u64, + reserved_end: u64, +) -> Vec<(u64, u64)> { + let mut out = Vec::new(); + if end <= reserved_start || start >= reserved_end { + out.push((start, end)); + return out; + } + if start < reserved_start { + out.push((start, reserved_start)); + } + if end > reserved_end { + out.push((reserved_end, end)); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + // RAM as arch_memory_regions() lays it out, big enough to hold the whole + // firmware carve-out plus a little usable memory above it. + fn test_mem() -> GuestMemoryMmap { + GuestMemoryMmap::from_ranges(&[( + layout::RAM_START, + (layout::KERNEL_START.raw_value() - layout::RAM_START.raw_value() + 0x10_0000) as usize, + )]) + .unwrap() + } + + fn read(mem: &GuestMemoryMmap, addr: u64, len: usize) -> Vec { + let mut out = vec![0u8; len]; + mem.read_slice(&mut out, GuestAddress(addr)).unwrap(); + out + } + + fn le64(bytes: &[u8]) -> u64 { + u64::from_le_bytes(bytes.try_into().unwrap()) + } + + fn le32(bytes: &[u8]) -> u32 { + u32::from_le_bytes(bytes.try_into().unwrap()) + } + + #[test] + fn crc32_matches_known_vector() { + assert_eq!(crc32(b"123456789"), 0xcbf4_3926); + } + + #[test] + fn reserved_span_is_punched_out_of_ram() { + // A region containing the whole reserved span splits in two. + assert_eq!( + subtract_reserved(0x4000_0000, 0x8000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x4040_0000, 0x8000_0000)] + ); + // A region entirely above it is untouched. + assert_eq!( + subtract_reserved(0x1_0000_0000, 0x2_0000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x1_0000_0000, 0x2_0000_0000)] + ); + // A region straddling the tail keeps only what is above. + assert_eq!( + subtract_reserved(0x4030_0000, 0x5000_0000, 0x4000_0000, 0x4040_0000), + vec![(0x4040_0000, 0x5000_0000)] + ); + } + + #[test] + fn system_table_header_is_well_formed() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let systab = read(&mem, handoff.systab_addr, EFI_SYSTEM_TABLE_SIZE as usize); + + assert_eq!(le64(&systab[0x00..0x08]), EFI_SYSTEM_TABLE_SIGNATURE); + assert_eq!(le32(&systab[0x08..0x0c]), EFI_2_70_SYSTEM_TABLE_REVISION); + assert_eq!(le32(&systab[0x0c..0x10]), EFI_SYSTEM_TABLE_SIZE as u32); + assert_eq!(le64(&systab[0x68..0x70]), CONFIG_ENTRY_COUNT); + + // UEFI 4.2: the checksum covers header_size bytes with crc32 zeroed. + let stored = le32(&systab[0x10..0x14]); + let mut zeroed = systab.clone(); + zeroed[0x10..0x14].fill(0); + assert_eq!(stored, crc32(&zeroed)); + + // firmware_vendor must point at NUL-terminated UTF-16LE. + let vendor_addr = le64(&systab[0x18..0x20]); + let vendor = read(&mem, vendor_addr, 34); + let utf16: Vec = vendor + .as_chunks::<2>() + .0 + .iter() + .map(|c| u16::from_le_bytes(*c)) + .take_while(|c| *c != 0) + .collect(); + assert_eq!(String::from_utf16(&utf16).unwrap(), "Cloud Hypervisor"); + } + + #[test] + fn configuration_table_points_at_every_structure() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let systab = read(&mem, handoff.systab_addr, EFI_SYSTEM_TABLE_SIZE as usize); + let table_addr = le64(&systab[0x70..0x78]); + let entries = read( + &mem, + table_addr, + (CONFIG_ENTRY_COUNT * CONFIG_ENTRY_SIZE) as usize, + ); + + let found: Vec<([u8; 16], u64)> = entries + .as_chunks::<{ CONFIG_ENTRY_SIZE as usize }>() + .0 + .iter() + .map(|e| (e[0..16].try_into().unwrap(), le64(&e[16..24]))) + .collect(); + + let lookup = |guid: [u8; 16]| { + found + .iter() + .find(|(g, _)| *g == guid) + .unwrap_or_else(|| panic!("missing configuration table entry")) + .1 + }; + + assert_eq!(lookup(ACPI_20_TABLE_GUID), layout::RSDP_POINTER.raw_value()); + assert_eq!(lookup(SMBIOS3_TABLE_GUID), layout::SMBIOS_START.raw_value()); + + // RT Properties must declare that nothing is supported, otherwise the + // guest will call into runtime services this handoff does not have. + let rt_props = read( + &mem, + lookup(EFI_RT_PROPERTIES_TABLE_GUID), + EFI_RT_PROPERTIES_TABLE_SIZE as usize, + ); + assert_eq!(u16::from_le_bytes([rt_props[0], rt_props[1]]), 1); + assert_eq!(le32(&rt_props[4..8]), 0); + + // The memreserve table starts empty with room for entries. + let memreserve = read( + &mem, + lookup(LINUX_EFI_MEMRESERVE_TABLE_GUID), + MEMRESERVE_HEADER_SIZE as usize, + ); + assert!(le32(&memreserve[0..4]) > 0, "no capacity for entries"); + assert_eq!(le32(&memreserve[4..8]), 0, "count must start at zero"); + assert_eq!(le64(&memreserve[8..16]), 0, "next must be null"); + } + + #[test] + fn memory_map_covers_ram_without_overlapping() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + assert_eq!(handoff.mmap_desc_size, EFI_MEMORY_DESCRIPTOR_SIZE as u32); + assert_eq!(handoff.mmap_desc_ver, EFI_MEMORY_DESCRIPTOR_VERSION); + + let raw = read(&mem, handoff.mmap_addr, handoff.mmap_size as usize); + let mut ranges: Vec<(u64, u64, u32)> = raw + .as_chunks::<{ EFI_MEMORY_DESCRIPTOR_SIZE as usize }>() + .0 + .iter() + .map(|d| { + let start = le64(&d[8..16]); + (start, start + le64(&d[24..32]) * PAGE_SIZE, le32(&d[0..4])) + }) + .collect(); + assert!(!ranges.is_empty()); + ranges.sort_by_key(|r| r.0); + + // A hole or an overlap both make the kernel reject the map, and the + // stub tree has no /memory node to fall back on. + for pair in ranges.windows(2) { + assert_eq!(pair[0].1, pair[1].0, "gap or overlap in the memory map"); + } + assert_eq!(ranges[0].0, layout::FDT_START.raw_value()); + assert_eq!(ranges.last().unwrap().1, mem.last_addr().raw_value() + 1); + + // Everything firmware owns has to be typed as such; handing the ACPI + // tables or the EFI structures back as conventional memory would let + // the guest allocate over them. + let firmware_end = layout::SMBIOS_START.raw_value() + layout::SMBIOS_MAX_SIZE; + for (start, end, typ) in &ranges { + if *start < firmware_end { + assert_ne!(*typ, EFI_CONVENTIONAL_MEMORY, "{start:#x}..{end:#x}"); + } else { + assert_eq!(*typ, EFI_CONVENTIONAL_MEMORY, "{start:#x}..{end:#x}"); + } + } + // The EFI structures outlive boot services: the kernel keeps appending + // to the memreserve table after they end. + let efi_region = ranges + .iter() + .find(|(s, _, _)| *s == layout::EFI_START.raw_value()) + .expect("EFI region missing from the map"); + assert_eq!(efi_region.2, EFI_RUNTIME_SERVICES_DATA); + } + + #[test] + fn metadata_stays_below_the_acpi_region() { + let mem = test_mem(); + let handoff = write_efi_tables(&mem, layout::RSDP_POINTER).unwrap(); + let acpi_base = layout::ACPI_START.raw_value(); + assert!(handoff.systab_addr < acpi_base); + assert!(handoff.mmap_addr + u64::from(handoff.mmap_size) < acpi_base); + } +} diff --git a/arch/src/aarch64/fdt.rs b/arch/src/aarch64/fdt.rs index ecd6030e3f..eea929a8b9 100644 --- a/arch/src/aarch64/fdt.rs +++ b/arch/src/aarch64/fdt.rs @@ -27,6 +27,7 @@ use vm_memory::{Address, Bytes, GuestMemoryBackend, GuestMemoryError, GuestMemor use super::super::{DeviceType, GuestMemoryMmap, InitramfsConfig}; use super::cache::{CacheTopologyInfo, read_cache_topology}; +use super::efi::EfiHandoff; use super::layout::{ GIC_V2M_COMPATIBLE, GICV2M_SPI_BASE, GICV2M_SPI_NUM, IRQ_BASE, MEM_32BIT_DEVICES_SIZE, MEM_32BIT_DEVICES_START, MEM_PCI_IO_SIZE, MEM_PCI_IO_START, PCI_HIGH_BASE, @@ -146,6 +147,48 @@ pub fn create_fdt( Ok(fdt_final) } +/// The device tree for ACPI boot: a `/chosen` node and nothing else. +/// +/// `dt_scan_depth1_nodes` in the arm64 kernel treats any depth-1 node other +/// than `chosen` as proof of a "real" DTB and turns ACPI off, so describing no +/// hardware here is what lets the guest take its hardware from ACPI instead — +/// and is why `acpi=force` is not needed on the command line. +pub fn create_stub_fdt( + cmdline: &str, + initrd: &Option, + efi: &EfiHandoff, +) -> FdtWriterResult> { + let mut fdt = FdtWriter::new().unwrap(); + + let root_node = fdt.begin_node("")?; + fdt.property_u32("#address-cells", ADDRESS_CELLS)?; + fdt.property_u32("#size-cells", SIZE_CELLS)?; + + let chosen_node = fdt.begin_node("chosen")?; + fdt.property_string("bootargs", cmdline)?; + if let Some(initrd_config) = initrd { + let initrd_start = initrd_config.address.raw_value(); + fdt.property_u64("linux,initrd-start", initrd_start)?; + fdt.property_u64("linux,initrd-end", initrd_start + initrd_config.size as u64)?; + } + fdt.property_u64("linux,uefi-system-table", efi.systab_addr)?; + fdt.property_u64("linux,uefi-mmap-start", efi.mmap_addr)?; + fdt.property_u32("linux,uefi-mmap-size", efi.mmap_size)?; + fdt.property_u32("linux,uefi-mmap-desc-size", efi.mmap_desc_size)?; + fdt.property_u32("linux,uefi-mmap-desc-ver", efi.mmap_desc_ver)?; + // Ubuntu's kernel makes `linux,uefi-secure-boot` a required property in + // efi_get_fdt_params(); without it the EFI handoff is abandoned, the memory + // map is never installed, and — since this tree has no /memory node — + // memblock comes up empty and paging_init panics with "Failed to allocate + // page table page". Mainline ignores the property. + fdt.property_u32("linux,uefi-secure-boot", 0)?; + fdt.end_node(chosen_node)?; + + fdt.end_node(root_node)?; + + fdt.finish() +} + pub fn write_fdt_to_memory(fdt_final: &[u8], guest_mem: &GuestMemoryMmap) -> Result<()> { // Write FDT to memory. guest_mem @@ -1063,9 +1106,92 @@ fn print_node(node: FdtNode<'_, '_>, n_spaces: usize) { mod tests { use std::collections::BTreeMap; + use vm_memory::GuestAddress; + use super::*; use crate::NumaNode; + fn test_handoff() -> EfiHandoff { + EfiHandoff { + systab_addr: 0x4010_1000, + mmap_addr: 0x4010_0000, + mmap_size: 160, + mmap_desc_size: 40, + mmap_desc_ver: 1, + } + } + + // The whole ACPI mode rests on this: dt_scan_depth1_nodes() in the arm64 + // kernel reads any depth-1 node other than `chosen` as proof of a real DTB + // and turns ACPI back off, which would silently put us back on the device + // tree with no hardware described in it. + #[test] + fn stub_fdt_has_only_a_chosen_node() { + let blob = create_stub_fdt("console=ttyAMA0", &None, &test_handoff()).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let root = fdt.find_node("/").unwrap(); + let children: Vec<&str> = root.children().map(|c| c.name).collect(); + assert_eq!(children, vec!["chosen"]); + } + + #[test] + fn stub_fdt_publishes_the_efi_handoff() { + let handoff = test_handoff(); + let blob = create_stub_fdt("console=ttyAMA0 rw", &None, &handoff).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let chosen = fdt.find_node("/chosen").unwrap(); + + let u64_prop = |name: &str| -> u64 { + let p = chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")); + BigEndian::read_u64(p.value) + }; + let u32_prop = |name: &str| -> u32 { + let p = chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")); + BigEndian::read_u32(p.value) + }; + + assert_eq!(u64_prop("linux,uefi-system-table"), handoff.systab_addr); + assert_eq!(u64_prop("linux,uefi-mmap-start"), handoff.mmap_addr); + assert_eq!(u32_prop("linux,uefi-mmap-size"), handoff.mmap_size); + assert_eq!( + u32_prop("linux,uefi-mmap-desc-size"), + handoff.mmap_desc_size + ); + assert_eq!(u32_prop("linux,uefi-mmap-desc-ver"), handoff.mmap_desc_ver); + + // Ubuntu's efi_get_fdt_params() treats this as required; dropping it + // aborts the handoff and panics the guest in paging_init. + assert_eq!(u32_prop("linux,uefi-secure-boot"), 0); + } + + #[test] + fn stub_fdt_carries_the_initramfs_when_there_is_one() { + let initrd = Some(InitramfsConfig { + address: GuestAddress(0x8000_0000), + size: 0x10_0000, + }); + let blob = create_stub_fdt("", &initrd, &test_handoff()).unwrap(); + let fdt = fdt_parser::Fdt::new(&blob).unwrap(); + let chosen = fdt.find_node("/chosen").unwrap(); + let prop = |name: &str| { + BigEndian::read_u64( + chosen + .properties() + .find(|p| p.name == name) + .unwrap_or_else(|| panic!("{name} missing")) + .value, + ) + }; + assert_eq!(prop("linux,initrd-start"), 0x8000_0000); + assert_eq!(prop("linux,initrd-end"), 0x8010_0000); + } + // Helper function to create a simple NumaNode for testing fn create_test_numa_node(cpus: Vec, device_id: Option) -> NumaNode { NumaNode { diff --git a/arch/src/aarch64/layout.rs b/arch/src/aarch64/layout.rs index a6858c761a..389a097737 100644 --- a/arch/src/aarch64/layout.rs +++ b/arch/src/aarch64/layout.rs @@ -116,6 +116,12 @@ pub const FDT_START: GuestAddress = RAM_START; /// documentation](https://www.kernel.org/doc/Documentation/arm64/booting.txt). pub const FDT_MAX_SIZE: u64 = 0x20_0000; +/// EFI handoff structures (system table, configuration table, memory map) for +/// ACPI boot. Carved from the upper half of the FDT reservation: the stub +/// device tree that mode uses is a few KiB against a 2 MiB window. +pub const EFI_START: GuestAddress = GuestAddress(RAM_START.0 + FDT_MAX_SIZE / 2); +pub const EFI_MAX_SIZE: u64 = FDT_MAX_SIZE / 2; + /// Put ACPI table above dtb pub const ACPI_START: GuestAddress = GuestAddress(RAM_START.0 + FDT_MAX_SIZE); const ACPI_SMBIOS_MAX_SIZE: u64 = 0x20_0000; diff --git a/arch/src/aarch64/mod.rs b/arch/src/aarch64/mod.rs index e92fa0663b..bd6afd8468 100644 --- a/arch/src/aarch64/mod.rs +++ b/arch/src/aarch64/mod.rs @@ -4,6 +4,9 @@ /// Module for cache info. pub mod cache; +/// Module for the synthesized EFI handoff that gives an ACPI-booted guest its +/// RSDP without firmware. +pub mod efi; /// Module for the flattened device tree. pub mod fdt; /// Layout for this aarch64 system. @@ -56,6 +59,10 @@ pub enum Error { /// Error initializing PMU for vcpu #[error("Error initializing PMU for vcpu")] VcpuInitPmu, + + /// Failed to write the EFI handoff structures. + #[error("Failed to write the EFI handoff structures")] + SetupEfi(#[source] efi::Error), } #[derive(Debug, Copy, Clone)] @@ -162,6 +169,33 @@ pub fn configure_system( Ok(()) } +/// The ACPI counterpart of [`configure_system`]: synthesize the EFI handoff the +/// arm64 EFI stub expects, then hand the guest a device tree that describes +/// nothing but where to find it. Everything else — CPUs, GIC, timer, PCI — comes +/// from the ACPI tables `create_acpi_tables()` has already written. +pub fn configure_system_acpi( + guest_mem: &GuestMemoryMmap, + cmdline: &str, + initrd: &Option, + rsdp_addr: GuestAddress, + smbios: Option<&smbios::SmbiosConfig>, +) -> super::Result<()> { + smbios::setup_smbios(guest_mem, smbios).map_err(Error::SmbiosSetup)?; + + let handoff = efi::write_efi_tables(guest_mem, rsdp_addr) + .map_err(|e| super::Error::PlatformSpecific(Error::SetupEfi(e)))?; + + let fdt_final = fdt::create_stub_fdt(cmdline, initrd, &handoff).map_err(|_| Error::SetupFdt)?; + + if log_enabled!(Level::Debug) { + fdt::print_fdt(&fdt_final); + } + + fdt::write_fdt_to_memory(&fdt_final, guest_mem).map_err(Error::WriteFdtToMemory)?; + + Ok(()) +} + /// Returns the memory address where the initramfs could be loaded. pub fn initramfs_load_addr( guest_mem: &GuestMemoryMmap, diff --git a/vmm/src/config.rs b/vmm/src/config.rs index 6f7fd47202..05ee3a7b8c 100644 --- a/vmm/src/config.rs +++ b/vmm/src/config.rs @@ -890,6 +890,10 @@ impl PlatformConfig { oem_strings=,chassis_asset_tag=" .to_string(); + if cfg!(target_arch = "aarch64") { + syntax.push_str(",acpi_boot=on|off"); + } + if cfg!(feature = "tdx") { syntax.push_str(",tdx=on|off"); } @@ -958,6 +962,8 @@ impl PlatformConfig { .add("iommufd") .add("iommufd_fd") .add("vfio_p2p_dma"); + #[cfg(target_arch = "aarch64")] + parser.add("acpi_boot"); for field in SMBIOS_STRING_FIELDS { parser.add(field.key); } @@ -1008,6 +1014,12 @@ impl PlatformConfig { .map_err(Error::ParsePlatform)? .unwrap_or(Toggle(false)) .0; + #[cfg(target_arch = "aarch64")] + let acpi_boot = parser + .convert::("acpi_boot") + .map_err(Error::ParsePlatform)? + .unwrap_or(Toggle(true)) + .0; let mut platform_config = PlatformConfig { num_pci_segments, @@ -1029,6 +1041,8 @@ impl PlatformConfig { #[cfg(feature = "sev_snp")] sev_snp, vfio_p2p_dma, + #[cfg(target_arch = "aarch64")] + acpi_boot, }; for field in SMBIOS_STRING_FIELDS { @@ -5879,6 +5893,8 @@ id=\"{id}\",pci_segment={pci_segment},queue_sizes={queue_sizes}" tdx: false, #[cfg(feature = "sev_snp")] sev_snp: false, + #[cfg(target_arch = "aarch64")] + acpi_boot: default_platformconfig_acpi_boot(), } } diff --git a/vmm/src/vm.rs b/vmm/src/vm.rs index a146ebd155..8a3e3b91c0 100644 --- a/vmm/src/vm.rs +++ b/vmm/src/vm.rs @@ -1864,7 +1864,7 @@ impl Vm { #[cfg(target_arch = "aarch64")] fn configure_system( &mut self, - _rsdp_addr: Option, + rsdp_addr: Option, _entry_addr: EntryPoint, ) -> Result<()> { let cmdline = Self::generate_cmdline( @@ -1933,13 +1933,28 @@ impl Vm { )) })?; - let smbios = self - .config - .lock() - .unwrap() - .platform - .as_ref() - .and_then(|p| p.smbios_config()); + let platform = self.config.lock().unwrap().platform.clone(); + let smbios = platform.as_ref().and_then(|p| p.smbios_config()); + + // Placed after the vgic and PMU setup above: init_pmu() arms + // KVM_ARM_VCPU_PMU_V3_INIT, without which KVM_RUN fails EINVAL on every + // vcpu that carries the PMU feature bit. + if platform.as_ref().is_none_or(|p| p.acpi_boot) { + // The ACPI tables are already in guest memory; all that is missing is + // a way for the guest to find them, so the device tree carries the EFI + // handoff instead of a hardware description. + let rsdp_addr = rsdp_addr.ok_or(Error::ConfigureSystem( + arch::Error::PlatformSpecific(arch::aarch64::Error::SetupFdt), + ))?; + return arch::aarch64::configure_system_acpi( + &mem, + cmdline.as_cstring().unwrap().to_str().unwrap(), + &initramfs_config, + rsdp_addr, + smbios.as_ref(), + ) + .map_err(Error::ConfigureSystem); + } arch::configure_system( &mem, diff --git a/vmm/src/vm_config.rs b/vmm/src/vm_config.rs index 87948b34c3..ee8b283cf2 100644 --- a/vmm/src/vm_config.rs +++ b/vmm/src/vm_config.rs @@ -122,6 +122,11 @@ pub fn default_platformconfig_iommu_address_width_bits() -> u8 { DEFAULT_IOMMU_ADDRESS_WIDTH_BITS } +#[cfg(target_arch = "aarch64")] +pub fn default_platformconfig_acpi_boot() -> bool { + true +} + pub fn default_platformconfig_vfio_p2p_dma() -> bool { true } @@ -166,6 +171,13 @@ pub struct PlatformConfig { pub iommufd_fd: Option, #[serde(default = "default_platformconfig_vfio_p2p_dma")] pub vfio_p2p_dma: bool, + /// Hand the guest ACPI through a synthesized EFI handoff instead of a + /// hardware-describing device tree. PCI hotplug is ACPI-only, so a + /// direct-kernel-booted aarch64 guest cannot see hot-added or ejected + /// devices without this. + #[cfg(target_arch = "aarch64")] + #[serde(default = "default_platformconfig_acpi_boot")] + pub acpi_boot: bool, } #[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))]