Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 8 additions & 19 deletions boot/init/src/boot.rs
Original file line number Diff line number Diff line change
@@ -1,5 +1,4 @@
//! Boot sequence: early mounts → resolve disks → overlay → network persist →
//! switch_root → exec.
//! Boot sequence: early mounts, resolve disks, overlay, persist network, switch_root, exec.

use std::fs;
use std::path::Path;
Expand All @@ -12,8 +11,7 @@ const LAYER_DIR: &str = "/l";
const COW_DIR: &str = "/cow";
const NEWROOT: &str = "/newroot";
const POLL_INTERVAL: Duration = Duration::from_millis(2);
// NICs probe in single-digit ms; a missing one must degrade to the DHCP
// fallback, not stall rootfs handoff for the full disk budget (10s).
// a missing NIC must degrade to the DHCP fallback, not stall handoff for the disk budget.
const NIC_TIMEOUT: Duration = Duration::from_millis(200);

/// Cumulative µs checkpoints since sandbox-init start.
Expand Down Expand Up @@ -44,8 +42,7 @@ impl Marks {

pub fn run() -> ! {
let mut marks = Marks::new();
// Best-effort: if devtmpfs fails there is no console either; later
// failures then power off silently, which is still the right end state.
// best-effort: without devtmpfs there is no console, so a later failure powers off silently.
let _ = sys::mount(
"devtmpfs",
"/dev",
Expand All @@ -57,8 +54,7 @@ pub fn run() -> ! {
let _ = sys::mount("proc", "/proc", Some("proc"), sys::MNT_SECURE, None);
let _ = sys::mount("sysfs", "/sys", Some("sysfs"), sys::MNT_SECURE, None);

// Start marker: kernel-relative and visible at production loglevel,
// where the kernel's own boot lines are suppressed. One console write.
// start marker, visible at production loglevel where the kernel's own boot lines are suppressed.
println!("sandbox-init: start at {}s", uptime());

let cmdline = fs::read_to_string("/proc/cmdline").unwrap_or_default();
Expand All @@ -71,13 +67,11 @@ pub fn run() -> ! {
sys::fatal(&err, cfg.debug);
}

// One deferred trace line (µs, cumulative since sandbox-init start):
// per-phase console writes would perturb exactly what they measure.
// one deferred line: per-phase console writes would perturb what they measure.
if cfg.trace {
println!("sandbox-init: trace{}", marks.render());
}
// Single marker line; boot-bench.sh keys on it. Uptime is kernel-relative,
// directly comparable with printk timestamps on the serial log.
// boot-bench.sh keys on this line; the uptime is comparable with printk timestamps.
println!(
"sandbox-init: rootfs ready at {}s, handing off to {}",
uptime(),
Expand Down Expand Up @@ -145,11 +139,7 @@ fn assemble(cfg: &BootCfg, marks: &mut Marks) -> Result<(), String> {
Ok(())
}

/// Materializes kernel ip= params (cocoon CNI static flow) as MAC-matched
/// networkd units in the new root — persistence only, nothing is configured
/// in the initramfs; networkd applies them once the real init is up. A NIC
/// that never shows up degrades that interface to the DHCP fallback instead
/// of failing the boot, matching the old init-bottom hook.
/// Persists kernel ip= params as MAC-matched networkd units in the new root; a missing NIC degrades to the DHCP fallback.
fn persist_network(cfg: &BootCfg) {
if cfg.ips.is_empty() {
return;
Expand All @@ -175,8 +165,7 @@ fn persist_network(cfg: &BootCfg) {
}
}

/// Resolves every NIC's MAC in one sysfs sweep per poll iteration, against one
/// shared deadline — a missing NIC must cost the timeout once, not once each.
/// Resolves every NIC MAC against one shared deadline, so a missing NIC costs the timeout once.
fn wait_nic_macs(devices: &[&str], timeout: Duration) -> Vec<Option<String>> {
let deadline = Instant::now() + timeout;
let mut found: Vec<Option<String>> = vec![None; devices.len()];
Expand Down
71 changes: 24 additions & 47 deletions boot/init/src/cfg.rs
Original file line number Diff line number Diff line change
@@ -1,13 +1,11 @@
//! Kernel cmdline contract, shared with cocoon (hypervisor/utils.go builds
//! the cmdline; this module is the consuming end).
//! Kernel cmdline contract shared with cocoon, whose hypervisor/utils.go builds the cmdline.

use std::fmt::Write as _;
use std::time::Duration;

pub const DEFAULT_INIT: &str = "/sbin/init";
const DEFAULT_TIMEOUT: Duration = Duration::from_secs(10);
// Cap: `Instant + Duration` panics on overflow, and a panic in PID 1 with
// panic=abort is a kernel panic — a hung VM instead of fatal()'s poweroff.
// cap the timeout: `Instant + Duration` panics on overflow, and a PID 1 panic is a kernel panic.
const MAX_TIMEOUT_SECS: u64 = 86_400;

#[derive(Debug, PartialEq, Eq)]
Expand All @@ -19,8 +17,7 @@ pub struct BootCfg {
/// Per-device wait budget.
pub timeout: Duration,
pub hostname: Option<String>,
/// Static per-NIC config from kernel ip= params (cocoon CNI flow).
/// Persisted as systemd-networkd units, never applied in the initramfs.
/// Static per-NIC config, persisted as networkd units and never applied in the initramfs.
pub ips: Vec<IpParam>,
/// Handoff target inside the assembled rootfs.
pub init: String,
Expand All @@ -30,16 +27,14 @@ pub struct BootCfg {
pub trace: bool,
}

/// One `ip=<addr>::<gw>:<mask>:<host>:<dev>:off[:dns0[:dns1]]` param
/// (the shape cocoon's BuildIPParams emits, one per NIC).
/// One `ip=<addr>::<gw>:<mask>:<host>:<dev>:off[:dns0[:dns1]]` param, as cocoon emits it per NIC.
#[derive(Debug, PartialEq, Eq)]
pub struct IpParam {
pub addr: String,
pub prefix: u8,
pub gateway: Option<String>,
pub dns: Vec<String>,
/// Initramfs-time device name (ethN) — only used to look up the MAC;
/// the persisted unit matches by MAC so rootfs udev renames don't matter.
/// Initramfs-time device name, used only to look up the MAC the persisted unit matches on.
pub device: String,
}

Expand All @@ -54,8 +49,7 @@ pub fn parse(cmdline: &str) -> Result<BootCfg, String> {
debug: false,
trace: false,
};
for tok in cmdline.split_ascii_whitespace() {
let (key, val) = tok.split_once('=').unwrap_or((tok, ""));
for (key, val) in params(cmdline) {
match key {
"cocoon.layers" => {
cfg.layers = val
Expand Down Expand Up @@ -92,32 +86,22 @@ pub fn parse(cmdline: &str) -> Result<BootCfg, String> {
Ok(cfg)
}

/// Debug check for the path where parse() itself failed and cfg.debug is
/// unavailable. Token handling mirrors parse() exactly: same value
/// predicate, last occurrence wins.
/// Debug check for the path where parse() itself failed and cfg.debug is unavailable.
pub fn debug_requested(cmdline: &str) -> bool {
let mut debug = false;
for tok in cmdline.split_ascii_whitespace() {
let (key, val) = tok.split_once('=').unwrap_or((tok, ""));
if key == "sandbox.debug" {
debug = debug_token(val);
}
}
debug
params(cmdline)
.rfind(|&(key, _)| key == "sandbox.debug")
.is_some_and(|(_, val)| debug_token(val))
}

/// Overlay mount data. Layer mountpoints are index-based (/l/0, /l/1, …) so
/// arbitrary serial strings can never break lowerdir parsing (':' or ',').
/// Overlay mount data; index-based layer mountpoints keep a serial string out of lowerdir parsing.
pub fn overlay_data(lower: &[String], cow_dir: &str) -> String {
format!(
"lowerdir={},upperdir={cow_dir}/upper,workdir={cow_dir}/work,index=on,redirect_dir=on,metacopy=on,xino=on",
lower.join(":")
)
}

/// systemd-networkd unit for one static NIC. MAC-matched (device names may
/// change once rootfs udev renames), named 10-<mac>.network so it sorts
/// before the image's 20-wired.network DHCP fallback.
/// systemd-networkd unit for one static NIC, named to sort before the image's DHCP fallback.
pub fn network_unit(ip: &IpParam, mac: &str) -> String {
let mut unit = format!(
"[Match]\nMACAddress={mac}\n\n[Network]\nAddress={}/{}\n",
Expand All @@ -132,6 +116,13 @@ pub fn network_unit(ip: &IpParam, mac: &str) -> String {
unit
}

/// Kernel cmdline tokens as (key, value); a bare token carries an empty value.
fn params(cmdline: &str) -> impl DoubleEndedIterator<Item = (&str, &str)> {
cmdline
.split_ascii_whitespace()
.map(|tok| tok.split_once('=').unwrap_or((tok, "")))
}

/// sandbox.debug value semantics, shared by parse() and debug_requested().
fn debug_token(val: &str) -> bool {
val.is_empty() || val == "1"
Expand Down Expand Up @@ -247,31 +238,17 @@ mod tests {
assert!(debug_requested("sandbox.debug=0 x sandbox.debug"));
}

#[test]
fn debug_requested_matches_parse() {
for tail in [
"",
"sandbox.debug",
"sandbox.debug=",
"sandbox.debug=1",
"sandbox.debug=0",
"sandbox.debug=1 sandbox.debug=0",
] {
let cmdline = format!("cocoon.layers=l0 cocoon.cow=cow {tail}");
assert_eq!(
parse(&cmdline).unwrap().debug,
debug_requested(&cmdline),
"divergence for {tail:?}"
);
}
}

#[test]
fn parse_debug_forms() {
let base = "cocoon.layers=l0 cocoon.cow=cow";
assert!(parse(&format!("{base} sandbox.debug")).unwrap().debug);
assert!(parse(&format!("{base} sandbox.debug=1")).unwrap().debug);
assert!(!parse(&format!("{base} sandbox.debug=0")).unwrap().debug);
assert!(
!parse(&format!("{base} sandbox.debug=1 sandbox.debug=0"))
.unwrap()
.debug
);
}

#[test]
Expand Down
6 changes: 2 additions & 4 deletions boot/init/src/main.rs
Original file line number Diff line number Diff line change
@@ -1,8 +1,6 @@
//! sandbox-init: the entire initramfs userland. Assembles the EROFS + overlay
//! rootfs described on the kernel cmdline and hands off to the real init.
//! sandbox-init: assembles the EROFS+overlay rootfs named on the kernel cmdline, then execs the real init.

// Off Linux only cfg's own tests use it; the bin compiles it dead so
// `cargo test` still covers the cmdline parsing on dev hosts.
// compiled dead off Linux so `cargo test` still covers cmdline parsing on a dev host.
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
mod cfg;

Expand Down
16 changes: 5 additions & 11 deletions boot/init/src/sys.rs
Original file line number Diff line number Diff line change
@@ -1,5 +1,4 @@
//! Thin wrappers over the handful of syscalls the boot path needs; all the
//! unsafe lives here.
//! Thin wrappers over the syscalls the boot path needs; all the unsafe lives here.

use std::ffi::CString;
use std::io;
Expand Down Expand Up @@ -56,12 +55,10 @@ pub fn sethostname(name: &str) -> Result<(), String> {
Ok(())
}

/// Re-root into the assembled overlay. The old-root layer mounts stay pinned
/// by the overlay's references; the ~1MiB initramfs is deliberately not freed
/// (recursive delete would cost more than the memory is worth).
/// Re-roots into the assembled overlay; the ~1MiB initramfs is deliberately never freed.
pub fn switch_root(newroot: &str) -> Result<(), String> {
std::env::set_current_dir(newroot).map_err(|err| format!("chdir {newroot}: {err}"))?;
mount(".", "/", None, libc::MS_MOVE, None)?;
move_mount(".", "/")?;
// SAFETY: the literal is a live 'static CStr for the duration of the call.
if unsafe { libc::chroot(c".".as_ptr()) } != 0 {
return Err(format!("chroot: {}", io::Error::last_os_error()));
Expand All @@ -86,8 +83,7 @@ pub fn exec_init(path: &str) -> String {
io::Error::last_os_error().to_string()
}

/// Route PID 1 stdio to /dev/console. The kernel already did this when the
/// cpio carries the console node; this also covers cpios built without it.
/// Routes PID 1 stdio to /dev/console, covering a cpio built without the console node.
pub fn claim_console() {
// SAFETY: the literal is a live 'static CStr; the fd is validated (>=0)
// before any dup2/close and only closed when it is not already a std stream.
Expand All @@ -104,9 +100,7 @@ pub fn claim_console() {
}
}

/// Terminal state: report, then either hand the operator a shell (debug
/// initramfs) or power off so the orchestrator sees a dead VM immediately
/// instead of a hung boot.
/// Terminal state: hands the operator a debug shell, else powers off so the VM dies visibly.
pub fn fatal(msg: &str, debug: bool) -> ! {
eprintln!("sandbox-init: FATAL: {msg}");
if debug {
Expand Down
8 changes: 0 additions & 8 deletions e2e/e2e_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -379,8 +379,6 @@ func TestWritableVolumeEndToEnd(t *testing.T) {
}
}

// The read-only leg runs second: it is admitted only once the writer's
// release has cleared the dirty marker.
func TestVolumeModeWireShape(t *testing.T) {
scratch := writeVolumeImage(t, "scratch.img", "scratch-bytes")
stack := startTenantStack(t, "node-token", nil,
Expand Down Expand Up @@ -523,8 +521,6 @@ func TestAttachOnlyVolumeWireShape(t *testing.T) {
}
}

// TestDirtyVolumeRefusesReader: the marker a crashed writer leaves behind
// (pre-created here) turns read-only claims into 409s over the wire.
func TestDirtyVolumeRefusesReader(t *testing.T) {
scratch := writeVolumeImage(t, "scratch.img", "scratch-bytes")
if err := os.WriteFile(scratch+".dirty", nil, 0o600); err != nil {
Expand Down Expand Up @@ -601,17 +597,13 @@ func writeVolumeImage(t *testing.T, name, content string) string {
return image
}

// assertNoDirtyMarker fails if the image carries the write-ahead marker: an
// attach-only claim makes no consistency promise, so it must never write one.
func assertNoDirtyMarker(t *testing.T, image, when string) {
t.Helper()
if _, err := os.Stat(image + ".dirty"); !errors.Is(err, os.ErrNotExist) {
t.Errorf("dirty marker at %s: stat=%v, want no marker", when, err)
}
}

// rawClaimResponse decodes the volume entries generically, so the assertion
// is the server's own JSON rather than the SDK's mirror of it.
type rawClaimResponse struct {
ID string `json:"id"`
Token string `json:"token"`
Expand Down
6 changes: 0 additions & 6 deletions e2e/fakeengine_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -16,8 +16,6 @@ import (
"github.com/cocoonstack/sandbox/sdk/go/silkd/silkdtest"
)

// fakeEngine replaces only the cocoon CLI: every "VM" is a silkdtest daemon
// behind a real hybrid-vsock UDS, so the data plane runs production code.
type fakeEngine struct {
real *engine.Engine
dir string
Expand Down Expand Up @@ -81,8 +79,6 @@ func (f *fakeEngine) SnapshotRemove(_ context.Context, _ string) error { return

func (f *fakeEngine) SnapshotList(_ context.Context) ([]string, error) { return nil, nil }

// Hibernate closes the VM's silkd listener and Restore brings a fresh one up,
// mirroring the stop/resume the control plane observes.
func (f *fakeEngine) Hibernate(ctx context.Context, name, _ string) error {
return f.Remove(ctx, name)
}
Expand Down Expand Up @@ -143,8 +139,6 @@ func (f *fakeEngine) SyncGuest(_ context.Context, _ string) error {
return nil
}

// Removals join the trace only for VMs that carried a volume, so warm-pool
// churn cannot perturb the order.
func (f *fakeEngine) volumeOpsLog() []string {
f.mu.Lock()
defer f.mu.Unlock()
Expand Down
Loading