From b31007d97e2e3a1c0827cfb506c5be1592bb1222 Mon Sep 17 00:00:00 2001 From: Erick Cestari Date: Fri, 18 Sep 2026 12:25:00 -0300 Subject: [PATCH 1/3] workloads/lnd: map coverage counters directly onto AFL's map Remap Go's libfuzzer counter section onto the AFL shared memory instead of copying it on every sync. Startup coverage no longer pollutes every map, and crashing or timed-out inputs now report coverage. The trigger/ack pipes stay as a liveness handshake: try_wait still sees a just-crashed LND as running. --- .../src/scenarios/encrypted_bytes.rs | 2 +- smite-scenarios/src/targets.rs | 5 +- smite-scenarios/src/targets/cln.rs | 2 +- smite-scenarios/src/targets/ldk.rs | 4 +- smite-scenarios/src/targets/lnd.rs | 47 +++--- workloads/lnd/Dockerfile | 7 +- workloads/lnd/align.ld | 7 + workloads/lnd/sancov.go | 154 ++++++------------ 8 files changed, 93 insertions(+), 135 deletions(-) create mode 100644 workloads/lnd/align.ld diff --git a/smite-scenarios/src/scenarios/encrypted_bytes.rs b/smite-scenarios/src/scenarios/encrypted_bytes.rs index 0d235b99..2a7609e8 100644 --- a/smite-scenarios/src/scenarios/encrypted_bytes.rs +++ b/smite-scenarios/src/scenarios/encrypted_bytes.rs @@ -66,7 +66,7 @@ impl Scenario for EncryptedBytesScenario { log::debug!("[{:?}] Target responded with pong", start.elapsed()); } - // Check if target is still alive (and trigger coverage sync for LND) + // Check if target is still alive if let Err(e) = self.target.check_alive() { log::debug!("[{:?}] check_alive: {e}", start.elapsed()); return ScenarioResult::Fail(Violation::Crashed.to_string()); diff --git a/smite-scenarios/src/targets.rs b/smite-scenarios/src/targets.rs index e7e3119b..d0b75695 100644 --- a/smite-scenarios/src/targets.rs +++ b/smite-scenarios/src/targets.rs @@ -82,9 +82,8 @@ pub trait Target: Sized { /// Check if target is still alive. Returns `Err(Crashed)` if dead. /// /// Implementation varies by target: - /// - LND: Pipe-based coverage sync (Go can't write to AFL shm directly) - /// - CLN/LDK: Process liveness check (C/Rust AFL instrumentation writes directly) - /// - Eclair: Process liveness check (Java agent writes directly via JNI shmat) + /// - LND: Pipe handshake (a just-crashed LND still looks alive to `try_wait`) + /// - CLN/LDK/Eclair: Process liveness check /// /// # Errors /// diff --git a/smite-scenarios/src/targets/cln.rs b/smite-scenarios/src/targets/cln.rs index d706b8c4..f38dfc15 100644 --- a/smite-scenarios/src/targets/cln.rs +++ b/smite-scenarios/src/targets/cln.rs @@ -1,7 +1,7 @@ //! CLN (Core Lightning) target implementation. //! //! CLN is written in C, so AFL instrumentation (via `afl-clang-fast`) writes -//! directly to shared memory. No coverage pipes are needed. +//! directly to shared memory. //! //! CLN uses a subdaemon architecture: `lightningd` spawns separate binaries //! (`lightning_connectd`, `lightning_gossipd`, etc.). Global subdaemons have diff --git a/smite-scenarios/src/targets/ldk.rs b/smite-scenarios/src/targets/ldk.rs index 9d5a5e05..70d7bf60 100644 --- a/smite-scenarios/src/targets/ldk.rs +++ b/smite-scenarios/src/targets/ldk.rs @@ -1,7 +1,7 @@ //! LDK target implementation. //! -//! Unlike LND, LDK is written in Rust so AFL instrumentation writes directly -//! to shared memory. No coverage pipes are needed. +//! LDK is written in Rust, so AFL instrumentation writes directly to shared +//! memory. use std::fs; use std::io::{BufRead, BufReader}; diff --git a/smite-scenarios/src/targets/lnd.rs b/smite-scenarios/src/targets/lnd.rs index d6613562..1e861286 100644 --- a/smite-scenarios/src/targets/lnd.rs +++ b/smite-scenarios/src/targets/lnd.rs @@ -57,25 +57,22 @@ impl LndConfig { } } -/// Pipes for LND coverage synchronization. +/// Pipes for the LND liveness handshake. /// -/// Go can't write directly to AFL's shared memory, so we use pipes: -/// 1. Scenario writes trigger byte -/// 2. LND copies coverage to AFL shared memory -/// 3. LND writes ack byte -/// 4. If scenario's ack read fails (EOF), LND crashed -struct CoveragePipes { +/// The scenario writes a trigger byte and LND echoes it back as an ack. EOF +/// instead of the ack means LND died. `try_wait` can't replace this: a dying +/// process closes its sockets before it becomes reapable, so right after a +/// crash it still looks alive. +struct LivenessPipes { trigger_write: PipeWriter, ack_read: PipeReader, } -impl CoveragePipes { - /// Triggers LND to copy coverage counters to AFL shared memory. - fn sync(&mut self) -> std::io::Result<()> { +impl LivenessPipes { + /// Blocks until LND acks the trigger byte. Fails if LND died. + fn check(&mut self) -> std::io::Result<()> { let mut buf = [0u8; 1]; - // Write 1 byte to trigger coverage copy self.trigger_write.write_all(&buf)?; - // Wait for coverage copy to finish (EOF = crash) self.ack_read.read_exact(&mut buf)?; Ok(()) } @@ -99,7 +96,7 @@ pub struct LndTarget { lnd: ManagedProcess, #[allow(dead_code)] // bitcoind shuts down on drop bitcoind: ManagedProcess, - coverage_pipes: Option, + liveness_pipes: Option, pubkey: secp256k1::PublicKey, addr: SocketAddr, bitcoin_cli: BitcoinCli, @@ -108,12 +105,12 @@ pub struct LndTarget { } impl LndTarget { - /// Starts LND and waits for it to be ready. Returns the process, coverage + /// Starts LND and waits for it to be ready. Returns the process, liveness /// pipes (if in fuzzing mode), and LND's identity pubkey. fn start_lnd( config: &LndConfig, data_dir: &Path, - ) -> Result<(ManagedProcess, Option, secp256k1::PublicKey), TargetError> { + ) -> Result<(ManagedProcess, Option, secp256k1::PublicKey), TargetError> { log::info!("Starting lnd..."); let lnd_dir = data_dir.join("lnd"); @@ -147,7 +144,7 @@ impl LndTarget { .stdout(Stdio::null()) .stderr(Stdio::null()); - // Set up coverage pipes if in fuzzing mode. We keep all four pipe ends alive + // Set up liveness pipes if in fuzzing mode. We keep all four pipe ends alive // until after spawn so the FDs are valid when the child forks. let pipe_ends = if std::env::var("__AFL_SHM_ID").is_ok() { let (trigger_read, trigger_write) = std::io::pipe()?; @@ -206,10 +203,10 @@ impl LndTarget { None }; - let lnd = ManagedProcess::spawn(&mut cmd, "lnd")?; + let mut lnd = ManagedProcess::spawn(&mut cmd, "lnd")?; // Extract parent-side pipe ends; child-side ends are dropped (closed) here - let coverage_pipes = pipe_ends.map(|(_, trigger_write, ack_read, _)| CoveragePipes { + let liveness_pipes = pipe_ends.map(|(_, trigger_write, ack_read, _)| LivenessPipes { trigger_write, ack_read, }); @@ -218,10 +215,13 @@ impl LndTarget { // block_height matches the initial blocks we generated. log::info!("Waiting for lnd to be ready and synced..."); for _ in 0..120 { + if !lnd.is_running() { + return Err(TargetError::StartFailed("lnd exited during startup".into())); + } if let Ok((pubkey, blockheight, synced_to_chain)) = Self::query_info(config, &lnd_dir) { if blockheight >= bitcoind::INITIAL_BLOCKS && synced_to_chain { log::info!("lnd synced (blockheight={blockheight})"); - return Ok((lnd, coverage_pipes, pubkey)); + return Ok((lnd, liveness_pipes, pubkey)); } log::debug!( "lnd not yet synced (blockheight={blockheight}, synced_to_chain={synced_to_chain})" @@ -287,7 +287,7 @@ impl Target for LndTarget { let (data_path, temp_dir) = bitcoind::resolve_data_dir()?; let (bitcoind, bitcoin_cli) = bitcoind::start(&config.bitcoind_config(), &data_path)?; - let (lnd, coverage_pipes, pubkey) = Self::start_lnd(&config, &data_path)?; + let (lnd, liveness_pipes, pubkey) = Self::start_lnd(&config, &data_path)?; let addr = SocketAddr::from(([127, 0, 0, 1], config.lnd_p2p_port)); log::info!("Both daemons are running, ready to fuzz"); @@ -295,7 +295,7 @@ impl Target for LndTarget { Ok(Self { lnd, bitcoind, - coverage_pipes, + liveness_pipes, pubkey, addr, bitcoin_cli, @@ -320,9 +320,8 @@ impl Target for LndTarget { } fn check_alive(&mut self) -> Result<(), TargetError> { - // If we have coverage pipes, sync triggers coverage copy AND detects crashes - if let Some(pipes) = &mut self.coverage_pipes { - pipes.sync().map_err(|_| TargetError::Crashed)?; + if let Some(pipes) = &mut self.liveness_pipes { + pipes.check().map_err(|_| TargetError::Crashed)?; } else { // No pipes (local mode) - just check process is running if !self.lnd.is_running() { diff --git a/workloads/lnd/Dockerfile b/workloads/lnd/Dockerfile index dc2671ef..4b767241 100644 --- a/workloads/lnd/Dockerfile +++ b/workloads/lnd/Dockerfile @@ -57,9 +57,12 @@ RUN wget https://bitcoincore.org/bin/bitcoin-core-${BITCOIN_VERSION}/bitcoin-${B WORKDIR /lnd RUN cd cmd/lncli && go build -# Copy sancov.go and build LND with coverage instrumentation +# Copy sancov.go and build LND with coverage instrumentation. align.ld +# page-aligns the coverage counters so sancov.go can map them onto AFL's map. COPY ./workloads/lnd/sancov.go /lnd/sancov.go -RUN cd cmd/lnd && CGO_ENABLED=1 go build -v -tags=libfuzzer -gcflags=all=-d=libfuzzer +COPY ./workloads/lnd/align.ld /lnd/align.ld +RUN cd cmd/lnd && CGO_ENABLED=1 go build -v -tags=libfuzzer -gcflags=all=-d=libfuzzer \ + -ldflags="-extldflags=-Wl,-T,/lnd/align.ld" # Copy smite workspace files and build all scenario binaries WORKDIR /smite diff --git a/workloads/lnd/align.ld b/workloads/lnd/align.ld new file mode 100644 index 00000000..73aea89d --- /dev/null +++ b/workloads/lnd/align.ld @@ -0,0 +1,7 @@ +/* Gives Go's libfuzzer counter section its own pages, so sancov.go can map + * AFL's shared memory over them without touching neighboring data. */ +SECTIONS +{ + .go.fuzzcntrs ALIGN(4096) : { *(.go.fuzzcntrs) . = ALIGN(4096); } +} +INSERT AFTER .bss; diff --git a/workloads/lnd/sancov.go b/workloads/lnd/sancov.go index dac197bd..cb7ef257 100644 --- a/workloads/lnd/sancov.go +++ b/workloads/lnd/sancov.go @@ -3,81 +3,64 @@ package lnd /* #cgo CFLAGS: -fPIC +#define _GNU_SOURCE #include #include #include -#include -#include #include +#include -static uint8_t *__coverage_map = NULL; -static size_t __coverage_map_size = 0; -static int __coverage_initialized = 0; - -// Track multiple counter regions (from different instrumented modules). -#define MAX_COUNTER_REGIONS 128 - -struct counter_region { - uint8_t *start; - uint8_t *end; -}; +static void fatal(const char *msg) { + fprintf(stderr, "sancov: %s\n", msg); + exit(1); +} -static struct counter_region __counter_regions[MAX_COUNTER_REGIONS]; -static size_t __num_regions = 0; -static size_t __total_counters = 0; +// Called once by Go's libfuzzer runtime with the bounds of the counter section. +// +// Maps AFL's shared memory over the counter section, so instrumented code +// writes coverage straight into AFL's map. Relies on align.ld giving the +// section its own pages. +void __sanitizer_cov_8bit_counters_init(char *start, char *end) { + static int initialized = 0; + size_t size = (size_t)(end - start); -static int __init_coverage_map(void) { - if (__coverage_initialized) { - return 1; + if (getenv("AFL_DUMP_MAP_SIZE")) { + printf("%zu\n", size); + exit(0); } const char *shm_id_str = getenv("__AFL_SHM_ID"); if (!shm_id_str) { - printf("Warning: __AFL_SHM_ID not set, coverage tracking disabled\n"); - return 0; + return; // Not fuzzing. } - const char *map_size_str = getenv("AFL_MAP_SIZE"); - if (!map_size_str) { - printf("Warning: AFL_MAP_SIZE not set, coverage tracking disabled\n"); - return 0; + if (initialized) { + fatal("counters registered twice"); } + initialized = 1; - int shm_id = atoi(shm_id_str); - if (shm_id < 0) { - printf("Warning: Invalid __AFL_SHM_ID value: %s\n", shm_id_str); - return 0; + size_t page = (size_t)sysconf(_SC_PAGESIZE); + if (page != 4096) { + fatal("page size is not 4096"); } - - __coverage_map = (uint8_t *)shmat(shm_id, NULL, 0); - if (__coverage_map == (void *)-1) { - printf("Warning: Failed to attach to shared memory segment %d\n", shm_id); - __coverage_map = NULL; - return 0; + if ((uintptr_t)start % page != 0) { + fatal("counter section is not page-aligned, was LND linked with align.ld?"); } - __coverage_map_size = (size_t)atoi(map_size_str); - - printf("Coverage map initialized: %p (size: %zu)\n", __coverage_map, - __coverage_map_size); - __coverage_initialized = 1; - return 1; -} - -// Copy all counter regions to AFL shared memory. -void sancov_copy_coverage_to_shmem(void) { - if (!__coverage_map || __num_regions == 0) { - return; + int shm_id = atoi(shm_id_str); + struct shmid_ds ds; + if (shmctl(shm_id, IPC_STAT, &ds) == -1) { + fatal("failed to stat the AFL shared memory segment"); + } + // shmat maps the whole segment, so a larger one would overwrite whatever + // follows the counter section. + if (ds.shm_segsz != size) { + fatal("AFL map size doesn't match the counter section"); } - size_t offset = 0; - for (size_t i = 0; i < __num_regions && offset < __coverage_map_size; ++i) { - size_t region_size = __counter_regions[i].end - __counter_regions[i].start; - size_t copy_size = region_size; - if (offset + copy_size > __coverage_map_size) { - copy_size = __coverage_map_size - offset; - } - memcpy(__coverage_map + offset, __counter_regions[i].start, copy_size); - offset += copy_size; + // Hits so far are lost, but nyx_get_fuzz_input zeroes the map before the + // snapshot anyway. + if (shmat(shm_id, start, SHM_REMAP) == (void *)-1) { + fatal("failed to map the AFL shared memory over the counters"); } } @@ -86,37 +69,6 @@ void __sanitizer_cov_pcs_init(const uintptr_t *pcs_beg, // PC table not used for AFL coverage. } -// Called by Go's libfuzzer instrumentation to register coverage counters. -// May be called multiple times if multiple modules are instrumented. -void __sanitizer_cov_8bit_counters_init(char *start, char *end) { - const char *dump_map_size_str = getenv("AFL_DUMP_MAP_SIZE"); - if (dump_map_size_str) { - printf("%zu\n", (size_t)(end - start)); - exit(0); - } - - __init_coverage_map(); - - if (__num_regions >= MAX_COUNTER_REGIONS) { - fprintf(stderr, "Error: Too many counter regions (max %d)\n", - MAX_COUNTER_REGIONS); - exit(1); - } - size_t region_size = end - start; - __counter_regions[__num_regions].start = (uint8_t *)start; - __counter_regions[__num_regions].end = (uint8_t *)end; - ++__num_regions; - __total_counters += region_size; - - printf("Registered counter region %zu: %zu counters\n", __num_regions, - region_size); - - if (__total_counters > __coverage_map_size) { - printf("Warning: Total counter size (%zu) exceeds map size (%zu)\n", - __total_counters, __coverage_map_size); - } -} - // Empty stubs for comparison tracing hooks. Go's libfuzzer instrumentation // emits calls to these, so we need to provide them to satisfy the linker. // Marked weak so they can be overridden by real implementations if desired. @@ -144,8 +96,10 @@ __attribute__((weak)) void __sanitizer_cov_trace_const_cmp4(uint32_t arg1, __attribute__((weak)) void __sanitizer_cov_trace_const_cmp8(uint64_t arg1, uint64_t arg2) {} -__attribute__((weak)) void __sanitizer_weak_hook_strcmp(const char *s1, - const char *s2) {} +__attribute__((weak)) void __sanitizer_weak_hook_strcmp(void *caller_pc, + const char *s1, + const char *s2, + int result) {} */ import "C" @@ -154,16 +108,16 @@ import ( ) // This file provides coverage tracking for Go programs built with -d=libfuzzer. -// It integrates with AFL's shared memory coverage tracking. +// The C code above maps the coverage counters onto AFL's shared memory, so no +// per-execution work is needed to report coverage. // -// Coverage sync is triggered via pipe IPC: -// - Scenario requests a sync by writing a byte to the trigger fd -// - Coverage loop reads from the trigger fd, copies coverage, and writes to the -// ack fd -// - Scenario reads from ack fd (synchronous handshake) +// The Go code answers the scenario's liveness handshake over pipes: +// - Scenario writes a byte to the trigger fd +// - We echo it on the ack fd +// - Scenario reads the ack; EOF means LND died func init() { - // Only start coverage loop if we're in fuzzing mode + // Only answer the handshake if we're in fuzzing mode if os.Getenv("__AFL_SHM_ID") == "" { return } @@ -171,8 +125,8 @@ func init() { // Any scenario that starts LND as a subprocess must set FDs as follows: // 3: read end of trigger pipe // 4: write end of ack pipe - triggerFile := os.NewFile(uintptr(3), "coverage_trigger") - ackFile := os.NewFile(uintptr(4), "coverage_ack") + triggerFile := os.NewFile(uintptr(3), "liveness_trigger") + ackFile := os.NewFile(uintptr(4), "liveness_ack") go func() { defer triggerFile.Close() @@ -180,15 +134,11 @@ func init() { buf := make([]byte, 1) for { - // Wait for request to copy coverage _, err := triggerFile.Read(buf) if err != nil { return // Pipe closed, exit loop } - C.sancov_copy_coverage_to_shmem() - - // Signal that coverage has been copied ackFile.Write(buf) } }() From 533df780a160dbf0f9d6af895c2089bd93411745 Mon Sep 17 00:00:00 2001 From: Erick Cestari Date: Sat, 26 Sep 2026 21:58:26 -0300 Subject: [PATCH 2/3] smite+smite-scenarios: treat exiting processes as not running try_wait only sees a crashed multithreaded process once it is reapable, after the kernel has closed its sockets, so a scenario can see the connection drop while the target still looks alive. Also check PF_EXITING, which the kernel sets before closing any file. Teardown keeps waiting for the reap through has_exited(). --- smite-scenarios/src/targets/cln.rs | 4 +- smite/src/process.rs | 72 +++++++++++++++++++++++++++--- 2 files changed, 67 insertions(+), 9 deletions(-) diff --git a/smite-scenarios/src/targets/cln.rs b/smite-scenarios/src/targets/cln.rs index f38dfc15..85cc1491 100644 --- a/smite-scenarios/src/targets/cln.rs +++ b/smite-scenarios/src/targets/cln.rs @@ -299,14 +299,14 @@ impl Drop for ClnTarget { // where the CLI has returned but lightningd hasn't exited yet. log::debug!("lightningd: waiting for process to exit"); let deadline = std::time::Instant::now() + Duration::from_secs(5); - while self.cln.is_running() && std::time::Instant::now() < deadline { + while !self.cln.has_exited() && std::time::Instant::now() < deadline { std::thread::sleep(Duration::from_millis(10)); } } else { log::debug!("lightningd: lightning-cli stop failed, falling back to SIGTERM"); } // ManagedProcess::drop handles cleanup. If lightningd already exited, - // is_running() returns false and no signal is sent. If the timeout + // has_exited() returns true and no signal is sent. If the timeout // expired, ManagedProcess sends SIGTERM as a fallback and targets the // whole process group so any lingering subdaemons are cleaned up too. } diff --git a/smite/src/process.rs b/smite/src/process.rs index 276176b3..8f9d9d76 100644 --- a/smite/src/process.rs +++ b/smite/src/process.rs @@ -3,6 +3,7 @@ //! Wraps [`std::process::Child`] with graceful shutdown support //! and automatic cleanup on drop. +use std::fs; use std::io; use std::os::unix::process::CommandExt; use std::process::{Child, Command, ExitStatus}; @@ -56,13 +57,22 @@ impl ManagedProcess { /// Checks if the process is still running (non-blocking). /// - /// Returns `true` if the process is running, `false` if it has exited - /// or if the status cannot be determined (e.g., already reaped). + /// Returns `false` as soon as the process starts exiting. [`has_exited`] + /// alone lags: a crashed multithreaded process only becomes reapable after + /// the kernel has closed its sockets, so peers see it die first. The lag + /// shows on multi-CPU hosts (local mode), not in the Nyx guest (Linux 4.15, + /// one vCPU). + /// + /// [`has_exited`]: Self::has_exited pub fn is_running(&mut self) -> bool { - // Ok(None) = still running - // Ok(Some(_)) = exited - // Err(_) = can't determine (e.g., ECHILD if already reaped) - treat as not running - matches!(self.child.try_wait(), Ok(None)) + !self.has_exited() && !is_exiting(self.pid()) + } + + /// Checks if the process has exited and been reaped (non-blocking). + /// + /// Also returns `true` if the status cannot be determined (e.g., ECHILD). + pub fn has_exited(&mut self) -> bool { + !matches!(self.child.try_wait(), Ok(None)) } /// Attempts graceful shutdown: SIGTERM, wait for timeout, then SIGKILL. @@ -141,7 +151,9 @@ impl ManagedProcess { impl Drop for ManagedProcess { fn drop(&mut self) { - if self.is_running() { + // Not `is_running`: an exiting process still needs reaping, and its + // process group may still need a signal. + if !self.has_exited() { log::debug!( "{}: dropping running process, attempting shutdown", self.name @@ -153,6 +165,30 @@ impl Drop for ManagedProcess { } } +/// `PF_EXITING` from the kernel's `include/linux/sched.h`. +const PF_EXITING: u64 = 0x4; + +/// Checks if the kernel has started tearing down process `pid`. +/// +/// Reads the main thread's `PF_EXITING` flag from `/proc//stat`. The +/// kernel sets it before closing any of the process's files. Returns `false` +/// if `/proc` can't be read; `scripts/setup-nyx.sh` mounts it in the Nyx guest. +fn is_exiting(pid: u32) -> bool { + let Ok(stat) = fs::read_to_string(format!("/proc/{pid}/stat")) else { + return false; + }; + // The command name may contain spaces and parens, so skip past its ')'. + let Some((_, fields)) = stat.rsplit_once(')') else { + return false; + }; + // Fields after the name: state, ppid, pgrp, session, tty_nr, tpgid, flags. + fields + .split_whitespace() + .nth(6) + .and_then(|flags| flags.parse::().ok()) + .is_some_and(|flags| flags & PF_EXITING != 0) +} + /// Sends SIGUSR1 to the process with the given `pid`. /// /// # Errors @@ -234,6 +270,28 @@ mod tests { ); } + #[test] + fn is_exiting_detects_zombie() { + assert!(!is_exiting(std::process::id())); + + // The ") " in the name checks that parsing skips the whole name. + let temp_dir = TempDir::new().unwrap(); + let bin = temp_dir.path().join("a) b"); + std::os::unix::fs::symlink("/bin/true", &bin).unwrap(); + + // Left unreaped, the exited child stays a zombie with PF_EXITING set. + let mut child = Command::new(&bin).spawn().unwrap(); + let pid = child.id(); + let deadline = Instant::now() + Duration::from_secs(1); + while !is_exiting(pid) && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(10)); + } + assert!(is_exiting(pid)); + + child.wait().unwrap(); + assert!(!is_exiting(pid)); + } + fn wait_for_pid_file(path: &Path) -> i32 { let deadline = Instant::now() + Duration::from_secs(1); while Instant::now() < deadline { From d5235ed5c089a79b38c745967b9aea70483c2c97 Mon Sep 17 00:00:00 2001 From: Erick Cestari Date: Sat, 26 Sep 2026 21:58:37 -0300 Subject: [PATCH 3/3] smite-nyx-sys+workloads/lnd: detect LND crashes with the crash handler Replaces the trigger/ack pipe handshake. The handler reports Go fatal errors before LND's sockets close, and reports now carry the Go traceback. ALLOW_HANDLER_OVERRIDE lets Go keep its own signal handlers, which it needs to recover nil dereferences. --- Cargo.lock | 1 - smite-nyx-sys/src/nyx-crash-handler.c | 11 ++- smite-scenarios/Cargo.toml | 1 - smite-scenarios/src/targets.rs | 6 -- smite-scenarios/src/targets/lnd.rs | 117 ++++---------------------- workloads/lnd/Dockerfile | 19 ++++- workloads/lnd/init.sh | 4 + workloads/lnd/sancov.go | 39 +++------ 8 files changed, 57 insertions(+), 141 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index c7a7ea06..f6656f14 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -698,7 +698,6 @@ version = "0.0.0" dependencies = [ "bitcoin", "hex", - "libc", "log", "postcard", "serde", diff --git a/smite-nyx-sys/src/nyx-crash-handler.c b/smite-nyx-sys/src/nyx-crash-handler.c index 5aea00bf..ab642990 100644 --- a/smite-nyx-sys/src/nyx-crash-handler.c +++ b/smite-nyx-sys/src/nyx-crash-handler.c @@ -9,6 +9,9 @@ /// Compile-time options: /// - -DCATCH_SIGNALS: Install our own signal handler for fatal signals, and /// block any attempts to override our signal handler. +/// - -DALLOW_HANDLER_OVERRIDE: With -DCATCH_SIGNALS, let the target replace our +/// handlers. For runtimes like Go that need their own and forward fatal +/// signals they don't handle to ours. /// - -DENABLE_NYX: Use nyx hypercalls to let nyx know that a crash has occured. /// If not set, crash reports are written to /tmp/smite-crash.log. /// - -DASAN_LOG_PATH=: Path to the ASan log file. @@ -34,7 +37,8 @@ #include "nyx.h" #endif -// Must match PANIC_LOG_PATH in workloads/ldk/src/main.rs. +// Must match PANIC_LOG_PATH in workloads/ldk/src/main.rs and +// workloads/lnd/sancov.go. #define PANIC_LOG_PATH "/tmp/smite-panic.log" #define ASAN_LOG_PATH "/tmp/asan.log" #define MAX_CUSTOM_BACKTRACE_SIZE 50 @@ -72,7 +76,8 @@ void append_target_log(const char *path) { return; } - char buffer[0x100000]; + // Static: Go calls our handler on its 32 KiB signal stack. + static char buffer[0x100000]; size_t bytes_read = fread(buffer, 1, sizeof(buffer) - 1, file); fclose(file); @@ -186,6 +191,7 @@ void __assert_perror_fail(int errnum, const char *file, unsigned int line, #ifdef CATCH_SIGNALS +#ifndef ALLOW_HANDLER_OVERRIDE int sigaction(int signum, const struct sigaction *act, struct sigaction *oldact) { int (*_sigaction)(int signum, const struct sigaction *act, @@ -204,6 +210,7 @@ int sigaction(int signum, const struct sigaction *act, return _sigaction(signum, act, oldact); } } +#endif void fault_handler(int signo, siginfo_t *info, void *extra) { char signal_msg[0x1000]; diff --git a/smite-scenarios/Cargo.toml b/smite-scenarios/Cargo.toml index 7ab5f907..ea607c55 100644 --- a/smite-scenarios/Cargo.toml +++ b/smite-scenarios/Cargo.toml @@ -22,7 +22,6 @@ thiserror.workspace = true bitcoin.workspace = true hex = "0.4" -libc.workspace = true serde.workspace = true serde_json.workspace = true tempfile = "3" diff --git a/smite-scenarios/src/targets.rs b/smite-scenarios/src/targets.rs index d0b75695..ce35c0b0 100644 --- a/smite-scenarios/src/targets.rs +++ b/smite-scenarios/src/targets.rs @@ -25,8 +25,6 @@ const CRASH_LOG_PATH: &str = "/tmp/smite-crash.log"; /// In Nyx mode, crashes are reported directly via hypercall and we never get to /// this point. In local mode, the crash handler writes crash data to a file. /// -/// Used by targets that have an external crash handler (CLN, Eclair). -/// /// # Errors /// /// Returns [`TargetError::Crashed`] if the crash log file exists. @@ -81,10 +79,6 @@ pub trait Target: Sized { /// Check if target is still alive. Returns `Err(Crashed)` if dead. /// - /// Implementation varies by target: - /// - LND: Pipe handshake (a just-crashed LND still looks alive to `try_wait`) - /// - CLN/LDK/Eclair: Process liveness check - /// /// # Errors /// /// Returns [`TargetError::Crashed`] if the target has crashed. diff --git a/smite-scenarios/src/targets/lnd.rs b/smite-scenarios/src/targets/lnd.rs index 1e861286..537bf4a4 100644 --- a/smite-scenarios/src/targets/lnd.rs +++ b/smite-scenarios/src/targets/lnd.rs @@ -1,9 +1,7 @@ //! LND target implementation. use std::fs; -use std::io::{PipeReader, PipeWriter, Read, Write}; use std::net::SocketAddr; -use std::os::unix::io::AsRawFd; use std::path::Path; use std::process::{Command, Stdio}; use std::time::Duration; @@ -14,7 +12,7 @@ use smite::bitcoin::BitcoinCli; use smite::process::ManagedProcess; use super::bitcoind; -use super::{Target, TargetError, TargetRpc}; +use super::{Target, TargetError, TargetRpc, check_crash_log}; /// Configuration for the LND target. pub struct LndConfig { @@ -57,27 +55,6 @@ impl LndConfig { } } -/// Pipes for the LND liveness handshake. -/// -/// The scenario writes a trigger byte and LND echoes it back as an ack. EOF -/// instead of the ack means LND died. `try_wait` can't replace this: a dying -/// process closes its sockets before it becomes reapable, so right after a -/// crash it still looks alive. -struct LivenessPipes { - trigger_write: PipeWriter, - ack_read: PipeReader, -} - -impl LivenessPipes { - /// Blocks until LND acks the trigger byte. Fails if LND died. - fn check(&mut self) -> std::io::Result<()> { - let mut buf = [0u8; 1]; - self.trigger_write.write_all(&buf)?; - self.ack_read.read_exact(&mut buf)?; - Ok(()) - } -} - /// RPC handle for interacting with LND node target. #[derive(Debug, Clone)] pub struct LndRpc; @@ -96,7 +73,6 @@ pub struct LndTarget { lnd: ManagedProcess, #[allow(dead_code)] // bitcoind shuts down on drop bitcoind: ManagedProcess, - liveness_pipes: Option, pubkey: secp256k1::PublicKey, addr: SocketAddr, bitcoin_cli: BitcoinCli, @@ -105,12 +81,12 @@ pub struct LndTarget { } impl LndTarget { - /// Starts LND and waits for it to be ready. Returns the process, liveness - /// pipes (if in fuzzing mode), and LND's identity pubkey. + /// Starts LND and waits for it to be ready. Returns the process and LND's + /// identity pubkey. fn start_lnd( config: &LndConfig, data_dir: &Path, - ) -> Result<(ManagedProcess, Option, secp256k1::PublicKey), TargetError> { + ) -> Result<(ManagedProcess, secp256k1::PublicKey), TargetError> { log::info!("Starting lnd..."); let lnd_dir = data_dir.join("lnd"); @@ -144,73 +120,15 @@ impl LndTarget { .stdout(Stdio::null()) .stderr(Stdio::null()); - // Set up liveness pipes if in fuzzing mode. We keep all four pipe ends alive - // until after spawn so the FDs are valid when the child forks. - let pipe_ends = if std::env::var("__AFL_SHM_ID").is_ok() { - let (trigger_read, trigger_write) = std::io::pipe()?; - let (ack_read, ack_write) = std::io::pipe()?; - - let trigger_fd = trigger_read.as_raw_fd(); - let ack_fd = ack_write.as_raw_fd(); - - // SAFETY: This closure runs in the child process after fork, before exec. - // We only call async-signal-safe libc functions (fcntl, dup2, close) and - // create io::Error from last_os_error() which just stores an i32 errno. - unsafe { - use std::os::unix::process::CommandExt; - - cmd.pre_exec(move || { - let mut t = trigger_fd; - let mut a = ack_fd; - - // Move FDs to safe range (>= 10) to avoid conflicts with targets 3 and 4. - // For example, if ack_fd were 3, dup2(trigger_fd, 3) would close it. - if t < 10 { - let new_t = libc::fcntl(t, libc::F_DUPFD, 10); - if new_t == -1 { - return Err(std::io::Error::last_os_error()); - } - libc::close(t); - t = new_t; - } - if a < 10 { - let new_a = libc::fcntl(a, libc::F_DUPFD, 10); - if new_a == -1 { - return Err(std::io::Error::last_os_error()); - } - libc::close(a); - a = new_a; - } - - // Assign to fixed FD numbers that LND's sancov.go expects - if libc::dup2(t, 3) == -1 { - return Err(std::io::Error::last_os_error()); - } - if libc::dup2(a, 4) == -1 { - return Err(std::io::Error::last_os_error()); - } - - // Close the intermediate FDs (dup2 doesn't close the source) - libc::close(t); - libc::close(a); - - Ok(()) - }); - } - - Some((trigger_read, trigger_write, ack_read, ack_write)) - } else { - None - }; + // LD_PRELOAD the crash handler to report crashes immediately (before + // process teardown closes TCP sockets). Go only raises a signal the + // handler sees at GOTRACEBACK=crash; by default it exits with status 2. + if let Ok(handler) = std::env::var("SMITE_CRASH_HANDLER") { + cmd.env("LD_PRELOAD", handler).env("GOTRACEBACK", "crash"); + } let mut lnd = ManagedProcess::spawn(&mut cmd, "lnd")?; - // Extract parent-side pipe ends; child-side ends are dropped (closed) here - let liveness_pipes = pipe_ends.map(|(_, trigger_write, ack_read, _)| LivenessPipes { - trigger_write, - ack_read, - }); - // Wait for LND to be ready and fully synced. We poll getinfo until // block_height matches the initial blocks we generated. log::info!("Waiting for lnd to be ready and synced..."); @@ -221,7 +139,7 @@ impl LndTarget { if let Ok((pubkey, blockheight, synced_to_chain)) = Self::query_info(config, &lnd_dir) { if blockheight >= bitcoind::INITIAL_BLOCKS && synced_to_chain { log::info!("lnd synced (blockheight={blockheight})"); - return Ok((lnd, liveness_pipes, pubkey)); + return Ok((lnd, pubkey)); } log::debug!( "lnd not yet synced (blockheight={blockheight}, synced_to_chain={synced_to_chain})" @@ -287,7 +205,7 @@ impl Target for LndTarget { let (data_path, temp_dir) = bitcoind::resolve_data_dir()?; let (bitcoind, bitcoin_cli) = bitcoind::start(&config.bitcoind_config(), &data_path)?; - let (lnd, liveness_pipes, pubkey) = Self::start_lnd(&config, &data_path)?; + let (lnd, pubkey) = Self::start_lnd(&config, &data_path)?; let addr = SocketAddr::from(([127, 0, 0, 1], config.lnd_p2p_port)); log::info!("Both daemons are running, ready to fuzz"); @@ -295,7 +213,6 @@ impl Target for LndTarget { Ok(Self { lnd, bitcoind, - liveness_pipes, pubkey, addr, bitcoin_cli, @@ -320,13 +237,9 @@ impl Target for LndTarget { } fn check_alive(&mut self) -> Result<(), TargetError> { - if let Some(pipes) = &mut self.liveness_pipes { - pipes.check().map_err(|_| TargetError::Crashed)?; - } else { - // No pipes (local mode) - just check process is running - if !self.lnd.is_running() { - return Err(TargetError::Crashed); - } + check_crash_log()?; + if !self.lnd.is_running() { + return Err(TargetError::Crashed); } Ok(()) } diff --git a/workloads/lnd/Dockerfile b/workloads/lnd/Dockerfile index 4b767241..a8d8db7a 100644 --- a/workloads/lnd/Dockerfile +++ b/workloads/lnd/Dockerfile @@ -78,6 +78,18 @@ RUN set -eu; for f in smite-scenarios/src/bin/lnd_*.rs; do \ cargo build -p smite-scenarios --bin "$(basename $f .rs)" --release --features nyx; \ done +# Build crash handler shared libraries, LD_PRELOADed into LND. Two variants: +# nyx-crash-handler.so - reports crashes via Nyx hypercalls +# crash-handler.so - calls _exit(1), for local reproduction +# -DALLOW_HANDLER_OVERRIDE lets Go install its own signal handlers, which it +# needs to turn nil dereferences into recoverable panics. Go forwards fatal +# signals it doesn't handle to ours. +RUN clang-${LLVM_V} -fPIC -DENABLE_NYX -DCATCH_SIGNALS -DALLOW_HANDLER_OVERRIDE \ + -DNO_PT_NYX -D_GNU_SOURCE \ + smite-nyx-sys/src/nyx-crash-handler.c -ldl -shared -o /nyx-crash-handler.so && \ + clang-${LLVM_V} -fPIC -DCATCH_SIGNALS -DALLOW_HANDLER_OVERRIDE -D_GNU_SOURCE \ + smite-nyx-sys/src/nyx-crash-handler.c -ldl -shared -o /crash-handler.so + # Runtime image FROM debian:bookworm-slim ARG SCENARIO @@ -96,9 +108,14 @@ COPY --from=builder /lnd/cmd/lncli/lncli /usr/local/bin/lncli COPY --from=builder /usr/local/bin/bitcoind /usr/local/bin/bitcoind COPY --from=builder /usr/local/bin/bitcoin-cli /usr/local/bin/bitcoin-cli -# Copy the lnd-scenario binary +# Copy crash handlers and the lnd-scenario binary +COPY --from=builder /nyx-crash-handler.so /crash-handler.so / COPY --from=builder /smite/target/release/lnd_${SCENARIO} /lnd-scenario +# Default to the local crash handler; init.sh overrides with the Nyx version. +# LndTarget forwards this as LD_PRELOAD on lnd only. +ENV SMITE_CRASH_HANDLER=/crash-handler.so + # Copy the init script COPY ./workloads/lnd/init.sh /init.sh RUN chmod +x /init.sh /lnd-scenario diff --git a/workloads/lnd/init.sh b/workloads/lnd/init.sh index a1ad12a9..e4519a4c 100644 --- a/workloads/lnd/init.sh +++ b/workloads/lnd/init.sh @@ -7,4 +7,8 @@ set -eu # Run the LND fuzzing harness export SMITE_NYX=1 export PATH=$PATH:/usr/local/bin + +# Override the default crash handler with the Nyx version, which reports +# crashes via Nyx hypercalls instead of _exit(1). +export SMITE_CRASH_HANDLER=/nyx-crash-handler.so /lnd-scenario > /init.log 2>&1 diff --git a/workloads/lnd/sancov.go b/workloads/lnd/sancov.go index cb7ef257..d18bbe92 100644 --- a/workloads/lnd/sancov.go +++ b/workloads/lnd/sancov.go @@ -104,42 +104,25 @@ __attribute__((weak)) void __sanitizer_weak_hook_strcmp(void *caller_pc, import "C" import ( + "fmt" "os" + "runtime/debug" ) // This file provides coverage tracking for Go programs built with -d=libfuzzer. // The C code above maps the coverage counters onto AFL's shared memory, so no // per-execution work is needed to report coverage. -// -// The Go code answers the scenario's liveness handshake over pipes: -// - Scenario writes a byte to the trigger fd -// - We echo it on the ack fd -// - Scenario reads the ack; EOF means LND died +// Must match PANIC_LOG_PATH in smite-nyx-sys/src/nyx-crash-handler.c. +const panicLogPath = "/tmp/smite-panic.log" + +// Copies fatal error reports where the LD_PRELOADed crash handler reads them, +// so crash reports include the Go traceback. LND's stderr is discarded. func init() { - // Only answer the handshake if we're in fuzzing mode - if os.Getenv("__AFL_SHM_ID") == "" { + f, err := os.Create(fmt.Sprintf("%s.%d", panicLogPath, os.Getpid())) + if err != nil { return } - - // Any scenario that starts LND as a subprocess must set FDs as follows: - // 3: read end of trigger pipe - // 4: write end of ack pipe - triggerFile := os.NewFile(uintptr(3), "liveness_trigger") - ackFile := os.NewFile(uintptr(4), "liveness_ack") - - go func() { - defer triggerFile.Close() - defer ackFile.Close() - - buf := make([]byte, 1) - for { - _, err := triggerFile.Read(buf) - if err != nil { - return // Pipe closed, exit loop - } - - ackFile.Write(buf) - } - }() + defer f.Close() // SetCrashOutput keeps its own duplicate. + debug.SetCrashOutput(f, debug.CrashOptions{}) }