Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion crates/fspy/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@ The injection process is almost identical on both platforms other than the envir

## Linux-specific implementation for fully static binaries

For fully static binaries (such as `esbuild`), `LD_PRELOAD` does not work. In this case, `seccomp_unotify` is used to intercept direct system calls. The handler is implemented in `src/unix/syscall_handler`.
For fully static binaries (such as `esbuild`), `LD_PRELOAD` does not work. In this case, `seccomp_unotify` is used to intercept direct system calls. The handler is implemented in `src/unix/syscall_handler`. It records accesses into the same shared-memory channel as the preload library (`fspy_shared::ipc::channel`).

## Linux musl implementation

Expand Down
1 change: 0 additions & 1 deletion crates/fspy/src/lib.rs
Original file line number Diff line number Diff line change
@@ -1,6 +1,5 @@
pub mod error;

#[cfg(not(target_env = "musl"))]
mod ipc;

#[cfg(unix)]
Expand Down
68 changes: 30 additions & 38 deletions crates/fspy/src/unix/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -4,13 +4,15 @@ mod syscall_handler;
#[cfg(target_os = "macos")]
mod macos_artifacts;

#[cfg(target_os = "linux")]
use std::sync::Arc;
use std::{io, path::Path};

#[cfg(target_os = "linux")]
use fspy_seccomp_unotify::supervisor::supervise;
use fspy_shared::ipc::PathAccess;
#[cfg(not(target_env = "musl"))]
use fspy_shared::ipc::{IpcStr, channel::channel};
use fspy_shared::ipc::IpcStr;
use fspy_shared::ipc::{PathAccess, channel::channel};
use fspy_shared_unix::{
exec::ExecResolveConfig,
payload::{Payload, encode_payload},
Expand All @@ -22,9 +24,10 @@ use syscall_handler::SyscallHandler;
use tokio::task::spawn_blocking;
use tokio_util::sync::CancellationToken;

#[cfg(not(target_env = "musl"))]
use crate::ipc::ChannelAccesses;
use crate::{ChildTermination, Command, TrackedChild, arena::PathAccessArena, error::SpawnError};
use crate::{
ChildTermination, Command, TrackedChild, arena::PathAccessArena, error::SpawnError,
ipc::ChannelAccesses,
};

#[derive(Debug)]
#[cfg_attr(
Expand Down Expand Up @@ -83,13 +86,18 @@ impl SpyImpl {
mut command: Command,
cancellation_token: CancellationToken,
) -> Result<TrackedChild, SpawnError> {
#[cfg(target_os = "linux")]
let supervisor = supervise::<SyscallHandler>().map_err(SpawnError::Supervisor)?;

#[cfg(not(target_env = "musl"))]
let ipc_receiver = channel(crate::ipc::shm_capacity(), allocator_api2::alloc::Global)
.map_err(SpawnError::ChannelCreation)?;

// The supervisor records the accesses it intercepts into the same
// channel as the preload library.
#[cfg(target_os = "linux")]
let supervisor = {
let ipc_sender = Arc::new(ipc_receiver.sender().map_err(SpawnError::ChannelCreation)?);
supervise(move || SyscallHandler::new(Arc::clone(&ipc_sender)))
.map_err(SpawnError::Supervisor)?
};

let payload = Payload {
#[cfg(not(target_env = "musl"))]
ipc_channel_conf: ipc_receiver.conf(),
Expand Down Expand Up @@ -163,26 +171,15 @@ impl SpyImpl {
}
};

let arenas = std::iter::once(exec_resolve_accesses);
// Stop the supervisor and collect path accesses from it.
// Stop the supervisor before closing the channel, so the
// accesses it intercepted are all recorded in the channel.
#[cfg(target_os = "linux")]
let arenas = arenas.chain(
supervisor
.stop()
.await?
.into_iter()
.map(syscall_handler::SyscallHandler::into_arena),
);
let arenas = arenas.collect::<Vec<_>>();
supervisor.stop().await?;

// Close the ipc channel after the child has exited.
// We are not interested in path accesses from descendants after the main child has exited.
#[cfg(not(target_env = "musl"))]
#[cfg(not(target_env = "musl"))]
let path_accesses = ChannelAccesses::try_from(ipc_receiver)
.map(|ipc_accesses| PathAccessIterable { arenas, ipc_accesses });
#[cfg(target_env = "musl")]
let path_accesses = Ok(PathAccessIterable { arenas });
.map(|ipc_accesses| PathAccessIterable { exec_resolve_accesses, ipc_accesses });

io::Result::Ok(ChildTermination { status, path_accesses })
})
Expand All @@ -193,24 +190,19 @@ impl SpyImpl {
}

pub struct PathAccessIterable {
arenas: Vec<PathAccessArena>,
#[cfg(not(target_env = "musl"))]
exec_resolve_accesses: PathAccessArena,
ipc_accesses: ChannelAccesses,
}

impl PathAccessIterable {
/// Iterates over the path accesses in the order they were made.
///
/// Accesses made at the same time by different threads or processes
/// appear in an unspecified order relative to each other.
pub fn iter(&self) -> impl Iterator<Item = PathAccess<'_>> {
let accesses_in_arena =
self.arenas.iter().flat_map(|arena| arena.borrow_accesses().iter()).copied();

#[cfg(not(target_env = "musl"))]
{
let accesses_in_shm = self.ipc_accesses.iter_path_accesses();
accesses_in_shm.chain(accesses_in_arena)
}
#[cfg(target_env = "musl")]
{
accesses_in_arena
}
// Resolving the program happens before the child is spawned.
let accesses_in_arena = self.exec_resolve_accesses.borrow_accesses().iter().copied();
let accesses_in_shm = self.ipc_accesses.iter_path_accesses();
accesses_in_arena.chain(accesses_in_shm)
}
}
31 changes: 16 additions & 15 deletions crates/fspy/src/unix/syscall_handler/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -9,33 +9,30 @@ use std::{
io,
os::unix::ffi::OsStrExt,
path::{Path, PathBuf},
sync::Arc,
};

use fspy_seccomp_unotify::{
impl_handler,
supervisor::handler::arg::{CStrPtr, Caller, Fd},
};
use fspy_shared::ipc::{AccessMode, PathAccess};

use crate::arena::PathAccessArena;
use fspy_shared::ipc::{AccessMode, PathAccess, channel::Sender};

const PATH_MAX: usize = libc::PATH_MAX as usize;

#[derive(Debug)]
/// Records the accesses of intercepted syscalls into the IPC channel.
///
/// The supervisor lets a syscall continue only after its handler returns, so
/// every record is published before the access it describes, as the channel
/// requires of its senders.
pub struct SyscallHandler {
arena: PathAccessArena,
ipc_sender: Arc<Sender>,
path_read_buf: [u8; PATH_MAX],
}

impl Default for SyscallHandler {
fn default() -> Self {
Self { arena: PathAccessArena::default(), path_read_buf: [0; PATH_MAX] }
}
}

impl SyscallHandler {
pub fn into_arena(self) -> PathAccessArena {
self.arena
pub const fn new(ipc_sender: Arc<Sender>) -> Self {
Self { ipc_sender, path_read_buf: [0; PATH_MAX] }
}

fn handle_open(
Expand All @@ -57,7 +54,7 @@ impl SyscallHandler {
}
path = Cow::Owned(resolved_path);
}
self.arena.add(PathAccess {
self.ipc_sender.send(&PathAccess {
mode: match flags & libc::O_ACCMODE {
libc::O_RDWR => AccessMode::READ | AccessMode::WRITE,
libc::O_WRONLY => AccessMode::WRITE,
Expand All @@ -68,9 +65,13 @@ impl SyscallHandler {
Ok(())
}

#[expect(
clippy::needless_pass_by_ref_mut,
reason = "same receiver as `handle_open`, which writes `path_read_buf`"
)]
fn handle_open_dir(&mut self, caller: Caller, fd: Fd) -> io::Result<()> {
let path = fd.get_path(caller)?;
self.arena.add(PathAccess {
self.ipc_sender.send(&PathAccess {
mode: AccessMode::READ_DIR,
path: OsStr::from_bytes(path.as_bytes()).into(),
});
Expand Down
4 changes: 4 additions & 0 deletions crates/fspy/src/windows/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,10 @@ pub struct PathAccessIterable {
}

impl PathAccessIterable {
/// Iterates over the path accesses in the order they were made.
///
/// Accesses made at the same time by different threads or processes
/// appear in an unspecified order relative to each other.
pub fn iter(&self) -> impl Iterator<Item = PathAccess<'_>> {
self.ipc_accesses.iter_path_accesses()
}
Expand Down
10 changes: 8 additions & 2 deletions crates/fspy_seccomp_unotify/src/supervisor/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -57,12 +57,18 @@ impl<H> Supervisor<H> {

/// Creates a new supervisor that listens for seccomp user notifications.
///
/// `new_handler` creates the handler for each notification fd a target
/// process sends.
///
/// # Panics
/// Panics if the seccomp filter cannot be compiled or the target architecture is unsupported.
///
/// # Errors
/// Returns an error if the temporary IPC socket cannot be created.
pub fn supervise<H: SeccompNotifyHandler + Default + Send + 'static>() -> io::Result<Supervisor<H>>
pub fn supervise<H, F>(mut new_handler: F) -> io::Result<Supervisor<H>>
where
H: SeccompNotifyHandler + Send + 'static,
F: FnMut() -> H + Send + 'static,
{
let notify_listener = tempfile::Builder::new()
.prefix("fspy_seccomp_notify")
Expand Down Expand Up @@ -104,7 +110,7 @@ pub fn supervise<H: SeccompNotifyHandler + Default + Send + 'static>() -> io::Re
let notify_fd = unsafe { OwnedFd::from_raw_fd(notify_fd) };
let mut listener = NotifyListener::try_from(notify_fd)?;

let mut handler = H::default();
let mut handler = new_handler();
let mut resp_buf = alloc_seccomp_notif_resp();

join_set.spawn(async move {
Expand Down
2 changes: 1 addition & 1 deletion crates/fspy_seccomp_unotify/tests/arg_types.rs
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ async fn run_in_pre_exec(
) -> Result<Vec<Syscall>, Box<dyn Error>> {
Ok(timeout(Duration::from_secs(5), async move {
let mut cmd = Command::new("/bin/echo");
let supervisor = supervise::<SyscallRecorder>()?;
let supervisor = supervise(SyscallRecorder::default)?;

let payload = supervisor.payload().clone();

Expand Down
72 changes: 65 additions & 7 deletions crates/fspy_shared/src/ipc/channel/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ use std::{env::temp_dir, ffi::OsStr, io, num::NonZeroUsize, path::PathBuf};
use allocator_api2::alloc::Allocator;
use fspy_nostd::Fat;
use fspy_nostd_alloc::OsCString;
use fspy_shm::Mapping;
use fspy_shm::{Mapping, ShmHandle};
use shm_io::{SealError, ShmReader, ShmWriter};

/// Reads the committed frames of a sealed channel; borrows the shared
Expand Down Expand Up @@ -74,7 +74,7 @@ pub fn channel<A: Allocator>(capacity: usize, allocator: A) -> io::Result<Receiv
));
}

Ok(Receiver { keeper, mapping })
Ok(Receiver { keeper, handle, mapping })
}

/// Encodes `path` as an owned NUL-terminated platform C string.
Expand Down Expand Up @@ -263,13 +263,15 @@ pub struct Receiver<A: Allocator> {
/// Keeps the shared memory's backing file alive for as long as senders
/// may attach.
keeper: ShmKeeper<A>,
/// Maps the region again for senders in this process.
handle: ShmHandle,
mapping: Mapping,
}

// SAFETY: `Receiver` only holds the mapping; it accesses it exclusively
// through the `shm_io` protocol in `close`, which synchronizes with senders
// via atomic operations. The mapping's address is stable and independently
// owned.
// SAFETY: `Receiver` only holds the mapping and the handle that maps it; it
// accesses the mapping exclusively through the `shm_io` protocol in `close`,
// which synchronizes with senders via atomic operations. The mapping's
// address is stable and independently owned.
unsafe impl<A: Allocator + Send> Send for Receiver<A> {}

// SAFETY: see the `Send` impl.
Expand All @@ -283,6 +285,30 @@ impl<A: Allocator> Receiver<A> {
ChannelConf { shm_id: IpcStr::from_os_c_str(self.keeper.path.as_c_str()) }
}

/// Creates a sender in this process.
///
/// The sender maps the region separately, so like a sender in another
/// process it stays usable after the receiver closes; its claims are
/// refused from then on.
///
/// # Errors
///
/// When the region cannot be mapped again.
///
/// # Panics
///
/// When the region cannot hold the protocol, which [`channel`] proved
/// it could.
pub fn sender(&self) -> io::Result<Sender> {
let mapping = self.handle.map().map_err(shm_error_to_io)?;
// SAFETY: `mapping` maps the region `channel` created
// zero-initialized and accessed only through the `shm_io` protocol
// since.
let writer = unsafe { ShmWriter::new(mapping, SLOTS) }
.expect("the shared-memory region cannot hold the channel");
Ok(Sender { writer })
}

/// Closes the channel and returns every committed frame, borrowed from
/// the shared mapping that moves into the returned [`FrameReader`].
///
Expand All @@ -307,7 +333,8 @@ impl<A: Allocator> Receiver<A> {
/// When the region cannot hold the protocol, which [`channel`] proved
/// it could before any sender saw it.
pub fn close(self) -> Result<FrameReader, RecordsLost> {
let Self { keeper, mapping } = self;
let Self { keeper, handle, mapping } = self;
drop(handle);
// SAFETY: `mapping` was created zero-initialized by `channel`, its
// address is stable and independently owned, and all attached
// processes access it only through the `shm_io` protocol.
Expand Down Expand Up @@ -509,6 +536,37 @@ mod tests {
);
}

/// A sender the receiver creates in its own process writes into the
/// same region as senders in other processes, and like an attached
/// sender it outlives the close with every claim refused.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn receiver_sender_shares_the_region_with_other_processes() {
let receiver = channel(CAPACITY, Global).unwrap();
let local_sender = receiver.sender().unwrap();

let mut frame = local_sender.writer.claim_frame(NonZeroUsize::new(2).unwrap()).unwrap();
frame.copy_from_slice(&[4, 2]);
frame.finish();

let conf = wincode::serialize(&receiver.conf()).unwrap();
let cmd = command_for_fn!(conf, |conf: Vec<u8>| {
let conf: ChannelConf = wincode::deserialize(&conf).unwrap();
let sender = conf.sender(Global).unwrap();
let mut frame = sender.writer.claim_frame(NonZeroUsize::new(2).unwrap()).unwrap();
frame.copy_from_slice(&[2, 4]);
frame.finish();
});
assert!(std::process::Command::from(cmd).status().unwrap().success());

let frames = receiver.close().unwrap();
assert!(frames.iter().collect::<Vec<_>>() == [&[4, 2], &[2, 4]]);

assert!(
local_sender.writer.claim_frame(NonZeroUsize::new(2).unwrap()).unwrap_err()
== shm_io::ClaimError::Closed
);
}

#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn concurrent_senders() {
let receiver = channel(CAPACITY, Global).unwrap();
Expand Down
1 change: 0 additions & 1 deletion crates/fspy_shared/src/ipc/mod.rs
Original file line number Diff line number Diff line change
@@ -1,4 +1,3 @@
#[cfg(not(target_env = "musl"))]
pub mod channel;
mod ipc_path;
use std::fmt::Debug;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -4,10 +4,7 @@ comment = """
`VP_RUN_INTERNAL_FSPY_SHM_CAPACITY` sizes the shared memory a tracked task reports its file accesses through. The task makes twenty thousand of them, and this run leaves 64 KiB for them once the descriptor table has taken its half-gibibyte of sparse address space, which is nowhere near enough.

The task runs to the end anyway: recording must never stop the program doing the work, so the accesses past that go unrecorded and the process carries on to a clean exit, printing its last line. What the run cannot claim is that it saw every file the task touched, so it is not cached, and the second run says the same rather than replaying an entry built from part of a trace.

Not on musl, which has no preload: those builds collect through the seccomp supervisor, on the runner's own side of the boundary, so they have no shared-memory channel to fill.
"""
cfg = 'not(target_env = "musl")'
steps = [
{ argv = [
"vt",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -4,8 +4,6 @@

The task runs to the end anyway: recording must never stop the program doing the work, so the accesses past that go unrecorded and the process carries on to a clean exit, printing its last line. What the run cannot claim is that it saw every file the task touched, so it is not cached, and the second run says the same rather than replaying an entry built from part of a trace.

Not on musl, which has no preload: those builds collect through the seccomp supervisor, on the runner's own side of the boundary, so they have no shared-memory channel to fill.

## `VP_RUN_INTERNAL_FSPY_SHM_CAPACITY=536936448 vt run -v stat`

512 MiB of table and 64 KiB of room for records
Expand Down
Loading