Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
400 changes: 399 additions & 1 deletion crates/sandlock-core/src/chroot/dispatch.rs

Large diffs are not rendered by default.

37 changes: 37 additions & 0 deletions crates/sandlock-core/src/chroot/resolve.rs
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,25 @@ pub fn confine(virtual_path: &str) -> PathBuf {
result
}

/// Byte-preserving variant of [`confine`] for Linux filesystem names that do
/// not originate in UTF-8 APIs, such as pathname AF_UNIX addresses.
pub fn confine_path(virtual_path: &Path) -> PathBuf {
use std::path::Component;

let mut result = PathBuf::from("/");
for component in virtual_path.components() {
match component {
Component::RootDir | Component::CurDir => {}
Component::ParentDir => {
result.pop();
}
Component::Normal(part) => result.push(part),
Component::Prefix(prefix) => result.push(prefix.as_os_str()),
}
}
result
}

/// Strip chroot root prefix from host path.
/// Returns None if host path is not under chroot root.
pub fn to_virtual_path(chroot_root: &Path, host_path: &Path) -> Option<PathBuf> {
Expand Down Expand Up @@ -197,9 +216,27 @@ pub fn resolve_chroot_mounts(mounts: &[(PathBuf, PathBuf)]) -> Vec<(PathBuf, Pat
#[cfg(test)]
mod tests {
use super::*;
use std::os::unix::ffi::OsStringExt;
use std::os::unix::fs::symlink;
use tempfile::TempDir;

#[test]
fn confine_path_preserves_non_utf8_components_and_clamps_parent() {
let raw = PathBuf::from(std::ffi::OsString::from_vec(
b"/allowed/\xff/../service/../../../../etc".to_vec(),
));
assert_eq!(
confine_path(&raw).as_os_str().as_encoded_bytes(),
b"/etc"
);

let raw = PathBuf::from(std::ffi::OsString::from_vec(b"/allowed/\xff/service".to_vec()));
assert_eq!(
confine_path(&raw).as_os_str().as_encoded_bytes(),
b"/allowed/\xff/service"
);
}

#[test]
fn resolve_chroot_root_none_is_ok_none() {
assert!(resolve_chroot_root(None).unwrap().is_none());
Expand Down
70 changes: 57 additions & 13 deletions crates/sandlock-core/src/freeze.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,11 @@
//! that share the calling task's `mm_struct` via
//! `clone(CLONE_VM)` without `CLONE_THREAD`.
//!
//! `freeze_sandbox_for_execve` closes both classes. When `policy_fn`
//! `freeze_sandbox_for_execve` covers tracked sandbox tasks in both classes.
//! It cannot discover a process outside `ProcessIndex` that independently
//! maps the same `MAP_SHARED` backing object; that writer class remains an
//! explicit limitation unless the backing object is otherwise controlled.
//! When `policy_fn`
//! is active, every fork-like syscall is traced for one ptrace
//! fork/clone/vfork event and the child is registered in
//! `ProcessIndex` before it can run user code. The exec freeze can
Expand All @@ -36,9 +40,12 @@
//! siblings and peers because `de_thread` will not run.
//!
//! Peer threads (different TGID) survive execve. The supervisor must
//! `PTRACE_DETACH` them after `NOTIF_SEND` so they can resume normal
//! execution. The freeze function returns the peer TID list for that
//! purpose; siblings are not returned because they need no follow-up.
//! eventually `PTRACE_DETACH` them so they can resume normal execution.
//! The current implementation detaches them after `NOTIF_SEND` succeeds,
//! but notification-send success is not documented here as proof that the
//! kernel has consumed execve's user-memory arguments. That ordering needs
//! a kernel-observable completion boundary before this can be claimed as a
//! complete TOCTOU guarantee.
//!
//! # Failure modes (strict)
//!
Expand Down Expand Up @@ -192,10 +199,7 @@ fn list_threads_of_tgid(tgid: i32) -> io::Result<Vec<i32>> {
let dir = fs::read_dir(format!("/proc/{}/task", tgid))?;
let mut tids = Vec::new();
for entry in dir {
let entry = match entry {
Ok(e) => e,
Err(_) => continue,
};
let entry = entry?;
let name = entry.file_name();
let name_str = match name.to_str() {
Some(s) => s,
Expand All @@ -208,6 +212,13 @@ fn list_threads_of_tgid(tgid: i32) -> io::Result<Vec<i32>> {
Ok(tids)
}

/// `ESRCH`/`ENOENT` mean the process or task directory disappeared during
/// enumeration. Other errors mean enumeration was incomplete and must not
/// be treated as a successful freeze.
fn task_directory_disappeared(error: &io::Error) -> bool {
matches!(error.raw_os_error(), Some(libc::ESRCH | libc::ENOENT))
}

/// Read the TGID containing `tid`, as an `io::Result` so a missing or
/// unparseable value aborts the freeze instead of silently narrowing it
/// to one task.
Expand Down Expand Up @@ -255,9 +266,13 @@ impl std::fmt::Display for FreezeError {
}
}

/// Freeze every sandbox thread that could mutate execve argv before
/// the supervisor reads it for `policy_fn` and before the kernel
/// re-reads it.
/// Freeze every enumerated, tracked sandbox thread that could mutate
/// execve argv before the supervisor reads it for `policy_fn`.
///
/// This cannot discover external processes outside `ProcessIndex` that
/// independently map the same shared backing object. The caller also must
/// not treat `NOTIF_SEND` success as proof that the kernel has consumed the
/// execve user-memory arguments.
///
/// Walks every TGID in `processes`, enumerates each TGID's threads via
/// `/proc/<tgid>/task/`, and `PTRACE_SEIZE` + `PTRACE_INTERRUPT`s
Expand Down Expand Up @@ -289,10 +304,23 @@ pub(crate) fn freeze_sandbox_for_execve(

for tgid in &tgids {
// /proc/<tgid>/task may disappear if the TGID exited between
// snapshot and walk — that's fine, no threads to freeze.
// snapshot and walk — that's fine. Any other enumeration error
// means the freeze would be incomplete, so roll back and fail.
let tids = match list_threads_of_tgid(*tgid) {
Ok(t) => t,
Err(_) => continue,
Err(e) if task_directory_disappeared(&e) => continue,
Err(e) => {
for t in &sibling_tids {
detach(*t);
}
for t in &peer_tids {
detach(*t);
}
return Err(FreezeError {
error: e,
pending_tids,
});
}
};
for tid in tids {
if tid == caller_tid {
Expand Down Expand Up @@ -420,6 +448,22 @@ mod tests {
assert!(!requires_freeze_on_continue(libc::SYS_connect));
}

#[test]
fn only_disappeared_task_directories_are_ignored() {
assert!(task_directory_disappeared(&io::Error::from_raw_os_error(
libc::ESRCH
)));
assert!(task_directory_disappeared(&io::Error::from_raw_os_error(
libc::ENOENT
)));
assert!(!task_directory_disappeared(&io::Error::from_raw_os_error(
libc::EACCES
)));
assert!(!task_directory_disappeared(&io::Error::from_raw_os_error(
libc::EIO
)));
}

/// Regression test for the cross-process TOCTOU concern raised on
/// issue #27 (Changaco): a peer process in the sandbox — different
/// TGID, possibly aliasing argv pages via shared memory — must also
Expand Down
29 changes: 15 additions & 14 deletions crates/sandlock-core/src/network/connect.rs
Original file line number Diff line number Diff line change
Expand Up @@ -12,11 +12,11 @@ use crate::seccomp::notif::NotifAction;
use crate::sys::structs::{SeccompNotif, ECONNREFUSED};

use super::materialize::{
named_unix_socket_path, parse_ip_from_sockaddr, parse_port_from_sockaddr,
classify_unix_addr, parse_ip_from_sockaddr, parse_port_from_sockaddr, UnixAddr,
set_port_in_sockaddr, sockaddr_is_ipv6,
};
use super::unix::connect_named_unix_on_behalf;
use super::verdict::{layered_destination_verdict, path_under_any};
use super::unix::{connect_named_unix_on_behalf, connect_pinned_unix_on_behalf};
use super::verdict::layered_destination_verdict;
use super::query_socket_protocol;

// ============================================================
Expand Down Expand Up @@ -128,17 +128,16 @@ pub(super) async fn connect_on_behalf(
// EACCES. The decision is made on `addr_bytes` (our immune copy) and we
// never return Continue on the deny path, so it is TOCTOU-safe.
// Abstract sockets (no path) are handled by the Landlock abstract scope.
match named_unix_socket_path(&addr_bytes) {
Some(path) if ctx.policy.has_unix_fs_gate => {
match classify_unix_addr(&addr_bytes) {
UnixAddr::Named(path) if ctx.policy.has_unix_fs_gate => {
if ctx.policy.chroot_root.is_some() {
// Chroot mode: the child's paths are virtual, so a lexical
// check against the (virtual) write grants is consistent,
// and host socket paths are absent from the chroot view
// anyway. Deny unless under a write grant.
if path_under_any(&path, &ctx.policy.chroot_writable) {
NotifAction::Continue
} else {
NotifAction::Errno(libc::EACCES)
let dup_fd = match crate::seccomp::notif::dup_fd_from_pid(notif.pid, sockfd) {
Ok(fd) => fd,
Err(e) => return NotifAction::Errno(e.raw_os_error().unwrap_or(libc::EBADF)),
};
match crate::chroot::dispatch::pin_named_unix_target(notif, &path, ctx).await {
Ok(pinned) => connect_pinned_unix_on_behalf(dup_fd, pinned),
Err(action) => action,
}
} else {
// Non-chroot: resolve the symlink-followed real target and
Expand All @@ -152,6 +151,9 @@ pub(super) async fn connect_on_behalf(
)
}
}
UnixAddr::Malformed(errno) if ctx.policy.has_unix_fs_gate => {
NotifAction::Errno(errno)
}
// Abstract/unnamed socket, non-AF_UNIX family, or gate disabled.
_ => NotifAction::Continue,
}
Expand Down Expand Up @@ -463,4 +465,3 @@ mod tests {
}

}

Loading
Loading