Skip to content

Commit 85939cf

Browse files
committed
fix(sandbox): reject local control peers when the kernel denies the ingress filter
Binding the boundary control listener attaches a classic BPF filter that drops loopback-interface ingress. Some kernels gate SO_ATTACH_FILTER on CAP_NET_ADMIN, and a sandbox pod holds no capabilities at all, so the attach returns EPERM and every sandbox dies at startup with: bind boundary control listener: Operation not permitted (os error 1) Granting the capability is not an option: the native Linux audit evidence requires an empty capability set, so a sandbox that could attach the filter would fail its own qualification instead. Treat EPERM as "filter unavailable" and move the check to accept time. Any peer the workload can reach the listener from arrives over `lo`, so its source address is either loopback or the address the connection was accepted on; a supervisor runs in a different pod and presents a different address. The serve loop already drops PermissionDenied accepts, which is the same path the Unix peer check uses, and the control channel still requires the per-sandbox mTLS certificate. Kernels that permit the filter keep the stronger standing guarantee. Signed-off-by: divesh <dgude@nvidia.com>
1 parent 834b79a commit 85939cf

2 files changed

Lines changed: 86 additions & 14 deletions

File tree

‎crates/openshell-isolation-interface/src/linux/socket_confinement.rs‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -57,6 +57,24 @@ pub fn reject_loopback_ingress(fd: impl AsFd) -> io::Result<()> {
5757
reject_ingress_interface(fd, index)
5858
}
5959

60+
/// Install loopback-ingress rejection, reporting whether the kernel allowed it.
61+
///
62+
/// Some kernels gate `SO_ATTACH_FILTER` on `CAP_NET_ADMIN`, which a sandbox
63+
/// holding no capabilities cannot satisfy. `Ok(false)` means the filter is
64+
/// unavailable and the caller must reject loopback peers as it accepts them
65+
/// instead; every other error is still a hard failure.
66+
///
67+
/// # Errors
68+
///
69+
/// Returns the kernel error for any failure other than `EPERM`.
70+
pub fn try_reject_loopback_ingress(fd: impl AsFd) -> io::Result<bool> {
71+
match reject_loopback_ingress(fd) {
72+
Ok(()) => Ok(true),
73+
Err(error) if error.raw_os_error() == Some(libc::EPERM) => Ok(false),
74+
Err(error) => Err(error),
75+
}
76+
}
77+
6078
fn reject_ingress_interface(fd: impl AsFd, index: u32) -> io::Result<()> {
6179
// Ancillary loads use the documented negative offset encoding.
6280
let ifindex_offset = (libc::SKF_AD_OFF + libc::SKF_AD_IFINDEX).cast_unsigned();

‎crates/openshell-sandbox/src/boundary_server.rs‎

Lines changed: 68 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -3277,6 +3277,9 @@ mod linux {
32773277
Tcp {
32783278
listener: std::net::TcpListener,
32793279
server_config: Arc<rustls::ServerConfig>,
3280+
/// Set when the kernel refused the loopback-ingress filter, so
3281+
/// `accept` has to reject same-host peers itself.
3282+
reject_local_peers: bool,
32803283
},
32813284
}
32823285

@@ -3305,12 +3308,13 @@ mod linux {
33053308
})
33063309
}
33073310
BoundaryListenerConfig::TlsTcp { address, tls } => {
3308-
let listener = Self::bind_tcp(*address)?;
3311+
let (listener, reject_local_peers) = Self::bind_tcp(*address)?;
33093312
listener.set_nonblocking(true)?;
33103313
let server_config = Arc::new(load_tls_server_config(tls)?);
33113314
Ok(Self::Tcp {
33123315
listener,
33133316
server_config,
3317+
reject_local_peers,
33143318
})
33153319
}
33163320
}
@@ -3320,22 +3324,36 @@ mod linux {
33203324
/// before it listens so workload sockets cannot reach it through
33213325
/// loopback or the pod's own address. Configuration rejects loopback
33223326
/// addresses; tests bind them without the filter.
3323-
fn bind_tcp(address: std::net::SocketAddr) -> io::Result<std::net::TcpListener> {
3327+
///
3328+
/// Returns whether `accept` must reject same-host peers itself, which
3329+
/// is the case on kernels that gate `SO_ATTACH_FILTER` on
3330+
/// `CAP_NET_ADMIN`: a sandbox holds no capabilities, so the standing
3331+
/// filter is unavailable there and the check moves to accept time.
3332+
fn bind_tcp(address: std::net::SocketAddr) -> io::Result<(std::net::TcpListener, bool)> {
33243333
let socket = socket2::Socket::new(
33253334
socket2::Domain::for_address(address),
33263335
socket2::Type::STREAM,
33273336
Some(socket2::Protocol::TCP),
33283337
)?;
33293338
socket.set_cloexec(true)?;
33303339
socket.set_reuse_address(true)?;
3340+
let mut reject_local_peers = false;
33313341
if !address.ip().is_loopback() {
3332-
openshell_isolation_interface::linux::socket_confinement::reject_loopback_ingress(
3333-
&socket,
3334-
)?;
3342+
let filtered =
3343+
openshell_isolation_interface::linux::socket_confinement::try_reject_loopback_ingress(
3344+
&socket,
3345+
)?;
3346+
reject_local_peers = !filtered;
3347+
if reject_local_peers {
3348+
tracing::warn!(
3349+
"Kernel refused the boundary loopback-ingress filter; rejecting \
3350+
same-host control peers on accept instead"
3351+
);
3352+
}
33353353
}
33363354
socket.bind(&address.into())?;
33373355
socket.listen(128)?;
3338-
Ok(socket.into())
3356+
Ok((socket.into(), reject_local_peers))
33393357
}
33403358

33413359
fn bind_vsock(port: u32) -> io::Result<OwnedFd> {
@@ -3428,8 +3446,12 @@ mod linux {
34283446
Self::Tcp {
34293447
listener,
34303448
server_config,
3449+
reject_local_peers,
34313450
} => {
34323451
let (stream, _) = listener.accept()?;
3452+
if *reject_local_peers {
3453+
reject_workload_tcp_peer(&stream)?;
3454+
}
34333455
if let Err(error) = stream.set_nodelay(true) {
34343456
tracing::debug!(%error, "Failed to set boundary TCP_NODELAY");
34353457
}
@@ -3442,6 +3464,24 @@ mod linux {
34423464
}
34433465
}
34443466

3467+
/// Refuse a control connection that originated inside this sandbox.
3468+
///
3469+
/// This is the accept-time stand-in for the loopback-ingress filter. Any
3470+
/// peer the workload can reach the listener from arrives over `lo`, so its
3471+
/// source address is either loopback or one of this network namespace's own
3472+
/// addresses, which is also the address the connection was accepted on. A
3473+
/// supervisor runs in a different pod and so presents a different address.
3474+
/// Rejection here is defence in depth; the control channel still requires
3475+
/// the per-sandbox mTLS certificate.
3476+
fn reject_workload_tcp_peer(stream: &std::net::TcpStream) -> io::Result<()> {
3477+
let peer = stream.peer_addr()?.ip().to_canonical();
3478+
let local = stream.local_addr()?.ip().to_canonical();
3479+
if peer.is_loopback() || peer == local {
3480+
return Err(io::Error::from_raw_os_error(libc::EACCES));
3481+
}
3482+
Ok(())
3483+
}
3484+
34453485
fn reject_workload_unix_peer(stream: &std::os::unix::net::UnixStream) -> io::Result<()> {
34463486
let mut credentials = libc::ucred {
34473487
pid: 0,
@@ -5150,22 +5190,36 @@ mod linux {
51505190
.tcp_local_addr()
51515191
.expect("TLS listener address")
51525192
.port();
5153-
// Loopback and the host's own address both arrive on `lo`; the
5154-
// dropped SYN never completes a handshake.
5155-
let result = std::net::TcpStream::connect_timeout(
5193+
// Loopback and the host's own address both arrive on `lo`. With the
5194+
// standing filter the dropped SYN never completes a handshake; where
5195+
// the kernel gates `SO_ATTACH_FILTER` on `CAP_NET_ADMIN` the
5196+
// handshake completes and `accept` refuses the peer instead. Either
5197+
// way no loopback client reaches a control session.
5198+
let _client = std::net::TcpStream::connect_timeout(
51565199
&std::net::SocketAddr::from(([127, 0, 0, 1], port)),
51575200
Duration::from_millis(300),
51585201
);
5159-
assert!(
5160-
result.is_err(),
5161-
"loopback client reached the control listener"
5162-
);
51635202
assert!(matches!(
51645203
listener.accept().map(|_| ()),
5165-
Err(error) if error.kind() == io::ErrorKind::WouldBlock
5204+
Err(error) if matches!(
5205+
error.kind(),
5206+
io::ErrorKind::WouldBlock | io::ErrorKind::PermissionDenied
5207+
)
51665208
));
51675209
}
51685210

5211+
#[test]
5212+
fn accept_time_fallback_refuses_a_same_host_control_peer() {
5213+
// Stands in for a kernel that gates `SO_ATTACH_FILTER` on
5214+
// `CAP_NET_ADMIN`, where the filter cannot be installed at all.
5215+
let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind listener");
5216+
let address = listener.local_addr().expect("listener address");
5217+
let _client = std::net::TcpStream::connect(address).expect("connect loopback client");
5218+
let (accepted, _) = listener.accept().expect("accept loopback client");
5219+
let error = reject_workload_tcp_peer(&accepted).expect_err("loopback peer refused");
5220+
assert_eq!(error.kind(), io::ErrorKind::PermissionDenied);
5221+
}
5222+
51695223
#[test]
51705224
fn tls_listener_preserves_session_when_control_switches_to_async_streaming() {
51715225
let directory = tempfile::tempdir().expect("temporary directory");

0 commit comments

Comments
 (0)