// End-to-end reproduction of the sysbox-fs 0.7.1 listener loss.
// A tracee traps getppid() to a seccomp user-notification listener; the
// supervisor runs sysbox-fs's connHandler loop (tracer.go) under notify_lock
// contention while its polling thread takes signals.
// loop=orig: any revents other than POLLIN ends the loop and closes the
// listener (sysbox-fs 0.7.1 and master).
// loop=patched: a lone POLLERR polls again.
// The tracee reports the first getppid() that fails with ENOSYS, the error a
// system container's processes get from mount() once its listener is closed.
//
// usage: chain <orig|patched> <seconds> <signal-interval-us>
#define _GNU_SOURCE
#include <errno.h>
#include <linux/filter.h>
#include <linux/seccomp.h>
#include <poll.h>
#include <pthread.h>
#include <signal.h>
#include <stdatomic.h>
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/ioctl.h>
#include <sys/prctl.h>
#include <sys/socket.h>
#include <sys/syscall.h>
#include <sys/wait.h>
#include <time.h>
#include <unistd.h>
static int listener, report_fd, patched;
static atomic_int stop, loop_ended;
static atomic_long n_pollerr;
static pthread_t poller;
static void on_sig(int s) { (void)s; }
static void send_fd(int sock, int fd) {
char c = 0;
struct iovec iov = {&c, 1};
char buf[CMSG_SPACE(sizeof(int))];
struct msghdr m = {.msg_iov = &iov, .msg_iovlen = 1, .msg_control = buf, .msg_controllen = sizeof buf};
struct cmsghdr *h = CMSG_FIRSTHDR(&m);
h->cmsg_level = SOL_SOCKET; h->cmsg_type = SCM_RIGHTS; h->cmsg_len = CMSG_LEN(sizeof(int));
memcpy(CMSG_DATA(h), &fd, sizeof(int));
if (sendmsg(sock, &m, 0) < 0) { perror("sendmsg"); exit(1); }
}
static int recv_fd(int sock) {
char c;
struct iovec iov = {&c, 1};
char buf[CMSG_SPACE(sizeof(int))];
struct msghdr m = {.msg_iov = &iov, .msg_iovlen = 1, .msg_control = buf, .msg_controllen = sizeof buf};
if (recvmsg(sock, &m, 0) <= 0) { perror("recvmsg"); exit(1); }
int fd;
memcpy(&fd, CMSG_DATA(CMSG_FIRSTHDR(&m)), sizeof(int));
return fd;
}
static void *tracee_loop(void *a) {
(void)a;
for (;;) {
if (syscall(__NR_getppid) < 0 && errno == ENOSYS) {
if (write(report_fd, "E", 1) != 1) _exit(1);
pause();
}
}
return NULL;
}
static void tracee(int sock) {
struct sock_filter f[] = {
BPF_STMT(BPF_LD | BPF_W | BPF_ABS, offsetof(struct seccomp_data, nr)),
BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_getppid, 0, 1),
BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF),
BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_ALLOW),
};
struct sock_fprog prog = {sizeof f / sizeof f[0], f};
if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0)) { perror("prctl"); exit(1); }
int fd = syscall(__NR_seccomp, SECCOMP_SET_MODE_FILTER, SECCOMP_FILTER_FLAG_NEW_LISTENER, &prog);
if (fd < 0) { perror("seccomp"); exit(1); }
send_fd(sock, fd);
close(fd);
for (int i = 0; i < 16; i++) { pthread_t t; pthread_create(&t, NULL, tracee_loop, NULL); }
pause();
}
// sysbox-fs's process() goroutines: receive and answer.
static void *responder(void *a) {
(void)a;
while (!stop && !loop_ended) {
struct seccomp_notif req; memset(&req, 0, sizeof req);
if (ioctl(listener, SECCOMP_IOCTL_NOTIF_RECV, &req) < 0) continue;
struct seccomp_notif_resp resp = {.id = req.id, .flags = SECCOMP_USER_NOTIF_FLAG_CONTINUE};
ioctl(listener, SECCOMP_IOCTL_NOTIF_SEND, &resp);
}
return NULL;
}
static void *lock_hammer(void *a) {
(void)a;
__u64 id = 1;
while (!stop && !loop_ended) ioctl(listener, SECCOMP_IOCTL_NOTIF_ID_VALID, &id);
return NULL;
}
// tracer.go connHandler, seccompUnusedNotif == true.
static void *conn_handler(void *a) {
(void)a;
for (;;) {
struct pollfd p = {listener, POLLIN, 0};
if (poll(&p, 1, -1) < 0) {
if (errno == EINTR) { if (stop) return NULL; continue; }
break;
}
if (stop) return NULL;
if (p.revents == POLLERR) {
n_pollerr++;
if (patched) continue;
}
if (p.revents != POLLIN) break;
// NotifReceive + process() are done by the responder threads here.
}
loop_ended = 1;
close(listener); // seccompSessionDelete
return NULL;
}
int main(int argc, char **argv) {
if (argc != 4 || (strcmp(argv[1], "orig") && strcmp(argv[1], "patched"))) {
fprintf(stderr, "usage: %s <orig|patched> <seconds> <signal-interval-us>\n", argv[0]);
return 2;
}
patched = !strcmp(argv[1], "patched");
int secs = atoi(argv[2]), interval = atoi(argv[3]);
int sv[2], rp[2];
if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) || pipe(rp)) { perror("setup"); return 1; }
pid_t child = fork();
if (child == 0) { close(sv[0]); close(rp[0]); report_fd = rp[1]; tracee(sv[1]); _exit(0); }
close(sv[1]); close(rp[1]);
listener = recv_fd(sv[0]);
struct sigaction sa; memset(&sa, 0, sizeof sa);
sa.sa_handler = on_sig;
sigaction(SIGUSR1, &sa, NULL);
pthread_t t;
for (int i = 0; i < 4; i++) pthread_create(&t, NULL, responder, NULL);
for (int i = 0; i < 4; i++) pthread_create(&t, NULL, lock_hammer, NULL);
pthread_create(&poller, NULL, conn_handler, NULL);
struct timespec start, now; clock_gettime(CLOCK_MONOTONIC, &start);
double enosys_at = -1;
long sent = 0;
for (;;) {
clock_gettime(CLOCK_MONOTONIC, &now);
double el = (now.tv_sec - start.tv_sec) + (now.tv_nsec - start.tv_nsec) / 1e9;
if (el >= secs) break;
struct pollfd r = {rp[0], POLLIN, 0};
if (poll(&r, 1, 0) == 1) { enosys_at = el; break; }
if (!loop_ended) { pthread_kill(poller, SIGUSR1); sent++; }
usleep(interval);
}
stop = 1;
if (!loop_ended) { pthread_kill(poller, SIGUSR1); pthread_join(poller, NULL); }
kill(child, SIGKILL);
waitpid(child, NULL, 0);
printf("loop=%s signals=%ld POLLERR=%ld listener_closed=%s tracee_ENOSYS=%s",
argv[1], sent, n_pollerr, loop_ended ? "yes" : "no", enosys_at >= 0 ? "yes" : "no");
if (enosys_at >= 0) printf(" (after %.2fs)", enosys_at);
printf("\n");
return 0;
}
Sysbox version: 0.7.1 (sysbox-fs commit c3d2ebc6). The same code is on sysbox-fs master (b3179b68).
Host: Ubuntu 24.04.4 LTS, Linux 7.0.0-1011-aws x86_64.
What happened
A system container running Docker worked for about 19 hours, then every
docker runanddocker pullinside it started failing:straceon the container's containerd showedmount(..., MS_BIND|MS_REC, NULL) = -1 ENOSYSandumount2(...) = -1 ENOSYS. The same bind mount from a newdocker execshell in that container succeeded, and the other Sysbox containers on the host were unaffected. sysbox-fs no longer held the seccomp notification fd it had logged for that container's init process, and it logged nothing when it closed it.Once a filter's listener is gone the kernel answers every trapped syscall with ENOSYS, so every process under the container's init process lost mount/umount emulation for the rest of its life. Restarting the inner dockerd from a
docker exec, which gets its own listener, brought it back.Cause
connHandlerinseccomp/tracer.goends the session, and so closes the seccomp fd, whenever poll() returns anything other than exactly POLLIN:The kernel's
seccomp_notify_poll()(kernel/seccomp.c) returns a lone POLLERR when it is interrupted while waiting for the filter's notify lock:That happens when the polling thread takes a signal while another thread holds the lock to receive or answer a notification. sysbox-fs receives signals all the time (SIGCHLD from its nsenter children, and Go's SIGURG), and on this host all of its threads had them unblocked, so a container with a lot of mount traffic eventually hits it. The fd is still healthy at that point. A session that is really over reports POLLHUP (the filter has no users left) or POLLNVAL.
Reproduction
The program below does not need Sysbox. A child traps getppid() to a user-notification listener. The parent runs connHandler's poll loop while other threads receive and answer notifications and the polling thread takes signals. With the current loop logic the listener is closed on the first POLLERR and the child's getppid() starts returning ENOSYS. With a lone POLLERR treated as transient, the listener survives:
Each variant gave the same result in 3 of 3 runs on the host above. With no signals, the same setup produces no POLLERR at all.
chain.c
Fix
Poll again on a lone POLLERR and keep ending the session on anything else. I'll open a pull request in sysbox-fs that references this issue.
This was investigated with an AI assistant (Claude), which also wrote the reproduction and the fix.