From 457721d7544d03b41c485972ce98813e9108b467 Mon Sep 17 00:00:00 2001 From: Joshua Covington Date: Sat, 26 Sep 2026 10:53:24 +0000 Subject: [PATCH 01/10] jail: drop the deferred path's own mount-propagation remount userns_wait_idmaps() remounts / with mountns_propagation() once the user namespace exists. On the deferred path the mount namespace is still owned by the initial user namespace at that point, so the call returns EPERM and the jail dies with "mount propagation failed". isolate_mountns() applies the same propagation to every CLONE_NEWNS jail before any user namespace handling, so drop the second call. Fixes: 6aa23a8b197e ("jail: give the container's namespaces to its own user namespace") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/jail.c | 6 ------ 1 file changed, 6 deletions(-) diff --git a/jail/jail.c b/jail/jail.c index 4204ee2..a4ec484 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -1853,12 +1853,6 @@ static int userns_wait_idmaps(void) return -1; } - if ((opts.namespace & CLONE_NEWNS) && - mount("none", "/", "none", mountns_propagation(), NULL)) { - ERROR("mount propagation failed: %m\n"); - return -1; - } - return 0; } From 66a33c82f755a835e4df6a467099a4f89a58a85b Mon Sep 17 00:00:00 2001 From: Daniel Golle Date: Mon, 28 Sep 2026 11:07:11 +0100 Subject: [PATCH 02/10] jail: keep locked flags on phase-2 read-only remounts remount_proc_sys_after_unshare() and remount_readonly_now() run in a mount namespace that does not own the mounts copied into it, where MNT_LOCK_{NOSUID,NODEV,NOEXEC} are set. A remount passing only MS_RDONLY asks to clear them and fails with EPERM, so /proc/sys and any OCI readonlyPath stay writable. Add bind_remount_readonly(), which ORs in the flags read back from mountinfo. The atime class is left out of that: path_mount() preserves it on a remount that names no atime flag, which needs no mountinfo lookup and cannot pick the wrong class when the path is not found. Fixes: d381289c2b35 ("jail: mask default sensitive /proc,/sys paths for plain jails") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/fs.c | 32 ++++++++++++++++++++++++-------- jail/fs.h | 1 + jail/jail.c | 10 +++------- 3 files changed, 28 insertions(+), 15 deletions(-) diff --git a/jail/fs.c b/jail/fs.c index 93670fe..8e6ab12 100644 --- a/jail/fs.c +++ b/jail/fs.c @@ -363,9 +363,13 @@ static void mountinfo_unescape(char *s) *w = '\0'; } +/* + * The flags a locked mount will not let a remount clear. The atime class is + * deliberately absent: path_mount() preserves it when the remount names none. + */ static unsigned long mountinfo_current_flags(const char *path) { - unsigned long flags = MS_RELATIME; + unsigned long flags = 0; bool found = false; FILE *f; char *line = NULL; @@ -403,12 +407,6 @@ static unsigned long mountinfo_current_flags(const char *path) this_flags |= MS_NODEV; else if (!strcmp(tok, "noexec")) this_flags |= MS_NOEXEC; - else if (!strcmp(tok, "noatime")) - this_flags |= MS_NOATIME; - else if (!strcmp(tok, "relatime")) - this_flags |= MS_RELATIME; - else if (!strcmp(tok, "nodiratime")) - this_flags |= MS_NODIRATIME; } /* last match wins: it's the topmost/effective entry */ @@ -419,11 +417,29 @@ static unsigned long mountinfo_current_flags(const char *path) fclose(f); if (!found) - return MS_RELATIME; + return 0; return flags; } +/* + * Self-bind @path and remount it read-only, preserving the flags already in + * effect. Mounts copied in by unshare(CLONE_NEWNS) under a userns that does + * not own them are MNT_LOCK_{NOSUID,NODEV,NOEXEC,ATIME}; a remount clearing + * any of those fails with EPERM. + */ +int bind_remount_readonly(const char *path, unsigned long flags) +{ + if (mount(path, path, "bind", MS_BIND | (flags & MS_REC), NULL)) + return -1; + + flags |= MS_REMOUNT | MS_BIND | MS_RDONLY | mountinfo_current_flags(path); + if (mount(path, path, "bind", flags, NULL)) + return -1; + + return 0; +} + static bool fs_userns; void jail_fs_set_userns(bool enabled) diff --git a/jail/fs.h b/jail/fs.h index a7f603f..ba6666c 100644 --- a/jail/fs.h +++ b/jail/fs.h @@ -45,6 +45,7 @@ int fs_mount_enable_idmap(const char *target, uint32_t uid, uint32_t gid); char *resolve_mount_source(const char *source); int add_mount_fd(int fd, const char *target, int error); int mask_path_now(const char *path); +int bind_remount_readonly(const char *path, unsigned long flags); /* open_tree()/mount_setattr() wrappers - no glibc wrappers yet. * Fields must match the kernel's struct mount_attr layout exactly diff --git a/jail/jail.c b/jail/jail.c index a4ec484..f2f2c57 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -2105,9 +2105,7 @@ static int remount_readonly_now(const char *path) if (stat(path, &s)) return 0; /* doesn't exist, nothing to restrict */ - if (mount(path, path, "bind", MS_BIND | MS_REC, NULL)) - return -1; - if (mount(path, path, "bind", MS_REMOUNT | MS_BIND | MS_RDONLY | MS_REC, NULL)) + if (bind_remount_readonly(path, MS_REC)) return -1; DEBUG("read-only path %s\n", path); @@ -2164,10 +2162,8 @@ static void remount_proc_sys_after_unshare(void) if (opts.namespace & CLONE_NEWNET) mount("/proc/sys/net", "/proc/self/net", "bind", MS_BIND, NULL); - if (mount("/proc/sys", "/proc/sys", "bind", MS_BIND, NULL)) - return; - if (mount("/proc/sys", "/proc/sys", "bind", MS_REMOUNT | MS_BIND | MS_RDONLY, NULL)) - WARNING("could not remount /proc/sys read-only\n"); + if (bind_remount_readonly("/proc/sys", 0)) + WARNING("could not remount /proc/sys read-only: %m\n"); if (opts.namespace & CLONE_NEWNET) mount("/proc/self/net", "/proc/sys/net", "bind", MS_MOVE, NULL); From 0925850d765b0293494359600cbc3b59a0fbd786 Mon Sep 17 00:00:00 2001 From: Joshua Covington Date: Sat, 26 Sep 2026 10:37:00 +0000 Subject: [PATCH 03/10] jail: keep locked flags when remounting a noafile mask The file masks for /proc/kcore, /proc/sysrq-trigger and the OCI maskedPaths bind the noafile and remount it with a hard-coded MS_RELATIME. procd mounts /tmp with MS_NOATIME and a user namespace from clone() locks the atime class of everything inherited into the jail's mount namespace, so the remount is EPERM: a critical mask fails mount_all() and an optional one is left read-write. Split the remount half of bind_remount_readonly() into remount_readonly() and use it for both noafile mask sites, naming no atime flag so the kernel keeps the class the bind inherited. Fixes: 6aa23a8b197e ("jail: give the container's namespaces to its own user namespace") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/fs.c | 17 ++++++++++------- jail/fs.h | 1 + 2 files changed, 11 insertions(+), 7 deletions(-) diff --git a/jail/fs.c b/jail/fs.c index 8e6ab12..c5da101 100644 --- a/jail/fs.c +++ b/jail/fs.c @@ -339,7 +339,7 @@ int mask_path_now(const char *path) } else { if (mount(JAIL_NOAFILE, path, "bind", MS_BIND, NULL)) return -1; - if (mount(JAIL_NOAFILE, path, "bind", MS_REMOUNT | MS_BIND | MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, NULL)) + if (remount_readonly(path, MS_NOSUID | MS_NOEXEC | MS_NODEV)) return -1; } @@ -428,16 +428,19 @@ static unsigned long mountinfo_current_flags(const char *path) * not own them are MNT_LOCK_{NOSUID,NODEV,NOEXEC,ATIME}; a remount clearing * any of those fails with EPERM. */ +int remount_readonly(const char *path, unsigned long flags) +{ + flags |= MS_REMOUNT | MS_BIND | MS_RDONLY | mountinfo_current_flags(path); + + return mount(NULL, path, NULL, flags, NULL); +} + int bind_remount_readonly(const char *path, unsigned long flags) { if (mount(path, path, "bind", MS_BIND | (flags & MS_REC), NULL)) return -1; - flags |= MS_REMOUNT | MS_BIND | MS_RDONLY | mountinfo_current_flags(path); - if (mount(path, path, "bind", flags, NULL)) - return -1; - - return 0; + return remount_readonly(path, flags); } static bool fs_userns; @@ -498,7 +501,7 @@ static int do_mount(const char *root, const char *orig_source, const char *targe if (mount(UJAIL_NOAFILE, new, "bind", MS_BIND, NULL)) return error; - if (mount(UJAIL_NOAFILE, new, "bind", MS_REMOUNT | MS_BIND | MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, NULL)) + if (remount_readonly(new, MS_NOSUID | MS_NOEXEC | MS_NODEV)) return error; } diff --git a/jail/fs.h b/jail/fs.h index ba6666c..186bd7b 100644 --- a/jail/fs.h +++ b/jail/fs.h @@ -45,6 +45,7 @@ int fs_mount_enable_idmap(const char *target, uint32_t uid, uint32_t gid); char *resolve_mount_source(const char *source); int add_mount_fd(int fd, const char *target, int error); int mask_path_now(const char *path); +int remount_readonly(const char *path, unsigned long flags); int bind_remount_readonly(const char *path, unsigned long flags); /* open_tree()/mount_setattr() wrappers - no glibc wrappers yet. From ac855770e3b2b0043447e83c87bbcbe52bc7daef Mon Sep 17 00:00:00 2001 From: Daniel Golle Date: Mon, 28 Sep 2026 11:08:48 +0100 Subject: [PATCH 04/10] jail: apply root_map_uid to a joined user namespace The euid taken before clone(), the owner of the staged device nodes, the overlay upper chown, the console hand-over and the devpts gid all key on CLONE_NEWUSER, which -j does not set, and root_map_uid is never derived for a joined namespace. The jail's root then finds the files prepared for it owned by nobody. Derive the mapping by reading /proc//uid_map of the pid -j named: uid_m_show() renders the outside column through the reader's user namespace, so reading it from here yields a host uid at any nesting depth. The namespace fd pins the namespace but not that pid, so the map counts only while /proc//ns/user still names the namespace ujail holds. Those checks key on jail_has_userns(), false for a namespace given as a path, where the map cannot be read and the ownership decisions would act on a guess. The securebits restore instead keys on the namespace being present, so one invocation always gets one securebits set. Fixes: acf36f2777ae ("jail: seteuid before clone(CLONE_NEWUSER)") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/jail.c | 116 ++++++++++++++++++++++++++++++++++++++++++++++++---- 1 file changed, 109 insertions(+), 7 deletions(-) diff --git a/jail/jail.c b/jail/jail.c index f2f2c57..f28193b 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -488,6 +488,23 @@ static inline bool userns_deferred(void) false; } +/* + * The jail's root maps to a host uid we know: our own user namespace, or a + * joined one whose map we managed to read. Without the map the ownership + * decisions below would act on a guess, so they stay as they were. + */ +static bool joined_userns_mapped; + +static inline bool jail_has_userns(void) +{ + return (opts.namespace & CLONE_NEWUSER) || joined_userns_mapped; +} + +static inline bool jail_in_userns(void) +{ + return (opts.namespace & CLONE_NEWUSER) || opts.setns.user != -1; +} + static inline bool has_namespaces(void) { return ((opts.setns.pid != -1) || @@ -1100,7 +1117,7 @@ static struct mknod_args default_devices[] = { static int prepare_jail_dev(void) { struct mknod_args **cur, *curdef; - uid_t base = (opts.namespace & CLONE_NEWUSER) ? opts.root_map_uid : 0; + uid_t base = jail_has_userns() ? opts.root_map_uid : 0; mode_t oldmask = umask(0); char path[PATH_MAX], *tmp; int consfd; @@ -1561,7 +1578,7 @@ static int build_jail_fs(void) return -1; } - jail_fs_set_userns((opts.namespace & CLONE_NEWUSER) || (opts.setns.user != -1)); + jail_fs_set_userns(jail_has_userns()); if (mount_all(jail_root, jail_dev)) { ERROR("mount_all() failed\n"); @@ -3402,7 +3419,7 @@ static void post_start_hook(void) /* restore securebits back to normal (and lock them if not in userns) */ if (opts.capset.apply) { - if (prctl(PR_SET_SECUREBITS, (opts.namespace & CLONE_NEWUSER)?0: + if (prctl(PR_SET_SECUREBITS, jail_in_userns() ? 0 : SECBIT_KEEP_CAPS_LOCKED|SECBIT_NO_SETUID_FIXUP_LOCKED|SECBIT_NOROOT_LOCKED)) { ERROR("prctl(PR_SET_SECUREBITS) failed: %m\n"); free_and_exit(EXIT_FAILURE); @@ -4162,6 +4179,9 @@ static int parseOCIlinuxns(struct blob_attr *msg) * The string argument is the reference PID followed by ':' and a * ',' separated list of namespaces to to join. */ +/* the pid -j resolved the user namespace from, for uid_map_root_read() */ +static pid_t setns_user_pid = -1; + static int jail_join_ns(char *arg) { pid_t pid; @@ -4208,6 +4228,8 @@ static int jail_join_ns(char *arg) return errno?:ESTALE; *setns = fd; + if (nstype == CLONE_NEWUSER) + setns_user_pid = pid; if (etmp) tmp = etmp; @@ -4218,6 +4240,83 @@ static int jail_join_ns(char *arg) return 0; } +/* + * What uid 0 of the joined user namespace maps to on the host. Read from + * here rather than from inside: uid_m_show() renders the outside column + * through the reader's user namespace, so ours gives a host uid at any + * nesting depth, and no process of ours has to enter a namespace the + * caller controls. + */ +static bool uid_map_root_read(pid_t pid, uid_t *host_uid) +{ + uint32_t inside, outside, count; + bool found = false; + char path[32]; + FILE *f; + + snprintf(path, sizeof(path), "/proc/%d/uid_map", pid); + f = fopen(path, "re"); + if (!f) + return false; + + while (fscanf(f, "%u %u %u", &inside, &outside, &count) == 3) { + if (inside || count < 1 || outside >= (uint32_t)-1) + continue; + + *host_uid = outside; + found = true; + break; + } + + fclose(f); + + return found; +} + +/* the pid is not pinned by the namespace fd we hold, so it can be recycled */ +static bool userns_pid_valid(pid_t pid) +{ + struct stat nsst, pidst; + char path[32]; + + if (fstat(opts.setns.user, &nsst)) + return false; + + snprintf(path, sizeof(path), "/proc/%d/ns/user", pid); + if (stat(path, &pidst)) + return false; + + return nsst.st_dev == pidst.st_dev && nsst.st_ino == pidst.st_ino; +} + +static bool userns_root_map(void) +{ + uid_t host_uid; + + if (setns_user_pid < 0) { + WARNING("the joined user namespace was given as a path, so its " + "root uid cannot be determined; keeping the ownership " + "the jail would have had without it\n"); + return false; + } + + if (!uid_map_root_read(setns_user_pid, &host_uid)) { + ERROR("cannot read the uid map of the joined user namespace\n"); + return false; + } + + if (!userns_pid_valid(setns_user_pid)) { + ERROR("pid %d no longer names the joined user namespace\n", + setns_user_pid); + return false; + } + + opts.root_map_uid = host_uid; + DEBUG("root of the joined user namespace is uid %d\n", host_uid); + + return true; +} + static void get_jail_root_user(bool is_gidmap, uint32_t container_id, uint32_t host_id, uint32_t size) { if (container_id == 0 && size >= 1) @@ -6845,6 +6944,9 @@ int main(int argc, char **argv) ulog_open(ULOG_SYSLOG, LOG_DAEMON, "jail"); } + if (opts.setns.user != -1) + joined_userns_mapped = userns_root_map(); + for (credidx = 0; credidx < n_cred_targets; credidx++) { ret = fs_mount_enable_idmap(cred_targets[credidx], opts.pw_uid > 0 ? (uint32_t)opts.pw_uid : 0, @@ -7144,7 +7246,7 @@ static void post_main(struct uloop_timeout *t) add_mount("shm", "/dev/shm", "tmpfs", MS_NOSUID | MS_NOEXEC | MS_NODEV, 0, "mode=1777,size=10%", -1); { - const char *ptsopts = (opts.namespace & CLONE_NEWUSER) ? + const char *ptsopts = jail_has_userns() ? "newinstance,ptmxmode=0666,mode=0620,gid=0" : "newinstance,ptmxmode=0666,mode=0620,gid=5"; @@ -7225,7 +7327,7 @@ static void post_main(struct uloop_timeout *t) free_and_exit(EXIT_FAILURE); } - if (opts.namespace & CLONE_NEWUSER) { + if (jail_has_userns()) { if (opts.overlaydir) { if (chown(opts.overlaydir, opts.root_map_uid, opts.root_map_uid)) { ERROR("chown(%s, %d, %d) failed: %m\n", @@ -7276,12 +7378,12 @@ static void post_main(struct uloop_timeout *t) close(parent_master); } else { console_fd = parent_master; - if ((opts.namespace & CLONE_NEWUSER) && seteuid(0)) { + if (jail_has_userns() && seteuid(0)) { ERROR("seteuid(0) failed: %m\n"); free_and_exit(EXIT_FAILURE); } pass_console(console_fd); - if ((opts.namespace & CLONE_NEWUSER) && seteuid(opts.root_map_uid)) { + if (jail_has_userns() && seteuid(opts.root_map_uid)) { ERROR("seteuid(%d) failed: %m\n", opts.root_map_uid); free_and_exit(EXIT_FAILURE); } From 7f1b26ce3759c11cbdf78bb5cf2a9e7cd3f25c2b Mon Sep 17 00:00:00 2001 From: Daniel Golle Date: Mon, 28 Sep 2026 11:09:47 +0100 Subject: [PATCH 05/10] jail: join an external userns after build_jail_fs() Mounting procfs needs CAP_SYS_ADMIN in the user namespace owning the pid namespace. A jail joining a user namespace by -j entered it at the top of exec_jail(), before any mount, while its pid namespace comes from clone() in the parent and belongs to the initial user namespace, so such a jail with -p failed with EPERM on its own /proc. Build the jail fs privileged and join in enter_userns(), where a deferred user namespace is created, so the join takes the same phase-2 path. That path applied its masks without looking at the result, so make remask_after_unshare() report them and keep the critical ones fatal, as mask_default_paths() does in phase 1. The /proc/sys read-only self-bind follows it into phase 2 on the same terms: fatal, and left alone when the bundle defines /proc/sys itself. Fixes: c482c5de77f4 ("jail: add support for referencing existing namespaces") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/jail.c | 172 +++++++++++++++++++++++++++++++--------------------- 1 file changed, 102 insertions(+), 70 deletions(-) diff --git a/jail/jail.c b/jail/jail.c index f28193b..1aea0b6 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -470,10 +470,17 @@ static char console_slave_name[64]; * Joining a namespace by path needs privilege in the user namespace owning it, * which our own user namespace would take away, so in that case it is created * after the joins instead of by clone(). crun makes the same distinction. + * + * A userns joined with -j (opts.setns.user) drops privilege the same way and + * never owns the pidns clone() created, so procfs must be mounted before + * entering it. Both cases are entered in enter_userns(), after build_jail_fs(). */ static inline bool userns_deferred(void) { - if (!(opts.namespace & CLONE_NEWUSER) || opts.setns.user != -1) + if (opts.setns.user != -1) + return true; + + if (!(opts.namespace & CLONE_NEWUSER)) return false; return (opts.setns.pid != -1) || @@ -1797,13 +1804,14 @@ static void free_and_exit(int ret) exit(ret); } +static int setns_open(unsigned long nstype); static void post_jail_fs(void); static void enter_userns(void); static int userns_wait_idmaps(void); #ifdef CLONE_NEWTIME static int timens_create(void); #endif -static void remask_after_unshare(void); +static int remask_after_unshare(void); static void remount_proc_sys_after_unshare(void); static void enter_jail_fs(void) { @@ -1861,13 +1869,26 @@ static int userns_wait_idmaps(void) return -1; } + return 0; +} + +/* become root in the userns just entered */ +static int userns_become_root(void) +{ if (setregid(0, 0) < 0 || setreuid(0, 0) < 0) { - ERROR("cannot become root in our user namespace: %m\n"); + ERROR("cannot become root in the user namespace: %m\n"); return -1; } + if (setgroups(0, NULL) < 0) { - ERROR("setgroups: %m\n"); - return -1; + /* setgroups=deny is permanent once gid_map is written; only a + * joined userns can have it */ + if (errno != EPERM || opts.setns.user == -1) { + ERROR("setgroups: %m\n"); + return -1; + } + WARNING("setgroups(0, NULL) denied by the joined userns; " + "continuing without dropping supplementary groups\n"); } return 0; @@ -1875,17 +1896,31 @@ static int userns_wait_idmaps(void) static void enter_userns(void) { + int ret; + if (!userns_deferred()) { post_jail_fs(); return; } - if (unshare(CLONE_NEWUSER)) { - ERROR("unshare(CLONE_NEWUSER) failed: %m\n"); - free_and_exit(-1); + if (opts.setns.user != -1) { + /* maps exist already, no handshake */ + ret = setns_open(CLONE_NEWUSER); + if (ret) { + ERROR("failed to join user namespace: %s\n", strerror(ret)); + free_and_exit(-1); + } + } else { + if (unshare(CLONE_NEWUSER)) { + ERROR("unshare(CLONE_NEWUSER) failed: %m\n"); + free_and_exit(-1); + } + + if (userns_wait_idmaps()) + free_and_exit(-1); } - if (userns_wait_idmaps()) + if (userns_become_root()) free_and_exit(-1); #ifdef CLONE_NEWTIME @@ -1899,7 +1934,9 @@ static void enter_userns(void) free_and_exit(-1); } if (opts.namespace & CLONE_NEWNS) { - remask_after_unshare(); + if (remask_after_unshare()) + free_and_exit(-1); + remount_proc_sys_after_unshare(); } @@ -2131,34 +2168,52 @@ static int remount_readonly_now(const char *path) /* re-apply masks inside the container's own mntns: the first pass is * locked once copied in after unshare(CLONE_NEWNS); best-effort */ -static void remask_after_unshare(void) +static int mask_paths_now(const char **paths, bool critical) { const char **p; + + for (p = paths; *p; p++) { + if (!mask_path_now(*p)) + continue; + + if (critical) { + ERROR("failed to mask sensitive path %s: %m\n", *p); + return -1; + } + + WARNING("could not mask optional path %s: %m\n", *p); + } + + return 0; +} + +static int remask_after_unshare(void) +{ char **dp; if (!opts.ocibundle) { - if (opts.procfs) { - for (p = proc_mask_critical; *p; p++) - mask_path_now(*p); - for (p = proc_mask_optional; *p; p++) - mask_path_now(*p); - } + if (opts.procfs && + (mask_paths_now(proc_mask_critical, true) || + mask_paths_now(proc_mask_optional, false))) + return -1; - if (opts.sysfs) { - for (p = sys_mask_critical; *p; p++) - mask_path_now(*p); - } + if (opts.sysfs && mask_paths_now(sys_mask_critical, true)) + return -1; } /* same reason as the default masks above: locked in phase 1, an * OCI bundle's own masks would outlive its unshare(CLONE_NEWNS). */ if (opts.oci_deferred_masked) for (dp = opts.oci_deferred_masked; *dp; dp++) - mask_path_now(*dp); + if (mask_path_now(*dp)) + WARNING("could not mask %s: %m\n", *dp); if (opts.oci_deferred_readonly) for (dp = opts.oci_deferred_readonly; *dp; dp++) - remount_readonly_now(*dp); + if (remount_readonly_now(*dp)) + WARNING("could not make %s read-only: %m\n", *dp); + + return 0; } /* /proc/sys is locked read-only for every procfs jail via a self-bind, @@ -2171,19 +2226,29 @@ static void remount_proc_sys_after_unshare(void) if (!(opts.procfs || opts.ocibundle)) return; + if (mount_is_defined("/proc/sys")) + return; if (stat("/proc/sys", &s)) return; /* the MS_MOVE below needs a mountpoint to move from, or it's a * silent EINVAL. */ - if (opts.namespace & CLONE_NEWNET) - mount("/proc/sys/net", "/proc/self/net", "bind", MS_BIND, NULL); + if ((opts.namespace & CLONE_NEWNET) && + mount("/proc/sys/net", "/proc/self/net", "bind", MS_BIND, NULL)) { + ERROR("failed to stash /proc/sys/net: %m\n"); + free_and_exit(-1); + } - if (bind_remount_readonly("/proc/sys", 0)) - WARNING("could not remount /proc/sys read-only: %m\n"); + if (bind_remount_readonly("/proc/sys", 0)) { + ERROR("failed to remount /proc/sys read-only: %m\n"); + free_and_exit(-1); + } - if (opts.namespace & CLONE_NEWNET) - mount("/proc/self/net", "/proc/sys/net", "bind", MS_MOVE, NULL); + if ((opts.namespace & CLONE_NEWNET) && + mount("/proc/self/net", "/proc/sys/net", "bind", MS_MOVE, NULL)) { + ERROR("failed to restore /proc/sys/net: %m\n"); + free_and_exit(-1); + } } static bool resolve_jail_user_gids(int primary_gid) @@ -3167,24 +3232,17 @@ static int exec_jail(void *arg) } /* - * Joining an external userns drops privilege immediately, so this has - * to run before it. A userns of our own owns the mount namespace it - * was created with and locks everything inherited into it, so there - * the detach neither works nor is needed. + * Must run before enter_userns() drops privilege over the inherited + * mounts. A userns from clone() owns its mntns and has everything + * inherited MNT_LOCKED, so there the detach is neither possible nor + * needed. */ - if ((opts.namespace & CLONE_NEWNS) && - (userns_deferred() || opts.setns.user != -1) && + if ((opts.namespace & CLONE_NEWNS) && userns_deferred() && detach_inherited_mounts()) { ERROR("failed to detach inherited mounts\n"); return EXIT_FAILURE; } - ret = setns_open(CLONE_NEWUSER); - if (ret) { - ERROR("failed to join user namespace: %s\n", strerror(ret)); - return EXIT_FAILURE; - } - buf[0] = 'i'; if (write(pipes[1], buf, 1) < 1) { ERROR("can't write to parent\n"); @@ -3206,14 +3264,9 @@ static int exec_jail(void *arg) false, recv_fds, nrecv, &extroot_idmap_fd, &overlay_idmap_fd); - if ((opts.namespace & CLONE_NEWUSER) && !userns_deferred() && - userns_wait_idmaps()) - return EXIT_FAILURE; - - if (opts.setns.user != -1 && (opts.namespace & CLONE_NEWNS) && - unshare(CLONE_NEWNS)) { - ERROR("unshare(CLONE_NEWNS) failed: %m\n"); - return EXIT_FAILURE; + if ((opts.namespace & CLONE_NEWUSER) && !userns_deferred()) { + if (userns_wait_idmaps() || userns_become_root()) + return EXIT_FAILURE; } if (opts.namespace & CLONE_NEWCGROUP) @@ -3225,27 +3278,6 @@ static int exec_jail(void *arg) free_and_exit(EXIT_FAILURE); } - if (opts.setns.user != -1) { - if (setregid(0, 0) < 0) { - ERROR("setgid\n"); - free_and_exit(EXIT_FAILURE); - } - if (setreuid(0, 0) < 0) { - ERROR("setuid\n"); - free_and_exit(EXIT_FAILURE); - } - if (setgroups(0, NULL) < 0) { - if (errno != EPERM) { - ERROR("setgroups\n"); - free_and_exit(EXIT_FAILURE); - } - WARNING("setgroups(0, NULL) denied by the joined " - "userns (setgroups=deny is permanent once a " - "gid_map is written); continuing without " - "dropping supplementary groups\n"); - } - } - #ifdef CLONE_NEWTIME if ((opts.namespace & CLONE_NEWTIME) && opts.setns.time == -1 && !userns_deferred() && timens_create()) From 88fe395b2cd5e368f7b67770eeb82a0dadcf1d6b Mon Sep 17 00:00:00 2001 From: Daniel Golle Date: Mon, 28 Sep 2026 11:11:36 +0100 Subject: [PATCH 06/10] jail: fail on an unresolvable -j specification jail_join_ns() reports a stale pid or an unknown namespace type, but the return value is discarded, so the jail starts with none of the requested namespaces and one that asked for a user namespace runs as host root in the initial one. Report the error and refuse to start. Two of the names it accepts never resolved: "mount" and "network" were used verbatim in /proc//ns/, where the links are "mnt" and "net", and "mnt" was not accepted at all, so the mount namespace could not be joined under any spelling. Keep the accepted names and their links in one table. Fixes: c482c5de77f4 ("jail: add support for referencing existing namespaces") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/jail.c | 69 +++++++++++++++++++++++++++++++++++------------------ 1 file changed, 46 insertions(+), 23 deletions(-) diff --git a/jail/jail.c b/jail/jail.c index 1aea0b6..f8ebce2 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -4130,27 +4130,45 @@ static const struct blobmsg_policy oci_linux_namespace_policy[] = { [OCI_LINUX_NAMESPACE_PATH] = { "path", BLOBMSG_TYPE_STRING }, }; -static int resolve_nstype(char *type) { - if (!strcmp("pid", type)) - return CLONE_NEWPID; - else if (!strcmp("network", type)) - return CLONE_NEWNET; - else if (!strcmp("net", type)) - return CLONE_NEWNET; - else if (!strcmp("mount", type)) - return CLONE_NEWNS; - else if (!strcmp("ipc", type)) - return CLONE_NEWIPC; - else if (!strcmp("uts", type)) - return CLONE_NEWUTS; - else if (!strcmp("user", type)) - return CLONE_NEWUSER; - else if (!strcmp("cgroup", type)) - return CLONE_NEWCGROUP; - else if (!strcmp("time", type)) - return CLONE_NEWTIME; - else - return 0; +/* the name -j accepts, and the /proc//ns/ link it actually lives under */ +static const struct { + const char *name; + const char *link; + int nstype; +} ns_types[] = { + { "pid", "pid", CLONE_NEWPID }, + { "net", "net", CLONE_NEWNET }, + { "network", "net", CLONE_NEWNET }, + { "mnt", "mnt", CLONE_NEWNS }, + { "mount", "mnt", CLONE_NEWNS }, + { "ipc", "ipc", CLONE_NEWIPC }, + { "uts", "uts", CLONE_NEWUTS }, + { "user", "user", CLONE_NEWUSER }, + { "cgroup", "cgroup", CLONE_NEWCGROUP }, + { "time", "time", CLONE_NEWTIME }, + { NULL, NULL, 0 }, +}; + +static int resolve_nstype(const char *type) +{ + int i; + + for (i = 0; ns_types[i].name; i++) + if (!strcmp(ns_types[i].name, type)) + return ns_types[i].nstype; + + return 0; +} + +static const char *nstype_link(int nstype) +{ + int i; + + for (i = 0; ns_types[i].name; i++) + if (ns_types[i].nstype == nstype) + return ns_types[i].link; + + return NULL; } static int parseOCIlinuxns(struct blob_attr *msg) @@ -4250,7 +4268,7 @@ static int jail_join_ns(char *arg) if (*setns != -1) return ENOTUNIQ; - if (asprintf(&nspath, "/proc/%d/ns/%s", pid, tmp) < 0) + if (asprintf(&nspath, "/proc/%d/ns/%s", pid, nstype_link(nstype)) < 0) return ENOMEM; fd = open(nspath, O_RDONLY); @@ -6740,7 +6758,12 @@ int main(int argc, char **argv) opts.hostname = strdup(optarg); break; case 'j': - jail_join_ns(optarg); + ret = jail_join_ns(optarg); + if (ret) { + ERROR("failed to join namespace: %s\n", + strerror(ret)); + return EXIT_FAILURE; + } break; case 'b': if (!opts.ocibundle) From 638c2ecfaf8a6f1a64614feadd550922ccef12b2 Mon Sep 17 00:00:00 2001 From: Daniel Golle Date: Mon, 28 Sep 2026 11:12:50 +0100 Subject: [PATCH 07/10] jail: let a joined namespace replace the created one Every non-OCI jail has CLONE_NEWPID and CLONE_NEWIPC added to its namespace set, and a join refused any namespace that set already named. procd appends -j after the flags that populate it, so a jail combining procfs, sysfs or netns with a -j entry for the same namespace was rejected, and clone(CLONE_NEWPID) after setns(CLONE_NEWPID) is EINVAL, so "-j :pid" never got past clone() either. Clear the create bit of every joined namespace once both the options and the bundle have been read, and fail when the pid namespace cannot be entered. A jail root has to be built in a mount namespace ujail owns, so an entry naming one is refused where a jail filesystem was asked for, and linux.namespaces entries keep their per-type uniqueness. Fixes: c482c5de77f4 ("jail: add support for referencing existing namespaces") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/jail.c | 44 ++++++++++++++++++++++++++++++++++++-------- 1 file changed, 36 insertions(+), 8 deletions(-) diff --git a/jail/jail.c b/jail/jail.c index f8ebce2..9c8324e 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -4171,6 +4171,9 @@ static const char *nstype_link(int nstype) return NULL; } +/* linux.namespaces entries are unique by type */ +static int oci_ns_seen; + static int parseOCIlinuxns(struct blob_attr *msg) { struct blob_attr *tb[__OCI_LINUX_NAMESPACE_MAX]; @@ -4187,18 +4190,20 @@ static int parseOCIlinuxns(struct blob_attr *msg) if (!nstype) return EINVAL; - if (opts.namespace & nstype) - return ENOTUNIQ; - setns = get_namespace_fd(nstype); if (!setns) return EFAULT; - if (*setns != -1) + if (oci_ns_seen & nstype) return ENOTUNIQ; + oci_ns_seen |= nstype; + if (tb[OCI_LINUX_NAMESPACE_PATH]) { + if (*setns != -1) + return ENOTUNIQ; + DEBUG("opening existing %s namespace from path %s\n", blobmsg_get_string(tb[OCI_LINUX_NAMESPACE_TYPE]), blobmsg_get_string(tb[OCI_LINUX_NAMESPACE_PATH])); @@ -4257,9 +4262,6 @@ static int jail_join_ns(char *arg) if (!nstype) return EINVAL; - if (opts.namespace & nstype) - return ENOTUNIQ; - setns = get_namespace_fd(nstype); if (!setns) @@ -6999,6 +7001,26 @@ int main(int argc, char **argv) ulog_open(ULOG_SYSLOG, LOG_DAEMON, "jail"); } + if (opts.setns.ns != -1 && (opts.namespace & CLONE_NEWNS)) { + ERROR("cannot build a jail filesystem in a joined mount namespace\n"); + ret = EXIT_FAILURE; + goto errout; + } + + /* clone(CLONE_NEWPID) after setns(CLONE_NEWPID) is EINVAL */ + if (opts.setns.net != -1) + opts.namespace &= ~CLONE_NEWNET; + if (opts.setns.ipc != -1) + opts.namespace &= ~CLONE_NEWIPC; + if (opts.setns.uts != -1) + opts.namespace &= ~CLONE_NEWUTS; + if (opts.setns.pid != -1) + opts.namespace &= ~CLONE_NEWPID; + if (opts.setns.user != -1) + opts.namespace &= ~CLONE_NEWUSER; + if (opts.setns.cgroup != -1) + opts.namespace &= ~CLONE_NEWCGROUP; + if (opts.setns.user != -1) joined_userns_mapped = userns_root_map(); @@ -7235,6 +7257,7 @@ static void netifd_restart_watch(void) static void post_main(struct uloop_timeout *t) { int child_status; + int nsret; if (apply_rlimits()) { ERROR("error applying resource limits\n"); @@ -7366,7 +7389,12 @@ static void post_main(struct uloop_timeout *t) if (opts.setns.pid != -1) { pidns_fd = ns_open_pid("pid", getpid()); - setns_open(CLONE_NEWPID); + nsret = setns_open(CLONE_NEWPID); + if (nsret) { + ERROR("failed to join pid namespace: %s\n", + strerror(nsret)); + free_and_exit(EXIT_FAILURE); + } } else { pidns_fd = -1; } From 56e1a2032e0d4accab6c591d05333598b3c16b88 Mon Sep 17 00:00:00 2001 From: Joshua Covington Date: Sun, 27 Sep 2026 19:06:54 +0000 Subject: [PATCH 08/10] jail: factor out the MS_* to MOUNT_ATTR_* translation Move the mount flag translation of idmap_tree_fd() into mountflags_to_attr(). No functional change: MOUNT_ATTR_RELATIME is 0 and __ATIME is only cleared when an atime flag is requested, so a clone with MNT_LOCK_ATIME keeps its atime mode. Signed-off-by: Joshua Covington --- jail/fs.c | 47 ++++++++++++++++++++++++++++------------------- 1 file changed, 28 insertions(+), 19 deletions(-) diff --git a/jail/fs.c b/jail/fs.c index c5da101..ca31293 100644 --- a/jail/fs.c +++ b/jail/fs.c @@ -126,6 +126,31 @@ unsigned long detect_atime_flag(const char *mountpoint) #define MOUNT_ATTR_NODIRATIME 0x00000080 #endif +/* MS_* -> MOUNT_ATTR_* for fsmount()/mount_setattr() */ +static unsigned mountflags_to_attr(unsigned long mountflags) +{ + unsigned attr = 0; + + if (mountflags & MS_RDONLY) + attr |= MOUNT_ATTR_RDONLY; + if (mountflags & MS_NOSUID) + attr |= MOUNT_ATTR_NOSUID; + if (mountflags & MS_NODEV) + attr |= MOUNT_ATTR_NODEV; + if (mountflags & MS_NOEXEC) + attr |= MOUNT_ATTR_NOEXEC; + if (mountflags & MS_NODIRATIME) + attr |= MOUNT_ATTR_NODIRATIME; + if (mountflags & MS_NOATIME) + attr |= MOUNT_ATTR_NOATIME; + else if (mountflags & MS_STRICTATIME) + attr |= MOUNT_ATTR_STRICTATIME; + else + attr |= MOUNT_ATTR_RELATIME; + + return attr; +} + int sys_openat2(int dfd, const char *path, struct open_how *how, size_t size) { return syscall(SYS_openat2, dfd, path, how, size); @@ -1335,26 +1360,10 @@ static int idmap_tree_fd(const char *source, int source_fd, int userns_fd, return -1; } - attr.attr_set = MOUNT_ATTR_IDMAP; - if (mountflags & MS_RDONLY) - attr.attr_set |= MOUNT_ATTR_RDONLY; - if (mountflags & MS_NOSUID) - attr.attr_set |= MOUNT_ATTR_NOSUID; - if (mountflags & MS_NODEV) - attr.attr_set |= MOUNT_ATTR_NODEV; - if (mountflags & MS_NOEXEC) - attr.attr_set |= MOUNT_ATTR_NOEXEC; - if (mountflags & MS_NODIRATIME) - attr.attr_set |= MOUNT_ATTR_NODIRATIME; - if (mountflags & (MS_NOATIME | MS_RELATIME | MS_STRICTATIME)) { + /* keep the (possibly locked) atime mode unless one is requested */ + attr.attr_set = MOUNT_ATTR_IDMAP | mountflags_to_attr(mountflags); + if (mountflags & (MS_NOATIME | MS_RELATIME | MS_STRICTATIME)) attr.attr_clr |= MOUNT_ATTR__ATIME; - if (mountflags & MS_NOATIME) - attr.attr_set |= MOUNT_ATTR_NOATIME; - else if (mountflags & MS_STRICTATIME) - attr.attr_set |= MOUNT_ATTR_STRICTATIME; - else - attr.attr_set |= MOUNT_ATTR_RELATIME; - } attr.propagation = propagation; attr.userns_fd = userns_fd; From 8c9df2b8ae3216c2c7339f4e0603c1abbe608216 Mon Sep 17 00:00:00 2001 From: Joshua Covington Date: Sun, 27 Sep 2026 19:07:46 +0000 Subject: [PATCH 09/10] jail: mount sysfs in the parent when the jail lacks a netns Mounting sysfs needs CAP_SYS_ADMIN in the userns owning the netns. A jail with a userns from clone() and no netns of its own gets EPERM for -s or an OCI sysfs mount and does not start. fsmount() sysfs in the parent before clone() with the queued flags and leave the fd for do_mount_fd() to move_mount(). The mount is not MNT_LOCK_READONLY: only lock_mnt_tree() on a copy into a less privileged userns sets it, and a clone-time userns makes no such copy. A jail keeping CAP_SYS_ADMIN can remount /sys read-write. Fixes: 6aa23a8b197e ("jail: give the container's namespaces to its own user namespace") Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/fs.c | 56 +++++++++++++++++++++++++++++++++++++++++++++++++++-- jail/fs.h | 23 ++++++++++++++++++++++ jail/jail.c | 5 +++++ 3 files changed, 82 insertions(+), 2 deletions(-) diff --git a/jail/fs.c b/jail/fs.c index ca31293..b3c35ea 100644 --- a/jail/fs.c +++ b/jail/fs.c @@ -337,6 +337,21 @@ int sys_move_mount(int from_dfd, const char *from_path, int to_dfd, const char * return syscall(SYS_move_mount, from_dfd, from_path, to_dfd, to_path, flags); } +int sys_fsopen(const char *fsname, unsigned flags) +{ + return syscall(SYS_fsopen, fsname, flags); +} + +int sys_fsconfig(int fd, unsigned cmd, const char *key, const void *value, int aux) +{ + return syscall(SYS_fsconfig, fd, cmd, key, value, aux); +} + +int sys_fsmount(int fd, unsigned flags, unsigned attr_flags) +{ + return syscall(SYS_fsmount, fd, flags, attr_flags); +} + int sys_mount_setattr(int dfd, const char *path, unsigned flags, struct ujail_mount_attr *attr, size_t size) { return syscall(SYS_mount_setattr, dfd, path, flags, attr, size); @@ -788,6 +803,39 @@ int add_mount_fd(int fd, const char *target, int error) return 0; } +/* + * sysfs needs CAP_SYS_ADMIN in the userns owning the netns. fsmount() the + * queued sysfs entries here, before clone(), for do_mount_fd(). + */ +int premount_sysfs(void) +{ + struct mount *m; + int fsfd, mfd; + + list_for_each_entry(m, &mounts_order, list) { + if (!m->filesystemtype || strcmp(m->filesystemtype, "sysfs") || + m->source_fd >= 0) + continue; + + fsfd = sys_fsopen("sysfs", FSOPEN_CLOEXEC); + if (fsfd < 0) + return -1; + if (sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0)) { + close(fsfd); + return -1; + } + mfd = sys_fsmount(fsfd, FSMOUNT_CLOEXEC, mountflags_to_attr(m->mountflags)); + close(fsfd); + if (mfd < 0) + return -1; + + m->source_fd = mfd; + DEBUG("pre-mounted sysfs for %s as fd:%d\n", m->target, mfd); + } + + return 0; +} + int add_mount_volume(const char *source, const char *target, int error) { struct mount *m; @@ -1609,6 +1657,7 @@ static int mount_one(const char *jailroot, const char *jail_dev, struct mount *m { char devtarget[PATH_MAX]; struct mount *outer; + int ret; if (m->mounted) return 0; @@ -1639,8 +1688,11 @@ static int mount_one(const char *jailroot, const char *jail_dev, struct mount *m if (m->idmap) return do_idmap_mount(jailroot, m) ? -1 : 0; - if (m->source_fd >= 0) - return do_mount_fd(jailroot, m->source_fd, m->target, m->error) ? -1 : 0; + if (m->source_fd >= 0) { + ret = do_mount_fd(jailroot, m->source_fd, m->target, m->error); + m->source_fd = -1; + return ret ? -1 : 0; + } if (do_mount(jailroot, m->source, m->target, m->filesystemtype, m->mountflags, m->propflags, m->optstr, m->error, m->inner, m->source_fd)) diff --git a/jail/fs.h b/jail/fs.h index 186bd7b..89eadad 100644 --- a/jail/fs.h +++ b/jail/fs.h @@ -59,6 +59,29 @@ int sys_open_tree(int dfd, const char *path, unsigned flags); int sys_move_mount(int from_dfd, const char *from_path, int to_dfd, const char *to_path, unsigned flags); int sys_mount_setattr(int dfd, const char *path, unsigned flags, struct ujail_mount_attr *attr, size_t size); +int sys_fsopen(const char *fsname, unsigned flags); +int sys_fsconfig(int fd, unsigned cmd, const char *key, const void *value, int aux); +int sys_fsmount(int fd, unsigned flags, unsigned attr_flags); +int premount_sysfs(void); + +#ifndef FSOPEN_CLOEXEC +#define FSOPEN_CLOEXEC 0x00000001 +#endif +#ifndef FSMOUNT_CLOEXEC +#define FSMOUNT_CLOEXEC 0x00000001 +#endif +#ifndef FSCONFIG_CMD_CREATE +#define FSCONFIG_CMD_CREATE 6 +#endif +#ifndef MOUNT_ATTR_NOSUID +#define MOUNT_ATTR_NOSUID 0x00000002 +#endif +#ifndef MOUNT_ATTR_NODEV +#define MOUNT_ATTR_NODEV 0x00000004 +#endif +#ifndef MOUNT_ATTR_NOEXEC +#define MOUNT_ATTR_NOEXEC 0x00000008 +#endif #ifndef OPEN_TREE_CLONE #define OPEN_TREE_CLONE 1 diff --git a/jail/jail.c b/jail/jail.c index 9c8324e..87bccfc 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -7493,6 +7493,11 @@ static void post_main(struct uloop_timeout *t) } } + /* a clone-time userns cannot mount sysfs of a foreign netns */ + if ((opts.namespace & CLONE_NEWUSER) && !userns_deferred() && + !(opts.namespace & CLONE_NEWNET) && premount_sysfs()) + WARNING("cannot mount sysfs for the jail: %m\n"); + prime_jail_mount(opts.extroot); prime_jail_mount(opts.overlaydir); for (size_t i = 0; i < (size_t)num_volume_sources; i++) From 4021843627a4a2f2c29446107bea56faa85a050a Mon Sep 17 00:00:00 2001 From: Daniel Golle Date: Mon, 28 Sep 2026 11:13:10 +0100 Subject: [PATCH 10/10] jail: log a failing mask mount The mask branch of do_mount() returns the caller's error code without a message, so a failure to mask a critical path such as /proc/kcore or /sys/firmware surfaces only as "mount_all() failed", and an optional one leaves no trace at all even though the path it was meant to hide stays visible. Report both, with the path and the errno. Signed-off-by: Joshua Covington Signed-off-by: Daniel Golle --- jail/fs.c | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/jail/fs.c b/jail/fs.c index b3c35ea..4b7704e 100644 --- a/jail/fs.c +++ b/jail/fs.c @@ -504,7 +504,7 @@ static int do_mount(const char *root, const char *orig_source, const char *targe char devpts_data[512]; const char *mount_data; char *source = (char *)orig_source; - int fd, ret = 0; + int fd, err, ret = 0; bool is_bind = (orig_mountflags & MS_BIND); bool is_mask = (source == (void *)(-1)); bool use_fd = false; @@ -534,15 +534,21 @@ static int do_mount(const char *root, const char *orig_source, const char *targe return 0; /* doesn't exists, nothing to mask */ if (S_ISDIR(s.st_mode)) {/* use empty 0-sized tmpfs for directories */ - if (mount("none", new, "tmpfs", MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, "size=0,mode=000")) - return error; + err = mount("none", new, "tmpfs", MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, "size=0,mode=000"); } else { /* mount-bind 0-sized file having mode 000 */ - if (mount(UJAIL_NOAFILE, new, "bind", MS_BIND, NULL)) - return error; + err = mount(UJAIL_NOAFILE, new, "bind", MS_BIND, NULL); + if (!err) + err = remount_readonly(new, MS_NOSUID | MS_NOEXEC | MS_NODEV); + } - if (remount_readonly(new, MS_NOSUID | MS_NOEXEC | MS_NODEV)) - return error; + if (err) { + if (error) + ERROR("failed to mask %s: %m\n", new); + else + WARNING("could not mask optional path %s: %m\n", + new); + return error; } DEBUG("masked path %s\n", new);