diff --git a/jail/fs.c b/jail/fs.c index 93670fe..4b7704e 100644 --- a/jail/fs.c +++ b/jail/fs.c @@ -126,6 +126,31 @@ unsigned long detect_atime_flag(const char *mountpoint) #define MOUNT_ATTR_NODIRATIME 0x00000080 #endif +/* MS_* -> MOUNT_ATTR_* for fsmount()/mount_setattr() */ +static unsigned mountflags_to_attr(unsigned long mountflags) +{ + unsigned attr = 0; + + if (mountflags & MS_RDONLY) + attr |= MOUNT_ATTR_RDONLY; + if (mountflags & MS_NOSUID) + attr |= MOUNT_ATTR_NOSUID; + if (mountflags & MS_NODEV) + attr |= MOUNT_ATTR_NODEV; + if (mountflags & MS_NOEXEC) + attr |= MOUNT_ATTR_NOEXEC; + if (mountflags & MS_NODIRATIME) + attr |= MOUNT_ATTR_NODIRATIME; + if (mountflags & MS_NOATIME) + attr |= MOUNT_ATTR_NOATIME; + else if (mountflags & MS_STRICTATIME) + attr |= MOUNT_ATTR_STRICTATIME; + else + attr |= MOUNT_ATTR_RELATIME; + + return attr; +} + int sys_openat2(int dfd, const char *path, struct open_how *how, size_t size) { return syscall(SYS_openat2, dfd, path, how, size); @@ -312,6 +337,21 @@ int sys_move_mount(int from_dfd, const char *from_path, int to_dfd, const char * return syscall(SYS_move_mount, from_dfd, from_path, to_dfd, to_path, flags); } +int sys_fsopen(const char *fsname, unsigned flags) +{ + return syscall(SYS_fsopen, fsname, flags); +} + +int sys_fsconfig(int fd, unsigned cmd, const char *key, const void *value, int aux) +{ + return syscall(SYS_fsconfig, fd, cmd, key, value, aux); +} + +int sys_fsmount(int fd, unsigned flags, unsigned attr_flags) +{ + return syscall(SYS_fsmount, fd, flags, attr_flags); +} + int sys_mount_setattr(int dfd, const char *path, unsigned flags, struct ujail_mount_attr *attr, size_t size) { return syscall(SYS_mount_setattr, dfd, path, flags, attr, size); @@ -339,7 +379,7 @@ int mask_path_now(const char *path) } else { if (mount(JAIL_NOAFILE, path, "bind", MS_BIND, NULL)) return -1; - if (mount(JAIL_NOAFILE, path, "bind", MS_REMOUNT | MS_BIND | MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, NULL)) + if (remount_readonly(path, MS_NOSUID | MS_NOEXEC | MS_NODEV)) return -1; } @@ -363,9 +403,13 @@ static void mountinfo_unescape(char *s) *w = '\0'; } +/* + * The flags a locked mount will not let a remount clear. The atime class is + * deliberately absent: path_mount() preserves it when the remount names none. + */ static unsigned long mountinfo_current_flags(const char *path) { - unsigned long flags = MS_RELATIME; + unsigned long flags = 0; bool found = false; FILE *f; char *line = NULL; @@ -403,12 +447,6 @@ static unsigned long mountinfo_current_flags(const char *path) this_flags |= MS_NODEV; else if (!strcmp(tok, "noexec")) this_flags |= MS_NOEXEC; - else if (!strcmp(tok, "noatime")) - this_flags |= MS_NOATIME; - else if (!strcmp(tok, "relatime")) - this_flags |= MS_RELATIME; - else if (!strcmp(tok, "nodiratime")) - this_flags |= MS_NODIRATIME; } /* last match wins: it's the topmost/effective entry */ @@ -419,11 +457,32 @@ static unsigned long mountinfo_current_flags(const char *path) fclose(f); if (!found) - return MS_RELATIME; + return 0; return flags; } +/* + * Self-bind @path and remount it read-only, preserving the flags already in + * effect. Mounts copied in by unshare(CLONE_NEWNS) under a userns that does + * not own them are MNT_LOCK_{NOSUID,NODEV,NOEXEC,ATIME}; a remount clearing + * any of those fails with EPERM. + */ +int remount_readonly(const char *path, unsigned long flags) +{ + flags |= MS_REMOUNT | MS_BIND | MS_RDONLY | mountinfo_current_flags(path); + + return mount(NULL, path, NULL, flags, NULL); +} + +int bind_remount_readonly(const char *path, unsigned long flags) +{ + if (mount(path, path, "bind", MS_BIND | (flags & MS_REC), NULL)) + return -1; + + return remount_readonly(path, flags); +} + static bool fs_userns; void jail_fs_set_userns(bool enabled) @@ -445,7 +504,7 @@ static int do_mount(const char *root, const char *orig_source, const char *targe char devpts_data[512]; const char *mount_data; char *source = (char *)orig_source; - int fd, ret = 0; + int fd, err, ret = 0; bool is_bind = (orig_mountflags & MS_BIND); bool is_mask = (source == (void *)(-1)); bool use_fd = false; @@ -475,15 +534,21 @@ static int do_mount(const char *root, const char *orig_source, const char *targe return 0; /* doesn't exists, nothing to mask */ if (S_ISDIR(s.st_mode)) {/* use empty 0-sized tmpfs for directories */ - if (mount("none", new, "tmpfs", MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, "size=0,mode=000")) - return error; + err = mount("none", new, "tmpfs", MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, "size=0,mode=000"); } else { /* mount-bind 0-sized file having mode 000 */ - if (mount(UJAIL_NOAFILE, new, "bind", MS_BIND, NULL)) - return error; + err = mount(UJAIL_NOAFILE, new, "bind", MS_BIND, NULL); + if (!err) + err = remount_readonly(new, MS_NOSUID | MS_NOEXEC | MS_NODEV); + } - if (mount(UJAIL_NOAFILE, new, "bind", MS_REMOUNT | MS_BIND | MS_RDONLY | MS_NOSUID | MS_NOEXEC | MS_NODEV | MS_RELATIME, NULL)) - return error; + if (err) { + if (error) + ERROR("failed to mask %s: %m\n", new); + else + WARNING("could not mask optional path %s: %m\n", + new); + return error; } DEBUG("masked path %s\n", new); @@ -744,6 +809,39 @@ int add_mount_fd(int fd, const char *target, int error) return 0; } +/* + * sysfs needs CAP_SYS_ADMIN in the userns owning the netns. fsmount() the + * queued sysfs entries here, before clone(), for do_mount_fd(). + */ +int premount_sysfs(void) +{ + struct mount *m; + int fsfd, mfd; + + list_for_each_entry(m, &mounts_order, list) { + if (!m->filesystemtype || strcmp(m->filesystemtype, "sysfs") || + m->source_fd >= 0) + continue; + + fsfd = sys_fsopen("sysfs", FSOPEN_CLOEXEC); + if (fsfd < 0) + return -1; + if (sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0)) { + close(fsfd); + return -1; + } + mfd = sys_fsmount(fsfd, FSMOUNT_CLOEXEC, mountflags_to_attr(m->mountflags)); + close(fsfd); + if (mfd < 0) + return -1; + + m->source_fd = mfd; + DEBUG("pre-mounted sysfs for %s as fd:%d\n", m->target, mfd); + } + + return 0; +} + int add_mount_volume(const char *source, const char *target, int error) { struct mount *m; @@ -1316,26 +1414,10 @@ static int idmap_tree_fd(const char *source, int source_fd, int userns_fd, return -1; } - attr.attr_set = MOUNT_ATTR_IDMAP; - if (mountflags & MS_RDONLY) - attr.attr_set |= MOUNT_ATTR_RDONLY; - if (mountflags & MS_NOSUID) - attr.attr_set |= MOUNT_ATTR_NOSUID; - if (mountflags & MS_NODEV) - attr.attr_set |= MOUNT_ATTR_NODEV; - if (mountflags & MS_NOEXEC) - attr.attr_set |= MOUNT_ATTR_NOEXEC; - if (mountflags & MS_NODIRATIME) - attr.attr_set |= MOUNT_ATTR_NODIRATIME; - if (mountflags & (MS_NOATIME | MS_RELATIME | MS_STRICTATIME)) { + /* keep the (possibly locked) atime mode unless one is requested */ + attr.attr_set = MOUNT_ATTR_IDMAP | mountflags_to_attr(mountflags); + if (mountflags & (MS_NOATIME | MS_RELATIME | MS_STRICTATIME)) attr.attr_clr |= MOUNT_ATTR__ATIME; - if (mountflags & MS_NOATIME) - attr.attr_set |= MOUNT_ATTR_NOATIME; - else if (mountflags & MS_STRICTATIME) - attr.attr_set |= MOUNT_ATTR_STRICTATIME; - else - attr.attr_set |= MOUNT_ATTR_RELATIME; - } attr.propagation = propagation; attr.userns_fd = userns_fd; @@ -1581,6 +1663,7 @@ static int mount_one(const char *jailroot, const char *jail_dev, struct mount *m { char devtarget[PATH_MAX]; struct mount *outer; + int ret; if (m->mounted) return 0; @@ -1611,8 +1694,11 @@ static int mount_one(const char *jailroot, const char *jail_dev, struct mount *m if (m->idmap) return do_idmap_mount(jailroot, m) ? -1 : 0; - if (m->source_fd >= 0) - return do_mount_fd(jailroot, m->source_fd, m->target, m->error) ? -1 : 0; + if (m->source_fd >= 0) { + ret = do_mount_fd(jailroot, m->source_fd, m->target, m->error); + m->source_fd = -1; + return ret ? -1 : 0; + } if (do_mount(jailroot, m->source, m->target, m->filesystemtype, m->mountflags, m->propflags, m->optstr, m->error, m->inner, m->source_fd)) diff --git a/jail/fs.h b/jail/fs.h index a7f603f..89eadad 100644 --- a/jail/fs.h +++ b/jail/fs.h @@ -45,6 +45,8 @@ int fs_mount_enable_idmap(const char *target, uint32_t uid, uint32_t gid); char *resolve_mount_source(const char *source); int add_mount_fd(int fd, const char *target, int error); int mask_path_now(const char *path); +int remount_readonly(const char *path, unsigned long flags); +int bind_remount_readonly(const char *path, unsigned long flags); /* open_tree()/mount_setattr() wrappers - no glibc wrappers yet. * Fields must match the kernel's struct mount_attr layout exactly @@ -57,6 +59,29 @@ int sys_open_tree(int dfd, const char *path, unsigned flags); int sys_move_mount(int from_dfd, const char *from_path, int to_dfd, const char *to_path, unsigned flags); int sys_mount_setattr(int dfd, const char *path, unsigned flags, struct ujail_mount_attr *attr, size_t size); +int sys_fsopen(const char *fsname, unsigned flags); +int sys_fsconfig(int fd, unsigned cmd, const char *key, const void *value, int aux); +int sys_fsmount(int fd, unsigned flags, unsigned attr_flags); +int premount_sysfs(void); + +#ifndef FSOPEN_CLOEXEC +#define FSOPEN_CLOEXEC 0x00000001 +#endif +#ifndef FSMOUNT_CLOEXEC +#define FSMOUNT_CLOEXEC 0x00000001 +#endif +#ifndef FSCONFIG_CMD_CREATE +#define FSCONFIG_CMD_CREATE 6 +#endif +#ifndef MOUNT_ATTR_NOSUID +#define MOUNT_ATTR_NOSUID 0x00000002 +#endif +#ifndef MOUNT_ATTR_NODEV +#define MOUNT_ATTR_NODEV 0x00000004 +#endif +#ifndef MOUNT_ATTR_NOEXEC +#define MOUNT_ATTR_NOEXEC 0x00000008 +#endif #ifndef OPEN_TREE_CLONE #define OPEN_TREE_CLONE 1 diff --git a/jail/jail.c b/jail/jail.c index 4204ee2..87bccfc 100644 --- a/jail/jail.c +++ b/jail/jail.c @@ -470,10 +470,17 @@ static char console_slave_name[64]; * Joining a namespace by path needs privilege in the user namespace owning it, * which our own user namespace would take away, so in that case it is created * after the joins instead of by clone(). crun makes the same distinction. + * + * A userns joined with -j (opts.setns.user) drops privilege the same way and + * never owns the pidns clone() created, so procfs must be mounted before + * entering it. Both cases are entered in enter_userns(), after build_jail_fs(). */ static inline bool userns_deferred(void) { - if (!(opts.namespace & CLONE_NEWUSER) || opts.setns.user != -1) + if (opts.setns.user != -1) + return true; + + if (!(opts.namespace & CLONE_NEWUSER)) return false; return (opts.setns.pid != -1) || @@ -488,6 +495,23 @@ static inline bool userns_deferred(void) false; } +/* + * The jail's root maps to a host uid we know: our own user namespace, or a + * joined one whose map we managed to read. Without the map the ownership + * decisions below would act on a guess, so they stay as they were. + */ +static bool joined_userns_mapped; + +static inline bool jail_has_userns(void) +{ + return (opts.namespace & CLONE_NEWUSER) || joined_userns_mapped; +} + +static inline bool jail_in_userns(void) +{ + return (opts.namespace & CLONE_NEWUSER) || opts.setns.user != -1; +} + static inline bool has_namespaces(void) { return ((opts.setns.pid != -1) || @@ -1100,7 +1124,7 @@ static struct mknod_args default_devices[] = { static int prepare_jail_dev(void) { struct mknod_args **cur, *curdef; - uid_t base = (opts.namespace & CLONE_NEWUSER) ? opts.root_map_uid : 0; + uid_t base = jail_has_userns() ? opts.root_map_uid : 0; mode_t oldmask = umask(0); char path[PATH_MAX], *tmp; int consfd; @@ -1561,7 +1585,7 @@ static int build_jail_fs(void) return -1; } - jail_fs_set_userns((opts.namespace & CLONE_NEWUSER) || (opts.setns.user != -1)); + jail_fs_set_userns(jail_has_userns()); if (mount_all(jail_root, jail_dev)) { ERROR("mount_all() failed\n"); @@ -1780,13 +1804,14 @@ static void free_and_exit(int ret) exit(ret); } +static int setns_open(unsigned long nstype); static void post_jail_fs(void); static void enter_userns(void); static int userns_wait_idmaps(void); #ifdef CLONE_NEWTIME static int timens_create(void); #endif -static void remask_after_unshare(void); +static int remask_after_unshare(void); static void remount_proc_sys_after_unshare(void); static void enter_jail_fs(void) { @@ -1844,19 +1869,26 @@ static int userns_wait_idmaps(void) return -1; } + return 0; +} + +/* become root in the userns just entered */ +static int userns_become_root(void) +{ if (setregid(0, 0) < 0 || setreuid(0, 0) < 0) { - ERROR("cannot become root in our user namespace: %m\n"); - return -1; - } - if (setgroups(0, NULL) < 0) { - ERROR("setgroups: %m\n"); + ERROR("cannot become root in the user namespace: %m\n"); return -1; } - if ((opts.namespace & CLONE_NEWNS) && - mount("none", "/", "none", mountns_propagation(), NULL)) { - ERROR("mount propagation failed: %m\n"); - return -1; + if (setgroups(0, NULL) < 0) { + /* setgroups=deny is permanent once gid_map is written; only a + * joined userns can have it */ + if (errno != EPERM || opts.setns.user == -1) { + ERROR("setgroups: %m\n"); + return -1; + } + WARNING("setgroups(0, NULL) denied by the joined userns; " + "continuing without dropping supplementary groups\n"); } return 0; @@ -1864,17 +1896,31 @@ static int userns_wait_idmaps(void) static void enter_userns(void) { + int ret; + if (!userns_deferred()) { post_jail_fs(); return; } - if (unshare(CLONE_NEWUSER)) { - ERROR("unshare(CLONE_NEWUSER) failed: %m\n"); - free_and_exit(-1); + if (opts.setns.user != -1) { + /* maps exist already, no handshake */ + ret = setns_open(CLONE_NEWUSER); + if (ret) { + ERROR("failed to join user namespace: %s\n", strerror(ret)); + free_and_exit(-1); + } + } else { + if (unshare(CLONE_NEWUSER)) { + ERROR("unshare(CLONE_NEWUSER) failed: %m\n"); + free_and_exit(-1); + } + + if (userns_wait_idmaps()) + free_and_exit(-1); } - if (userns_wait_idmaps()) + if (userns_become_root()) free_and_exit(-1); #ifdef CLONE_NEWTIME @@ -1888,7 +1934,9 @@ static void enter_userns(void) free_and_exit(-1); } if (opts.namespace & CLONE_NEWNS) { - remask_after_unshare(); + if (remask_after_unshare()) + free_and_exit(-1); + remount_proc_sys_after_unshare(); } @@ -2111,9 +2159,7 @@ static int remount_readonly_now(const char *path) if (stat(path, &s)) return 0; /* doesn't exist, nothing to restrict */ - if (mount(path, path, "bind", MS_BIND | MS_REC, NULL)) - return -1; - if (mount(path, path, "bind", MS_REMOUNT | MS_BIND | MS_RDONLY | MS_REC, NULL)) + if (bind_remount_readonly(path, MS_REC)) return -1; DEBUG("read-only path %s\n", path); @@ -2122,34 +2168,52 @@ static int remount_readonly_now(const char *path) /* re-apply masks inside the container's own mntns: the first pass is * locked once copied in after unshare(CLONE_NEWNS); best-effort */ -static void remask_after_unshare(void) +static int mask_paths_now(const char **paths, bool critical) { const char **p; + + for (p = paths; *p; p++) { + if (!mask_path_now(*p)) + continue; + + if (critical) { + ERROR("failed to mask sensitive path %s: %m\n", *p); + return -1; + } + + WARNING("could not mask optional path %s: %m\n", *p); + } + + return 0; +} + +static int remask_after_unshare(void) +{ char **dp; if (!opts.ocibundle) { - if (opts.procfs) { - for (p = proc_mask_critical; *p; p++) - mask_path_now(*p); - for (p = proc_mask_optional; *p; p++) - mask_path_now(*p); - } + if (opts.procfs && + (mask_paths_now(proc_mask_critical, true) || + mask_paths_now(proc_mask_optional, false))) + return -1; - if (opts.sysfs) { - for (p = sys_mask_critical; *p; p++) - mask_path_now(*p); - } + if (opts.sysfs && mask_paths_now(sys_mask_critical, true)) + return -1; } /* same reason as the default masks above: locked in phase 1, an * OCI bundle's own masks would outlive its unshare(CLONE_NEWNS). */ if (opts.oci_deferred_masked) for (dp = opts.oci_deferred_masked; *dp; dp++) - mask_path_now(*dp); + if (mask_path_now(*dp)) + WARNING("could not mask %s: %m\n", *dp); if (opts.oci_deferred_readonly) for (dp = opts.oci_deferred_readonly; *dp; dp++) - remount_readonly_now(*dp); + if (remount_readonly_now(*dp)) + WARNING("could not make %s read-only: %m\n", *dp); + + return 0; } /* /proc/sys is locked read-only for every procfs jail via a self-bind, @@ -2162,21 +2226,29 @@ static void remount_proc_sys_after_unshare(void) if (!(opts.procfs || opts.ocibundle)) return; + if (mount_is_defined("/proc/sys")) + return; if (stat("/proc/sys", &s)) return; /* the MS_MOVE below needs a mountpoint to move from, or it's a * silent EINVAL. */ - if (opts.namespace & CLONE_NEWNET) - mount("/proc/sys/net", "/proc/self/net", "bind", MS_BIND, NULL); + if ((opts.namespace & CLONE_NEWNET) && + mount("/proc/sys/net", "/proc/self/net", "bind", MS_BIND, NULL)) { + ERROR("failed to stash /proc/sys/net: %m\n"); + free_and_exit(-1); + } - if (mount("/proc/sys", "/proc/sys", "bind", MS_BIND, NULL)) - return; - if (mount("/proc/sys", "/proc/sys", "bind", MS_REMOUNT | MS_BIND | MS_RDONLY, NULL)) - WARNING("could not remount /proc/sys read-only\n"); + if (bind_remount_readonly("/proc/sys", 0)) { + ERROR("failed to remount /proc/sys read-only: %m\n"); + free_and_exit(-1); + } - if (opts.namespace & CLONE_NEWNET) - mount("/proc/self/net", "/proc/sys/net", "bind", MS_MOVE, NULL); + if ((opts.namespace & CLONE_NEWNET) && + mount("/proc/self/net", "/proc/sys/net", "bind", MS_MOVE, NULL)) { + ERROR("failed to restore /proc/sys/net: %m\n"); + free_and_exit(-1); + } } static bool resolve_jail_user_gids(int primary_gid) @@ -3160,24 +3232,17 @@ static int exec_jail(void *arg) } /* - * Joining an external userns drops privilege immediately, so this has - * to run before it. A userns of our own owns the mount namespace it - * was created with and locks everything inherited into it, so there - * the detach neither works nor is needed. + * Must run before enter_userns() drops privilege over the inherited + * mounts. A userns from clone() owns its mntns and has everything + * inherited MNT_LOCKED, so there the detach is neither possible nor + * needed. */ - if ((opts.namespace & CLONE_NEWNS) && - (userns_deferred() || opts.setns.user != -1) && + if ((opts.namespace & CLONE_NEWNS) && userns_deferred() && detach_inherited_mounts()) { ERROR("failed to detach inherited mounts\n"); return EXIT_FAILURE; } - ret = setns_open(CLONE_NEWUSER); - if (ret) { - ERROR("failed to join user namespace: %s\n", strerror(ret)); - return EXIT_FAILURE; - } - buf[0] = 'i'; if (write(pipes[1], buf, 1) < 1) { ERROR("can't write to parent\n"); @@ -3199,14 +3264,9 @@ static int exec_jail(void *arg) false, recv_fds, nrecv, &extroot_idmap_fd, &overlay_idmap_fd); - if ((opts.namespace & CLONE_NEWUSER) && !userns_deferred() && - userns_wait_idmaps()) - return EXIT_FAILURE; - - if (opts.setns.user != -1 && (opts.namespace & CLONE_NEWNS) && - unshare(CLONE_NEWNS)) { - ERROR("unshare(CLONE_NEWNS) failed: %m\n"); - return EXIT_FAILURE; + if ((opts.namespace & CLONE_NEWUSER) && !userns_deferred()) { + if (userns_wait_idmaps() || userns_become_root()) + return EXIT_FAILURE; } if (opts.namespace & CLONE_NEWCGROUP) @@ -3218,27 +3278,6 @@ static int exec_jail(void *arg) free_and_exit(EXIT_FAILURE); } - if (opts.setns.user != -1) { - if (setregid(0, 0) < 0) { - ERROR("setgid\n"); - free_and_exit(EXIT_FAILURE); - } - if (setreuid(0, 0) < 0) { - ERROR("setuid\n"); - free_and_exit(EXIT_FAILURE); - } - if (setgroups(0, NULL) < 0) { - if (errno != EPERM) { - ERROR("setgroups\n"); - free_and_exit(EXIT_FAILURE); - } - WARNING("setgroups(0, NULL) denied by the joined " - "userns (setgroups=deny is permanent once a " - "gid_map is written); continuing without " - "dropping supplementary groups\n"); - } - } - #ifdef CLONE_NEWTIME if ((opts.namespace & CLONE_NEWTIME) && opts.setns.time == -1 && !userns_deferred() && timens_create()) @@ -3412,7 +3451,7 @@ static void post_start_hook(void) /* restore securebits back to normal (and lock them if not in userns) */ if (opts.capset.apply) { - if (prctl(PR_SET_SECUREBITS, (opts.namespace & CLONE_NEWUSER)?0: + if (prctl(PR_SET_SECUREBITS, jail_in_userns() ? 0 : SECBIT_KEEP_CAPS_LOCKED|SECBIT_NO_SETUID_FIXUP_LOCKED|SECBIT_NOROOT_LOCKED)) { ERROR("prctl(PR_SET_SECUREBITS) failed: %m\n"); free_and_exit(EXIT_FAILURE); @@ -4091,29 +4130,50 @@ static const struct blobmsg_policy oci_linux_namespace_policy[] = { [OCI_LINUX_NAMESPACE_PATH] = { "path", BLOBMSG_TYPE_STRING }, }; -static int resolve_nstype(char *type) { - if (!strcmp("pid", type)) - return CLONE_NEWPID; - else if (!strcmp("network", type)) - return CLONE_NEWNET; - else if (!strcmp("net", type)) - return CLONE_NEWNET; - else if (!strcmp("mount", type)) - return CLONE_NEWNS; - else if (!strcmp("ipc", type)) - return CLONE_NEWIPC; - else if (!strcmp("uts", type)) - return CLONE_NEWUTS; - else if (!strcmp("user", type)) - return CLONE_NEWUSER; - else if (!strcmp("cgroup", type)) - return CLONE_NEWCGROUP; - else if (!strcmp("time", type)) - return CLONE_NEWTIME; - else - return 0; +/* the name -j accepts, and the /proc//ns/ link it actually lives under */ +static const struct { + const char *name; + const char *link; + int nstype; +} ns_types[] = { + { "pid", "pid", CLONE_NEWPID }, + { "net", "net", CLONE_NEWNET }, + { "network", "net", CLONE_NEWNET }, + { "mnt", "mnt", CLONE_NEWNS }, + { "mount", "mnt", CLONE_NEWNS }, + { "ipc", "ipc", CLONE_NEWIPC }, + { "uts", "uts", CLONE_NEWUTS }, + { "user", "user", CLONE_NEWUSER }, + { "cgroup", "cgroup", CLONE_NEWCGROUP }, + { "time", "time", CLONE_NEWTIME }, + { NULL, NULL, 0 }, +}; + +static int resolve_nstype(const char *type) +{ + int i; + + for (i = 0; ns_types[i].name; i++) + if (!strcmp(ns_types[i].name, type)) + return ns_types[i].nstype; + + return 0; +} + +static const char *nstype_link(int nstype) +{ + int i; + + for (i = 0; ns_types[i].name; i++) + if (ns_types[i].nstype == nstype) + return ns_types[i].link; + + return NULL; } +/* linux.namespaces entries are unique by type */ +static int oci_ns_seen; + static int parseOCIlinuxns(struct blob_attr *msg) { struct blob_attr *tb[__OCI_LINUX_NAMESPACE_MAX]; @@ -4130,18 +4190,20 @@ static int parseOCIlinuxns(struct blob_attr *msg) if (!nstype) return EINVAL; - if (opts.namespace & nstype) - return ENOTUNIQ; - setns = get_namespace_fd(nstype); if (!setns) return EFAULT; - if (*setns != -1) + if (oci_ns_seen & nstype) return ENOTUNIQ; + oci_ns_seen |= nstype; + if (tb[OCI_LINUX_NAMESPACE_PATH]) { + if (*setns != -1) + return ENOTUNIQ; + DEBUG("opening existing %s namespace from path %s\n", blobmsg_get_string(tb[OCI_LINUX_NAMESPACE_TYPE]), blobmsg_get_string(tb[OCI_LINUX_NAMESPACE_PATH])); @@ -4172,6 +4234,9 @@ static int parseOCIlinuxns(struct blob_attr *msg) * The string argument is the reference PID followed by ':' and a * ',' separated list of namespaces to to join. */ +/* the pid -j resolved the user namespace from, for uid_map_root_read() */ +static pid_t setns_user_pid = -1; + static int jail_join_ns(char *arg) { pid_t pid; @@ -4197,9 +4262,6 @@ static int jail_join_ns(char *arg) if (!nstype) return EINVAL; - if (opts.namespace & nstype) - return ENOTUNIQ; - setns = get_namespace_fd(nstype); if (!setns) @@ -4208,7 +4270,7 @@ static int jail_join_ns(char *arg) if (*setns != -1) return ENOTUNIQ; - if (asprintf(&nspath, "/proc/%d/ns/%s", pid, tmp) < 0) + if (asprintf(&nspath, "/proc/%d/ns/%s", pid, nstype_link(nstype)) < 0) return ENOMEM; fd = open(nspath, O_RDONLY); @@ -4218,6 +4280,8 @@ static int jail_join_ns(char *arg) return errno?:ESTALE; *setns = fd; + if (nstype == CLONE_NEWUSER) + setns_user_pid = pid; if (etmp) tmp = etmp; @@ -4228,6 +4292,83 @@ static int jail_join_ns(char *arg) return 0; } +/* + * What uid 0 of the joined user namespace maps to on the host. Read from + * here rather than from inside: uid_m_show() renders the outside column + * through the reader's user namespace, so ours gives a host uid at any + * nesting depth, and no process of ours has to enter a namespace the + * caller controls. + */ +static bool uid_map_root_read(pid_t pid, uid_t *host_uid) +{ + uint32_t inside, outside, count; + bool found = false; + char path[32]; + FILE *f; + + snprintf(path, sizeof(path), "/proc/%d/uid_map", pid); + f = fopen(path, "re"); + if (!f) + return false; + + while (fscanf(f, "%u %u %u", &inside, &outside, &count) == 3) { + if (inside || count < 1 || outside >= (uint32_t)-1) + continue; + + *host_uid = outside; + found = true; + break; + } + + fclose(f); + + return found; +} + +/* the pid is not pinned by the namespace fd we hold, so it can be recycled */ +static bool userns_pid_valid(pid_t pid) +{ + struct stat nsst, pidst; + char path[32]; + + if (fstat(opts.setns.user, &nsst)) + return false; + + snprintf(path, sizeof(path), "/proc/%d/ns/user", pid); + if (stat(path, &pidst)) + return false; + + return nsst.st_dev == pidst.st_dev && nsst.st_ino == pidst.st_ino; +} + +static bool userns_root_map(void) +{ + uid_t host_uid; + + if (setns_user_pid < 0) { + WARNING("the joined user namespace was given as a path, so its " + "root uid cannot be determined; keeping the ownership " + "the jail would have had without it\n"); + return false; + } + + if (!uid_map_root_read(setns_user_pid, &host_uid)) { + ERROR("cannot read the uid map of the joined user namespace\n"); + return false; + } + + if (!userns_pid_valid(setns_user_pid)) { + ERROR("pid %d no longer names the joined user namespace\n", + setns_user_pid); + return false; + } + + opts.root_map_uid = host_uid; + DEBUG("root of the joined user namespace is uid %d\n", host_uid); + + return true; +} + static void get_jail_root_user(bool is_gidmap, uint32_t container_id, uint32_t host_id, uint32_t size) { if (container_id == 0 && size >= 1) @@ -6619,7 +6760,12 @@ int main(int argc, char **argv) opts.hostname = strdup(optarg); break; case 'j': - jail_join_ns(optarg); + ret = jail_join_ns(optarg); + if (ret) { + ERROR("failed to join namespace: %s\n", + strerror(ret)); + return EXIT_FAILURE; + } break; case 'b': if (!opts.ocibundle) @@ -6855,6 +7001,29 @@ int main(int argc, char **argv) ulog_open(ULOG_SYSLOG, LOG_DAEMON, "jail"); } + if (opts.setns.ns != -1 && (opts.namespace & CLONE_NEWNS)) { + ERROR("cannot build a jail filesystem in a joined mount namespace\n"); + ret = EXIT_FAILURE; + goto errout; + } + + /* clone(CLONE_NEWPID) after setns(CLONE_NEWPID) is EINVAL */ + if (opts.setns.net != -1) + opts.namespace &= ~CLONE_NEWNET; + if (opts.setns.ipc != -1) + opts.namespace &= ~CLONE_NEWIPC; + if (opts.setns.uts != -1) + opts.namespace &= ~CLONE_NEWUTS; + if (opts.setns.pid != -1) + opts.namespace &= ~CLONE_NEWPID; + if (opts.setns.user != -1) + opts.namespace &= ~CLONE_NEWUSER; + if (opts.setns.cgroup != -1) + opts.namespace &= ~CLONE_NEWCGROUP; + + if (opts.setns.user != -1) + joined_userns_mapped = userns_root_map(); + for (credidx = 0; credidx < n_cred_targets; credidx++) { ret = fs_mount_enable_idmap(cred_targets[credidx], opts.pw_uid > 0 ? (uint32_t)opts.pw_uid : 0, @@ -7088,6 +7257,7 @@ static void netifd_restart_watch(void) static void post_main(struct uloop_timeout *t) { int child_status; + int nsret; if (apply_rlimits()) { ERROR("error applying resource limits\n"); @@ -7154,7 +7324,7 @@ static void post_main(struct uloop_timeout *t) add_mount("shm", "/dev/shm", "tmpfs", MS_NOSUID | MS_NOEXEC | MS_NODEV, 0, "mode=1777,size=10%", -1); { - const char *ptsopts = (opts.namespace & CLONE_NEWUSER) ? + const char *ptsopts = jail_has_userns() ? "newinstance,ptmxmode=0666,mode=0620,gid=0" : "newinstance,ptmxmode=0666,mode=0620,gid=5"; @@ -7219,7 +7389,12 @@ static void post_main(struct uloop_timeout *t) if (opts.setns.pid != -1) { pidns_fd = ns_open_pid("pid", getpid()); - setns_open(CLONE_NEWPID); + nsret = setns_open(CLONE_NEWPID); + if (nsret) { + ERROR("failed to join pid namespace: %s\n", + strerror(nsret)); + free_and_exit(EXIT_FAILURE); + } } else { pidns_fd = -1; } @@ -7235,7 +7410,7 @@ static void post_main(struct uloop_timeout *t) free_and_exit(EXIT_FAILURE); } - if (opts.namespace & CLONE_NEWUSER) { + if (jail_has_userns()) { if (opts.overlaydir) { if (chown(opts.overlaydir, opts.root_map_uid, opts.root_map_uid)) { ERROR("chown(%s, %d, %d) failed: %m\n", @@ -7286,12 +7461,12 @@ static void post_main(struct uloop_timeout *t) close(parent_master); } else { console_fd = parent_master; - if ((opts.namespace & CLONE_NEWUSER) && seteuid(0)) { + if (jail_has_userns() && seteuid(0)) { ERROR("seteuid(0) failed: %m\n"); free_and_exit(EXIT_FAILURE); } pass_console(console_fd); - if ((opts.namespace & CLONE_NEWUSER) && seteuid(opts.root_map_uid)) { + if (jail_has_userns() && seteuid(opts.root_map_uid)) { ERROR("seteuid(%d) failed: %m\n", opts.root_map_uid); free_and_exit(EXIT_FAILURE); } @@ -7318,6 +7493,11 @@ static void post_main(struct uloop_timeout *t) } } + /* a clone-time userns cannot mount sysfs of a foreign netns */ + if ((opts.namespace & CLONE_NEWUSER) && !userns_deferred() && + !(opts.namespace & CLONE_NEWNET) && premount_sysfs()) + WARNING("cannot mount sysfs for the jail: %m\n"); + prime_jail_mount(opts.extroot); prime_jail_mount(opts.overlaydir); for (size_t i = 0; i < (size_t)num_volume_sources; i++)