diff --git a/docs/internals.md b/docs/internals.md index 799ccb89..86526868 100644 --- a/docs/internals.md +++ b/docs/internals.md @@ -1874,6 +1874,30 @@ final component may not exist yet. `path_translate_at()` picks one by flags; in the `openat2(RESOLVE_NO_SYMLINKS)` precheck described below, and walks relative paths from their descriptor with the same clamp applied in the walk. +### Folding `//` And `.` + +Linux steps over `//` runs and `.` components in the walk every path syscall +shares, so `//sys/bus`, `/./sys/bus` and `/sys/./bus` are `/sys/bus` to all of +them. The intercepts match literal prefixes, so `path_translate_at()` folds an +absolute name once, before any of them reads it. + +Two things stay as written. `..` is not folded, because Linux applies it to +what the component before it resolved to. And a `.` that is the last component +keeps its place, because that is where Linux gives it a meaning of its own: +`rmdir("d/.")` is `EINVAL` where `rmdir("d/")` removes `d`. + +What a final `.` or a trailing slash means to a lookup is a requirement: the +name has to resolve to a directory, following a final symlink to get there. The +host walk applies it for itself. The intercepts match names literally, so +`proc_intercept_open()`, `proc_intercept_stat_at()` and +`proc_intercept_readlink()` take the requirement off the name once, dispatch on +the bare name, and enforce it on the answer: a served directory answers as +itself, anything else answers `ENOTDIR`, and `readlink` of a served directory +answers `EINVAL`. The gates in `path.c` read the bare name for the same reason. + +`tests/test-path-fold.c` holds every respelling of a name to the answer its +canonical spelling gets, and both endings to what Linux makes of them. + ### Clamping `..` At The Guest Root A guest resolves `..` against its own root, and Linux clamps it there: `/..` diff --git a/mk/tests.mk b/mk/tests.mk index a52074fd..a428e7ab 100644 --- a/mk/tests.mk +++ b/mk/tests.mk @@ -41,7 +41,7 @@ ELFUSE_HOST_NOFILE_MIN ?= $(shell bash "$(CURDIR)/tests/test-config.sh" --host-n test-sysroot-name-unique \ test-sysroot-name-relative \ test-nosysroot-literal-names test-sysroot-outside-names \ - test-sysroot-root test-usb-sysfs test-usb-sysfs-sysroot \ + test-sysroot-root test-path-fold test-usb-sysfs test-usb-sysfs-sysroot \ test-usb-sysfs-matrix \ test-usb-sysfs-overflow test-usbdev-ioctl test-usbdev-faults \ test-usbdev-urb-loopback test-usbdev-ioctl-loopback \ @@ -278,6 +278,7 @@ $(call run-host-unit,test-usb-desc-host,USB descriptor blob walk unit test) $(call run-host-unit,test-usbdev-urb-host,usbdevfs URB bookkeeping unit test) $(call run-host-unit,test-elf-headers-host,ELF header validation unit test) $(call run-host-unit,test-gdbstub-host,buffered GDB session regression) +$(call run-lane,test-path-fold,one answer per object however its path is spelled) $(call run-lane,test-usb-sysfs,synthetic USB tree contract) $(call run-lane,test-usb-sysfs-sysroot,synthetic USB /sys sharing a populated sysroot) $(call run-lane,test-usb-sysfs-matrix,every /sys and /dev/bus entry point against every path class) @@ -1695,6 +1696,13 @@ test-casefold-host: $(BUILD_DIR)/test-casefold-host test-casefold-walk-host: $(BUILD_DIR)/test-casefold-walk-host $(BUILD_DIR)/test-casefold-walk-host +## Hold every respelling of a name to the canonical spelling's answer +# The matrix runs this against the reference kernel too. This lane adds the +# fixture, so the synthetic USB names are present on a machine with no device +# and do not agree merely by being absent under every spelling. +test-path-fold: $(ELFUSE_BIN) $(TEST_DIR)/test-path-fold + ELFUSE_USB_FIXTURE=1 $(ELFUSE_BIN) $(TEST_DIR)/test-path-fold + ## Assert the synthetic USB tree's contract and its two agreeing views # The device-dependent half only runs against whatever is attached, so the lane # prints the device count rather than letting an empty bus read as full cover. diff --git a/src/runtime/procemu.c b/src/runtime/procemu.c index fa5b5017..eaaf5cde 100644 --- a/src/runtime/procemu.c +++ b/src/runtime/procemu.c @@ -2292,10 +2292,10 @@ static int proc_open_mounts_node(const char *path) * guest and the buffer, so a split would be a jump table by another name. */ /* NOLINTNEXTLINE(readability-function-size) */ -int proc_intercept_open(const guest_t *g, - const char *path, - int linux_flags, - int mode) +static int intercept_open_dispatch(const guest_t *g, + const char *path, + int linux_flags, + int mode) { /* /dev/ptmx -> host /dev/ptmx + keepalive slave (see pty_open_master). * O_PATH is path-only on Linux: it must not run the device open hook or @@ -2387,7 +2387,7 @@ int proc_intercept_open(const guest_t *g, * intercepted too or callers that probe then enumerate see inconsistent * Linux-visible behavior. */ - if (!strcmp(path, "/dev/pts") || !strcmp(path, "/dev/pts/")) + if (!strcmp(path, "/dev/pts")) return pty_open_pts_dir(linux_flags); /* /dev/pts/N -> the macOS slave path captured at /dev/ptmx open time. @@ -2420,7 +2420,7 @@ int proc_intercept_open(const guest_t *g, * "self" symlink. The DIR* created from this allows getdents64 to enumerate * /proc like a real procfs. Cleaned up via atexit. */ - if (!strcmp(path, "/proc") || !strcmp(path, "/proc/")) { + if (!strcmp(path, "/proc")) { const char *dir = ensure_proc_tmpdir(g); if (!dir) return -1; @@ -2428,7 +2428,7 @@ int proc_intercept_open(const guest_t *g, } /* /proc/self -> directory fd for the PID subdirectory */ - if (!strcmp(path, "/proc/self") || !strcmp(path, "/proc/self/")) { + if (!strcmp(path, "/proc/self")) { const char *dir = ensure_proc_tmpdir(g); if (!dir) return -1; @@ -3093,7 +3093,9 @@ int proc_intercept_stat(const char *path, struct stat *st) return proc_intercept_stat_at(path, st, false); } -int proc_intercept_stat_at(const char *path, struct stat *st, bool follow) +static int intercept_stat_dispatch(const char *path, + struct stat *st, + bool follow) { /* Intercept stat for /proc paths emulated via proc_intercept_open. Without * this, runtime libraries that probe a file's existence via stat() before @@ -3124,8 +3126,16 @@ int proc_intercept_stat_at(const char *path, struct stat *st, bool follow) return 0; } + /* The mount table open() synthesizes; a stat that said ENOENT for a name + * open() then served was one object with two answers. + */ + if (!strcmp(path, "/etc/mtab")) { + stat_fill_proc_file(st, 0444, path); + return 0; + } + /* /dev/shm is a directory */ - if (!strcmp(path, "/dev/shm") || !strcmp(path, "/dev/shm/")) { + if (!strcmp(path, "/dev/shm")) { stat_fill_proc_dir(st, 01777, 2, path); /* sticky bit, like real /dev/shm */ return 0; @@ -3149,7 +3159,7 @@ int proc_intercept_stat_at(const char *path, struct stat *st, bool follow) * passes. The numeric tail must round-trip with /dev/ttysN via the open * intercept (see proc_intercept_open). */ - if (!strcmp(path, "/dev/pts") || !strcmp(path, "/dev/pts/")) { + if (!strcmp(path, "/dev/pts")) { stat_fill_proc_dir(st, 0755, 2, path); return 0; } @@ -3209,18 +3219,15 @@ int proc_intercept_stat_at(const char *path, struct stat *st, bool follow) } /* /proc and /proc/ are directories */ - if (!strcmp(path, "/proc") || !strcmp(path, "/proc/")) { + if (!strcmp(path, "/proc")) { stat_fill_proc_dir(st, 0555, 3, path); return 0; } { - char pidbuf[32], pidslash[32]; + char pidbuf[32]; snprintf(pidbuf, sizeof(pidbuf), "/proc/%lld", (long long) proc_get_pid()); - snprintf(pidslash, sizeof(pidslash), "/proc/%lld/", - (long long) proc_get_pid()); - if (!strcmp(path, pidbuf) || !strcmp(path, pidslash) || - !strcmp(path, "/proc/self") || !strcmp(path, "/proc/self/")) { + if (!strcmp(path, pidbuf) || !strcmp(path, "/proc/self")) { stat_fill_proc_dir(st, 0555, 3, path); return 0; } @@ -3439,7 +3446,9 @@ static int proc_readlink_self_exe(char *buf, size_t bufsiz) return (int) len; } -int proc_intercept_readlink(const char *path, char *buf, size_t bufsiz) +static int intercept_readlink_dispatch(const char *path, + char *buf, + size_t bufsiz) { { char alias[LINUX_PATH_MAX]; @@ -3447,7 +3456,7 @@ int proc_intercept_readlink(const char *path, char *buf, size_t bufsiz) if (aliased < 0) return -1; if (aliased > 0) - return proc_intercept_readlink(alias, buf, bufsiz); + return intercept_readlink_dispatch(alias, buf, bufsiz); } if (!strcmp(path, "/proc/self/exe")) @@ -3696,3 +3705,88 @@ int proc_intercept_write(int guest_fd, pthread_mutex_unlock(&oom_write_lock); return rc; } + +/* What the requirement answers for @path once the dispatcher has been asked: 0 + * when it names an intercepted directory, -1 with errno for an intercepted file + * or a lookup that failed, PROC_NOT_INTERCEPTED when the name is not ours and + * the host applies the requirement itself. + */ +static int intercept_require_dir(const char *stripped) +{ + struct stat st; + int rc = intercept_stat_dispatch(stripped, &st, true); + if (rc != 0) + return rc; + if (!S_ISDIR(st.st_mode)) { + errno = ENOTDIR; + return -1; + } + return 0; +} + +int proc_intercept_open(const guest_t *g, + const char *path, + int linux_flags, + int mode) +{ + /* The directory requirement of a trailing slash or ".": the name has to + * resolve to a directory, following a final symlink to get there. The + * dispatchers match names literally, so it is taken off here, once, and + * enforced on what they answer. + */ + char buf[LINUX_PATH_MAX]; + const char *name = path_bare_name(path, buf, sizeof(buf)); + if (name != path) { + int rc = intercept_require_dir(name); + if (rc != 0) + return rc; + linux_flags &= ~LINUX_O_NOFOLLOW; + } + return intercept_open_dispatch(g, name, linux_flags, mode); +} + +int proc_intercept_stat_at(const char *path, struct stat *st, bool follow) +{ + char buf[LINUX_PATH_MAX]; + const char *name = path_bare_name(path, buf, sizeof(buf)); + if (name == path) + return intercept_stat_dispatch(path, st, follow); + int rc = intercept_stat_dispatch(name, st, true); + if (rc == 0 && !S_ISDIR(st->st_mode)) { + errno = ENOTDIR; + return -1; + } + return rc; +} + +int proc_intercept_readlink(const char *path, char *buf, size_t bufsiz) +{ + char name_buf[LINUX_PATH_MAX]; + const char *name = path_bare_name(path, name_buf, sizeof(name_buf)); + if (name != path) { + /* The requirement resolves the name to a directory, which is no link. + */ + int rc = intercept_require_dir(name); + if (rc == 0) { + errno = EINVAL; + return -1; + } + return rc; + } + + int rc = intercept_readlink_dispatch(path, buf, bufsiz); + if (rc != PROC_NOT_INTERCEPTED) + return rc; + + /* An object an intercept serves and no readlink arm names is not a link: + * Linux answers EINVAL for it, where the host, asked for a name it does not + * have, answered ENOENT. + */ + struct stat st; + if (intercept_stat_dispatch(path, &st, false) == 0 && + !S_ISLNK(st.st_mode)) { + errno = EINVAL; + return -1; + } + return PROC_NOT_INTERCEPTED; +} diff --git a/src/syscall/fs-stat.c b/src/syscall/fs-stat.c index 910f6783..e57fca81 100644 --- a/src/syscall/fs-stat.c +++ b/src/syscall/fs-stat.c @@ -623,11 +623,23 @@ static int64_t sys_statfs_impl(guest_t *g, if (tx.fuse_path) return -LINUX_ENOSYS; - if (statfs_path_is_proc(tx.intercept_path)) { - if (proc_path_is_symlink(tx.intercept_path)) { + /* The directory requirement of a trailing slash or ".", applied before the + * classification below, which reads the bare name: a served file under it + * is ENOTDIR, and a served directory is classified as itself. + */ + char bare[LINUX_PATH_MAX]; + const char *name = path_bare_name(tx.intercept_path, bare, sizeof(bare)); + if (name != tx.intercept_path && + path_might_use_stat_intercept(tx.intercept_path)) { + struct stat st; + if (proc_intercept_stat_at(tx.intercept_path, &st, true) == -1) + return linux_errno(); + } + + if (statfs_path_is_proc(name)) { + if (proc_path_is_symlink(name)) { char link[LINUX_PATH_MAX]; - int len = proc_intercept_readlink(tx.intercept_path, link, - sizeof(link) - 1); + int len = proc_intercept_readlink(name, link, sizeof(link) - 1); if (len < 0) return linux_errno(); link[len] = '\0'; @@ -635,8 +647,7 @@ static int64_t sys_statfs_impl(guest_t *g, } struct stat mac_st; - int intercepted = - proc_intercept_stat_at(tx.intercept_path, &mac_st, true); + int intercepted = proc_intercept_stat_at(name, &mac_st, true); if (intercepted == 0) { linux_statfs_t lin_st; fill_proc_statfs(&lin_st); @@ -667,7 +678,7 @@ static int64_t sys_statfs_impl(guest_t *g, * report the same ENOENT a real Linux kernel would for the rest. */ char sys_abs[LINUX_PATH_MAX]; - if (statfs_path_is_sysfs(tx.intercept_path, sys_abs, sizeof(sys_abs))) { + if (statfs_path_is_sysfs(name, sys_abs, sizeof(sys_abs))) { /* Classifying the folded name and then probing the raw one asked two * different questions of two different spellings: "/sys/../sys" is * sysfs, but no intercept and no host backing carries that literal @@ -676,7 +687,7 @@ static int64_t sys_statfs_impl(guest_t *g, * fallback and the answer all describe the same object. Folding is * idempotent, so this recurses at most once. */ - if (strcmp(sys_abs, tx.intercept_path) != 0) + if (strcmp(sys_abs, name) != 0) return sys_statfs_impl(g, sys_abs, buf_gva, depth + 1); bool exists = sys_abs[4] == '\0'; /* "/sys" itself */ @@ -699,7 +710,7 @@ static int64_t sys_statfs_impl(guest_t *g, return 0; } - int dev_bus = statfs_dev_bus_class(tx.intercept_path); + int dev_bus = statfs_dev_bus_class(name); if (dev_bus == -1) return linux_errno(); if (dev_bus == 0) { @@ -716,7 +727,7 @@ static int64_t sys_statfs_impl(guest_t *g, * glibc closes a perfectly good master -- breaking Unix98 pty allocation * for every glibc program. Answer from the virtual filesystem instead. */ - devpts_class_t devpts = statfs_devpts_class(tx.intercept_path); + devpts_class_t devpts = statfs_devpts_class(name); if (devpts == DEVPTS_ABSENT) return -LINUX_ENOENT; if (devpts == DEVPTS_MOUNT) { @@ -731,8 +742,7 @@ static int64_t sys_statfs_impl(guest_t *g, * on the leaf would follow a symlink onto the host and leak the host fs * identity, so answer synthetically; lstat is the nofollow existence probe. */ - bool shm_root = !strcmp(tx.intercept_path, "/dev/shm") || - !strcmp(tx.intercept_path, "/dev/shm/"); + bool shm_root = !strcmp(name, "/dev/shm"); if (tx.is_dev_shm || shm_root) { const char *shm_dir = proc_get_shm_dir(); if (!shm_dir) diff --git a/src/syscall/fs.c b/src/syscall/fs.c index b9244bd4..5255fdcd 100644 --- a/src/syscall/fs.c +++ b/src/syscall/fs.c @@ -2707,7 +2707,8 @@ int64_t sys_chdir(guest_t *g, uint64_t path_gva) errno = saved_errno; return linux_errno(); } - proc_cwd_set_virtual(virt); + char cwd_buf[LINUX_PATH_MAX]; + proc_cwd_set_virtual(path_bare_name(virt, cwd_buf, sizeof(cwd_buf))); return 0; } @@ -2723,7 +2724,9 @@ int64_t sys_chdir(guest_t *g, uint64_t path_gva) return -LINUX_ENOTDIR; if (chdir(tx.host_path) < 0) return linux_errno(); - proc_cwd_set_virtual(tx.intercept_path); + char cwd_buf[LINUX_PATH_MAX]; + proc_cwd_set_virtual( + path_bare_name(tx.intercept_path, cwd_buf, sizeof(cwd_buf))); return 0; } @@ -2739,7 +2742,9 @@ int64_t sys_chdir(guest_t *g, uint64_t path_gva) close_keep_errno(fd); if (chdir_rc < 0) return linux_errno(); - proc_cwd_set_virtual(tx.guest_path); + char cwd_buf[LINUX_PATH_MAX]; + proc_cwd_set_virtual( + path_bare_name(tx.guest_path, cwd_buf, sizeof(cwd_buf))); return 0; } @@ -2756,9 +2761,17 @@ int64_t sys_chdir(guest_t *g, uint64_t path_gva) * A name the intercept does not claim falls through to the host chdir * below, unchanged. */ - if (tx.intercept_path && - (path_prefix_match(tx.intercept_path, "/sys", 4) || - path_prefix_match(tx.intercept_path, "/dev/bus", 8))) { + if (tx.intercept_path && path_might_use_stat_intercept(tx.intercept_path)) { + /* A name an intercept serves as something other than a directory is + * ENOTDIR, whatever the host holds under that spelling. + */ + struct stat st; + int stat_rc = proc_intercept_stat_at(tx.intercept_path, &st, true); + if (stat_rc == -1) + return linux_errno(); + if (stat_rc == 0 && !S_ISDIR(st.st_mode)) + return -LINUX_ENOTDIR; + int host_fd = proc_intercept_open(g, tx.intercept_path, LINUX_O_DIRECTORY, 0); if (host_fd >= 0) { @@ -2774,7 +2787,9 @@ int64_t sys_chdir(guest_t *g, uint64_t path_gva) errno = saved_errno; return linux_errno(); } - proc_cwd_set_virtual(virt_path); + char cwd_buf[LINUX_PATH_MAX]; + proc_cwd_set_virtual( + path_bare_name(virt_path, cwd_buf, sizeof(cwd_buf))); return 0; } if (host_fd == -1) diff --git a/src/syscall/path.c b/src/syscall/path.c index a3e1369e..7ad64762 100644 --- a/src/syscall/path.c +++ b/src/syscall/path.c @@ -46,10 +46,45 @@ bool path_prefix_match(const char *path, const char *prefix, size_t plen) #define SYSFS_PREFIX "/sys" #define DEV_USB_PREFIX "/dev/bus" +static size_t bare_len(const char *path) +{ + size_t end = strlen(path); + for (;;) { + if (end >= 2 && path[end - 1] == '.' && path[end - 2] == '/') + end -= 1; + else if (end >= 2 && path[end - 1] == '/') + end -= 1; + else + break; + } + return end; +} + +bool path_dir_required(const char *path) +{ + return bare_len(path) != strlen(path); +} + +const char *path_bare_name(const char *path, char *buf, size_t bufsz) +{ + size_t end = bare_len(path); + if (end == strlen(path) || end + 1 > bufsz) + return path; + memcpy(buf, path, end); + buf[end] = '\0'; + return buf; +} + +/* Both gates read the name without its directory requirement, which the + * intercepts enforce for themselves: an exact-match test below would otherwise + * miss "/dev/fuse/" and hand a name the intercepts serve to the host. + */ bool path_might_use_open_intercept(const char *path) { if (!path || path[0] != '/') return false; + char bare[LINUX_PATH_MAX]; + path = path_bare_name(path, bare, sizeof(bare)); if (!strncmp(path, "/proc", 5)) return true; @@ -186,6 +221,8 @@ bool path_might_use_stat_intercept(const char *path) { if (!path || path[0] != '/') return false; + char bare[LINUX_PATH_MAX]; + path = path_bare_name(path, bare, sizeof(bare)); if (!strncmp(path, "/proc", 5)) return true; @@ -208,7 +245,8 @@ bool path_might_use_stat_intercept(const char *path) if (path_prefix_match(path, DEV_USB_PREFIX, sizeof(DEV_USB_PREFIX) - 1)) return true; - return false; + /* Synthesized on open, so it has to exist for stat too. */ + return !strcmp(path, "/etc/mtab"); } int path_check_intercept_access(const struct stat *st, int mode, int flags) @@ -489,6 +527,62 @@ static bool resolve_fd_magiclink_host_path(const char *path, return true; } +/* Whether an absolute name holds anything path_fold_components would remove: a + * "//" run, or a "." component with a slash after it. + */ +static bool path_has_foldable(const char *p) +{ + for (; *p; p++) { + if (p[0] == '/' && (p[1] == '/' || (p[1] == '.' && p[2] == '/'))) + return true; + } + return false; +} + +/* Fold "//" runs and "." components out of an absolute name, the way the Linux + * path walk steps over them. + * + * A "." that is the last component stays, with the slash after it if it has + * one, because the last component is where Linux gives it a meaning of its own: + * rmdir("d/.") is EINVAL where rmdir("d/") removes d. ".." is left alone, + * because Linux applies it to what the component before it resolved to, which + * is not a lexical question. + * + * @out may be @path: the result is never longer than what has been read. + */ +static void path_fold_components(const char *path, char *out) +{ + const char *p = path; + size_t w = 0; + + for (;;) { + bool slash = *p == '/'; + while (*p == '/') + p++; + if (!*p) { + if (slash) + out[w++] = '/'; + break; + } + + const char *seg = p; + while (*p && *p != '/') + p++; + size_t n = (size_t) (p - seg); + + const char *next = p; + while (*next == '/') + next++; + if (n == 1 && seg[0] == '.' && *next) + continue; + + out[w++] = '/'; + memmove(out + w, seg, n); + w += n; + } + out[w] = '\0'; +} + int path_translate_at(guest_fd_t dirfd, const char *path, unsigned int flags, @@ -534,6 +628,24 @@ int path_translate_at(guest_fd_t dirfd, } } + /* Linux folds "//" runs and "." components in the one walk every path + * syscall shares, so no filesystem is handed them. The intercepts match + * literal prefixes, here and behind this function, so the name is folded + * here once rather than by each of them. A relative name needs nothing: the + * two resolvers above fold what they join, and what they decline goes to + * the host as written. + */ + if (tx->guest_path[0] == '/' && path_has_foldable(tx->guest_path)) { + char *folded = + tx->guest_path == tx->proc_path ? tx->proc_path : tx->guest_buf; + if (folded == tx->guest_path || + strlen(tx->guest_path) < sizeof(tx->guest_buf)) { + path_fold_components(tx->guest_path, folded); + tx->guest_path = folded; + tx->intercept_path = folded; + } + } + /* A /sys walk that passes through one of the synthetic USB `subsystem` * symlinks is rewritten to the canonical guest spelling of where it lands, * before anything decides whose name it is. The links exist only in the @@ -573,7 +685,8 @@ int path_translate_at(guest_fd_t dirfd, * which must force nofollow on the host call; see dev_shm_resolve_path() * for that invariant. */ - if (!strncmp(tx->guest_path, "/dev/shm/", 9) && tx->guest_path[9] != '\0') { + if (!strncmp(tx->guest_path, "/dev/shm/", 9) && tx->guest_path[9] != '\0' && + !path_dir_required(tx->guest_path + 8)) { if (proc_dev_shm_resolve(tx->guest_path + 9, tx->host_buf, sizeof(tx->host_buf)) < 0) return -1; @@ -582,6 +695,27 @@ int path_translate_at(guest_fd_t dirfd, return 0; } + /* "/dev/stdout/" names the directory the descriptor's file would be, which + * a pipe, a terminal or a regular file is not. The host walk applies that + * rule to every other name; a magic link it never sees, and on a pipe it + * has no path to see it by. + */ + char link_bare[LINUX_PATH_MAX]; + const char *link_name = + path_bare_name(tx->guest_path, link_bare, sizeof(link_bare)); + if (link_name != tx->guest_path && tx->guest_path[0] == '/') { + host_fd_ref_t ref; + if (path_fd_magiclink_open(link_name, &ref) == 0) { + struct stat st; + bool is_dir = fstat(ref.fd, &st) == 0 && S_ISDIR(st.st_mode); + host_fd_ref_close(&ref); + if (!is_dir) { + errno = ENOTDIR; + return -1; + } + } + } + /* Only host_path moves; guest_path and intercept_path keep the /proc * spelling. open, stat and readlink never reach host_path for these paths: * proc_intercept_open dups the descriptor, proc_intercept_stat fstats it, @@ -609,7 +743,7 @@ int path_translate_at(guest_fd_t dirfd, */ if (tx->guest_path[0] == '/' && !(flags & (PATH_TR_NOFOLLOW | PATH_TR_CREATE)) && - resolve_fd_magiclink_host_path(tx->guest_path, tx->host_buf, + resolve_fd_magiclink_host_path(link_name, tx->host_buf, sizeof(tx->host_buf))) { tx->host_path = tx->host_buf; return 0; diff --git a/src/syscall/path.h b/src/syscall/path.h index 2f53b738..34c20b52 100644 --- a/src/syscall/path.h +++ b/src/syscall/path.h @@ -86,6 +86,18 @@ static inline int path_translation_at_flags(const path_translation_t *tx, */ bool path_prefix_match(const char *path, const char *prefix, size_t plen); +/* Whether @path ends in the directory requirement Linux attaches to a trailing + * slash or a final ".": the name has to resolve to a directory, and a final + * symlink is followed to get there. + */ +bool path_dir_required(const char *path); + +/* @path without that requirement, into @buf when there was one to take off, or + * @path itself when there was not. The intercepts match names literally and + * enforce the requirement for themselves; getcwd reports the directory. + */ +const char *path_bare_name(const char *path, char *buf, size_t bufsz); + /* Advance *pathp to the next '/'-separated component, skipping empty segments * from repeated slashes. * diff --git a/tests/test-matrix.sh b/tests/test-matrix.sh index dd726ec6..b06f0d9d 100755 --- a/tests/test-matrix.sh +++ b/tests/test-matrix.sh @@ -806,6 +806,7 @@ run_unit_tests() printf "\n/proc and /dev\n" test_check "$runner" "test-proc" "0 failed" "$bindir/test-proc" test_check "$runner" "test-sysfs-cpu" "0 failed" "$bindir/test-sysfs-cpu" + test_check "$runner" "test-path-fold" "0 failed" "$bindir/test-path-fold" test_rc "$runner" "test-procfs" 0 "$bindir/test-procfs" test_rc "$runner" "test-procfs-exec" 0 "$bindir/test-procfs-exec" test_rc "$runner" "test-proc-limits" 0 "$bindir/test-proc-limits" @@ -1527,9 +1528,15 @@ run_suite() # stop at its neighbors rather than extend page tables over them. No fixture, # not in either skip list, so it runs in both lanes. 253 and 228, observed here # at 295 and 273. +# +# Both went up by two for test-shim-sigreturn-x8 and test-path-fold. The first +# was registered with one binary run by hand behind it rather than a lane, so +# its floor was left for a run that observed one; the second holds every +# respelling of a path to the answer its canonical spelling gets. No fixture, +# not in either skip list. 255 and 230, observed here at 297 and 275. EXPECTED_BASELINES=( - "elfuse-aarch64|253|0" - "qemu-aarch64|228|0" + "elfuse-aarch64|255|0" + "qemu-aarch64|230|0" "elfuse-x86_64:apple-m1-m2|71|0" "elfuse-x86_64:apple-m3-plus|71|0" "elfuse-x86_64:apple-unknown|71|0" diff --git a/tests/test-path-fold.c b/tests/test-path-fold.c new file mode 100644 index 00000000..451835e0 --- /dev/null +++ b/tests/test-path-fold.c @@ -0,0 +1,458 @@ +/* + * test-path-fold.c -- one object, one answer, however the path is spelled. + * + * Copyright 2026 elfuse contributors + * SPDX-License-Identifier: Apache-2.0 + * + * Linux steps over "//" runs and "." components in the path walk every path + * syscall shares, so "//sys/bus", "/./sys/bus", "/sys/./bus" and "/sys//bus" + * are "/sys/bus" to all of them. elfuse serves part of the namespace from + * intercepts that match literal prefixes, and a name that reaches one unfolded + * is answered by whoever its spelling happens to match: one entry point serves + * it and the next reports ENOENT. + * + * Each name below is asked at every entry point under its canonical spelling, + * and then again under six respellings. The canonical answer is the expected + * one, so nothing here is recorded: a name this kernel or this build does not + * have answers ENOENT under every spelling, and that agrees too. + * + * The last component is different: a "." or a slash that ends a name is not + * noise but a requirement that the name resolve to a directory, and rmdir + * refuses the "." outright. Each name is asked under both endings too, and held + * to what Linux makes of them: a directory answers as itself, anything else + * answers ENOTDIR. The second half asserts the creation side against the errno + * Linux gives, with a symlink target beside it, which is stored as written + * because it is not walked. + * + * Plain Linux path semantics throughout, so the reference kernel adjudicates it + * and the test is registered in tests/test-matrix.sh. + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +static int failures; +static int checked; + +/* An answer, rendered: what kind of object and, where the entry point has more + * to say than that, what it said. Two spellings agree when the strings do. + */ +#define ANSWER_MAX 96 + +static void render_errno(char *out, int err) +{ + snprintf(out, ANSWER_MAX, "E%d", err); +} + +static void render_stat(char *out, const struct stat *st) +{ + snprintf(out, ANSWER_MAX, "mode=%o rdev=%u:%u", (unsigned) st->st_mode, + major(st->st_rdev), minor(st->st_rdev)); +} + +typedef void (*entry_fn)(const char *path, char *out); + +static void do_stat(const char *path, char *out) +{ + struct stat st; + if (stat(path, &st) < 0) + render_errno(out, errno); + else + render_stat(out, &st); +} + +static void do_lstat(const char *path, char *out) +{ + struct stat st; + if (lstat(path, &st) < 0) + render_errno(out, errno); + else + render_stat(out, &st); +} + +static void do_fstatat(const char *path, char *out) +{ + struct stat st; + if (fstatat(AT_FDCWD, path, &st, AT_SYMLINK_NOFOLLOW) < 0) + render_errno(out, errno); + else + render_stat(out, &st); +} + +static void do_access(const char *path, char *out) +{ + if (access(path, F_OK) < 0) + render_errno(out, errno); + else + snprintf(out, ANSWER_MAX, "ok"); +} + +/* O_NONBLOCK and O_NOCTTY so a device node neither parks the test nor becomes + * its terminal. The descriptor's own identity is part of the answer: a name + * served by one spelling and passed to the host by another opens either way and + * only the fstat tells them apart. + */ +static void do_open(const char *path, char *out) +{ + int fd = open(path, O_RDONLY | O_NONBLOCK | O_NOCTTY | O_CLOEXEC); + if (fd < 0) { + render_errno(out, errno); + return; + } + struct stat st; + if (fstat(fd, &st) < 0) + render_errno(out, errno); + else + render_stat(out, &st); + close(fd); +} + +static void do_statfs(const char *path, char *out) +{ + struct statfs sf; + if (statfs(path, &sf) < 0) + render_errno(out, errno); + else + snprintf(out, ANSWER_MAX, "type=%lx", (unsigned long) sf.f_type); +} + +static void do_readlink(const char *path, char *out) +{ + char buf[ANSWER_MAX - 8]; + ssize_t n = readlink(path, buf, sizeof(buf) - 1); + if (n < 0) { + render_errno(out, errno); + return; + } + buf[n] = '\0'; + snprintf(out, ANSWER_MAX, "-> %s", buf); +} + +/* The number of entries, which is what a listing served by the wrong owner gets + * wrong. + */ +static void do_getdents(const char *path, char *out) +{ + DIR *d = opendir(path); + if (!d) { + render_errno(out, errno); + return; + } + int n = 0; + while (readdir(d)) + n++; + closedir(d); + snprintf(out, ANSWER_MAX, "%d entries", n); +} + +/* chdir and what getcwd then reports, in a child so the cwd of the test does + * not move. getcwd names the directory, never the spelling that reached it. + */ +static void do_chdir(const char *path, char *out) +{ + int fds[2]; + if (pipe(fds) < 0) { + render_errno(out, errno); + return; + } + pid_t pid = fork(); + if (pid < 0) { + render_errno(out, errno); + close(fds[0]); + close(fds[1]); + return; + } + if (pid == 0) { + char msg[ANSWER_MAX]; + close(fds[0]); + if (chdir(path) < 0) { + render_errno(msg, errno); + } else { + char cwd[ANSWER_MAX - 8]; + if (!getcwd(cwd, sizeof(cwd))) + render_errno(msg, errno); + else + snprintf(msg, sizeof(msg), "cwd %s", cwd); + } + if (write(fds[1], msg, strlen(msg) + 1) < 0) + _exit(1); + _exit(0); + } + close(fds[1]); + ssize_t n = read(fds[0], out, ANSWER_MAX - 1); + close(fds[0]); + out[n > 0 ? n : 0] = '\0'; + int status; + waitpid(pid, &status, 0); +} + +static const struct { + const char *name; + entry_fn fn; +} entries[] = { + {"stat", do_stat}, {"lstat", do_lstat}, + {"fstatat", do_fstatat}, {"access", do_access}, + {"open", do_open}, {"statfs", do_statfs}, + {"readlink", do_readlink}, {"getdents", do_getdents}, + {"chdir", do_chdir}, +}; + +#define N_ENTRIES (sizeof(entries) / sizeof(entries[0])) + +/* Six respellings of an absolute name with at least two components. The last + * two interleave a "." and a "//": stepping over the "." in "/.//x" uncovers a + * "//" a finished first pass no longer sees, and "/././/x" is the same a turn + * deeper. + */ +#define N_SPELLINGS 6 + + +static int respell(const char *canon, int which, char *out, size_t outsz) +{ + const char *second = strchr(canon + 1, '/'); + if (!second) + return -1; + int head = (int) (second - canon); + int n; + switch (which) { + case 0: + n = snprintf(out, outsz, "/%s", canon); + break; + case 1: + n = snprintf(out, outsz, "/.%s", canon); + break; + case 2: + n = snprintf(out, outsz, "%.*s/.%s", head, canon, second); + break; + case 3: + n = snprintf(out, outsz, "%.*s/%s", head, canon, second); + break; + case 4: + n = snprintf(out, outsz, "/./%s", canon); + break; + default: + n = snprintf(out, outsz, "/././%s", canon); + break; + } + return n > 0 && (size_t) n < outsz ? 0 : -1; +} + +/* What the canonical spelling says a respelling should answer. For the six + * respellings that is the canonical answer itself. For "/." and + * "/" it is the directory @canon resolves to, or ENOTDIR when it is not + * one; a final symlink is followed to get there, so lstat and readlink answer + * for the directory, not the link. + */ +static void expect(size_t e, const char *canon, bool trailing, char *want) +{ + const char *name = entries[e].name; + struct stat st, lst; + + if (!trailing || stat(canon, &st) < 0) { + entries[e].fn(canon, want); + return; + } + bool is_link = lstat(canon, &lst) == 0 && S_ISLNK(lst.st_mode); + if (!S_ISDIR(st.st_mode)) + render_errno(want, ENOTDIR); + else if (is_link && (!strcmp(name, "lstat") || !strcmp(name, "fstatat"))) + do_stat(canon, want); + else if (is_link && !strcmp(name, "readlink")) + render_errno(want, EINVAL); + else + entries[e].fn(canon, want); +} + +/* Ask @path, with the canonical answer taken on both sides of it. Some of these + * directories list live state: /dev/fd lists the caller's descriptors and + * /dev/shm what every process of the user has made, either of which can change + * between two adjacent calls for reasons that have nothing to do with a + * spelling. A canonical answer that moved during the sample is no answer to + * hold the respelling to, so the sample is taken again; a respelling that + * differs from a canonical answer that held still fails. + */ +#define SAMPLE_TRIES 8 + +static void check_one(size_t e, + const char *canon, + const char *path, + bool trailing) +{ + char want[ANSWER_MAX], got[ANSWER_MAX], after[ANSWER_MAX]; + + checked++; + for (int t = 0; t < SAMPLE_TRIES; t++) { + expect(e, canon, trailing, want); + entries[e].fn(path, got); + expect(e, canon, trailing, after); + if (strcmp(want, after) != 0) + continue; + if (strcmp(want, got) != 0) { + printf("FAIL %s %s: %s, want %s as %s answers\n", entries[e].name, + path, got, want, canon); + failures++; + } + return; + } + printf("FAIL %s %s: %s never answered the same twice in %d samples\n", + entries[e].name, path, canon, SAMPLE_TRIES); + failures++; +} + +static void check_name(const char *canon) +{ + static const char *const endings[] = {"/.", "/"}; + char path[PATH_MAX]; + + for (size_t e = 0; e < N_ENTRIES; e++) { + for (int s = 0; s < N_SPELLINGS; s++) + if (respell(canon, s, path, sizeof(path)) == 0) + check_one(e, canon, path, false); + for (size_t s = 0; s < 2; s++) + if (snprintf(path, sizeof(path), "%s%s", canon, endings[s]) < + (int) sizeof(path)) + check_one(e, canon, path, true); + } +} + +/* First entry of @dir that is not "." or "..", as "/". */ +static int first_entry(const char *dir, char *out, size_t outsz) +{ + DIR *d = opendir(dir); + if (!d) + return -1; + struct dirent *de; + int rc = -1; + while ((de = readdir(d))) { + if (!strcmp(de->d_name, ".") || !strcmp(de->d_name, "..")) + continue; + int n = snprintf(out, outsz, "%s/%s", dir, de->d_name); + if (n > 0 && (size_t) n < outsz) + rc = 0; + break; + } + closedir(d); + return rc; +} + +static void expect_errno(const char *what, int rc, int want) +{ + int got = rc < 0 ? errno : 0; + checked++; + if (got != want) { + printf("FAIL %s: errno %d, want %d\n", what, got, want); + failures++; + } +} + +/* The creation side. A "." that ends the name keeps the meaning Linux gives it + * there, and a link target is not walked at all. + */ +static void check_last_component(void) +{ + char dir[] = "/tmp/test-path-fold-XXXXXX"; + if (!mkdtemp(dir)) { + perror("mkdtemp"); + failures++; + return; + } + + char file[PATH_MAX], sub[PATH_MAX], path[PATH_MAX]; + snprintf(file, sizeof(file), "%s/f", dir); + snprintf(sub, sizeof(sub), "%s/d", dir); + int fd = open(file, O_CREAT | O_WRONLY | O_CLOEXEC, 0600); + if (fd < 0 || mkdir(sub, 0700) < 0) { + perror("setup"); + failures++; + return; + } + close(fd); + + snprintf(path, sizeof(path), "%s//./d/.", dir); + expect_errno("rmdir of a directory spelled d/.", rmdir(path), EINVAL); + + snprintf(path, sizeof(path), "%s/.//d/", dir); + expect_errno("rmdir of a directory spelled d/", rmdir(path), 0); + + /* A link's target is data, not a path being walked: it is stored as written + * and read back the same, noise included. + */ + static const char target[] = "/tmp//no/./such/."; + char link[PATH_MAX], back[sizeof(target) + 8]; + snprintf(link, sizeof(link), "%s//./l", dir); + expect_errno("symlink with a noisy target", symlink(target, link), 0); + ssize_t n = readlink(link, back, sizeof(back) - 1); + back[n > 0 ? n : 0] = '\0'; + checked++; + if (strcmp(back, target) != 0) { + printf("FAIL readlink: target reads back \"%s\", want \"%s\"\n", back, + target); + failures++; + } + expect_errno("unlink of the link", unlink(link), 0); + + snprintf(path, sizeof(path), "%s/./f", dir); + expect_errno("unlink through an interior .", unlink(path), 0); + expect_errno("rmdir of the scratch directory", rmdir(dir), 0); +} + +int main(void) +{ + /* One name from each owner the namespace has: procfs, the host's own /dev, + * the devpts and shm directories, sysfs and its CPU subtree, the synthetic + * USB trees, the synthesized /etc files, and a directory nothing + * intercepts. The USB names are absent on a machine with no device and no + * fixture, which agrees under every spelling like any other absence. + */ + static const char *const names[] = { + "/proc/self/status", + "/proc/self/exe", + "/proc/sys/kernel", + "/dev/null", + "/dev/urandom", + "/dev/pts", + "/dev/shm", + "/dev/fd", + "/dev/stdout", + "/sys/devices", + "/sys/devices/system/cpu/online", + "/sys/bus/usb/devices", + "/dev/bus/usb", + "/etc/passwd", + "/etc/mtab", + "/tmp/no-such-name-here", + }; + + for (size_t i = 0; i < sizeof(names) / sizeof(names[0]); i++) + check_name(names[i]); + + /* A device directory and the node inside it, discovered: their numbers are + * the host's, and the node is two components below the literal the layer + * that serves it matches. + */ + char busdir[PATH_MAX], node[PATH_MAX], sysdev[PATH_MAX]; + if (first_entry("/dev/bus/usb", busdir, sizeof(busdir)) == 0) { + check_name(busdir); + if (first_entry(busdir, node, sizeof(node)) == 0) + check_name(node); + } + if (first_entry("/sys/bus/usb/devices", sysdev, sizeof(sysdev)) == 0) + check_name(sysdev); + + check_last_component(); + + if (!failures) + printf("PASS: %d answers agree with the canonical spelling\n", checked); + printf("%d failed\n", failures); + return failures ? 1 : 0; +}