Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 24 additions & 0 deletions docs/internals.md
Original file line number Diff line number Diff line change
Expand Up @@ -1874,6 +1874,30 @@ final component may not exist yet. `path_translate_at()` picks one by flags;
in the `openat2(RESOLVE_NO_SYMLINKS)` precheck described below, and walks
relative paths from their descriptor with the same clamp applied in the walk.

### Folding `//` And `.`

Linux steps over `//` runs and `.` components in the walk every path syscall
shares, so `//sys/bus`, `/./sys/bus` and `/sys/./bus` are `/sys/bus` to all of
them. The intercepts match literal prefixes, so `path_translate_at()` folds an
absolute name once, before any of them reads it.

Two things stay as written. `..` is not folded, because Linux applies it to
what the component before it resolved to. And a `.` that is the last component
keeps its place, because that is where Linux gives it a meaning of its own:
`rmdir("d/.")` is `EINVAL` where `rmdir("d/")` removes `d`.

What a final `.` or a trailing slash means to a lookup is a requirement: the
name has to resolve to a directory, following a final symlink to get there. The
host walk applies it for itself. The intercepts match names literally, so
`proc_intercept_open()`, `proc_intercept_stat_at()` and
`proc_intercept_readlink()` take the requirement off the name once, dispatch on
the bare name, and enforce it on the answer: a served directory answers as
itself, anything else answers `ENOTDIR`, and `readlink` of a served directory
answers `EINVAL`. The gates in `path.c` read the bare name for the same reason.

`tests/test-path-fold.c` holds every respelling of a name to the answer its
canonical spelling gets, and both endings to what Linux makes of them.

### Clamping `..` At The Guest Root

A guest resolves `..` against its own root, and Linux clamps it there: `/..`
Expand Down
10 changes: 9 additions & 1 deletion mk/tests.mk
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,7 @@ ELFUSE_HOST_NOFILE_MIN ?= $(shell bash "$(CURDIR)/tests/test-config.sh" --host-n
test-sysroot-name-unique \
test-sysroot-name-relative \
test-nosysroot-literal-names test-sysroot-outside-names \
test-sysroot-root test-usb-sysfs test-usb-sysfs-sysroot \
test-sysroot-root test-path-fold test-usb-sysfs test-usb-sysfs-sysroot \
test-usb-sysfs-matrix \
test-usb-sysfs-overflow test-usbdev-ioctl test-usbdev-faults \
test-usbdev-urb-loopback test-usbdev-ioctl-loopback \
Expand Down Expand Up @@ -278,6 +278,7 @@ $(call run-host-unit,test-usb-desc-host,USB descriptor blob walk unit test)
$(call run-host-unit,test-usbdev-urb-host,usbdevfs URB bookkeeping unit test)
$(call run-host-unit,test-elf-headers-host,ELF header validation unit test)
$(call run-host-unit,test-gdbstub-host,buffered GDB session regression)
$(call run-lane,test-path-fold,one answer per object however its path is spelled)
$(call run-lane,test-usb-sysfs,synthetic USB tree contract)
$(call run-lane,test-usb-sysfs-sysroot,synthetic USB /sys sharing a populated sysroot)
$(call run-lane,test-usb-sysfs-matrix,every /sys and /dev/bus entry point against every path class)
Expand Down Expand Up @@ -1695,6 +1696,13 @@ test-casefold-host: $(BUILD_DIR)/test-casefold-host
test-casefold-walk-host: $(BUILD_DIR)/test-casefold-walk-host
$(BUILD_DIR)/test-casefold-walk-host

## Hold every respelling of a name to the canonical spelling's answer
# The matrix runs this against the reference kernel too. This lane adds the
# fixture, so the synthetic USB names are present on a machine with no device
# and do not agree merely by being absent under every spelling.
test-path-fold: $(ELFUSE_BIN) $(TEST_DIR)/test-path-fold
ELFUSE_USB_FIXTURE=1 $(ELFUSE_BIN) $(TEST_DIR)/test-path-fold

## Assert the synthetic USB tree's contract and its two agreeing views
# The device-dependent half only runs against whatever is attached, so the lane
# prints the device count rather than letting an empty bus read as full cover.
Expand Down
130 changes: 112 additions & 18 deletions src/runtime/procemu.c
Original file line number Diff line number Diff line change
Expand Up @@ -2292,10 +2292,10 @@ static int proc_open_mounts_node(const char *path)
* guest and the buffer, so a split would be a jump table by another name.
*/
/* NOLINTNEXTLINE(readability-function-size) */
int proc_intercept_open(const guest_t *g,
const char *path,
int linux_flags,
int mode)
static int intercept_open_dispatch(const guest_t *g,
const char *path,
int linux_flags,
int mode)
{
/* /dev/ptmx -> host /dev/ptmx + keepalive slave (see pty_open_master).
* O_PATH is path-only on Linux: it must not run the device open hook or
Expand Down Expand Up @@ -2387,7 +2387,7 @@ int proc_intercept_open(const guest_t *g,
* intercepted too or callers that probe then enumerate see inconsistent
* Linux-visible behavior.
*/
if (!strcmp(path, "/dev/pts") || !strcmp(path, "/dev/pts/"))
if (!strcmp(path, "/dev/pts"))
return pty_open_pts_dir(linux_flags);

/* /dev/pts/N -> the macOS slave path captured at /dev/ptmx open time.
Expand Down Expand Up @@ -2420,15 +2420,15 @@ int proc_intercept_open(const guest_t *g,
* "self" symlink. The DIR* created from this allows getdents64 to enumerate
* /proc like a real procfs. Cleaned up via atexit.
*/
if (!strcmp(path, "/proc") || !strcmp(path, "/proc/")) {
if (!strcmp(path, "/proc")) {
const char *dir = ensure_proc_tmpdir(g);
if (!dir)
return -1;
return proc_open_dir_fd(dir, linux_flags);
}

/* /proc/self -> directory fd for the PID subdirectory */
if (!strcmp(path, "/proc/self") || !strcmp(path, "/proc/self/")) {
if (!strcmp(path, "/proc/self")) {
const char *dir = ensure_proc_tmpdir(g);
if (!dir)
return -1;
Expand Down Expand Up @@ -3093,7 +3093,9 @@ int proc_intercept_stat(const char *path, struct stat *st)
return proc_intercept_stat_at(path, st, false);
}

int proc_intercept_stat_at(const char *path, struct stat *st, bool follow)
static int intercept_stat_dispatch(const char *path,
struct stat *st,
bool follow)
{
/* Intercept stat for /proc paths emulated via proc_intercept_open. Without
* this, runtime libraries that probe a file's existence via stat() before
Expand Down Expand Up @@ -3124,8 +3126,16 @@ int proc_intercept_stat_at(const char *path, struct stat *st, bool follow)
return 0;
}

/* The mount table open() synthesizes; a stat that said ENOENT for a name
* open() then served was one object with two answers.
*/
if (!strcmp(path, "/etc/mtab")) {
stat_fill_proc_file(st, 0444, path);
return 0;
}

/* /dev/shm is a directory */
if (!strcmp(path, "/dev/shm") || !strcmp(path, "/dev/shm/")) {
if (!strcmp(path, "/dev/shm")) {
stat_fill_proc_dir(st, 01777, 2,
path); /* sticky bit, like real /dev/shm */
return 0;
Expand All @@ -3149,7 +3159,7 @@ int proc_intercept_stat_at(const char *path, struct stat *st, bool follow)
* passes. The numeric tail must round-trip with /dev/ttysN via the open
* intercept (see proc_intercept_open).
*/
if (!strcmp(path, "/dev/pts") || !strcmp(path, "/dev/pts/")) {
if (!strcmp(path, "/dev/pts")) {
stat_fill_proc_dir(st, 0755, 2, path);
return 0;
}
Expand Down Expand Up @@ -3209,18 +3219,15 @@ int proc_intercept_stat_at(const char *path, struct stat *st, bool follow)
}

/* /proc and /proc/<our_pid> are directories */
if (!strcmp(path, "/proc") || !strcmp(path, "/proc/")) {
if (!strcmp(path, "/proc")) {
stat_fill_proc_dir(st, 0555, 3, path);
return 0;
}
{
char pidbuf[32], pidslash[32];
char pidbuf[32];
snprintf(pidbuf, sizeof(pidbuf), "/proc/%lld",
(long long) proc_get_pid());
snprintf(pidslash, sizeof(pidslash), "/proc/%lld/",
(long long) proc_get_pid());
if (!strcmp(path, pidbuf) || !strcmp(path, pidslash) ||
!strcmp(path, "/proc/self") || !strcmp(path, "/proc/self/")) {
if (!strcmp(path, pidbuf) || !strcmp(path, "/proc/self")) {
stat_fill_proc_dir(st, 0555, 3, path);
return 0;
}
Expand Down Expand Up @@ -3439,15 +3446,17 @@ static int proc_readlink_self_exe(char *buf, size_t bufsiz)
return (int) len;
}

int proc_intercept_readlink(const char *path, char *buf, size_t bufsiz)
static int intercept_readlink_dispatch(const char *path,
char *buf,
size_t bufsiz)
{
{
char alias[LINUX_PATH_MAX];
int aliased = proc_alias_self(path, alias, sizeof(alias));
if (aliased < 0)
return -1;
if (aliased > 0)
return proc_intercept_readlink(alias, buf, bufsiz);
return intercept_readlink_dispatch(alias, buf, bufsiz);
}

if (!strcmp(path, "/proc/self/exe"))
Expand Down Expand Up @@ -3696,3 +3705,88 @@ int proc_intercept_write(int guest_fd,
pthread_mutex_unlock(&oom_write_lock);
return rc;
}

/* What the requirement answers for @path once the dispatcher has been asked: 0
* when it names an intercepted directory, -1 with errno for an intercepted file
* or a lookup that failed, PROC_NOT_INTERCEPTED when the name is not ours and
* the host applies the requirement itself.
*/
static int intercept_require_dir(const char *stripped)
{
struct stat st;
int rc = intercept_stat_dispatch(stripped, &st, true);
if (rc != 0)
return rc;
if (!S_ISDIR(st.st_mode)) {
errno = ENOTDIR;
return -1;
}
return 0;
}

int proc_intercept_open(const guest_t *g,
const char *path,
int linux_flags,
int mode)
{
/* The directory requirement of a trailing slash or ".": the name has to
* resolve to a directory, following a final symlink to get there. The
* dispatchers match names literally, so it is taken off here, once, and
* enforced on what they answer.
*/
char buf[LINUX_PATH_MAX];
const char *name = path_bare_name(path, buf, sizeof(buf));
if (name != path) {
int rc = intercept_require_dir(name);
if (rc != 0)
return rc;
linux_flags &= ~LINUX_O_NOFOLLOW;
}
return intercept_open_dispatch(g, name, linux_flags, mode);
}

int proc_intercept_stat_at(const char *path, struct stat *st, bool follow)
{
char buf[LINUX_PATH_MAX];
const char *name = path_bare_name(path, buf, sizeof(buf));
if (name == path)
return intercept_stat_dispatch(path, st, follow);
int rc = intercept_stat_dispatch(name, st, true);
if (rc == 0 && !S_ISDIR(st->st_mode)) {
errno = ENOTDIR;
return -1;
}
return rc;
}

int proc_intercept_readlink(const char *path, char *buf, size_t bufsiz)
{
char name_buf[LINUX_PATH_MAX];
const char *name = path_bare_name(path, name_buf, sizeof(name_buf));
if (name != path) {
/* The requirement resolves the name to a directory, which is no link.
*/
int rc = intercept_require_dir(name);
if (rc == 0) {
errno = EINVAL;
return -1;
}
return rc;
}

int rc = intercept_readlink_dispatch(path, buf, bufsiz);
if (rc != PROC_NOT_INTERCEPTED)
return rc;

/* An object an intercept serves and no readlink arm names is not a link:
* Linux answers EINVAL for it, where the host, asked for a name it does not
* have, answered ENOENT.
*/
struct stat st;
if (intercept_stat_dispatch(path, &st, false) == 0 &&
!S_ISLNK(st.st_mode)) {
errno = EINVAL;
return -1;
}
return PROC_NOT_INTERCEPTED;
}
34 changes: 22 additions & 12 deletions src/syscall/fs-stat.c
Original file line number Diff line number Diff line change
Expand Up @@ -623,20 +623,31 @@ static int64_t sys_statfs_impl(guest_t *g,
if (tx.fuse_path)
return -LINUX_ENOSYS;

if (statfs_path_is_proc(tx.intercept_path)) {
if (proc_path_is_symlink(tx.intercept_path)) {
/* The directory requirement of a trailing slash or ".", applied before the
* classification below, which reads the bare name: a served file under it
* is ENOTDIR, and a served directory is classified as itself.
*/
char bare[LINUX_PATH_MAX];
const char *name = path_bare_name(tx.intercept_path, bare, sizeof(bare));
if (name != tx.intercept_path &&
path_might_use_stat_intercept(tx.intercept_path)) {
struct stat st;
if (proc_intercept_stat_at(tx.intercept_path, &st, true) == -1)
return linux_errno();
}

if (statfs_path_is_proc(name)) {
if (proc_path_is_symlink(name)) {
char link[LINUX_PATH_MAX];
int len = proc_intercept_readlink(tx.intercept_path, link,
sizeof(link) - 1);
int len = proc_intercept_readlink(name, link, sizeof(link) - 1);
if (len < 0)
return linux_errno();
link[len] = '\0';
return sys_statfs_impl(g, link, buf_gva, depth + 1);
}

struct stat mac_st;
int intercepted =
proc_intercept_stat_at(tx.intercept_path, &mac_st, true);
int intercepted = proc_intercept_stat_at(name, &mac_st, true);
if (intercepted == 0) {
linux_statfs_t lin_st;
fill_proc_statfs(&lin_st);
Expand Down Expand Up @@ -667,7 +678,7 @@ static int64_t sys_statfs_impl(guest_t *g,
* report the same ENOENT a real Linux kernel would for the rest.
*/
char sys_abs[LINUX_PATH_MAX];
if (statfs_path_is_sysfs(tx.intercept_path, sys_abs, sizeof(sys_abs))) {
if (statfs_path_is_sysfs(name, sys_abs, sizeof(sys_abs))) {
/* Classifying the folded name and then probing the raw one asked two
* different questions of two different spellings: "/sys/../sys" is
* sysfs, but no intercept and no host backing carries that literal
Expand All @@ -676,7 +687,7 @@ static int64_t sys_statfs_impl(guest_t *g,
* fallback and the answer all describe the same object. Folding is
* idempotent, so this recurses at most once.
*/
if (strcmp(sys_abs, tx.intercept_path) != 0)
if (strcmp(sys_abs, name) != 0)
return sys_statfs_impl(g, sys_abs, buf_gva, depth + 1);

bool exists = sys_abs[4] == '\0'; /* "/sys" itself */
Expand All @@ -699,7 +710,7 @@ static int64_t sys_statfs_impl(guest_t *g,
return 0;
}

int dev_bus = statfs_dev_bus_class(tx.intercept_path);
int dev_bus = statfs_dev_bus_class(name);
if (dev_bus == -1)
return linux_errno();
if (dev_bus == 0) {
Expand All @@ -716,7 +727,7 @@ static int64_t sys_statfs_impl(guest_t *g,
* glibc closes a perfectly good master -- breaking Unix98 pty allocation
* for every glibc program. Answer from the virtual filesystem instead.
*/
devpts_class_t devpts = statfs_devpts_class(tx.intercept_path);
devpts_class_t devpts = statfs_devpts_class(name);
if (devpts == DEVPTS_ABSENT)
return -LINUX_ENOENT;
if (devpts == DEVPTS_MOUNT) {
Expand All @@ -731,8 +742,7 @@ static int64_t sys_statfs_impl(guest_t *g,
* on the leaf would follow a symlink onto the host and leak the host fs
* identity, so answer synthetically; lstat is the nofollow existence probe.
*/
bool shm_root = !strcmp(tx.intercept_path, "/dev/shm") ||
!strcmp(tx.intercept_path, "/dev/shm/");
bool shm_root = !strcmp(name, "/dev/shm");
if (tx.is_dev_shm || shm_root) {
const char *shm_dir = proc_get_shm_dir();
if (!shm_dir)
Expand Down
Loading
Loading