diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index 558bcb2bfe1..5b327c87940 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -62,7 +62,7 @@ jobs: fetch-depth: 0 # CMake and Ninja are what bootstrap builds LLVM and clang with, pinned to - # the versions Ubuntu 24.04 released (src/main.rs's `ALSO_USED`). + # the versions Ubuntu 24.04 released. - name: disk, QEMU, CMake and Ninja run: | sudo rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc \ diff --git a/README.md b/README.md index 670bfac6b13..6aa11f736d9 100644 --- a/README.md +++ b/README.md @@ -240,8 +240,7 @@ ninja`); on Debian and Ubuntu they are `build-essential`, `python3`, `cmake` and `ninja-build`. `cargo run` names anything it needs and cannot find, before it does anything -else — including the Python that only the toolchain bootstrap runs, which -costs that bootstrap rather than the build. +else. Everything this project depends on that it did not write is named where it is carried: `NOTICE` lists every committed third-party file with its hash, upstream and licence. diff --git a/issues/build/a-finder-file-in-a-store-directory-panics-its-sweep.md b/issues/build/a-finder-file-in-a-store-directory-panics-its-sweep.md new file mode 100644 index 00000000000..3ee324dca57 --- /dev/null +++ b/issues/build/a-finder-file-in-a-store-directory-panics-its-sweep.md @@ -0,0 +1,16 @@ +--- +status: open +kind: defect +opened: 2026-09-30 +--- + +# A Finder file in a store directory panics its sweep + +`keystore::sweep_by` takes every entry of a store directory as `` or +`.`, so a `.DS_Store` the macOS Finder writes there names the key +`""`, and `buildlock::keyed_idle` then opens the lock directory itself: +`build lock: open /.git/toyos-build-locks/llvm/: Is a directory (os +error 21)`. It happened after an LLVM placement, whose product was whole, so +the next build went on; the sweep had removed nothing. + +**Exit**: a sweep passes over a store entry that names no key, with a test. diff --git a/issues/build/a-rust-std-binary-cannot-link-the-cxx-runtime.md b/issues/build/a-rust-std-binary-cannot-link-the-cxx-runtime.md new file mode 100644 index 00000000000..e586514d080 --- /dev/null +++ b/issues/build/a-rust-std-binary-cannot-link-the-cxx-runtime.md @@ -0,0 +1,27 @@ +--- +status: open +kind: defect +opened: 2026-09-30 +--- + +# A Rust std binary cannot link the C++ runtime + +LLVM is C++, so a rustc that carries it links the C++ runtime into +`librustc_driver.so` beside std. The runtime #637 builds into the C sysroot, +`lib/libc++.a` (sysroot `de0de8ee7862147a`), does not link there. Linking +rustc's LLVM wrapper (`compiler/rustc_llvm/llvm-wrapper`, whole), the 71 LLVM +archives rustc links, `libc++.a` and the Rust sysroot's archives with `ld.lld +-shared` gives: + +- **Two unwinders.** `libc++.a` carries LLVM's libunwind + (`LIBCXXABI_ENABLE_STATIC_UNWINDER`) and std carries the `unwinding` crate, + and both define the Itanium `_Unwind_*` interface: 17 duplicate symbols, + `_Unwind_Resume` among them, at libunwind's `UnwindLevel1.c` and at + `unwinding`'s `src/unwinder/mod.rs:346`. +- **Local-exec thread-locals.** libc++abi's `eh_globals` is reached through + `R_X86_64_TPOFF32`, twice `cannot be used with -shared`: the runtime is + compiled as position-independent executable code, not for a shared object. + +**Exit**: a Rust std shared object and a Rust std executable each link the C++ +runtime with one unwinder and no relocation error, and a guest case runs a Rust +std program in which C++ code throws and catches an exception. diff --git a/issues/build/libc-headers-are-written-by-hand-and-drift-from-its-definitions.md b/issues/build/libc-headers-are-written-by-hand-and-drift-from-its-definitions.md index 9e955da48fb..6e6a255611a 100644 --- a/issues/build/libc-headers-are-written-by-hand-and-drift-from-its-definitions.md +++ b/issues/build/libc-headers-are-written-by-hand-and-drift-from-its-definitions.md @@ -6,7 +6,7 @@ opened: 2026-09-27 # libc's C headers are written by hand, and nothing holds them to its definitions -`userland/libc/include/` is 32 headers typed beside the Rust that defines what +`userland/libc/include/` is headers typed beside the Rust that defines what they declare, and no build or test compares the two. clang, compiling doomgeneric for the first time, found two places they had already parted: @@ -20,8 +20,7 @@ doomgeneric for the first time, found two places they had already parted: Both are fixed in the headers. The class is not: a signature changed in `userland/libc/src` changes no header, and the C sysroot ships whatever the headers say. The corpus shows the -surface is also incomplete — `stdint.h` has no `least`/`fast` types, so clang's -own `stdatomic.h` does not compile (`124_atomic_counter`), and `pthread.h` and +surface is also incomplete — `pthread.h` and `signal.h` stop short of `PTHREAD_PROCESS_SHARED` and `SIGUSR1`. **Generating them with cbindgen was priced and not taken**: it is a new diff --git a/issues/build/libc-printf-re-encodes-every-non-ascii-byte-of-its-format.md b/issues/build/libc-printf-re-encodes-every-non-ascii-byte-of-its-format.md deleted file mode 100644 index 84dfab50b23..00000000000 --- a/issues/build/libc-printf-re-encodes-every-non-ascii-byte-of-its-format.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# libc's `printf` re-encodes every non-ASCII byte of its format string - -`userland/libc/src/printf.rs:269` copies a format's literal bytes with -`w.write_char(fmt[i] as char)`, which turns each byte of a UTF-8 sequence into -the Latin-1 character of the same number and writes that character's UTF-8. So -`printf("привет=%g\n", x)` writes `пÑ\u{80}ивеÑ\u{82}=…`: the bytes are -doubled and wrong. A `%s` argument is not affected. - -Found by TinyCC's `83_utf8_in_identifiers`, which toyos-cc never compiled and -clang does; it is declined in `tests/toyos.rs`'s `NOT_RUN` with this entry named. - -**Exit**: a format's bytes reach the output unchanged, and -`83_utf8_in_identifiers` runs. diff --git a/issues/build/libc-pthread-keys-and-pthread-self-answer-for-the-process.md b/issues/build/libc-pthread-keys-and-pthread-self-answer-for-the-process.md deleted file mode 100644 index 257436dadb0..00000000000 --- a/issues/build/libc-pthread-keys-and-pthread-self-answer-for-the-process.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-28 ---- - -# libc's pthread keys and `pthread_self` answer for the process, not the thread - -`userland/libc/src/pthread.rs` indexes `pthread_getspecific` and -`pthread_setspecific` by `thread_index()`, which is `getpid() % 64`, and -`pthread_self` returns the pid. Every thread of a process reads and writes one -row of `TLS_VALUES` and gets one `pthread_t`: a key set in one thread is seen by -all of them, and `pthread_equal(pthread_self(), t)` holds for any two threads -of a process. Nothing says so; a C program that keeps per-thread state in a key -shares it. - -**Exit**: a key's value and `pthread_self` are the calling thread's, and a guest -C test that sets a key in one thread and reads it from another reds when they -are shared. diff --git a/issues/build/libcxx-is-built-without-std-filesystem.md b/issues/build/libcxx-is-built-without-std-filesystem.md new file mode 100644 index 00000000000..ff743a45db6 --- /dev/null +++ b/issues/build/libcxx-is-built-without-std-filesystem.md @@ -0,0 +1,15 @@ +--- +status: open +kind: defect +opened: 2026-09-30 +--- + +# libc++ for ToyOS is built without `std::filesystem` + +`src/libcxx.rs` configures the C++ runtime with `LIBCXX_ENABLE_FILESYSTEM=OFF`, +so a ToyOS C++ program that includes `` does not compile. libc++'s +filesystem is written on the POSIX directory, `stat` and path calls, and libc +has no `dirent.h`. + +**Exit**: libc carries what libc++'s `src/filesystem` calls, the option goes, +and a guest test lists a directory through `std::filesystem`. diff --git a/issues/build/libcxx-takes-the-generic-locale-path-on-toyos.md b/issues/build/libcxx-takes-the-generic-locale-path-on-toyos.md new file mode 100644 index 00000000000..278b48b885d --- /dev/null +++ b/issues/build/libcxx-takes-the-generic-locale-path-on-toyos.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-30 +--- + +# libc++ takes its generic locale path on ToyOS + +`libcxx/include/__locale_dir/locale_base_api.h` in `ToyOSOrg/llvm-project` +sends ToyOS down its fallback arm, the one it marks temporary: libc++ reaches +the locale through global `_l` names and `bsd_locale_fallbacks.h`, which is why +libc's `locale.rs` carries every `_l` function libc++ calls. Upstream moves +each platform to a header of its own under `__locale_dir/support/`, and a new +platform is asked for one. + +ToyOS's is Fuchsia's shape: one C locale, `uselocale` around the calls that +read it, and `support/no_locale/`'s characters and number readers. + +**Exit**: `__locale_dir/support/toyos.h`, its `__toyos__` arm in +`locale_base_api.h`, and its entries in `libcxx/include/CMakeLists.txt` and +the module map, in the fork; libc's `_l` functions that only the fallback +called go. It moves the LLVM key, so every host builds LLVM again: it rides +with the next change to the fork's `src/llvm-project`. diff --git a/issues/build/the-build-runs-host-tools-outside-rust-and-qemu.md b/issues/build/the-build-runs-host-tools-outside-rust-and-qemu.md index 77ff0585707..9d993d77588 100644 --- a/issues/build/the-build-runs-host-tools-outside-rust-and-qemu.md +++ b/issues/build/the-build-runs-host-tools-outside-rust-and-qemu.md @@ -14,11 +14,12 @@ arrives and is not one. M4 and M5 are stages of `issues/build/toyos-builds-itsel | tool | runs | verdict | exit | |---|---|---|---| -| Python | LLVM's CMake, whenever this host builds an LLVM (`src/llvm.rs`) | admitted: no Rust tool does the job, LLVM's CMake requires one (`find_package(Python3 … REQUIRED)` in `rust/src/llvm-project/llvm/CMakeLists.txt`) | M5 runs it in the guest | -| CMake | rustc's bootstrap, for LLVM and clang; `src/llvm.rs`, for the LLVM's key | admitted: no Rust tool does the job, LLVM, clang and LLD are described in CMake, and upstream's only other descriptions are a GN overlay it does not support and a Bazel one | M5 runs it in the guest | +| Python | LLVM's CMake, whenever this host builds an LLVM (`src/llvm.rs`); the C++ runtime's CMake, in every sysroot build (`src/libcxx.rs`), and its build, which runs `libcxx/utils/generate_iwyu_mapping.py` for a header it installs | admitted: no Rust tool does the job, LLVM's CMake requires one (`find_package(Python3 … REQUIRED)` in `rust/src/llvm-project/llvm/CMakeLists.txt`), and so does the runtimes', unconditionally (`runtimes/CMakeLists.txt`): configured as `src/libcxx.rs` does with no Python on `PATH` or in CMake's search, it found none and stopped, exit 1, and with only `python3` added it configured, exit 0 | M5 runs it in the guest | +| CMake | rustc's bootstrap, for LLVM and clang; `src/llvm.rs`, for the LLVM's key; `src/libcxx.rs`, for the C++ runtime in every sysroot build | admitted: no Rust tool does the job, LLVM, clang and LLD are described in CMake, and upstream's only other descriptions are a GN overlay it does not support and a Bazel one | M5 runs it in the guest | | Ninja | runs the build CMake generates for LLVM | refused: a Rust tool does it, n2 (`github.com/evmar/n2` at `b1fead5`), named `ninja` as its README directs for CMake: CMake's Ninja generator configured `rust/src/llvm-project/llvm` for it and it built `llvm-tblgen`, exit 0 each; named `n2`, CMake refuses its version, exit 1 | rustc's bootstrap builds LLVM under n2 | +| `sh` running LLVM's `config.guess`, and the POSIX tools and `cc` it runs | LLVM's CMake, whenever this host builds an LLVM, and the C++ runtime's, in every sysroot build, ask it the host's triple, unconditionally (`get_host_triple` in `rust/src/llvm-project/llvm/cmake/modules/GetHostTriple.cmake`, which runs `sh` by name) | refused: a Rust tool does the shell's part, brush 0.4.0: on the development host (macOS, arm64) `config.guess` printed `/bin/sh`'s triple under it, `arm64-apple-darwin27.0.0`, exit 0 each. The script runs `sed`, `uname`, `mktemp`, `grep`, `rm`, `rmdir` and `cc` there under either shell, and that `cc` is the `cc` rows'. Five of the other six are refused, uutils' doing each: under brush with sed 0.2.0, grep 0.2.0 and coreutils 0.12.0's `mktemp`, `rm` and `rmdir`, and nothing else on `PATH` but the host's `uname` and `cc`, it printed that triple, exit 0. `uname` is admitted: coreutils 0.12.0's answers `-p` with `unknown` where macOS's answers `arm`, and `config.guess` reads that as PowerPC, `powerpc-apple-darwin27.0.0`, exit 0 | CMake finds brush as its `sh`, uutils' `sed`, `grep`, `mktemp`, `rm` and `rmdir`, and a Rust `uname` that answers `-p` as the host's does; or M5 runs it in the guest | | `git` for worktrees, submodules, checkouts, fixtures and rustc's bootstrap | adds, removes and prunes worktrees (`src/worktree.rs`, `src/sysroot.rs`); updates submodules (`src/lib.rs`, `src/sysroot.rs`, `src/licence.rs`, `src/release.rs`); fetches the fork from the primary's and checks it out (`src/sysroot.rs`); fast-forwards the primary (`src/sync.rs`); makes the tests' fixture repositories; runs inside rustc's bootstrap | admitted: no Rust tool does the job, gitoxide 0.85 adds, removes and prunes no worktree, updates no submodule, stages, resets and pushes nothing, checks out only a fresh clone and fetches a local path by spawning `git`; a fixture must be what `git` makes, and bootstrap runs `git` itself | M4 runs it in the guest | -| `git` for reads, a config write, and clones and fetches over HTTPS | `rev-parse`, `show-ref`, `for-each-ref`, `rev-list`, `log`, `branch --contains`, `merge-base`, `ls-tree`, `ls-files`, `cat-file`, `config --get-regexp`, `worktree list`, `status`, `diff`, `ls-remote` and `grep`, in the build system and its tests; `config --global --add safe.directory` in the nightly's containers; `src/sync.rs`'s fetch of `origin`; every workflow's checkout | refused: a Rust tool does it, gitoxide 0.85, which reads refs, objects, the index, config, worktrees and status, adds a value to a config file and writes it (gix-config 0.58's `File::section_mut_or_create_new`, `SectionMut::push`, `File::write_to`), walks history, diffs, and lists, fetches and clones a remote over HTTPS; `grep` is a search of the files its index names | those are gitoxide's | +| `git` for reads, a config write, a commit's paths written out, and clones and fetches over HTTPS | `rev-parse`, `show-ref`, `for-each-ref`, `rev-list`, `log`, `branch --contains`, `merge-base`, `ls-tree`, `ls-files`, `cat-file`, `config --get-regexp`, `worktree list`, `status`, `diff`, `ls-remote` and `grep`, in the build system and its tests; `config --global --add safe.directory` in the nightly's containers; `checkout -- ` through an index of its own, which writes the C++ runtime's sources out of the LLVM commit into the stored LLVM (`src/llvm.rs`); `src/sync.rs`'s fetch of `origin`; every workflow's checkout | refused: a Rust tool does it, gitoxide 0.85, which reads refs, objects, the index, config, worktrees and status, adds a value to a config file and writes it (gix-config 0.58's `File::section_mut_or_create_new`, `SectionMut::push`, `File::write_to`), walks history, diffs, and lists, fetches and clones a remote over HTTPS; `grep` is a search of the files its index names; and gitoxide's CLI 0.59 (gix 0.88) wrote the runtimes' sources of LLVM `849da7d6` into an empty directory, each path's tree through `gix rev parse`, `gix index from-tree` and `gix free index checkout-exclusive`, exit 0 each: the 18759 files `git` writes there, byte for byte and mode for mode | those are gitoxide's | | `cc`, `c++` and `ar` on a Linux host, `build-essential` on the nightly's runners | rustc links every host binary through `cc`; `cc` and `c++` compile LLVM, clang, LLD and `rustc_llvm` (`src/llvm.rs` names both to bootstrap) and `ring`'s C for `tests/https-server-host` and `tests/https-fetch-host`; `ar` archives what `cc::Build` compiles | admitted: no Rust tool compiles C or C++, or takes rustc's host link | M5: no host in the loop | | the toolchain's own `clang`, `llvm-ar`, `rust-lld` and `llvm-config`, built from `ToyOSOrg/llvm-project` | rustc links every guest binary with `rust-lld`; `clang` compiles the C corpus and `hello.c` (`tests/common/compile.rs`, `tests/common/clang.rs`) and, with `llvm-ar`, doomgeneric through `cc::Build` (`src/clang.rs`); rustc's bootstrap asks `llvm-config` how to link LLVM | admitted: our fork's C++, which ToyOS can one day build and run; no Rust tool compiles C, `cc::Build` archives with an `ar`, bootstrap reads LLVM through `llvm-config`, and `CLAUDE.md` links everything with `rust-lld` | M5: no host in the loop | | `ovmf-generic` | the UEFI firmware of the nightly's guest containers (`src/firmware.rs`), packaged by Debian apart from QEMU | admitted: QEMU's own firmware, and no Rust firmware does its job | the instrument's QEMU carries its own firmware | diff --git a/issues/build/the-cxx-runtime-is-rebuilt-with-every-sysroot.md b/issues/build/the-cxx-runtime-is-rebuilt-with-every-sysroot.md new file mode 100644 index 00000000000..1b5b31fed0e --- /dev/null +++ b/issues/build/the-cxx-runtime-is-rebuilt-with-every-sysroot.md @@ -0,0 +1,25 @@ +--- +status: open +kind: tooling +opened: 2026-09-30 +--- + +# The C++ runtime is rebuilt with every sysroot + +`src/sysroot.rs` builds each target's libc++, libc++abi and libunwind into the +C sysroot after libc, so every change the sysroot key sees, an edit to +`toyos-abi/src`, `toyos/src` or libc's `src/` among them, configures and +builds both targets' runtimes again. + +The runtime is a function of the LLVM key, libc's `include/` and +`src/libcxx.rs` only if no configure probe links: the runtimes' `try_compile` +builds an executable by default, which links `CMAKE_SYSROOT`'s +`libtoyos_c.a`, so today libc's archive decides probe answers too. +`CMAKE_TRY_COMPILE_TARGET_TYPE=STATIC_LIBRARY` stops the link, but a probe +that only a link answers (`check_function_exists`, `check_library_exists`) +then answers yes for anything: its configure's results are owed a comparison +against today's before it is set. + +**Exit**: with the probes shown not to need a link, the runtime is a store +product of its own, keyed on the LLVM key, libc's `include/` and +`src/libcxx.rs`, and a sysroot build takes it (the store is #629's). diff --git a/issues/build/the-cxx-runtime-names-toyos-to-cmake-as-unix.md b/issues/build/the-cxx-runtime-names-toyos-to-cmake-as-unix.md new file mode 100644 index 00000000000..a2c8085f6f4 --- /dev/null +++ b/issues/build/the-cxx-runtime-names-toyos-to-cmake-as-unix.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-30 +--- + +# The C++ runtime's configure names ToyOS to CMake as `UNIX` + +CMake has no platform module for ToyOS, so `src/libcxx.rs` configures the +runtimes with `CMAKE_SYSTEM_NAME=ToyOS` and sets `UNIX=ON` beside it: the one +fact a platform module would give that the runtimes' build branches on. CMake +says so on every configure, once per check it runs: `System is unknown to +cmake, create: Platform/ToyOS to use this system`. Everything else a platform +module sets (library prefixes and suffixes, search paths, the shared-library +flags) CMake leaves at its defaults, which the runtimes' static-only build +happens not to read. + +**Exit**: CMake knows ToyOS, a `Modules/Platform/ToyOS.cmake` that sets what +its Unix-like neighbours set, and `UNIX=ON` goes from `src/libcxx.rs` with the +line every configure prints. diff --git a/issues/build/toyos-builds-itself.md b/issues/build/toyos-builds-itself.md index 8267404b85e..739c158b12b 100644 --- a/issues/build/toyos-builds-itself.md +++ b/issues/build/toyos-builds-itself.md @@ -13,22 +13,14 @@ with lld, one build of one fork, `ToyOSOrg/llvm-project`. The C library stays `userland/libc`, ours. Each stage lands on x86-64 first and on AArch64 one step behind, on `issues/kernel/toyos-runs-on-arm64.md`'s track. -- **M1 — clang cross-built, toyos-cc gone.** The host builds LLVM, clang and - lld from the fork as part of the toolchain; a C program compiled by that - clang against libc's C sysroot runs in QEMU, doomgeneric is built by it, and - toyos-cc is deleted. *Exit*: `c_hello`, `doom_frames` and the C corpus green - on clang, and toyos-cc's crate, tests and image row gone. Left for AArch64: clang's driver knows `aarch64-unknown-toyos`, - an AArch64 userland build makes its C sysroot and doomgeneric compiles - against it; no AArch64 C program has been linked and run. - **M2 — clang and lld as a package inside ToyOS; toyos-ld gone.** clang, lld and their runtime built *for* ToyOS on the host and installed by `/system/bin/pkg`; `clang hello.c && ./a.out` works in the guest. *Exit*: the in-guest compile-and-run test passes, and toyos-ld — today the only linker a ToyOS process can run — is deleted with its crate, its `[programs]` row and its tests. -- **M3 — libc++, and an LLVM-backed rustc inside ToyOS.** libc++, libc++abi - and libunwind for ToyOS; a C++ program with exceptions and threads runs; the - hosted rustc carries LLVM instead of Cranelift. *Exit*: that C++ test, and a +- **M3 — an LLVM-backed rustc inside ToyOS.** The + hosted rustc carries LLVM instead of Cranelift. *Exit*: a Rust program compiled and run inside ToyOS by the hosted rustc. - **M4 — ToyOS builds ToyOS byte-identical to the host.** cargo, rustc and clang in the guest build a userland program and then the kernel, and the @@ -47,8 +39,7 @@ and the network stack under it (`issues/hardware/the-lan-is-not-yet-production-g gigabyte of toolchain, and threads and `mmap` mature enough for LLVM (`issues/kernel/std-and-libc-drop-the-answer-thread-join-gives.md`). M2 and M4 also need libc to start a child process -(`issues/kernel/a-childs-end-is-an-event-and-a-parent-takes-its-children-down.md`). M3 needs -locale support or libc++'s no-localization build. M4 needs git in the guest, storage durable and fast +(`issues/kernel/a-childs-end-is-an-event-and-a-parent-takes-its-children-down.md`). M4 needs git in the guest, storage durable and fast enough for an LLVM build tree (`issues/filesystem/storage-is-layers-and-a-role-is-a-filesystem.md`), and memory beyond what 2 MiB process pages allow diff --git a/rust b/rust index c4c65e3e87a..aca5f527fcb 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit c4c65e3e87ae4c1b49ff3cc0004d8827a412d2e2 +Subproject commit aca5f527fcb5dc3b694e33a0f0a7312d8d29aa60 diff --git a/src/lib.rs b/src/lib.rs index 256afb5fff7..ced7fc6a450 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -26,6 +26,7 @@ pub mod kernelkeys; pub mod keystore; pub mod lan; pub mod libc; +pub mod libcxx; pub mod llvm; pub mod licence; pub mod metal; diff --git a/src/libcxx.rs b/src/libcxx.rs new file mode 100644 index 00000000000..a9027628d2a --- /dev/null +++ b/src/libcxx.rs @@ -0,0 +1,169 @@ +//! The C++ runtime of a C sysroot (`clang::CSysroot`): LLVM's libc++, +//! libc++abi and libunwind, built by CMake and n2 ([`ninja`]) from the +//! runtimes' sources the LLVM carries (`src/llvm.rs`), with that LLVM's clang, +//! against the C library the sysroot already holds (`src/libc.rs`). +//! +//! **One archive, `lib/libc++.a`, is the whole runtime**: libc++abi is linked +//! into it and libunwind into that, so the `-lc++` the ToyOS driver names for a +//! C++ link is all it needs, and its headers are `include/c++/v1`, where the +//! driver looks. [`OPTIONS`] is the configuration, each choice with its reason. + +use std::fs; +use std::path::{Path, PathBuf}; +use std::process::Command; + +use crate::arch::Arch; +use crate::clang::CSysroot; + +/// This file, which the release tag hashes. +pub(crate) const SOURCE: &str = file!(); + +/// What of `src/llvm-project` the runtimes' build reads: the runtimes, the CMake +/// modules they share with LLVM, and LLVM's libc, whose number parsing libc++ +/// compiles in. +pub(crate) const SOURCES: [&str; 7] = ["runtimes", "cmake", "llvm/cmake", "libunwind", "libcxxabi", "libcxx", "libc"]; + +/// The runtimes' CMake options beyond the target's and the tools'. +pub(crate) const OPTIONS: [(&str, &str); 20] = [ + ("CMAKE_BUILD_TYPE", "Release"), + // CMake has no platform module for ToyOS, and `UNIX` is how it says what + // the runtimes' build needs to know of one: an ELF system with POSIX threads + // (`issues/build/the-cxx-runtime-names-toyos-to-cmake-as-unix.md`). + ("CMAKE_SYSTEM_NAME", "ToyOS"), + ("UNIX", "ON"), + ("LLVM_ENABLE_RUNTIMES", "libunwind;libcxxabi;libcxx"), + ("LLVM_INCLUDE_TESTS", "OFF"), + ("LLVM_INCLUDE_DOCS", "OFF"), + // The C library is a static archive, so the C++ runtime is one too. + ("LIBUNWIND_ENABLE_SHARED", "OFF"), + ("LIBCXXABI_ENABLE_SHARED", "OFF"), + ("LIBCXX_ENABLE_SHARED", "OFF"), + ("LIBCXXABI_ENABLE_STATIC_UNWINDER", "ON"), + ("LIBCXX_ENABLE_STATIC_ABI_LIBRARY", "ON"), + ("LIBUNWIND_INSTALL_LIBRARY", "OFF"), + ("LIBUNWIND_INSTALL_HEADERS", "OFF"), + ("LIBCXXABI_INSTALL_LIBRARY", "OFF"), + ("LIBCXX_INSTALL_MODULES", "OFF"), + // The C library runs `thread_local` destructors (`__cxa_thread_atexit_impl`). + ("LIBCXXABI_HAS_CXA_THREAD_ATEXIT_IMPL", "ON"), + // The C library has no `dladdr`, so libunwind names no function it unwinds. + ("LIBUNWIND_ADDITIONAL_COMPILE_FLAGS", "-D_LIBUNWIND_USE_DLADDR=0"), + // The C library has no directory, `stat` or path surface for it + // (`issues/build/libcxx-is-built-without-std-filesystem.md`). + ("LIBCXX_ENABLE_FILESYSTEM", "OFF"), + ("LIBCXX_INCLUDE_BENCHMARKS", "OFF"), + ("LIBCXX_INCLUDE_TESTS", "OFF"), +]; + +/// cargo's install of n2, the Rust Ninja, but for its `--root`: without its +/// default jemalloc, which is C. +const N2: [&str; 7] = [ + "install", + "--locked", + "--no-default-features", + "--git", + "https://github.com/evmar/n2", + "--rev", + "b1fead52ccda0c497d816696f23f4099c3e8ec1f", +]; + +/// Where [`N2`] installs under `root`: a directory every argument names, so a +/// moved pin or flag names an empty one. +fn installed(root: &Path) -> PathBuf { + root.join("target/n2").join(crate::sysroot::short(N2.join("\0").as_bytes())) +} + +/// [`N2`], installed under `root` unless it is there, by the name `ninja`, under +/// which it speaks the Ninja CMake asks for. +pub fn ninja(root: &Path) -> PathBuf { + let dir = installed(root); + let ninja = dir.join("bin/ninja"); + // The link is made only once cargo has installed what it names. + if !ninja.is_file() { + let status = Command::new("cargo") + .args(N2) + .arg("--root") + .arg(&dir) + .env_remove("RUSTFLAGS") + .status() + .unwrap_or_else(|e| panic!("cargo failed to launch: {e}")); + assert!(status.success(), "n2 did not install into {}", dir.display()); + std::os::unix::fs::symlink("n2", &ninja).unwrap_or_else(|e| panic!("symlink {} -> n2: {e}", ninja.display())); + assert!(ninja.is_file(), "n2 installed into {}, and {} names no file", dir.display(), ninja.display()); + } + ninja +} + +/// Build `arch`'s C++ runtime from the runtimes' `sources` into `c`, which holds +/// the C library already, under `ninja`, in `scratch`, which it removes when +/// done. +pub fn build(c: &CSysroot, arch: Arch, sources: &Path, ninja: &Path, scratch: &Path) { + eprintln!("Building the C++ runtime for {} under {}", c.target, ninja.display()); + if scratch.exists() { + fs::remove_dir_all(scratch).unwrap_or_else(|e| panic!("remove {}: {e}", scratch.display())); + } + fs::create_dir_all(scratch).unwrap_or_else(|e| panic!("create {}: {e}", scratch.display())); + // What CMake runs after archiving: LLVM's archiver is `ranlib` when run by + // that name, which is all the link LLVM installs as `llvm-ranlib` is. + let ranlib = scratch.join("llvm-ranlib"); + std::os::unix::fs::symlink(&c.ar, &ranlib) + .unwrap_or_else(|e| panic!("symlink {} -> {}: {e}", ranlib.display(), c.ar.display())); + let path = |p: &Path| p.display().to_string(); + let mut definitions: Vec<(String, String)> = OPTIONS.iter().map(|(n, v)| (n.to_string(), v.to_string())).collect(); + definitions.push(("CMAKE_SYSTEM_PROCESSOR".into(), arch.name().into())); + for lang in ["C", "CXX", "ASM"] { + definitions.push((format!("CMAKE_{lang}_COMPILER"), path(&c.clang))); + definitions.push((format!("CMAKE_{lang}_COMPILER_TARGET"), c.target.into())); + } + definitions.push(("CMAKE_AR".into(), path(&c.ar))); + definitions.push(("CMAKE_RANLIB".into(), path(&ranlib))); + definitions.push(("CMAKE_SYSROOT".into(), path(&c.dir))); + definitions.push(("CMAKE_INSTALL_PREFIX".into(), path(&c.dir))); + definitions.push(("CMAKE_MAKE_PROGRAM".into(), path(ninja))); + + let mut configure = Command::new("cmake"); + configure.args(["-G", "Ninja", "-Wno-dev", "-S"]).arg(sources.join("runtimes")).arg("-B").arg(scratch); + configure.args(definitions.iter().map(|(name, value)| format!("-D{name}={value}"))); + run(configure, "configuration", c.target); + let mut install = Command::new(ninja); + install.arg("-C").arg(scratch).arg("install"); + run(install, "build", c.target); + + let archive = c.dir.join("lib/libc++.a"); + assert!(archive.is_file(), "the C++ runtime for {} was built, and installed no {}", c.target, archive.display()); + fs::remove_dir_all(scratch).unwrap_or_else(|e| panic!("remove {}: {e}", scratch.display())); +} + +/// Run `command` seeing nothing of this process's environment but what the +/// LLVM build does (`llvm::clear`), and refuse its failure with all it said. +fn run(mut command: Command, what: &str, target: &str) { + crate::llvm::clear(&mut command); + let out = command.output().unwrap_or_else(|e| panic!("run {command:?}: {e}")); + assert!( + out.status.success(), + "the C++ runtime's {what} for {target} failed ({}):\n{}{}", + out.status, + String::from_utf8_lossy(&out.stdout), + String::from_utf8_lossy(&out.stderr), + ); +} + +#[cfg(test)] +mod tests { + use super::*; + use toyos_tmpdir::TempDir; + + /// **n2 is installed once**: with the link and n2 in the directory the pin + /// names, no cargo runs, and the n2 there is the one returned. + #[test] + fn an_installed_n2_is_not_installed_again() { + let root = TempDir::new("n2"); + let bin = installed(&root).join("bin"); + fs::create_dir_all(&bin).unwrap(); + fs::write(bin.join("n2"), b"").unwrap(); + std::os::unix::fs::symlink("n2", bin.join("ninja")).unwrap(); + + assert_eq!(ninja(&root), bin.join("ninja")); + assert!(fs::read(bin.join("n2")).unwrap().is_empty(), "the installed n2 was replaced"); + } +} diff --git a/src/llvm.rs b/src/llvm.rs index 4c66bbfce78..c6146e84d7e 100644 --- a/src/llvm.rs +++ b/src/llvm.rs @@ -8,9 +8,11 @@ //! refused), the bootstrap configuration below, [`RECIPE`], and the tools the //! host builds it with ([`host_tools`]). `rust/build/llvm//` in the primary //! is bootstrap's install of that LLVM and its clang, with its LLD in `bin/` -//! beside `llvm-config`, made by whichever build first needs it ([`resolve`]), -//! and stored only when it was built from what the key names. Once its -//! [`SOURCE`] file exists it is read-only, its directories as well as its files. +//! beside `llvm-config` and in `src/` the runtimes' sources the C++ runtime is +//! built from (`src/libcxx.rs`) as its commit holds them, made by whichever +//! build first needs it ([`resolve`]), and stored only when it was built from +//! what the key names. Once its [`SOURCE`] file exists it is read-only, its +//! directories as well as its files. //! Every compiler build, the primary's and a worktree's own, names it as the //! host's `llvm-config` with `llvm-has-rust-patches`, so bootstrap builds no //! LLVM and takes LLD from beside it as `rust-lld`; `clang::provision` copies its @@ -47,7 +49,7 @@ use crate::toolchain::{self, host_triple}; /// What changes how a key's sources become an LLVM and is none of the other /// parts: the build's targets and what is kept of it. Moving it moves every key. const RECIPE: &str = "bootstrap build of src/llvm-project/llvm and src/llvm-project/lld; the install's bin, \ - include and lib, and lld in bin, read-only; 2"; + include and lib, and lld in bin, and the runtimes' sources in src, read-only; 3"; /// What of the caller's environment the LLVM build, and every tool its key /// asks, sees: @@ -132,7 +134,7 @@ fn key_of(fork: &Path, recipe: &str, config: &str, tools: &str) -> String { } /// Give `command` nothing of this process's environment but [`ENVIRONMENT`]. -fn clear(command: &mut Command) { +pub(crate) fn clear(command: &mut Command) { command.env_clear(); for name in ENVIRONMENT { if let Some(value) = std::env::var_os(name) { @@ -208,14 +210,29 @@ fn choose(root: &Path, rust_dir: &Path, fork: &Path, build: impl Fn(&Path) -> Pa Llvm { dir, _using: using } } +/// [`resolve`], recording nothing for `root`: what a sysroot build reads of the +/// LLVM its compiler links, whose record is that compiler's +/// (`compiler::choose`). +pub fn held(root: &Path, rust_dir: &Path, fork: &Path) -> Llvm { + held_with(root, rust_dir, fork, build_in_fork) +} + +/// [`held`] with the build passed in, as [`choose`] takes it. +fn held_with(root: &Path, rust_dir: &Path, fork: &Path, build: impl Fn(&Path) -> PathBuf) -> Llvm { + let key = key(fork); + let dir = store(rust_dir).join(&key); + let using = crate::buildlock::keyed_made(root, Keyed::Llvm, &key, || defect(&dir), || place(fork, &key, &dir, &build)); + Llvm { dir, _using: using } +} + /// Why `dir` is not a finished LLVM, if it is not. fn defect(dir: &Path) -> Option { if !dir.join(SOURCE).is_file() { return Some(format!("{} carries no {SOURCE}", dir.display())); } - let kept = KEPT.iter().map(|k| dir.join(k)).filter(|p| !p.is_dir()); + let kept = KEPT.iter().map(|k| dir.join(k)).chain(crate::libcxx::SOURCES.iter().map(|s| dir.join("src").join(s))); let tools = TOOLS.iter().map(|t| dir.join(t)).filter(|p| !p.is_file()); - let gone: Vec = kept.chain(tools).map(|p| p.display().to_string()).collect(); + let gone: Vec = kept.filter(|p| !p.is_dir()).chain(tools).map(|p| p.display().to_string()).collect(); (!gone.is_empty()).then(|| format!("{} carries no {}", dir.display(), gone.join(", "))) } @@ -270,6 +287,7 @@ fn place(fork: &Path, key: &str, dir: &Path, build: &impl Fn(&Path) -> PathBuf) built_from.trim(), fork.display(), ); + check_out_committed(&checkout, &commit, &crate::libcxx::SOURCES, &partial.join("src")); fs::write(partial.join(SOURCE), format!("{key}\n")) .unwrap_or_else(|e| panic!("write {}: {e}", partial.join(SOURCE).display())); read_only(&partial); @@ -278,6 +296,31 @@ fn place(fork: &Path, key: &str, dir: &Path, build: &impl Fn(&Path) -> PathBuf) fs::remove_dir_all(&built).unwrap_or_else(|e| panic!("remove {}: {e}", built.display())); } +/// Write `paths` as `commit` holds them, from the repository at `checkout`, +/// under `dest`: through an index of their own and with no sparse pattern, so +/// nothing the checkout holds beside the commit, tracked, ignored or left out, +/// reaches them. +fn check_out_committed(checkout: &Path, commit: &str, paths: &[&str], dest: &Path) { + fs::create_dir_all(dest).unwrap_or_else(|e| panic!("create {}: {e}", dest.display())); + let index = toyos_tmpdir::TempDir::new("llvm-runtimes-index"); + let out = Command::new("git") + .env("GIT_INDEX_FILE", index.join("index")) + .args(["-c", "core.sparseCheckout=false", "--work-tree"]) + .arg(dest) + .args(["checkout", commit, "--"]) + .args(paths) + .current_dir(checkout) + .output() + .unwrap_or_else(|e| panic!("run git in {}: {e}", checkout.display())); + assert!( + out.status.success(), + "git checkout {commit} -- {paths:?} into {} in {}: {}", + dest.display(), + checkout.display(), + String::from_utf8_lossy(&out.stderr).trim(), + ); +} + /// Take write permission from every file and directory under `dir`, and from /// `dir`. fn read_only(dir: &Path) { @@ -431,6 +474,9 @@ mod tests { git(&checkout, &["init", "-q"]); } write(&checkout.join("llvm/CMakeLists.txt"), content); + for source in crate::libcxx::SOURCES { + write(&checkout.join(source).join("CMakeLists.txt"), &format!("the {source} of {content}")); + } git(&checkout, &["add", "-A"]); let tree = git(&checkout, &["write-tree"]); let out = Command::new("git") @@ -490,6 +536,7 @@ mod tests { assert_eq!(fs::read_to_string(la.dir.join("bin/lld")).unwrap(), "the lld"); assert_eq!(fs::read_link(la.dir.join("bin/clang")).unwrap(), Path::new("clang-22")); assert!(la.dir.join("lib/clang/22/include/stddef.h").is_file()); + assert_eq!(fs::read_to_string(la.dir.join("src/libcxx/CMakeLists.txt")).unwrap(), "the libcxx of A", "the runtimes' sources are not the commit's"); assert!(!la.dir.join("build").exists(), "CMake's tree was kept"); assert!(!a.join("rust/build/toyos-llvm").exists(), "the build directory outlived the placement"); @@ -536,7 +583,7 @@ mod tests { fake_build(fork) }; let dir = choose(&a, &rust_dir, &a.join("rust"), counted).dir; - for (lost, made) in [("bin/lld", 2), ("lib", 3)] { + for (lost, made) in [("bin/lld", 2), ("lib", 3), ("src/libcxxabi", 4)] { keystore::writable(&dir); let lost = dir.join(lost); if lost.is_dir() { @@ -812,6 +859,37 @@ mod tests { assert!(stored.iter().all(|n| n.to_string_lossy().ends_with(".partial")), "stored: {stored:?}"); } + /// **The runtimes' sources are the commit's**: made while the LLVM was + /// built, an edit to a file the commit holds is refused and nothing is + /// stored, and a file the checkout ignores is not stored with it. + #[test] + fn the_runtimes_sources_are_the_commit_s() { + let scratch = Scratch::new("llvm-runtimes"); + let (_primary, rust_dir, [_same, a, _b]) = estate_built(&scratch); + let fork = a.join("rust"); + let checkout = fork.join(LLVM); + write(&checkout.join(".git/info/exclude"), "*.pyc\n"); + + let editing = |fork: &Path| { + write(&fork.join(LLVM).join("libcxx/CMakeLists.txt"), "an edit no commit holds"); + fake_build(fork) + }; + let said = refusal("an edit to the runtimes' sources was stored", || { + choose(&a, &rust_dir, &fork, editing); + }); + assert!(said.contains("holds changes no commit does"), "{said}"); + git(&checkout, &["checkout", "-q", "--", "libcxx"]); + + let ignored = |fork: &Path| { + write(&fork.join(LLVM).join("libcxx/utils/cache.pyc"), "what no commit holds"); + fake_build(fork) + }; + let dir = choose(&a, &rust_dir, &fork, ignored).dir; + assert_eq!(fs::read_to_string(dir.join("src/libcxx/CMakeLists.txt")).unwrap(), "the libcxx of A"); + assert!(checkout.join("libcxx/utils/cache.pyc").is_file()); + assert!(!dir.join("src/libcxx/utils").exists(), "a file the checkout ignores was stored"); + } + const WORKTREE: &str = "TOYOS_LLVM_TEST_WORKTREE"; const RUST_DIR: &str = "TOYOS_LLVM_TEST_RUST_DIR"; const ROLE: &str = "TOYOS_LLVM_TEST_ROLE"; @@ -917,6 +995,17 @@ mod tests { assert_eq!(defect(&dir), None, "the sweep took an LLVM the worktree that resolved it names"); } + /// **An LLVM held for a sysroot build is not recorded**: the record follows + /// the compiler's, so a worktree whose compiler is the primary's names none. + #[test] + fn a_held_llvm_is_not_recorded() { + let scratch = Scratch::new("llvm-held"); + let (_primary, rust_dir, [_same, a, _b]) = estate_built(&scratch); + let llvm = held_with(&a, &rust_dir, &a.join("rust"), fake_build); + assert_eq!(defect(&llvm.dir), None); + assert_eq!(keystore::recorded(&a, Keyed::Llvm), None, "a held LLVM was recorded"); + } + /// **A build directory whose compiler links the host's LLVM keeps none of /// its own**: bootstrap's LLVM and LLD, `download-ci-llvm`'s and its /// downloads go, and what a stopped removal left; the rest stays. diff --git a/src/main.rs b/src/main.rs index 3ed298f68d6..61ae13771fd 100644 --- a/src/main.rs +++ b/src/main.rs @@ -23,8 +23,14 @@ const REQUIRED: &[Tool] = &[ Tool { any: &["cc"], why: "rustc links every host binary through it; no guest binary" }, Tool { any: &["cmake"], - why: "every build keys the host's LLVM on its `--version`, and rustc's bootstrap \ - configures LLVM and clang with it; `brew install cmake` on macOS", + why: "every build keys the host's LLVM on its `--version`, rustc's bootstrap \ + configures LLVM and clang with it, and every sysroot build the C++ runtime; \ + `brew install cmake` on macOS", + }, + Tool { + any: &["python3", "python"], + why: "rust/x runs rustc's bootstrap, which is Python, and the C++ runtime's CMake \ + requires a Python 3 and runs it in every sysroot build", }, ]; @@ -37,25 +43,16 @@ const REQUIRED: &[Tool] = &[ /// host has not built that LLVM. Both are host tools like `cc`, and never in a /// guest: on macOS from Homebrew, on CI's toolchain runner at the versions /// `.github/workflows` pins. -const ALSO_USED: &[Tool] = &[ - Tool { - any: &["python3", "python", "py", "python2", "uv"], - why: "rust/x runs rustc's bootstrap, which is Python — a clean clone and \ - every toolchain change need one", - }, - Tool { - any: &["ninja"], - why: "rustc's bootstrap builds LLVM and clang with it, under CMake; `brew install \ - ninja` on macOS", - }, -]; +const ALSO_USED: &[Tool] = &[Tool { + any: &["ninja"], + why: "rustc's bootstrap builds LLVM and clang with it, under CMake; `brew install \ + ninja` on macOS", +}]; /// Where the OS would find `name`, if anywhere. /// /// A `PATH` scan and not a `--version` run: it is what `Command::new` does -/// anyway, and one name above must not be executed — asking macOS for `py` -/// opens the Command Line Tools installer, which is why `rust/x` searches -/// `python3` ahead of it. +/// anyway. fn executable_on_path(name: &str) -> bool { use std::os::unix::fs::PermissionsExt; let Some(path) = env::var_os("PATH") else { diff --git a/src/release.rs b/src/release.rs index 0f9d910b4ca..050598c85e0 100644 --- a/src/release.rs +++ b/src/release.rs @@ -26,7 +26,7 @@ fn trees() -> Vec<&'static str> { std::iter::once("rust") .chain(crate::sysroot::SYSROOT_SOURCES) .chain(crate::sysroot::SYSROOT_MANIFESTS) - .chain([crate::clang::SOURCE, file!()]) + .chain([crate::clang::SOURCE, crate::libcxx::SOURCE, file!()]) .collect() } @@ -463,7 +463,7 @@ mod tests { assert!(out.status.success(), "git {args:?}: {}", String::from_utf8_lossy(&out.stderr)); } - /// A commit to the C toolchain's declarations or headers moves the tag. + /// A commit to the C and C++ toolchain's declarations or headers moves the tag. #[test] fn the_tag_moves_with_the_c_toolchain() { let repo = TempDir::new("release-tag"); @@ -493,9 +493,13 @@ mod tests { let riscv = r#"targets = \"AArch64;RISCV;X86\""#; assert!(clang.contains(tools) && clang.contains(targets), "src/clang.rs no longer declares what this mutates"); let with_objdump = clang.replace(tools, objdump); + let cxx = fs::read_to_string(here.join(crate::libcxx::SOURCE)).unwrap(); + let (no_fs, fs_on) = (r#"("LIBCXX_ENABLE_FILESYSTEM", "OFF")"#, r#"("LIBCXX_ENABLE_FILESYSTEM", "ON")"#); + assert!(cxx.contains(no_fs), "src/libcxx.rs no longer declares what this mutates"); let mutations = [ (crate::clang::SOURCE, with_objdump.clone()), (crate::clang::SOURCE, with_objdump.replace(targets, riscv)), + (crate::libcxx::SOURCE, cxx.replace(no_fs, fs_on)), ("userland/libc/include/placeholder", "y".to_string()), ]; for (path, text) in mutations { diff --git a/src/sysroot.rs b/src/sysroot.rs index 684ae2e9cbd..f2f5044afb5 100644 --- a/src/sysroot.rs +++ b/src/sysroot.rs @@ -27,7 +27,8 @@ //! compiler is a worktree's own; the sysroot key's (`buildlock::keyed_*`), with //! this worktree's build lock put down; then, to build, this worktree's //! exclusively (its fork build directory is written); then, if the compiler is -//! the primary's, the global one shared, because it is read. +//! the primary's, the global one shared, because it is read; then the key of the +//! compiler's LLVM, held in use while the C++ runtime is built from its sources. //! //! A sysroot no worktree names any more is removed by `keystore::sweep`, which //! `--worktree remove` runs: each build records the key it used in its @@ -62,9 +63,10 @@ const SOURCES: &str = "SOURCES"; /// What changes how a key's sources become a sysroot and is none of them: the /// std build's recipe below. Moving it moves every key. -const RECIPE: &str = "bootstrap stage-0 local rebuild, profile compiler, no LLVM, \ +const RECIPE: &str = "bootstrap stage-0 local rebuild, profile compiler, no LLVM or Ninja, \ libtoyos_c merged, libraries from the stamp, linked by rust-lld, \ - a C sysroot of libc's staticlib and headers per target; 5"; + a C sysroot of libc's staticlib and headers per target, and its C++ runtime \ + built under n2 from the runtimes' sources of the compiler's LLVM; 8"; /// Every sysroot on this host. pub fn sysroots_dir(rust_dir: &Path) -> PathBuf { @@ -189,7 +191,11 @@ fn source_files(checkout: &Path, paths: &[&str], out: &mut Vec) { /// compiled by `compiler`. pub fn key(root: &Path, compiler: &Compiler, fork: &Path) -> String { let parts = [ - format!("{RECIPE}; cargo {STAGE0_CARGO}; targets {}", GUEST_TARGETS.join(" ")), + format!( + "{RECIPE}; cargo {STAGE0_CARGO}; targets {}; C++ runtime {:?}", + GUEST_TARGETS.join(" "), + crate::libcxx::OPTIONS + ), witness(root), tree_identity(fork, &["library", "src/bootstrap"]), compiler.identity(), @@ -318,13 +324,13 @@ pub fn ensure(root: &Path, rust_dir: &Path, lock: &mut Held) -> Sysroot { let dir = sysroots_dir(rust_dir).join(&key); crate::keystore::record(root, Keyed::Sysroot, &key); - let using = lock.without_shared(|| held(root, &key, &dir, || build(root, &compiler, &fork, &key, &dir))); + let using = lock.without_shared(|| held(root, &key, &dir, || build(root, rust_dir, &compiler, &fork, &key, &dir))); Sysroot { dir, primary_compiler: compiler.primary, _using: Some(using) } } /// Make the sysroot `key` names at `dir`, from `root`'s sources and the std fork /// at `fork`, with `compiler`. The caller holds the key's lock. -fn build(root: &Path, compiler: &Compiler, fork: &Path, key: &str, dir: &Path) { +fn build(root: &Path, rust_dir: &Path, compiler: &Compiler, fork: &Path, key: &str, dir: &Path) { let what = format!("building sysroot {key}"); let _worktree = buildlock::worktree_exclusive(root, &what); // Only the primary's compiler is rebuilt in place; one of a worktree's own @@ -343,6 +349,13 @@ fn build(root: &Path, compiler: &Compiler, fork: &Path, key: &str, dir: &Path) { crate::libc::build_c(root, partial, &libc_target, arch); } let _ = fs::remove_dir_all(&libc_target); + let llvm = crate::llvm::held(root, rust_dir, fork); + let ninja = crate::libcxx::ninja(root); + for arch in Arch::ALL { + let scratch = dir.with_extension(format!("libcxx-{}", arch.name())); + let c = crate::clang::CSysroot::of(partial, arch); + crate::libcxx::build(&c, arch, &llvm.dir.join("src"), &ninja, &scratch); + } // The sources the key named are the ones built, or this is not that key's. let again = self::key(root, compiler, fork); @@ -518,7 +531,9 @@ fn place_std(stamp: &Path, lib: &Path) { /// The linker is the compiler's own `rust-lld`, named by path so that which sysroot /// a stage-0 build searches for tools decides nothing; and no rpath, which bootstrap /// spells as a C driver's `-Wl,` arguments that a linker run directly refuses. -/// No LLVM: std builds none, and the profile's `download-ci-llvm` fetches one. +/// No LLVM: std builds none, and the profile's `download-ci-llvm` fetches one; +/// and no Ninja, which bootstrap otherwise demands on `PATH` for the LLVM it does +/// not build. fn std_config(compiler: &Path, cargo: &Path, build_dir: &Path, host: &str) -> String { let targets = GUEST_TARGETS.iter().map(|t| format!("\"{t}\"")).collect::>().join(", "); let linker = toolchain::rust_lld(compiler); @@ -540,6 +555,7 @@ target = [{targets}] [llvm] download-ci-llvm = false +ninja = false [rust] lld = false @@ -818,12 +834,13 @@ mod tests { assert!(kept.iter().all(|f| f.is_file()), "the same compiler's build went"); } - /// **A std build fetches no LLVM**: it builds none, and the `compiler` - /// profile would download one. + /// **A std build fetches no LLVM and asks for no Ninja**: it builds none, + /// the `compiler` profile would download one, and bootstrap would refuse it + /// with no `ninja` on `PATH`. #[test] - fn a_std_build_downloads_no_llvm() { + fn a_std_build_downloads_no_llvm_and_asks_for_no_ninja() { let config = std_config(Path::new("/c"), Path::new("/cargo"), Path::new("/b"), "h"); - assert!(config.contains("\n[llvm]\ndownload-ci-llvm = false\n"), "{config}"); + assert!(config.contains("\n[llvm]\ndownload-ci-llvm = false\nninja = false\n"), "{config}"); } /// **A switch that cannot remove the other compiler's build fails and does diff --git a/tests/common/clang.rs b/tests/common/clang.rs index a0d45102394..1fd380f04ca 100644 --- a/tests/common/clang.rs +++ b/tests/common/clang.rs @@ -1,6 +1,6 @@ -//! A C program compiled and linked by the toolchain's clang — the ToyOS driver -//! `ToyOSOrg/llvm-project` carries — judged as the loader sees the file, and -//! then run on ToyOS. +//! A C program and a C++ program compiled and linked by the toolchain's clang — +//! the ToyOS driver `ToyOSOrg/llvm-project` carries — judged as the loader sees +//! the file, and then run on ToyOS. use std::fs; use std::process::Command; @@ -72,3 +72,46 @@ pub fn c_hello(rust_bins: &[(String, Vec)]) -> Result<(), String> { eprintln!(" [c_hello] {SAYS}"); Ok(()) } + +const CXX_RUNTIME: &str = "tests/cxx/runtime.cpp"; +const CXX_RUNTIME_EXPECT: &str = "tests/cxx/runtime.expect"; + +/// Gate: a C++ program — libc++'s containers, strings and streams, exceptions, +/// threads and their destructors — compiled and linked by one clang +/// invocation, prints on ToyOS what [`CXX_RUNTIME_EXPECT`] holds. +pub fn cxx_runtime(rust_bins: &[(String, Vec)]) -> Result<(), String> { + let root = compile::repo_root(); + let c = compile::c_sysroot(); + let out = super::lane::dir().join("cxx-runtime"); + let built = Command::new(&c.clang) + .arg("--driver-mode=g++") + .args(c.args()) + .args(["-std=c++17", "-O2"]) + .arg(root.join(CXX_RUNTIME)) + .arg("-o") + .arg(&out) + .output() + .map_err(|e| format!("run {}: {e}", c.clang.display()))?; + if !built.status.success() { + return Err(format!("clang could not build {CXX_RUNTIME}:\n{}", String::from_utf8_lossy(&built.stderr))); + } + let elf = fs::read(&out).map_err(|e| format!("{}: {e}", out.display()))?; + judge_elf(&elf)?; + let expected = fs::read_to_string(root.join(CXX_RUNTIME_EXPECT)).map_err(|e| format!("{CXX_RUNTIME_EXPECT}: {e}"))?; + + let config = root.join("tests/testcases"); + let c_tests = [("cxx_runtime".to_string(), elf)]; + let mut qemu = QemuInstance::boot_with_options(&config, &c_tests, rust_bins, BootOptions::default()); + let result = qemu.run_test("test_c_cxx_runtime", Duration::from_secs(60)); + if let Some(err) = &result.error { + return Err(format!("{err}\n{}", result.stdout)); + } + if result.exit_code != Some(0) { + return Err(format!("{CXX_RUNTIME} exited {:?}:\n{}", result.exit_code, result.stdout)); + } + if let Some(mismatch) = super::console::c_verdict(&result.stdout, &expected).mismatch { + return Err(mismatch); + } + eprintln!(" [cxx_runtime] {} lines, as expected", expected.lines().count()); + Ok(()) +} diff --git a/tests/common/mod.rs b/tests/common/mod.rs index 80ad8764bab..8158bc6d87e 100644 --- a/tests/common/mod.rs +++ b/tests/common/mod.rs @@ -2,8 +2,8 @@ pub mod audio; /// blockd: the NVMe driver in userland, judged off its disk and the device's /// own trace. pub mod blockd; -/// The C toolchain, end to end: a program the toolchain's clang built, judged -/// as the loader reads it and then run. +/// The C and C++ toolchain, end to end: a program the toolchain's clang built, +/// judged as the loader reads it and then run. pub mod clang; pub mod clock; pub mod lane; diff --git a/tests/cxx/runtime.cpp b/tests/cxx/runtime.cpp new file mode 100644 index 00000000000..52177afc575 --- /dev/null +++ b/tests/cxx/runtime.cpp @@ -0,0 +1,374 @@ +// What `cxx_runtime` compiles with the toolchain's clang and runs on ToyOS: +// libc++'s containers, strings and streams, exceptions through frames with +// destructors, threads with their thread_local and static destructors, and +// libc's threads, keys, mutexes, condition variables and exit handlers. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { + +std::atomic thread_locals_destroyed{0}; +int exit_handlers_ran = 0; + +struct Farewell { + ~Farewell() { + std::printf("static destructor ran after %d exit handlers and %d thread_local destructors\n", + exit_handlers_ran, thread_locals_destroyed.load()); + } +} farewell; + +struct PerThread { + int id = 0; + ~PerThread() { thread_locals_destroyed.fetch_add(1); } +}; + +thread_local PerThread per_thread; + +struct Custom : std::runtime_error { + using std::runtime_error::runtime_error; +}; + +int thrower(int depth) { + std::vector held{"destroyed", "while", "unwinding"}; + if (depth == 0) + throw Custom("thrown from depth 0"); + return thrower(depth - 1) + static_cast(held.size()); +} + +std::mutex hand_m; +std::condition_variable hand_cv; +bool handed = false; +bool released = false; +pthread_t handed_over; + +void* handed_main(void*) { + std::unique_lock lock(hand_m); + handed_over = pthread_self(); + handed = true; + hand_cv.notify_all(); + hand_cv.wait(lock, [] { return released; }); + return reinterpret_cast(42); +} + +void* joiner_main(void* out) { + pthread_t handle; + { + std::unique_lock lock(hand_m); + hand_cv.wait(lock, [] { return handed; }); + handle = handed_over; + } + void* got = nullptr; + int rc = pthread_join(handle, &got); + *static_cast(out) = rc == 0 ? reinterpret_cast(got) : -rc; + return nullptr; +} + +void* exits_with_42(void*) { + pthread_exit(reinterpret_cast(42)); +} + +void* returns_null(void*) { + return nullptr; +} + +std::mutex detached_m; +std::condition_variable detached_cv; +int detached_ran = 0; + +void* detached_main(void*) { + std::lock_guard lock(detached_m); + detached_ran++; + detached_cv.notify_one(); + return nullptr; +} + +template +struct Counted { + ~Counted() { exit_handlers_ran++; } +}; + +template +void register_one() { + static Counted counted; +} + +template +void register_all(std::integer_sequence) { + (register_one(), ...); +} + +const char* errno_name(int e) { + switch (e) { + case 0: + return "0"; + case EPERM: + return "EPERM"; + case EINVAL: + return "EINVAL"; + case EDEADLK: + return "EDEADLK"; + case EAGAIN: + return "EAGAIN"; + case ETIMEDOUT: + return "ETIMEDOUT"; + default: + return "another error"; + } +} + +} // namespace + +int main() { + std::vector v(10); + std::iota(v.begin(), v.end(), 1); + int sum = std::accumulate(v.begin(), v.end(), 0); + std::string s = "hello"; + s += ", ToyOS"; + std::cout << "vector sum " << sum << ", back " << v.back() << ", string " << s << " (" << s.size() << ")\n"; + + try { + thrower(5); + } catch (const std::runtime_error& e) { + std::cout << "caught " << e.what() << "\n"; + } + try { + try { + throw 42; + } catch (int n) { + std::cout << "caught int " << n << ", rethrowing\n"; + throw; + } + } catch (int n) { + std::cout << "caught rethrown int " << n << "\n"; + } + try { + (void)std::vector().at(3); + } catch (const std::out_of_range&) { + std::cout << "caught out_of_range from at\n"; + } + try { + (void)std::stoi("not a number"); + } catch (const std::invalid_argument&) { + std::cout << "caught invalid_argument from stoi\n"; + } + try { + (void)std::stoi("99999999999"); + } catch (const std::out_of_range&) { + std::cout << "caught out_of_range from stoi\n"; + } + + std::vector partial(4); + std::vector workers; + for (int t = 0; t < 4; t++) { + workers.emplace_back([t, &partial] { + per_thread.id = t + 1; + long acc = 0; + for (long i = t; i < 1000; i += 4) + acc += i; + partial[t] = acc; + }); + } + for (auto& w : workers) + w.join(); + std::cout << "threads summed " << std::accumulate(partial.begin(), partial.end(), 0L) << "\n"; + std::cout << "thread_local destructors ran " << thread_locals_destroyed.load() << "\n"; + + std::mutex m; + std::condition_variable cv; + int stage = 0; + std::thread ping([&] { + std::unique_lock lock(m); + cv.wait(lock, [&] { return stage == 1; }); + stage = 2; + cv.notify_one(); + }); + { + std::lock_guard lock(m); + stage = 1; + } + cv.notify_one(); + { + std::unique_lock lock(m); + cv.wait(lock, [&] { return stage == 2; }); + } + ping.join(); + std::cout << "condition variable handshake done\n"; + + std::exception_ptr carried; + std::thread failing([&] { + try { + throw std::logic_error("from a thread"); + } catch (...) { + carried = std::current_exception(); + } + }); + failing.join(); + try { + std::rethrow_exception(carried); + } catch (const std::logic_error& e) { + std::cout << "rethrew " << e.what() << "\n"; + } + + pthread_key_t key; + pthread_key_create(&key, nullptr); + pthread_setspecific(key, &key); + void* seen = &seen; + std::thread::id other; + std::thread reader([&] { + seen = pthread_getspecific(key); + other = std::this_thread::get_id(); + }); + reader.join(); + std::cout << "a key set in main reads " << (seen == nullptr ? "null" : "set") << " in another thread and " + << (pthread_getspecific(key) == &key ? "set" : "lost") << " in main; thread ids " + << (other != std::this_thread::get_id() ? "differ" : "match") << "\n"; + + intptr_t joined = 0; + pthread_t joiner, handed_thread; + pthread_create(&joiner, nullptr, joiner_main, &joined); + pthread_create(&handed_thread, nullptr, handed_main, nullptr); + { + std::lock_guard lock(hand_m); + released = true; + } + hand_cv.notify_all(); + pthread_join(joiner, nullptr); + std::cout << "a thread joined by a third on its own pthread_self, before or after its creator returned, gave " << joined << "\n"; + + pthread_t exiting; + void* exited = nullptr; + pthread_create(&exiting, nullptr, exits_with_42, nullptr); + pthread_join(exiting, &exited); + std::cout << "pthread_exit handed its join " << reinterpret_cast(exited) << "\n"; + + // 64 stacks of 64 MiB each way, more than the guest's memory + // (`tests/common/qemu.rs`'s `-m`): each is freed, or a create runs out. + constexpr int stacks = 64; + pthread_attr_t big; + pthread_attr_init(&big); + pthread_attr_setstacksize(&big, size_t{64} << 20); + int joined_stacks = 0; + int detached_stacks = 0; + int refused = 0; + while (joined_stacks < stacks && refused == 0) { + pthread_t t; + refused = pthread_create(&t, &big, returns_null, nullptr); + if (refused == 0) { + pthread_join(t, nullptr); + joined_stacks++; + } + } + pthread_attr_setdetachstate(&big, PTHREAD_CREATE_DETACHED); + while (detached_stacks < stacks && refused == 0) { + pthread_t t; + refused = pthread_create(&t, &big, detached_main, nullptr); + if (refused == 0) + detached_stacks++; + } + { + std::unique_lock lock(detached_m); + detached_cv.wait(lock, [&] { return detached_ran == detached_stacks; }); + } + std::cout << "threads with 64 MiB stacks: " << joined_stacks << " joined, " << detached_stacks + << " detached, the last create answering " << errno_name(refused) << "\n"; + + std::recursive_mutex recursive; + recursive.lock(); + bool again = recursive.try_lock(); + if (again) + recursive.unlock(); + recursive.unlock(); + pthread_mutexattr_t checking; + pthread_mutexattr_init(&checking); + pthread_mutexattr_settype(&checking, PTHREAD_MUTEX_ERRORCHECK); + pthread_mutex_t checked; + pthread_mutex_init(&checked, &checking); + pthread_mutex_lock(&checked); + int relock = pthread_mutex_lock(&checked); + pthread_mutex_unlock(&checked); + std::cout << "a recursive mutex takes a second lock: " << (again ? "yes" : "no") + << "; an error-checking one answers " << errno_name(relock) << "\n"; + + char small[4]; + int whole = std::snprintf(small, sizeof small, "%d", 123456); + std::cout << "snprintf into 4 bytes answers " << whole << " and holds " << small << "\n"; + int printed = std::printf("%04100d\n", 42); + std::cout << "printf printed " << printed << " bytes\n"; + pthread_attr_t huge; + pthread_attr_init(&huge); + int no_stack = pthread_attr_setstacksize(&huge, SIZE_MAX); + std::cout << "a stack of SIZE_MAX bytes: " << errno_name(no_stack) << "; getentropy of nothing: " + << getentropy(nullptr, 0) << "\n"; + + std::ostringstream out; + out << std::stod("2.5") * 4 << ' ' << std::to_string(-17) << ' ' << std::stoull("18446744073709551615"); + std::cout << "stream " << out.str() << "\n"; + std::wstring wide = L"wide " + std::to_wstring(123); + std::cout << "wide length " << wide.size() << ", last " << static_cast(wide.back()) << "\n"; + std::map counts; + for (const char* word : {"a", "b", "a"}) + counts[word]++; + std::cout << "map a=" << counts["a"] << " b=" << counts["b"] << "\n"; + + { + std::mutex tm; + std::condition_variable tcv; + std::unique_lock lock(tm); + bool woke = tcv.wait_for(lock, std::chrono::milliseconds(10), [] { return false; }); + bool ready = false; + std::thread notifier([&] { + std::lock_guard g(tm); + ready = true; + tcv.notify_one(); + }); + auto notified = std::cv_status::no_timeout; + while (!ready && notified == std::cv_status::no_timeout) + notified = tcv.wait_until(lock, std::chrono::steady_clock::now() + std::chrono::seconds(30)); + lock.unlock(); + notifier.join(); + std::cout << "a wait nobody ends " << (woke ? "was woken" : "timed out") << ", a notified one answered " + << (notified == std::cv_status::no_timeout ? "no_timeout" : "timeout") << "\n"; + } + + std::cout << "a condition wait on a mutex it does not hold answers"; + for (int type : {PTHREAD_MUTEX_ERRORCHECK, PTHREAD_MUTEX_RECURSIVE}) { + pthread_mutexattr_t attr; + pthread_mutexattr_init(&attr); + pthread_mutexattr_settype(&attr, type); + pthread_mutex_t unheld; + pthread_mutex_init(&unheld, &attr); + pthread_cond_t cond; + pthread_cond_init(&cond, nullptr); + timespec at; + clock_gettime(CLOCK_REALTIME, &at); + at.tv_sec += 1; + int timed = pthread_cond_timedwait(&cond, &unheld, &at); + int plain = pthread_cond_wait(&cond, &unheld); + std::cout << (type == PTHREAD_MUTEX_ERRORCHECK ? " error-checking " : ", recursive ") << errno_name(timed) + << " and " << errno_name(plain); + } + std::cout << "\n"; + + register_all(std::make_integer_sequence{}); + per_thread.id = 0; + std::cout << "done\n"; + return 0; +} diff --git a/tests/cxx/runtime.expect b/tests/cxx/runtime.expect new file mode 100644 index 00000000000..f4421ae4fcb --- /dev/null +++ b/tests/cxx/runtime.expect @@ -0,0 +1,27 @@ +vector sum 55, back 10, string hello, ToyOS (12) +caught thrown from depth 0 +caught int 42, rethrowing +caught rethrown int 42 +caught out_of_range from at +caught invalid_argument from stoi +caught out_of_range from stoi +threads summed 499500 +thread_local destructors ran 4 +condition variable handshake done +rethrew from a thread +a key set in main reads null in another thread and set in main; thread ids differ +a thread joined by a third on its own pthread_self, before or after its creator returned, gave 42 +pthread_exit handed its join 42 +threads with 64 MiB stacks: 64 joined, 64 detached, the last create answering 0 +a recursive mutex takes a second lock: yes; an error-checking one answers EDEADLK +snprintf into 4 bytes answers 6 and holds 123 +00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000042 +printf printed 4101 bytes +a stack of SIZE_MAX bytes: EINVAL; getentropy of nothing: 0 +stream 10 -17 18446744073709551615 +wide length 8, last 3 +map a=2 b=1 +a wait nobody ends timed out, a notified one answered no_timeout +a condition wait on a mutex it does not hold answers error-checking EPERM and EPERM, recursive EPERM and EPERM +done +static destructor ran after 40 exit handlers and 5 thread_local destructors diff --git a/tests/toyos.rs b/tests/toyos.rs index d9134cf074e..540d3802f7d 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -645,6 +645,9 @@ const MACHINE_TESTS: &[(&str, Sched)] = &[ // One C program through the toolchain's clang, read by the loader's decoder, // and one boot to run it. ("c_hello", Sched::Parallel), + // One C++ program through the same clang and the C++ runtime its sysroot + // carries, and one boot. + ("cxx_runtime", Sched::Parallel), // One boot and one number, with no clock in the verdict: the frames are // counted in game tics, whatever the host's speed. ("doom_frames", Sched::Parallel), @@ -2499,11 +2502,6 @@ const NOT_RUN: &[NotRun] = &[ stage: Stage::Built, why: Why::Declined("AArch64's argument-passing corners, run here on x86-64: its `long double` lines print 0.0, for libc's reason in 22_floating_point"), }, - NotRun { - case: "83_utf8_in_identifiers", - stage: Stage::Built, - why: Why::Declined("its identifiers are UTF-8 and so is its `printf` format, and libc's `printf` writes each byte of a format as a character of its own (issues/build/libc-printf-re-encodes-every-non-ascii-byte-of-its-format.md): `привет` arrives as `пÑ\u{80}ивеÑ\u{82}`"), - }, NotRun { case: "95_bitfields", stage: Stage::Built, @@ -2589,26 +2587,16 @@ const NOT_RUN: &[NotRun] = &[ stage: Stage::Refused("definition 'alias_int' cannot also be an alias"), why: Why::Declined("it gives a symbol a definition and an alias at once, which TinyCC allows and clang refuses"), }, - NotRun { - case: "124_atomic_counter", - stage: Stage::Refused("unknown type name 'uint_least16_t'"), - why: Why::Declined("C11 atomics: clang's `stdatomic.h` needs the `least` types `stdint.h` does not define"), - }, NotRun { case: "125_atomic_misc", - stage: Stage::Refused("unknown type name 'uint_least16_t'"), - why: Why::Declined("C11 atomics, as 124_atomic_counter"), + stage: Stage::NoLink("main"), + why: Why::Declined("each of its `main`s is behind a `test_*` -D the harness does not pass, so the file preprocesses to no `main`"), }, NotRun { case: "128_run_atexit", stage: Stage::NoLink("on_exit"), why: Why::Declined("`on_exit`, a glibc extension libc does not define, and a -D per configuration to have a main at all"), }, - NotRun { - case: "136_atomic_gcc_style", - stage: Stage::Refused("unknown type name 'uint_least16_t'"), - why: Why::Declined("C11 atomics, as 124_atomic_counter"), - }, ]; /// Discover C tests by scanning tests/testcases/tinycc/*.c. @@ -10403,6 +10391,7 @@ fn run_machine_test( "pci_claim_caps_truncated" => faults::claim_caps_truncated(), // Body in `tests/common/clang.rs`. "c_hello" => common::clang::c_hello(rust_bins), + "cxx_runtime" => common::clang::cxx_runtime(rust_bins), "doom_frames" => doom_frames(rust_bins), "metal_sim_compositor" => { metal_sim_compositor(group_boot(held, METAL_SIM_DESKTOP, || { diff --git a/toyos-libc-copies/Cargo.toml b/toyos-libc-copies/Cargo.toml index 49816bb3e1e..55b2b580851 100644 --- a/toyos-libc-copies/Cargo.toml +++ b/toyos-libc-copies/Cargo.toml @@ -1,17 +1,18 @@ [package] name = "toyos-libc-copies" -description = "A host differential test of userland libc's architecture module: its copies, fills and square roots against core's." +description = "A host differential test of userland libc's self-contained modules: its copies, fills and square roots against core's, its UTF-8 reader against core's, and its number reader and long double widening against the host C library and compiler-builtins." version = "0.1.0" edition = "2021" license = "MIT OR Apache-2.0" publish = false -# `userland/libc/src/arch/`'s copies, fills and square roots, compiled for the -# host this runs on and held against Rust's own. A package of its own because -# libc cross-compiles and exports `memcpy`: this compiles only the architecture -# module, the way `kernel-loom` compiles kernel files. `std-runtime` is libc's -# own switch, on here so the module's `_start` stays out of a host binary that -# has an entry of its own. +# `userland/libc/src/`'s architecture module, UTF-8 reader and number reader, +# compiled for the host this runs on and held against Rust's own and the host C +# library's. A package of its own because libc cross-compiles and exports +# `memcpy`: this compiles only those modules, the way `kernel-loom` compiles +# kernel files. `std-runtime` is libc's own switch, on here so the +# architecture module's `_start` stays out of a host binary that has an entry +# of its own. [features] default = ["std-runtime"] std-runtime = [] diff --git a/toyos-libc-copies/src/errno_codes.rs b/toyos-libc-copies/src/errno_codes.rs new file mode 100644 index 00000000000..a004fd7f013 --- /dev/null +++ b/toyos-libc-copies/src/errno_codes.rs @@ -0,0 +1,31 @@ +//! libc's one list of errno codes (`errno.rs`) against `include/errno.h`, the +//! numbers a C program compares them with: every code the library answers +//! with is the header's, by name and value. + +use std::collections::HashMap; + +#[test] +fn every_code_libc_answers_with_is_errno_h_s() { + let header = include_str!("../../userland/libc/include/errno.h"); + let defined: HashMap<&str, i32> = header + .lines() + .filter_map(|line| line.strip_prefix("#define ")) + .filter_map(|rest| { + let mut words = rest.split_whitespace(); + Some((words.next()?, words.next()?.parse().ok()?)) + }) + .collect(); + let list = include_str!("../../userland/libc/src/errno.rs"); + let codes: Vec<(&str, i32)> = list + .lines() + .filter_map(|line| line.strip_prefix("pub(crate) const ")) + .map(|rest| { + let (name, value) = rest.split_once(": i32 = ").unwrap_or_else(|| panic!("errno.rs: {rest}")); + (name, value.trim_end_matches(';').parse().unwrap_or_else(|_| panic!("errno.rs: {rest}"))) + }) + .collect(); + assert!(codes.len() > 20, "errno.rs lists {} codes", codes.len()); + for (name, value) in codes { + assert_eq!(defined.get(name), Some(&value), "{name}"); + } +} diff --git a/toyos-libc-copies/src/lib.rs b/toyos-libc-copies/src/lib.rs index 371c8d880e3..386cee5b3e0 100644 --- a/toyos-libc-copies/src/lib.rs +++ b/toyos-libc-copies/src/lib.rs @@ -1,12 +1,34 @@ -//! libc's architecture module on the host, differentially: every copy and fill -//! it has, over every length to 300 and every source and destination offset to -//! 20, against `copy_within` and `fill`, with the two buffers overlapping both -//! ways; and its square roots against `f64::sqrt` and `f32::sqrt`. Each host -//! architecture checks its own module. +//! libc's modules that read and set nothing but what they are handed, on the +//! host, differentially. The architecture module: every copy and fill it has, +//! over every length to 300 and every source and destination offset to 20, +//! against `copy_within` and `fill`, with the two buffers overlapping both ways; +//! and its square roots against `f64::sqrt` and `f32::sqrt`. Each host +//! architecture checks its own module. The UTF-8 reader against +//! `core::str::from_utf8`, the number reader against the host C library's, +//! AArch64's `long double` widening against compiler-builtins', and the errno +//! codes against `include/errno.h`. + +#[cfg(test)] +extern crate alloc; #[cfg(test)] #[path = "../../userland/libc/src/arch/mod.rs"] mod arch; +#[cfg(test)] +#[path = "../../userland/libc/src/strtonum.rs"] +mod strtonum; +#[cfg(test)] +#[path = "../../userland/libc/src/utf8.rs"] +mod utf8; + +#[cfg(test)] +mod errno_codes; +#[cfg(all(test, target_arch = "aarch64"))] +mod long_double; +#[cfg(test)] +mod strtonum_differential; +#[cfg(test)] +mod utf8_differential; #[cfg(test)] mod tests { diff --git a/toyos-libc-copies/src/long_double.rs b/toyos-libc-copies/src/long_double.rs new file mode 100644 index 00000000000..3c08cf78c3c --- /dev/null +++ b/toyos-libc-copies/src/long_double.rs @@ -0,0 +1,61 @@ +//! AArch64's `strtold`, the arch module's own, against compiler-builtins' +//! `__extenddftf2`, the conversion a compiler emits for `(long double)x`: the +//! binary128 each leaves in `q0` for the `double` the host's `strtod` reads, +//! over the specials, every exponent's edges and seeded random bit patterns. + +use std::ffi::CString; + +unsafe extern "C" { + /// The arch module's: this binary defines it, ahead of the host's. + fn strtold(); + fn __extenddftf2(); +} + +/// `q0`'s bits after `f` is called with `x0` and `d0` as given. +fn q0(f: unsafe extern "C" fn(), x0: usize, d0: f64) -> u128 { + let (lo, hi): (u64, u64); + // SAFETY: both callees take at most `x0`, `x1` and `d0` under the C ABI, + // and `x0` is a NUL-terminated string where `strtold` reads one. + unsafe { + core::arch::asm!( + "blr {f}", + "fmov x0, d0", + "mov x1, v0.d[1]", + f = in(reg) f, + inout("x0") x0 => lo, + inout("x1") 0usize => hi, + inout("d0") d0 => _, + clobber_abi("C"), + ); + } + u128::from(hi) << 64 | u128::from(lo) +} + +fn judge(x: f64) { + let text = CString::new(format!("{x:e}")).unwrap(); + let ours = q0(strtold, text.as_ptr() as usize, 0.0); + let widened = q0(__extenddftf2, 0, x); + assert_eq!(ours, widened, "{x:e} ({:#018x}): {ours:#034x}, compiler-builtins' {widened:#034x}", x.to_bits()); +} + +#[test] +fn strtold_widens_as_compiler_builtins_does() { + for x in [0.0, -0.0, 1.0, -2.5, f64::MIN_POSITIVE, f64::MAX, f64::MIN, f64::INFINITY, f64::NEG_INFINITY, f64::NAN] { + judge(x); + } + for exponent in 0..0x7ff_u64 { + for fraction in [0, 1, 1 << 51, (1 << 52) - 1] { + judge(f64::from_bits(exponent << 52 | fraction)); + } + } + let mut state = 0x853c_49e6_748f_ea9b_u64; + for _ in 0..20_000 { + state ^= state << 13; + state ^= state >> 7; + state ^= state << 17; + let x = f64::from_bits(state); + if !x.is_nan() { + judge(x); + } + } +} diff --git a/toyos-libc-copies/src/strtonum_differential.rs b/toyos-libc-copies/src/strtonum_differential.rs new file mode 100644 index 00000000000..b817416c901 --- /dev/null +++ b/toyos-libc-copies/src/strtonum_differential.rs @@ -0,0 +1,226 @@ +//! libc's number reader against the host C library's `strtod`, `strtof`, +//! `strtol` and `strtoul`: the value, bit for bit, where it ends, and the +//! `ERANGE` and `EINVAL` C requires, over a corpus of the grammar's corners and +//! seeded random numbers, hexadecimal ones with a rounding tie in half of them. + +use std::ffi::{c_int, CString}; + +use crate::strtonum::{self, Read, Refusal}; + +unsafe extern "C" { + fn strtod(s: *const u8, end: *mut *mut u8) -> f64; + fn strtof(s: *const u8, end: *mut *mut u8) -> f32; + fn strtol(s: *const u8, end: *mut *mut u8, base: c_int) -> i64; + fn strtoul(s: *const u8, end: *mut *mut u8, base: c_int) -> u64; + #[cfg_attr(target_os = "macos", link_name = "__error")] + #[cfg_attr(target_os = "linux", link_name = "__errno_location")] + fn errno_location() -> *mut c_int; +} + +/// The host's `ERANGE` and `EINVAL`, which macOS and Linux number alike. +const ERANGE: c_int = 34; +const EINVAL: c_int = 22; + +/// What the host's reader answers for `s`: the value, where it ends, `errno`. +fn host(s: &CString, read: impl Fn(*const u8, *mut *mut u8) -> T) -> (T, usize, c_int) { + let mut end = std::ptr::null_mut(); + // SAFETY: the host's errno slot is this thread's. + unsafe { *errno_location() = 0 }; + let value = read(s.as_ptr().cast(), &mut end); + // SAFETY: as above. + (value, end as usize - s.as_ptr() as usize, unsafe { *errno_location() }) +} + +/// What errno a refusal is, as the host numbers it. +fn errno_of(refused: &Option) -> c_int { + match refused { + None => 0, + Some(Refusal::Base) => EINVAL, + Some(Refusal::Range) => ERANGE, + } +} + +/// Both NaN of one sign, or the same bits. +fn same(a: f64, b: f64) -> bool { + if a.is_nan() || b.is_nan() { + a.is_nan() && b.is_nan() && a.is_sign_negative() == b.is_sign_negative() + } else { + a.to_bits() == b.to_bits() + } +} + +/// `s` as libc reads it through both its code units, which must agree. +fn ours>(s: &CString) -> Read { + let wide: Vec = s.as_bytes_with_nul().iter().map(|&b| b.into()).collect(); + // SAFETY: both strings are NUL-terminated. + let (narrow, wide) = unsafe { (strtonum::float::(s.as_ptr().cast()), strtonum::float::<_, F>(wide.as_ptr())) }; + assert!( + same(narrow.value.into(), wide.value.into()) && narrow.end == wide.end && errno_of(&narrow.refused) == errno_of(&wide.refused), + "{s:?}: the wide reading differs" + ); + narrow +} + +/// `s` read as a `double` and a `float`, against the host. `ERANGE` is held +/// against the host's only for an overflow: C leaves an underflow's to each +/// library. +fn judge_float(s: &str) { + let c = CString::new(s).unwrap(); + // SAFETY: `c` is NUL-terminated. + let (value, end, errno) = host(&c, |s, e| unsafe { strtod(s, e) }); + let read = ours::(&c); + assert!(same(read.value, value), "{s:?}: {:e} ({:#x}), the host's {value:e} ({:#x})", read.value, read.value.to_bits(), value.to_bits()); + assert_eq!(read.end, end, "{s:?}: where it ends"); + if value.is_infinite() && !s.to_ascii_lowercase().contains("inf") { + assert_eq!((errno_of(&read.refused), errno), (ERANGE, ERANGE), "{s:?}: an overflow"); + } + // SAFETY: as above. + let (value, end, _) = host(&c, |s, e| unsafe { strtof(s, e) }); + let read = ours::(&c); + assert!(same(read.value.into(), value.into()), "{s:?}: float {:e}, the host's {value:e}", read.value); + assert_eq!(read.end, end, "{s:?}: where the float ends"); +} + +/// `s` read as `long` and `unsigned long` in `base`, against the host. Where a +/// base C defines finds no number, POSIX lets `errno` say `EINVAL` or nothing, +/// and it is not compared. +fn judge_int(s: &str, base: i32) { + let c = CString::new(s).unwrap(); + let errno = |read_errno: c_int, end: usize| if end == 0 && (base == 0 || (2..=36).contains(&base)) { 0 } else { read_errno }; + // SAFETY: NUL-terminated. + let read = unsafe { strtonum::signed::(c.as_ptr().cast(), base) }; + // SAFETY: as above. + let (value, end, host_errno) = host(&c, |s, e| unsafe { strtol(s, e, base) }); + assert_eq!((read.value, read.end, errno(errno_of(&read.refused), read.end)), (value, end, errno(host_errno, end)), "strtol({s:?}, {base})"); + // SAFETY: as above. + let read = unsafe { strtonum::unsigned::(c.as_ptr().cast(), base) }; + // SAFETY: as above. + let (value, end, host_errno) = host(&c, |s, e| unsafe { strtoul(s, e, base) }); + assert_eq!((read.value, read.end, errno(errno_of(&read.refused), read.end)), (value, end, errno(host_errno, end)), "strtoul({s:?}, {base})"); +} + +/// A seeded xorshift, so a red names the case that reproduces it. +struct Rng(u64); + +impl Rng { + fn next(&mut self) -> u64 { + self.0 ^= self.0 << 13; + self.0 ^= self.0 >> 7; + self.0 ^= self.0 << 17; + self.0 + } + + fn below(&mut self, n: u64) -> u64 { + self.next() % n + } + + fn hex(&mut self) -> char { + char::from_digit(self.below(16) as u32, 16).unwrap() + } +} + +#[test] +fn the_grammar_s_corners_agree_with_the_host() { + for s in [ + "", " ", "+", "-", ".", "e5", "0x", "0X", "0x.", "0x.p1", "0xp1", "0x1p", "0x1p+", "0x1p-x", "1e", "1e+", + "1e-x", ".5", "5.", "-.5e-3", " \t\n\x0b\x0c\r+1.5", "1.5xyz", "inf", "-INFINITY", "infinit", "infx", "nan", + "-nan", "NaN(12_ab)", "nan(", "nan()", "0x1.8", "0X1.8P1", "0x.8p1", "0x1P-2", "00x1p0", "0x0p0", + "-0x0p0", "0.000", "-0", "1e309", "-1e309", "1e-400", "4.9e-324", "2.4703282292062328e-324", + "2.4703282292062327e-324", "1.7976931348623157e308", "1.7976931348623158e308", "1.7976931348623159e308", + "0x1.fffffffffffffp1023", "0x1.fffffffffffff7p1023", "0x1.fffffffffffff8p1023", "0x1p1024", "0x1p-1074", + "0x1p-1075", "0x1.0000000000001p-1075", "0x1.8p-1075", "0x1p-1076", "0x0.0000000000001p-1022", + "0x0.00000000000008p-1022", "0x0.00000000000018p-1022", "0x1.00000000000008p0", "0x1.00000000000018p0", + "0x1.000000000000080000000000000000001p0", "0x1.0000000000000800000p0", "0x10000000000000080p0", + "0x10000000000000180p-4", "0x.000000000000000000000000000000000000000001p200", "0x1p-99999999999999999999", + "0x1p99999999999999999999", "0x1.000001p0", "0x1.0000008p0", "0x1.0000018p0", "0x1.fffffe8p127", "0x1p-149", + "0x1p-150", "0x1.8p-150", "123456789012345678901234567890", "0.1", "0.30000000000000004", "9007199254740993", + "1e23", "8.589973e9", + ] { + judge_float(s); + } + for (s, base) in [ + ("", 10), ("0x", 16), ("0x", 0), ("0xg", 16), ("0x1g", 0), (" -0x10", 0), ("010", 0), ("08", 0), ("0", 0), + ("+7", 8), ("-", 10), ("9223372036854775807", 10), ("9223372036854775808", 10), ("-9223372036854775808", 10), + ("-9223372036854775809", 10), ("18446744073709551615", 10), ("18446744073709551616", 10), ("-1", 10), + ("-18446744073709551615", 10), ("zz", 36), ("Zz", 36), ("z", 35), ("1", 1), ("1", 37), ("1", -1), + ("7fffffffffffffff", 16), ("ffffffffffffffffff", 16), (" \t12 34", 10), + ] { + judge_int(s, base); + } +} + +/// Hexadecimal numbers with up to 30 digits, exponents across both types' +/// ranges and past them, and in half of them a `double`'s 53 bits followed by +/// exactly half a unit, or half and a sticky bit, the ties to even. +#[test] +fn random_hexadecimal_numbers_round_as_the_host_s() { + let mut rng = Rng(0x9e37_79b9_7f4a_7c15); + for _ in 0..200_000 { + let mut s = String::new(); + if rng.below(4) == 0 { + s.push('-'); + } + s.push_str(if rng.below(2) == 0 { "0x" } else { "0X" }); + if rng.below(2) == 0 { + s.push('1'); + s.push('.'); + (0..13).for_each(|_| s.push(rng.hex())); + s.push('8'); + (0..rng.below(4)).for_each(|_| s.push('0')); + if rng.below(2) == 0 { + s.push('1'); + } + } else { + let digits = 1 + rng.below(30); + let point = rng.below(digits + 1); + for i in 0..digits { + if i == point { + s.push('.'); + } + s.push(rng.hex()); + } + } + let exponent = rng.below(2300) as i64 - 1150; + s.push_str(&format!("{}{exponent}", if rng.below(2) == 0 { 'p' } else { 'P' })); + if rng.below(8) == 0 { + s.push('z'); + } + judge_float(&s); + } +} + +/// Decimal numbers with up to 25 digits and exponents across `double`'s range. +#[test] +fn random_decimal_numbers_round_as_the_host_s() { + let mut rng = Rng(0x2545_f491_4f6c_dd1d); + for _ in 0..50_000 { + let digits = 1 + rng.below(25); + let point = rng.below(digits + 1); + let mut s = String::new(); + for i in 0..digits { + if i == point { + s.push('.'); + } + s.push(char::from_digit(rng.below(10) as u32, 10).unwrap()); + } + s.push_str(&format!("e{}", rng.below(700) as i64 - 350)); + judge_float(&s); + } +} + +/// Integers in every base C defines, and in base 0 with each prefix. +#[test] +fn random_integers_read_as_the_host_s() { + let mut rng = Rng(0xd1b5_4a32_d192_ed03); + for _ in 0..50_000 { + let base = [0, 2, 8, 10, 16, 36][rng.below(6) as usize]; + let mut s = String::from([" ", "", "-", "+"][rng.below(4) as usize]); + if base == 0 || base == 16 { + s.push_str(["", "0x", "0X", "0"][rng.below(4) as usize]); + } + for _ in 0..1 + rng.below(24) { + s.push(char::from_digit(rng.below(36) as u32, 36).unwrap()); + } + judge_int(&s, base); + } +} diff --git a/toyos-libc-copies/src/utf8_differential.rs b/toyos-libc-copies/src/utf8_differential.rs new file mode 100644 index 00000000000..a4362a3f678 --- /dev/null +++ b/toyos-libc-copies/src/utf8_differential.rs @@ -0,0 +1,80 @@ +//! libc's UTF-8 reader against `core::str::from_utf8`, over every sequence of +//! one to four bytes: each is fed a byte at a time, and where the reader ends a +//! character, refuses, or waits for more, `from_utf8` of the bytes so far must +//! say the same. A sequence's verdict is its first decided prefix's, so the +//! walk extends only the prefixes both call undecided, and every byte after +//! each of them. + +use crate::utf8::{Byte, MbState}; + +#[derive(Debug, PartialEq)] +enum Verdict { + /// A character ends at the last byte: its code point. + Char(u32), + /// No continuation makes the bytes a character. + Refused, + /// Some continuation does. + Prefix, +} + +/// What the reader says of the last of `bytes`, fed from the initial state; +/// every byte before it left a prefix. +fn reader(bytes: &[u8]) -> Verdict { + let mut st = MbState::INITIAL; + let (last, before) = bytes.split_last().expect("one byte at least"); + for &b in before { + assert!(matches!(st.feed(b), Byte::Continues), "{bytes:02x?}: decided before its last byte"); + } + match st.feed(*last) { + Byte::Ends(cp) => Verdict::Char(cp), + Byte::Refused => Verdict::Refused, + Byte::Continues => Verdict::Prefix, + } +} + +/// What `from_utf8` says of `bytes`, whose every proper prefix it calls +/// incomplete: `error_len` is `None` exactly when the input ended inside a +/// sequence some continuation completes. +fn oracle(bytes: &[u8]) -> Verdict { + match core::str::from_utf8(bytes) { + Ok(s) => Verdict::Char(s.chars().next().expect("one byte at least") as u32), + Err(e) if e.error_len().is_some() => Verdict::Refused, + Err(_) => Verdict::Prefix, + } +} + +#[test] +fn every_sequence_to_four_bytes_agrees_with_from_utf8() { + let (mut prefixes, mut judged, mut chars) = (vec![Vec::new()], 0u64, 0u32); + while let Some(prefix) = prefixes.pop() { + for b in 0..=u8::MAX { + let mut bytes = prefix.clone(); + bytes.push(b); + let verdict = oracle(&bytes); + assert_eq!(reader(&bytes), verdict, "{bytes:02x?}"); + judged += 1; + match verdict { + Verdict::Prefix => { + assert!(bytes.len() < 4, "{bytes:02x?}: four bytes and still a prefix"); + prefixes.push(bytes); + } + Verdict::Char(_) => chars += 1, + Verdict::Refused => {} + } + } + } + // Every scalar value but none of the surrogates, each in one form. + assert_eq!(chars, 0x11_0000 - 0x800); + assert!(judged > 0x11_0000, "{judged}"); +} + +/// A refusal leaves the initial state: the next byte starts afresh. +#[test] +fn a_refusal_starts_the_next_character_afresh() { + let mut st = MbState::INITIAL; + for b in [0xe0, 0x80] { + st.feed(b); + } + assert!(!st.is_partial()); + assert!(matches!(st.feed(b'a'), Byte::Ends(0x61))); +} diff --git a/userland/libc/include/arpa/inet.h b/userland/libc/include/arpa/inet.h index 46a8c572c07..145d3a58ab8 100644 --- a/userland/libc/include/arpa/inet.h +++ b/userland/libc/include/arpa/inet.h @@ -3,8 +3,16 @@ #include +#ifdef __cplusplus +extern "C" { +#endif + int inet_pton(int af, const char *src, void *dst); const char *inet_ntop(int af, const void *src, char *dst, socklen_t size); uint32_t inet_addr(const char *cp); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/assert.h b/userland/libc/include/assert.h index 3871493735a..e4e1cbc9084 100644 --- a/userland/libc/include/assert.h +++ b/userland/libc/include/assert.h @@ -1,6 +1,10 @@ #ifndef _ASSERT_H #define _ASSERT_H +#ifdef __cplusplus +extern "C" { +#endif + void __assert_fail(const char *expr, const char *file, int line, const char *func); #ifdef NDEBUG @@ -9,4 +13,8 @@ void __assert_fail(const char *expr, const char *file, int line, const char *fun #define assert(expr) ((expr) ? (void)0 : __assert_fail(#expr, __FILE__, __LINE__, __func__)) #endif +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/bits/types/FILE.h b/userland/libc/include/bits/types/FILE.h new file mode 100644 index 00000000000..e48a2b39933 --- /dev/null +++ b/userland/libc/include/bits/types/FILE.h @@ -0,0 +1,6 @@ +#ifndef _BITS_TYPES_FILE_H +#define _BITS_TYPES_FILE_H + +typedef struct _FILE FILE; + +#endif diff --git a/userland/libc/include/bits/types/locale_t.h b/userland/libc/include/bits/types/locale_t.h new file mode 100644 index 00000000000..82d872f6cc6 --- /dev/null +++ b/userland/libc/include/bits/types/locale_t.h @@ -0,0 +1,6 @@ +#ifndef _BITS_TYPES_LOCALE_T_H +#define _BITS_TYPES_LOCALE_T_H + +typedef struct __locale_struct *locale_t; + +#endif diff --git a/userland/libc/include/bits/types/mbstate_t.h b/userland/libc/include/bits/types/mbstate_t.h new file mode 100644 index 00000000000..4a8e6bcbcd7 --- /dev/null +++ b/userland/libc/include/bits/types/mbstate_t.h @@ -0,0 +1,9 @@ +#ifndef _BITS_TYPES_MBSTATE_T_H +#define _BITS_TYPES_MBSTATE_T_H + +typedef struct { + unsigned int __bits; + unsigned int __needed; +} mbstate_t; + +#endif diff --git a/userland/libc/include/bits/types/wint_t.h b/userland/libc/include/bits/types/wint_t.h new file mode 100644 index 00000000000..ca481fb4fd0 --- /dev/null +++ b/userland/libc/include/bits/types/wint_t.h @@ -0,0 +1,8 @@ +#ifndef _BITS_TYPES_WINT_T_H +#define _BITS_TYPES_WINT_T_H + +typedef __WINT_TYPE__ wint_t; + +#define WEOF ((wint_t)-1) + +#endif diff --git a/userland/libc/include/ctype.h b/userland/libc/include/ctype.h index 76a806ed659..36edcdb4d7e 100644 --- a/userland/libc/include/ctype.h +++ b/userland/libc/include/ctype.h @@ -1,9 +1,16 @@ #ifndef _CTYPE_H #define _CTYPE_H +#include + +#ifdef __cplusplus +extern "C" { +#endif + int isalpha(int c); int isdigit(int c); int isalnum(int c); +int isblank(int c); int isspace(int c); int isupper(int c); int islower(int c); @@ -15,4 +22,23 @@ int isgraph(int c); int toupper(int c); int tolower(int c); +int isalpha_l(int c, locale_t loc); +int isdigit_l(int c, locale_t loc); +int isalnum_l(int c, locale_t loc); +int isblank_l(int c, locale_t loc); +int isspace_l(int c, locale_t loc); +int isupper_l(int c, locale_t loc); +int islower_l(int c, locale_t loc); +int isprint_l(int c, locale_t loc); +int ispunct_l(int c, locale_t loc); +int isxdigit_l(int c, locale_t loc); +int iscntrl_l(int c, locale_t loc); +int isgraph_l(int c, locale_t loc); +int toupper_l(int c, locale_t loc); +int tolower_l(int c, locale_t loc); + +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/dlfcn.h b/userland/libc/include/dlfcn.h index 900f341e679..83383e88c64 100644 --- a/userland/libc/include/dlfcn.h +++ b/userland/libc/include/dlfcn.h @@ -1,6 +1,10 @@ #ifndef _DLFCN_H #define _DLFCN_H +#ifdef __cplusplus +extern "C" { +#endif + #define RTLD_LAZY 0x1 #define RTLD_NOW 0x2 #define RTLD_GLOBAL 0x100 @@ -12,4 +16,8 @@ void *dlsym(void *handle, const char *symbol); int dlclose(void *handle); char *dlerror(void); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/errno.h b/userland/libc/include/errno.h index 7fc819d6df5..6662185d19f 100644 --- a/userland/libc/include/errno.h +++ b/userland/libc/include/errno.h @@ -1,6 +1,10 @@ #ifndef _ERRNO_H #define _ERRNO_H +#ifdef __cplusplus +extern "C" { +#endif + int *__errno_location(void); #define errno (*__errno_location()) @@ -28,28 +32,49 @@ int *__errno_location(void); #define ENFILE 23 #define EMFILE 24 #define ENOTTY 25 +#define ETXTBSY 26 #define EFBIG 27 #define ENOSPC 28 #define ESPIPE 29 #define EROFS 30 +#define EMLINK 31 #define EPIPE 32 #define EDOM 33 #define ERANGE 34 +#define EDEADLK 35 +#define ENOLCK 37 #define ENOSYS 38 #define ELOOP 40 +#define ENOMSG 42 +#define EIDRM 43 +#define ENOSTR 60 +#define ENODATA 61 +#define ETIME 62 +#define ENOSR 63 +#define ENOLINK 67 +#define EPROTO 71 +#define EBADMSG 74 +#define EOVERFLOW 75 +#define EILSEQ 84 #define ENAMETOOLONG 36 #define ENOTEMPTY 39 #define EWOULDBLOCK EAGAIN #define EINPROGRESS 115 #define EALREADY 114 #define ENOTSOCK 88 +#define EDESTADDRREQ 89 #define EMSGSIZE 90 #define EPROTOTYPE 91 #define ENOPROTOOPT 92 +#define EPROTONOSUPPORT 93 +#define EOPNOTSUPP 95 +#define ENOTSUP EOPNOTSUPP #define EAFNOSUPPORT 97 #define EADDRINUSE 98 #define EADDRNOTAVAIL 99 +#define ENETDOWN 100 #define ENETUNREACH 101 +#define ENETRESET 102 #define ECONNABORTED 103 #define ECONNRESET 104 #define ENOBUFS 105 @@ -57,5 +82,13 @@ int *__errno_location(void); #define ENOTCONN 107 #define ETIMEDOUT 110 #define ECONNREFUSED 111 +#define EHOSTUNREACH 113 +#define ECANCELED 125 +#define EOWNERDEAD 130 +#define ENOTRECOVERABLE 131 + +#ifdef __cplusplus +} +#endif #endif diff --git a/userland/libc/include/fcntl.h b/userland/libc/include/fcntl.h index 95c0b4eab97..1d07513137c 100644 --- a/userland/libc/include/fcntl.h +++ b/userland/libc/include/fcntl.h @@ -3,6 +3,10 @@ #include +#ifdef __cplusplus +extern "C" { +#endif + #define O_RDONLY 0x0000 #define O_WRONLY 0x0001 #define O_RDWR 0x0002 @@ -24,4 +28,8 @@ int open(const char *path, int flags, ...); int fcntl(int fd, int cmd, ...); int creat(const char *path, int mode); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/glob.h b/userland/libc/include/glob.h index d9cdf154de5..07508c33dee 100644 --- a/userland/libc/include/glob.h +++ b/userland/libc/include/glob.h @@ -3,6 +3,10 @@ #include +#ifdef __cplusplus +extern "C" { +#endif + typedef struct { size_t gl_pathc; char **gl_pathv; @@ -23,4 +27,8 @@ typedef struct { int glob(const char *pattern, int flags, int (*errfunc)(const char *, int), glob_t *pglob); void globfree(glob_t *pglob); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/inttypes.h b/userland/libc/include/inttypes.h index d1383ba5702..018ea12cfd6 100644 --- a/userland/libc/include/inttypes.h +++ b/userland/libc/include/inttypes.h @@ -3,35 +3,113 @@ #include -#define PRId8 "d" -#define PRId16 "d" -#define PRId32 "d" -#define PRId64 "ld" - -#define PRIi8 "i" -#define PRIi16 "i" -#define PRIi32 "i" -#define PRIi64 "li" - -#define PRIu8 "u" -#define PRIu16 "u" -#define PRIu32 "u" -#define PRIu64 "lu" - -#define PRIx8 "x" -#define PRIx16 "x" -#define PRIx32 "x" -#define PRIx64 "lx" - -#define PRIX8 "X" -#define PRIX16 "X" -#define PRIX32 "X" -#define PRIX64 "lX" - -#define PRIo8 "o" -#define PRIo16 "o" -#define PRIo32 "o" -#define PRIo64 "lo" +#include + +#ifdef __cplusplus +extern "C" { +#endif + +typedef struct { + intmax_t quot; + intmax_t rem; +} imaxdiv_t; + +intmax_t imaxabs(intmax_t j); +imaxdiv_t imaxdiv(intmax_t numer, intmax_t denom); +intmax_t strtoimax(const char *nptr, char **endptr, int base); +uintmax_t strtoumax(const char *nptr, char **endptr, int base); +intmax_t wcstoimax(const wchar_t *nptr, wchar_t **endptr, int base); +uintmax_t wcstoumax(const wchar_t *nptr, wchar_t **endptr, int base); + +#define PRId8 __INT8_FMTd__ +#define PRId16 __INT16_FMTd__ +#define PRId32 __INT32_FMTd__ +#define PRId64 __INT64_FMTd__ +#define PRIdLEAST8 __INT_LEAST8_FMTd__ +#define PRIdLEAST16 __INT_LEAST16_FMTd__ +#define PRIdLEAST32 __INT_LEAST32_FMTd__ +#define PRIdLEAST64 __INT_LEAST64_FMTd__ +#define PRIdFAST8 __INT_FAST8_FMTd__ +#define PRIdFAST16 __INT_FAST16_FMTd__ +#define PRIdFAST32 __INT_FAST32_FMTd__ +#define PRIdFAST64 __INT_FAST64_FMTd__ +#define PRIdMAX __INTMAX_FMTd__ +#define PRIdPTR __INTPTR_FMTd__ + +#define PRIi8 __INT8_FMTi__ +#define PRIi16 __INT16_FMTi__ +#define PRIi32 __INT32_FMTi__ +#define PRIi64 __INT64_FMTi__ +#define PRIiLEAST8 __INT_LEAST8_FMTi__ +#define PRIiLEAST16 __INT_LEAST16_FMTi__ +#define PRIiLEAST32 __INT_LEAST32_FMTi__ +#define PRIiLEAST64 __INT_LEAST64_FMTi__ +#define PRIiFAST8 __INT_FAST8_FMTi__ +#define PRIiFAST16 __INT_FAST16_FMTi__ +#define PRIiFAST32 __INT_FAST32_FMTi__ +#define PRIiFAST64 __INT_FAST64_FMTi__ +#define PRIiMAX __INTMAX_FMTi__ +#define PRIiPTR __INTPTR_FMTi__ + +#define PRIo8 __UINT8_FMTo__ +#define PRIo16 __UINT16_FMTo__ +#define PRIo32 __UINT32_FMTo__ +#define PRIo64 __UINT64_FMTo__ +#define PRIoLEAST8 __UINT_LEAST8_FMTo__ +#define PRIoLEAST16 __UINT_LEAST16_FMTo__ +#define PRIoLEAST32 __UINT_LEAST32_FMTo__ +#define PRIoLEAST64 __UINT_LEAST64_FMTo__ +#define PRIoFAST8 __UINT_FAST8_FMTo__ +#define PRIoFAST16 __UINT_FAST16_FMTo__ +#define PRIoFAST32 __UINT_FAST32_FMTo__ +#define PRIoFAST64 __UINT_FAST64_FMTo__ +#define PRIoMAX __UINTMAX_FMTo__ +#define PRIoPTR __UINTPTR_FMTo__ + +#define PRIu8 __UINT8_FMTu__ +#define PRIu16 __UINT16_FMTu__ +#define PRIu32 __UINT32_FMTu__ +#define PRIu64 __UINT64_FMTu__ +#define PRIuLEAST8 __UINT_LEAST8_FMTu__ +#define PRIuLEAST16 __UINT_LEAST16_FMTu__ +#define PRIuLEAST32 __UINT_LEAST32_FMTu__ +#define PRIuLEAST64 __UINT_LEAST64_FMTu__ +#define PRIuFAST8 __UINT_FAST8_FMTu__ +#define PRIuFAST16 __UINT_FAST16_FMTu__ +#define PRIuFAST32 __UINT_FAST32_FMTu__ +#define PRIuFAST64 __UINT_FAST64_FMTu__ +#define PRIuMAX __UINTMAX_FMTu__ +#define PRIuPTR __UINTPTR_FMTu__ + +#define PRIx8 __UINT8_FMTx__ +#define PRIx16 __UINT16_FMTx__ +#define PRIx32 __UINT32_FMTx__ +#define PRIx64 __UINT64_FMTx__ +#define PRIxLEAST8 __UINT_LEAST8_FMTx__ +#define PRIxLEAST16 __UINT_LEAST16_FMTx__ +#define PRIxLEAST32 __UINT_LEAST32_FMTx__ +#define PRIxLEAST64 __UINT_LEAST64_FMTx__ +#define PRIxFAST8 __UINT_FAST8_FMTx__ +#define PRIxFAST16 __UINT_FAST16_FMTx__ +#define PRIxFAST32 __UINT_FAST32_FMTx__ +#define PRIxFAST64 __UINT_FAST64_FMTx__ +#define PRIxMAX __UINTMAX_FMTx__ +#define PRIxPTR __UINTPTR_FMTx__ + +#define PRIX8 __UINT8_FMTX__ +#define PRIX16 __UINT16_FMTX__ +#define PRIX32 __UINT32_FMTX__ +#define PRIX64 __UINT64_FMTX__ +#define PRIXLEAST8 __UINT_LEAST8_FMTX__ +#define PRIXLEAST16 __UINT_LEAST16_FMTX__ +#define PRIXLEAST32 __UINT_LEAST32_FMTX__ +#define PRIXLEAST64 __UINT_LEAST64_FMTX__ +#define PRIXFAST8 __UINT_FAST8_FMTX__ +#define PRIXFAST16 __UINT_FAST16_FMTX__ +#define PRIXFAST32 __UINT_FAST32_FMTX__ +#define PRIXFAST64 __UINT_FAST64_FMTX__ +#define PRIXMAX __UINTMAX_FMTX__ +#define PRIXPTR __UINTPTR_FMTX__ #define SCNd32 "d" #define SCNd64 "ld" @@ -42,4 +120,8 @@ #define SCNx32 "x" #define SCNx64 "lx" +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/limits.h b/userland/libc/include/limits.h index a939aee8e7d..7a882576299 100644 --- a/userland/libc/include/limits.h +++ b/userland/libc/include/limits.h @@ -20,6 +20,9 @@ #define LLONG_MAX LONG_MAX #define ULLONG_MAX ULONG_MAX +/* UTF-8, the encoding of the one locale there is. */ +#define MB_LEN_MAX 4 + #define PATH_MAX 4096 #define NAME_MAX 255 diff --git a/userland/libc/include/link.h b/userland/libc/include/link.h index 2eef0a3203d..5f0f6aee3a0 100644 --- a/userland/libc/include/link.h +++ b/userland/libc/include/link.h @@ -4,6 +4,10 @@ #include #include +#ifdef __cplusplus +extern "C" { +#endif + #define ElfW(type) Elf64_##type /* The first four fields of glibc's layout. There is no dlpi_adds or @@ -19,4 +23,8 @@ struct dl_phdr_info { int dl_iterate_phdr(int (*callback)(struct dl_phdr_info *info, size_t size, void *data), void *data); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/locale.h b/userland/libc/include/locale.h new file mode 100644 index 00000000000..24cda314b7d --- /dev/null +++ b/userland/libc/include/locale.h @@ -0,0 +1,70 @@ +#ifndef _LOCALE_H +#define _LOCALE_H + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +/* The C locale is the only locale: every name that means it (C, POSIX, + C.UTF-8 and the empty default) names it, and its encoding is UTF-8. */ + +struct lconv { + char *decimal_point; + char *thousands_sep; + char *grouping; + char *int_curr_symbol; + char *currency_symbol; + char *mon_decimal_point; + char *mon_thousands_sep; + char *mon_grouping; + char *positive_sign; + char *negative_sign; + char int_frac_digits; + char frac_digits; + char p_cs_precedes; + char p_sep_by_space; + char n_cs_precedes; + char n_sep_by_space; + char p_sign_posn; + char n_sign_posn; + char int_p_cs_precedes; + char int_p_sep_by_space; + char int_n_cs_precedes; + char int_n_sep_by_space; + char int_p_sign_posn; + char int_n_sign_posn; +}; + +#define LC_CTYPE 0 +#define LC_NUMERIC 1 +#define LC_TIME 2 +#define LC_COLLATE 3 +#define LC_MONETARY 4 +#define LC_MESSAGES 5 +#define LC_ALL 6 + +#define LC_CTYPE_MASK (1 << LC_CTYPE) +#define LC_NUMERIC_MASK (1 << LC_NUMERIC) +#define LC_TIME_MASK (1 << LC_TIME) +#define LC_COLLATE_MASK (1 << LC_COLLATE) +#define LC_MONETARY_MASK (1 << LC_MONETARY) +#define LC_MESSAGES_MASK (1 << LC_MESSAGES) +#define LC_ALL_MASK (LC_CTYPE_MASK | LC_NUMERIC_MASK | LC_TIME_MASK | LC_COLLATE_MASK | LC_MONETARY_MASK | LC_MESSAGES_MASK) + +#define LC_GLOBAL_LOCALE ((locale_t)-1) + +char *setlocale(int category, const char *locale); +struct lconv *localeconv(void); +locale_t newlocale(int category_mask, const char *locale, locale_t base); +locale_t duplocale(locale_t loc); +void freelocale(locale_t loc); +locale_t uselocale(locale_t newloc); + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/userland/libc/include/math.h b/userland/libc/include/math.h index 88b3555ff0b..4883ff8f1c9 100644 --- a/userland/libc/include/math.h +++ b/userland/libc/include/math.h @@ -1,9 +1,44 @@ #ifndef _MATH_H #define _MATH_H +#ifdef __cplusplus +extern "C" { +#endif + +typedef float float_t; +typedef double double_t; + #define HUGE_VAL (__builtin_huge_val()) -#define INFINITY (__builtin_inf()) -#define NAN (__builtin_nan("")) +#define HUGE_VALF (__builtin_huge_valf()) +#define HUGE_VALL (__builtin_huge_vall()) +#define INFINITY (__builtin_inff()) +#define NAN (__builtin_nanf("")) + +#define FP_NAN 0 +#define FP_INFINITE 1 +#define FP_ZERO 2 +#define FP_SUBNORMAL 3 +#define FP_NORMAL 4 + +#define FP_ILOGB0 (-2147483647 - 1) +#define FP_ILOGBNAN (-2147483647 - 1) + +#define MATH_ERRNO 1 +#define MATH_ERREXCEPT 2 +#define math_errhandling MATH_ERRNO + +#define fpclassify(x) __builtin_fpclassify(FP_NAN, FP_INFINITE, FP_NORMAL, FP_SUBNORMAL, FP_ZERO, x) +#define isfinite(x) __builtin_isfinite(x) +#define isinf(x) __builtin_isinf(x) +#define isnan(x) __builtin_isnan(x) +#define isnormal(x) __builtin_isnormal(x) +#define signbit(x) __builtin_signbit(x) +#define isgreater(x, y) __builtin_isgreater(x, y) +#define isgreaterequal(x, y) __builtin_isgreaterequal(x, y) +#define isless(x, y) __builtin_isless(x, y) +#define islessequal(x, y) __builtin_islessequal(x, y) +#define islessgreater(x, y) __builtin_islessgreater(x, y) +#define isunordered(x, y) __builtin_isunordered(x, y) double floor(double x); double ceil(double x); @@ -34,8 +69,8 @@ float fabsf(float x); double ldexp(double x, int exp); double frexp(double x, int *exp); -int isnan(double x); -int isinf(double x); -int isfinite(double x); +#ifdef __cplusplus +} +#endif #endif diff --git a/userland/libc/include/netdb.h b/userland/libc/include/netdb.h index 8d3e4c1ea9f..8e9e08bb2cc 100644 --- a/userland/libc/include/netdb.h +++ b/userland/libc/include/netdb.h @@ -4,6 +4,10 @@ #include #include +#ifdef __cplusplus +extern "C" { +#endif + struct addrinfo { int ai_flags; int ai_family; @@ -32,4 +36,8 @@ int getaddrinfo(const char *node, const char *service, void freeaddrinfo(struct addrinfo *res); const char *gai_strerror(int errcode); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/netinet/in.h b/userland/libc/include/netinet/in.h index d844ae0cbe1..66a6bf8dabb 100644 --- a/userland/libc/include/netinet/in.h +++ b/userland/libc/include/netinet/in.h @@ -4,6 +4,10 @@ #include #include +#ifdef __cplusplus +extern "C" { +#endif + #define INADDR_ANY ((uint32_t)0x00000000) #define INADDR_LOOPBACK ((uint32_t)0x7f000001) #define INADDR_NONE ((uint32_t)0xffffffff) @@ -28,4 +32,8 @@ uint16_t ntohs(uint16_t netshort); uint32_t htonl(uint32_t hostlong); uint32_t ntohl(uint32_t netlong); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/poll.h b/userland/libc/include/poll.h index a13f579f024..d95af4b501e 100644 --- a/userland/libc/include/poll.h +++ b/userland/libc/include/poll.h @@ -1,6 +1,10 @@ #ifndef _POLL_H #define _POLL_H +#ifdef __cplusplus +extern "C" { +#endif + #define POLLIN 0x001 #define POLLPRI 0x002 #define POLLOUT 0x004 @@ -16,4 +20,8 @@ struct pollfd { int poll(struct pollfd *fds, unsigned int nfds, int timeout); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/pthread.h b/userland/libc/include/pthread.h index c3b623faec0..6d47539b84e 100644 --- a/userland/libc/include/pthread.h +++ b/userland/libc/include/pthread.h @@ -2,12 +2,20 @@ #define _PTHREAD_H #include +#include #include +#ifdef __cplusplus +extern "C" { +#endif + typedef unsigned long pthread_t; typedef struct { unsigned int __state; + unsigned int __kind; + unsigned long __owner; + unsigned long __count; } pthread_mutex_t; typedef struct { @@ -28,9 +36,9 @@ typedef unsigned long pthread_rwlockattr_t; typedef unsigned long pthread_attr_t; typedef unsigned int pthread_key_t; -#define PTHREAD_MUTEX_INITIALIZER { 0 } +#define PTHREAD_MUTEX_INITIALIZER { 0, 0, 0, 0 } #define PTHREAD_COND_INITIALIZER { 0 } -#define PTHREAD_RWLOCK_INITIALIZER { { 0 } } +#define PTHREAD_RWLOCK_INITIALIZER { PTHREAD_MUTEX_INITIALIZER } #define PTHREAD_ONCE_INIT { 0 } #define PTHREAD_MUTEX_NORMAL 0 @@ -41,11 +49,15 @@ typedef unsigned int pthread_key_t; #define PTHREAD_CREATE_JOINABLE 0 #define PTHREAD_CREATE_DETACHED 1 +#define PTHREAD_KEYS_MAX 128 +#define PTHREAD_DESTRUCTOR_ITERATIONS 4 + /* Thread */ int pthread_create(pthread_t *thread, const pthread_attr_t *attr, void *(*start_routine)(void *), void *arg); int pthread_join(pthread_t thread, void **retval); int pthread_detach(pthread_t thread); +__attribute__((__noreturn__)) void pthread_exit(void *retval); pthread_t pthread_self(void); int pthread_equal(pthread_t t1, pthread_t t2); @@ -59,10 +71,13 @@ int pthread_mutex_destroy(pthread_mutex_t *mutex); int pthread_mutexattr_init(pthread_mutexattr_t *attr); int pthread_mutexattr_destroy(pthread_mutexattr_t *attr); int pthread_mutexattr_settype(pthread_mutexattr_t *attr, int type); +int pthread_mutexattr_gettype(const pthread_mutexattr_t *attr, int *type); /* Condition variable */ int pthread_cond_init(pthread_cond_t *cond, const pthread_condattr_t *attr); int pthread_cond_wait(pthread_cond_t *cond, pthread_mutex_t *mutex); +int pthread_cond_timedwait(pthread_cond_t *cond, pthread_mutex_t *mutex, + const struct timespec *abstime); int pthread_cond_signal(pthread_cond_t *cond); int pthread_cond_broadcast(pthread_cond_t *cond); int pthread_cond_destroy(pthread_cond_t *cond); @@ -93,4 +108,8 @@ int pthread_attr_setstacksize(pthread_attr_t *attr, size_t stacksize); int pthread_attr_getstacksize(const pthread_attr_t *attr, size_t *stacksize); int pthread_attr_setdetachstate(pthread_attr_t *attr, int detachstate); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/sched.h b/userland/libc/include/sched.h new file mode 100644 index 00000000000..aff38687725 --- /dev/null +++ b/userland/libc/include/sched.h @@ -0,0 +1,14 @@ +#ifndef _SCHED_H +#define _SCHED_H + +#ifdef __cplusplus +extern "C" { +#endif + +int sched_yield(void); + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/userland/libc/include/semaphore.h b/userland/libc/include/semaphore.h index 751e527271e..c1a5041f268 100644 --- a/userland/libc/include/semaphore.h +++ b/userland/libc/include/semaphore.h @@ -1,6 +1,10 @@ #ifndef _SEMAPHORE_H #define _SEMAPHORE_H +#ifdef __cplusplus +extern "C" { +#endif + typedef struct { int value; } sem_t; @@ -10,4 +14,8 @@ int sem_wait(sem_t *sem); int sem_post(sem_t *sem); int sem_destroy(sem_t *sem); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/setjmp.h b/userland/libc/include/setjmp.h index 58fe73b13e5..0015bfc6c33 100644 --- a/userland/libc/include/setjmp.h +++ b/userland/libc/include/setjmp.h @@ -1,10 +1,18 @@ #ifndef _SETJMP_H #define _SETJMP_H +#ifdef __cplusplus +extern "C" { +#endif + /* x86-64: rbx, rbp, r12-r15, rsp, rip = 8 registers */ typedef long jmp_buf[8]; int setjmp(jmp_buf env); void longjmp(jmp_buf env, int val); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/signal.h b/userland/libc/include/signal.h index 9b523773129..8f1e6100488 100644 --- a/userland/libc/include/signal.h +++ b/userland/libc/include/signal.h @@ -1,6 +1,10 @@ #ifndef _SIGNAL_H #define _SIGNAL_H +#ifdef __cplusplus +extern "C" { +#endif + #define SIGHUP 1 #define SIGINT 2 #define SIGQUIT 3 @@ -90,4 +94,8 @@ int kill(int pid, int sig); #define REG_RSP 15 #define REG_RIP 16 +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/stdint.h b/userland/libc/include/stdint.h index 56bfeb30bab..c2d1a444e9e 100644 --- a/userland/libc/include/stdint.h +++ b/userland/libc/include/stdint.h @@ -1,46 +1,122 @@ #ifndef _STDINT_H #define _STDINT_H -typedef signed char int8_t; -typedef short int16_t; -typedef int int32_t; -typedef long int64_t; - -typedef unsigned char uint8_t; -typedef unsigned short uint16_t; -typedef unsigned int uint32_t; -typedef unsigned long uint64_t; - -typedef long intptr_t; -typedef unsigned long uintptr_t; - -typedef long intmax_t; -typedef unsigned long uintmax_t; - -#define INT8_MIN (-128) -#define INT16_MIN (-32768) -#define INT32_MIN (-2147483647-1) -#define INT64_MIN (-9223372036854775807L-1) - -#define INT8_MAX 127 -#define INT16_MAX 32767 -#define INT32_MAX 2147483647 -#define INT64_MAX 9223372036854775807L - -#define UINT8_MAX 255 -#define UINT16_MAX 65535 -#define UINT32_MAX 4294967295U -#define UINT64_MAX 18446744073709551615UL - -#define SIZE_MAX UINT64_MAX -#define INTPTR_MIN INT64_MIN -#define INTPTR_MAX INT64_MAX -#define UINTPTR_MAX UINT64_MAX -#define INTMAX_MIN INT64_MIN -#define INTMAX_MAX INT64_MAX -#define UINTMAX_MAX UINT64_MAX - -#define PTRDIFF_MIN INT64_MIN -#define PTRDIFF_MAX INT64_MAX +/* Every type and limit is the compiler's own for the target, so one header + serves every architecture ToyOS runs on. */ + +typedef __INT8_TYPE__ int8_t; +typedef __INT16_TYPE__ int16_t; +typedef __INT32_TYPE__ int32_t; +typedef __INT64_TYPE__ int64_t; + +typedef __UINT8_TYPE__ uint8_t; +typedef __UINT16_TYPE__ uint16_t; +typedef __UINT32_TYPE__ uint32_t; +typedef __UINT64_TYPE__ uint64_t; + +typedef __INT_LEAST8_TYPE__ int_least8_t; +typedef __INT_LEAST16_TYPE__ int_least16_t; +typedef __INT_LEAST32_TYPE__ int_least32_t; +typedef __INT_LEAST64_TYPE__ int_least64_t; + +typedef __UINT_LEAST8_TYPE__ uint_least8_t; +typedef __UINT_LEAST16_TYPE__ uint_least16_t; +typedef __UINT_LEAST32_TYPE__ uint_least32_t; +typedef __UINT_LEAST64_TYPE__ uint_least64_t; + +typedef __INT_FAST8_TYPE__ int_fast8_t; +typedef __INT_FAST16_TYPE__ int_fast16_t; +typedef __INT_FAST32_TYPE__ int_fast32_t; +typedef __INT_FAST64_TYPE__ int_fast64_t; + +typedef __UINT_FAST8_TYPE__ uint_fast8_t; +typedef __UINT_FAST16_TYPE__ uint_fast16_t; +typedef __UINT_FAST32_TYPE__ uint_fast32_t; +typedef __UINT_FAST64_TYPE__ uint_fast64_t; + +typedef __INTPTR_TYPE__ intptr_t; +typedef __UINTPTR_TYPE__ uintptr_t; + +typedef __INTMAX_TYPE__ intmax_t; +typedef __UINTMAX_TYPE__ uintmax_t; + +#define INT8_MAX __INT8_MAX__ +#define INT16_MAX __INT16_MAX__ +#define INT32_MAX __INT32_MAX__ +#define INT64_MAX __INT64_MAX__ +#define INT8_MIN (-INT8_MAX - 1) +#define INT16_MIN (-INT16_MAX - 1) +#define INT32_MIN (-INT32_MAX - 1) +#define INT64_MIN (-INT64_MAX - 1) +#define UINT8_MAX __UINT8_MAX__ +#define UINT16_MAX __UINT16_MAX__ +#define UINT32_MAX __UINT32_MAX__ +#define UINT64_MAX __UINT64_MAX__ + +#define INT_LEAST8_MAX __INT_LEAST8_MAX__ +#define INT_LEAST16_MAX __INT_LEAST16_MAX__ +#define INT_LEAST32_MAX __INT_LEAST32_MAX__ +#define INT_LEAST64_MAX __INT_LEAST64_MAX__ +#define INT_LEAST8_MIN (-INT_LEAST8_MAX - 1) +#define INT_LEAST16_MIN (-INT_LEAST16_MAX - 1) +#define INT_LEAST32_MIN (-INT_LEAST32_MAX - 1) +#define INT_LEAST64_MIN (-INT_LEAST64_MAX - 1) +#define UINT_LEAST8_MAX __UINT_LEAST8_MAX__ +#define UINT_LEAST16_MAX __UINT_LEAST16_MAX__ +#define UINT_LEAST32_MAX __UINT_LEAST32_MAX__ +#define UINT_LEAST64_MAX __UINT_LEAST64_MAX__ + +#define INT_FAST8_MAX __INT_FAST8_MAX__ +#define INT_FAST16_MAX __INT_FAST16_MAX__ +#define INT_FAST32_MAX __INT_FAST32_MAX__ +#define INT_FAST64_MAX __INT_FAST64_MAX__ +#define INT_FAST8_MIN (-INT_FAST8_MAX - 1) +#define INT_FAST16_MIN (-INT_FAST16_MAX - 1) +#define INT_FAST32_MIN (-INT_FAST32_MAX - 1) +#define INT_FAST64_MIN (-INT_FAST64_MAX - 1) +#define UINT_FAST8_MAX __UINT_FAST8_MAX__ +#define UINT_FAST16_MAX __UINT_FAST16_MAX__ +#define UINT_FAST32_MAX __UINT_FAST32_MAX__ +#define UINT_FAST64_MAX __UINT_FAST64_MAX__ + +#define INTPTR_MAX __INTPTR_MAX__ +#define INTPTR_MIN (-INTPTR_MAX - 1) +#define UINTPTR_MAX __UINTPTR_MAX__ + +#define INTMAX_MAX __INTMAX_MAX__ +#define INTMAX_MIN (-INTMAX_MAX - 1) +#define UINTMAX_MAX __UINTMAX_MAX__ + +#define PTRDIFF_MAX __PTRDIFF_MAX__ +#define PTRDIFF_MIN (-PTRDIFF_MAX - 1) +#define SIZE_MAX __SIZE_MAX__ + +#define SIG_ATOMIC_MAX __SIG_ATOMIC_MAX__ +#define SIG_ATOMIC_MIN (-SIG_ATOMIC_MAX - 1) + +#define WCHAR_MAX __WCHAR_MAX__ +#ifdef __WCHAR_UNSIGNED__ +#define WCHAR_MIN 0 +#else +#define WCHAR_MIN (-WCHAR_MAX - 1) +#endif + +#define WINT_MAX __WINT_MAX__ +#ifdef __WINT_UNSIGNED__ +#define WINT_MIN 0 +#else +#define WINT_MIN (-WINT_MAX - 1) +#endif + +#define INT8_C(c) __INT8_C(c) +#define INT16_C(c) __INT16_C(c) +#define INT32_C(c) __INT32_C(c) +#define INT64_C(c) __INT64_C(c) +#define UINT8_C(c) __UINT8_C(c) +#define UINT16_C(c) __UINT16_C(c) +#define UINT32_C(c) __UINT32_C(c) +#define UINT64_C(c) __UINT64_C(c) +#define INTMAX_C(c) __INTMAX_C(c) +#define UINTMAX_C(c) __UINTMAX_C(c) #endif diff --git a/userland/libc/include/stdio.h b/userland/libc/include/stdio.h index e8a53cf47ca..27f39a1c48f 100644 --- a/userland/libc/include/stdio.h +++ b/userland/libc/include/stdio.h @@ -3,8 +3,11 @@ #include #include +#include -typedef struct _FILE FILE; +#ifdef __cplusplus +extern "C" { +#endif extern FILE *stdin; extern FILE *stdout; @@ -26,6 +29,8 @@ int vprintf(const char *fmt, va_list ap); int vfprintf(FILE *stream, const char *fmt, va_list ap); int vsprintf(char *str, const char *fmt, va_list ap); int vsnprintf(char *str, size_t size, const char *fmt, va_list ap); +int asprintf(char **strp, const char *fmt, ...); +int vasprintf(char **strp, const char *fmt, va_list ap); int sscanf(const char *str, const char *fmt, ...); FILE *fopen(const char *path, const char *mode); @@ -59,4 +64,8 @@ FILE *tmpfile(void); void perror(const char *s); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/stdlib.h b/userland/libc/include/stdlib.h index 4154e12549d..e0bebc26f82 100644 --- a/userland/libc/include/stdlib.h +++ b/userland/libc/include/stdlib.h @@ -3,30 +3,51 @@ #include +#include + +#ifdef __cplusplus +extern "C" { +#endif + #define EXIT_SUCCESS 0 #define EXIT_FAILURE 1 #define RAND_MAX 2147483647 +/* The C locale is the only one, and its encoding is UTF-8. */ +#define MB_CUR_MAX ((size_t)4) + +typedef struct { int quot; int rem; } div_t; +typedef struct { long quot; long rem; } ldiv_t; +typedef struct { long long quot; long long rem; } lldiv_t; + void *malloc(size_t size); void *calloc(size_t nmemb, size_t size); void *realloc(void *ptr, size_t size); void free(void *ptr); +void *aligned_alloc(size_t alignment, size_t size); +int posix_memalign(void **memptr, size_t alignment, size_t size); -void exit(int status); -void _exit(int status); -void _Exit(int status); -void abort(void); +__attribute__((__noreturn__)) void exit(int status); +__attribute__((__noreturn__)) void _exit(int status); +__attribute__((__noreturn__)) void _Exit(int status); +__attribute__((__noreturn__)) void abort(void); int atexit(void (*func)(void)); int system(const char *command); double atof(const char *s); int atoi(const char *s); long atol(const char *s); +long long atoll(const char *s); long strtol(const char *s, char **endptr, int base); unsigned long strtoul(const char *s, char **endptr, int base); long long strtoll(const char *s, char **endptr, int base); unsigned long long strtoull(const char *s, char **endptr, int base); +float strtof(const char *s, char **endptr); double strtod(const char *s, char **endptr); +long double strtold(const char *s, char **endptr); +float strtof_l(const char *s, char **endptr, locale_t loc); +double strtod_l(const char *s, char **endptr, locale_t loc); +long double strtold_l(const char *s, char **endptr, locale_t loc); char *getenv(const char *name); int setenv(const char *name, const char *value, int overwrite); @@ -37,8 +58,22 @@ void *bsearch(const void *key, const void *base, size_t nmemb, size_t size, int int abs(int j); long labs(long j); +long long llabs(long long j); +div_t div(int numer, int denom); +ldiv_t ldiv(long numer, long denom); +lldiv_t lldiv(long long numer, long long denom); + +int mblen(const char *s, size_t n); +int mbtowc(wchar_t *pwc, const char *s, size_t n); +int wctomb(char *s, wchar_t wc); +size_t mbstowcs(wchar_t *dest, const char *src, size_t n); +size_t wcstombs(char *dest, const wchar_t *src, size_t n); int rand(void); void srand(unsigned int seed); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/string.h b/userland/libc/include/string.h index 4c9e10be5c0..c909fa07de8 100644 --- a/userland/libc/include/string.h +++ b/userland/libc/include/string.h @@ -2,6 +2,11 @@ #define _STRING_H #include +#include + +#ifdef __cplusplus +extern "C" { +#endif void *memcpy(void *dest, const void *src, size_t n); void *memmove(void *dest, const void *src, size_t n); @@ -16,11 +21,16 @@ char *strcat(char *dest, const char *src); char *strncat(char *dest, const char *src, size_t n); int strcmp(const char *s1, const char *s2); int strncmp(const char *s1, const char *s2, size_t n); +int strcoll(const char *s1, const char *s2); +int strcoll_l(const char *s1, const char *s2, locale_t loc); +size_t strxfrm(char *dest, const char *src, size_t n); +size_t strxfrm_l(char *dest, const char *src, size_t n, locale_t loc); char *strchr(const char *s, int c); char *strrchr(const char *s, int c); char *strstr(const char *haystack, const char *needle); char *strdup(const char *s); char *strerror(int errnum); +int strerror_r(int errnum, char *buf, size_t buflen); size_t strspn(const char *s, const char *accept); size_t strcspn(const char *s, const char *reject); char *strpbrk(const char *s, const char *accept); @@ -29,4 +39,8 @@ char *strtok(char *str, const char *delim); int strcasecmp(const char *s1, const char *s2); int strncasecmp(const char *s1, const char *s2, size_t n); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/sys/mman.h b/userland/libc/include/sys/mman.h index 7c5d0f10d2d..edda25e8ff4 100644 --- a/userland/libc/include/sys/mman.h +++ b/userland/libc/include/sys/mman.h @@ -3,6 +3,10 @@ #include +#ifdef __cplusplus +extern "C" { +#endif + #define PROT_NONE 0x0 #define PROT_READ 0x1 #define PROT_WRITE 0x2 @@ -20,4 +24,8 @@ void *mmap(void *addr, size_t length, int prot, int flags, int fd, long offset); int munmap(void *addr, size_t length); int mprotect(void *addr, size_t len, int prot); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/sys/random.h b/userland/libc/include/sys/random.h new file mode 100644 index 00000000000..e5de1c93098 --- /dev/null +++ b/userland/libc/include/sys/random.h @@ -0,0 +1,16 @@ +#ifndef _SYS_RANDOM_H +#define _SYS_RANDOM_H + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +int getentropy(void *buffer, size_t length); + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/userland/libc/include/sys/socket.h b/userland/libc/include/sys/socket.h index f1921150dea..a6466c1744c 100644 --- a/userland/libc/include/sys/socket.h +++ b/userland/libc/include/sys/socket.h @@ -4,6 +4,10 @@ #include #include +#ifdef __cplusplus +extern "C" { +#endif + typedef unsigned int socklen_t; #define AF_UNSPEC 0 @@ -55,4 +59,8 @@ int getsockopt(int fd, int level, int optname, void *optval, socklen_t *optlen); int getpeername(int fd, struct sockaddr *addr, socklen_t *addrlen); int getsockname(int fd, struct sockaddr *addr, socklen_t *addrlen); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/sys/stat.h b/userland/libc/include/sys/stat.h index bf24651b0b5..4a9f1c4793a 100644 --- a/userland/libc/include/sys/stat.h +++ b/userland/libc/include/sys/stat.h @@ -4,6 +4,10 @@ #include #include +#ifdef __cplusplus +extern "C" { +#endif + struct stat { dev_t st_dev; ino_t st_ino; @@ -63,4 +67,8 @@ int mkdir(const char *path, mode_t mode); int chmod(const char *path, mode_t mode); mode_t umask(mode_t mask); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/sys/time.h b/userland/libc/include/sys/time.h index 7787e3f03ca..2dec877691f 100644 --- a/userland/libc/include/sys/time.h +++ b/userland/libc/include/sys/time.h @@ -3,6 +3,10 @@ #include +#ifdef __cplusplus +extern "C" { +#endif + struct timeval { long tv_sec; long tv_usec; @@ -15,4 +19,8 @@ struct timezone { int gettimeofday(struct timeval *tv, struct timezone *tz); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/sys/wait.h b/userland/libc/include/sys/wait.h index 0a9ea00c67d..06faf36813f 100644 --- a/userland/libc/include/sys/wait.h +++ b/userland/libc/include/sys/wait.h @@ -3,6 +3,10 @@ #include +#ifdef __cplusplus +extern "C" { +#endif + #define WNOHANG 1 #define WUNTRACED 2 @@ -14,4 +18,8 @@ pid_t wait(int *status); pid_t waitpid(pid_t pid, int *status, int options); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/time.h b/userland/libc/include/time.h index 8d67fd836a0..4fb11f9e61a 100644 --- a/userland/libc/include/time.h +++ b/userland/libc/include/time.h @@ -2,10 +2,16 @@ #define _TIME_H #include +#include + +#ifdef __cplusplus +extern "C" { +#endif typedef long time_t; typedef long clock_t; typedef long suseconds_t; +typedef int clockid_t; struct tm { int tm_sec; @@ -31,7 +37,7 @@ struct timespec { time_t time(time_t *t); clock_t clock(void); -int clock_gettime(int clk_id, struct timespec *tp); +int clock_gettime(clockid_t clk_id, struct timespec *tp); int nanosleep(const struct timespec *req, struct timespec *rem); struct tm *localtime(const time_t *timer); @@ -41,8 +47,13 @@ struct tm *gmtime_r(const time_t *timer, struct tm *result); time_t mktime(struct tm *tm); double difftime(time_t t1, time_t t0); size_t strftime(char *s, size_t max, const char *fmt, const struct tm *tm); +size_t strftime_l(char *s, size_t max, const char *fmt, const struct tm *tm, locale_t loc); unsigned int sleep(unsigned int seconds); int usleep(unsigned int usec); +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/unistd.h b/userland/libc/include/unistd.h index 4b5c401a7ff..fca9c36896a 100644 --- a/userland/libc/include/unistd.h +++ b/userland/libc/include/unistd.h @@ -4,6 +4,10 @@ #include #include +#ifdef __cplusplus +extern "C" { +#endif + #define STDIN_FILENO 0 #define STDOUT_FILENO 1 #define STDERR_FILENO 2 @@ -46,4 +50,8 @@ long sysconf(int name); #define W_OK 2 #define X_OK 1 +#ifdef __cplusplus +} +#endif + #endif diff --git a/userland/libc/include/wchar.h b/userland/libc/include/wchar.h index cc063ff0826..a2210f148d2 100644 --- a/userland/libc/include/wchar.h +++ b/userland/libc/include/wchar.h @@ -2,10 +2,74 @@ #define _WCHAR_H #include +#include +#include +#include +#include +#include +#include -#ifndef _WCHAR_T_DEFINED -#define _WCHAR_T_DEFINED -/* wchar_t is a builtin type in our compiler */ +#ifdef __cplusplus +extern "C" { +#endif + +struct tm; + +size_t wcslen(const wchar_t *s); +wchar_t *wcscpy(wchar_t *dest, const wchar_t *src); +wchar_t *wcsncpy(wchar_t *dest, const wchar_t *src, size_t n); +wchar_t *wcscat(wchar_t *dest, const wchar_t *src); +wchar_t *wcsncat(wchar_t *dest, const wchar_t *src, size_t n); +int wcscmp(const wchar_t *a, const wchar_t *b); +int wcsncmp(const wchar_t *a, const wchar_t *b, size_t n); +int wcscoll(const wchar_t *a, const wchar_t *b); +int wcscoll_l(const wchar_t *a, const wchar_t *b, locale_t loc); +size_t wcsxfrm(wchar_t *dest, const wchar_t *src, size_t n); +size_t wcsxfrm_l(wchar_t *dest, const wchar_t *src, size_t n, locale_t loc); +wchar_t *wcschr(const wchar_t *s, wchar_t c); +wchar_t *wcsrchr(const wchar_t *s, wchar_t c); +size_t wcsspn(const wchar_t *s, const wchar_t *accept); +size_t wcscspn(const wchar_t *s, const wchar_t *reject); +wchar_t *wcspbrk(const wchar_t *s, const wchar_t *accept); +wchar_t *wcsstr(const wchar_t *haystack, const wchar_t *needle); +wchar_t *wcstok(wchar_t *s, const wchar_t *delim, wchar_t **save); + +wchar_t *wmemchr(const wchar_t *s, wchar_t c, size_t n); +int wmemcmp(const wchar_t *a, const wchar_t *b, size_t n); +wchar_t *wmemcpy(wchar_t *dest, const wchar_t *src, size_t n); +wchar_t *wmemmove(wchar_t *dest, const wchar_t *src, size_t n); +wchar_t *wmemset(wchar_t *dest, wchar_t c, size_t n); + +long wcstol(const wchar_t *s, wchar_t **endptr, int base); +unsigned long wcstoul(const wchar_t *s, wchar_t **endptr, int base); +long long wcstoll(const wchar_t *s, wchar_t **endptr, int base); +unsigned long long wcstoull(const wchar_t *s, wchar_t **endptr, int base); +float wcstof(const wchar_t *s, wchar_t **endptr); +double wcstod(const wchar_t *s, wchar_t **endptr); +long double wcstold(const wchar_t *s, wchar_t **endptr); + +wint_t btowc(int c); +int wctob(wint_t c); +int mbsinit(const mbstate_t *ps); +size_t mbrlen(const char *s, size_t n, mbstate_t *ps); +size_t mbrtowc(wchar_t *pwc, const char *s, size_t n, mbstate_t *ps); +size_t wcrtomb(char *s, wchar_t wc, mbstate_t *ps); +size_t mbsrtowcs(wchar_t *dest, const char **src, size_t len, mbstate_t *ps); +size_t wcsrtombs(char *dest, const wchar_t **src, size_t len, mbstate_t *ps); +size_t mbsnrtowcs(wchar_t *dest, const char **src, size_t nms, size_t len, mbstate_t *ps); +size_t wcsnrtombs(char *dest, const wchar_t **src, size_t nwc, size_t len, mbstate_t *ps); + +wint_t fgetwc(FILE *stream); +wint_t getwc(FILE *stream); +wint_t ungetwc(wint_t wc, FILE *stream); +wint_t fputwc(wchar_t wc, FILE *stream); +wint_t putwc(wchar_t wc, FILE *stream); + +int swprintf(wchar_t *s, size_t n, const wchar_t *fmt, ...); +int vswprintf(wchar_t *s, size_t n, const wchar_t *fmt, va_list ap); + +#ifdef __cplusplus +} #endif #endif diff --git a/userland/libc/include/wctype.h b/userland/libc/include/wctype.h new file mode 100644 index 00000000000..dd4088e59ae --- /dev/null +++ b/userland/libc/include/wctype.h @@ -0,0 +1,56 @@ +#ifndef _WCTYPE_H +#define _WCTYPE_H + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +typedef unsigned long wctype_t; +typedef unsigned long wctrans_t; + +int iswalnum(wint_t wc); +int iswalpha(wint_t wc); +int iswblank(wint_t wc); +int iswcntrl(wint_t wc); +int iswdigit(wint_t wc); +int iswgraph(wint_t wc); +int iswlower(wint_t wc); +int iswprint(wint_t wc); +int iswpunct(wint_t wc); +int iswspace(wint_t wc); +int iswupper(wint_t wc); +int iswxdigit(wint_t wc); +int iswctype(wint_t wc, wctype_t desc); +wctype_t wctype(const char *name); +wint_t towlower(wint_t wc); +wint_t towupper(wint_t wc); +wint_t towctrans(wint_t wc, wctrans_t desc); +wctrans_t wctrans(const char *name); + +int iswalnum_l(wint_t wc, locale_t loc); +int iswalpha_l(wint_t wc, locale_t loc); +int iswblank_l(wint_t wc, locale_t loc); +int iswcntrl_l(wint_t wc, locale_t loc); +int iswdigit_l(wint_t wc, locale_t loc); +int iswgraph_l(wint_t wc, locale_t loc); +int iswlower_l(wint_t wc, locale_t loc); +int iswprint_l(wint_t wc, locale_t loc); +int iswpunct_l(wint_t wc, locale_t loc); +int iswspace_l(wint_t wc, locale_t loc); +int iswupper_l(wint_t wc, locale_t loc); +int iswxdigit_l(wint_t wc, locale_t loc); +int iswctype_l(wint_t wc, wctype_t desc, locale_t loc); +wctype_t wctype_l(const char *name, locale_t loc); +wint_t towlower_l(wint_t wc, locale_t loc); +wint_t towupper_l(wint_t wc, locale_t loc); +wint_t towctrans_l(wint_t wc, wctrans_t desc, locale_t loc); +wctrans_t wctrans_l(const char *name, locale_t loc); + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/userland/libc/src/arch/aarch64.rs b/userland/libc/src/arch/aarch64.rs index 011358f11ae..151ea40adfd 100644 --- a/userland/libc/src/arch/aarch64.rs +++ b/userland/libc/src/arch/aarch64.rs @@ -117,3 +117,83 @@ pub(crate) fn sqrt_f32(x: f32) -> f32 { unsafe { core::arch::asm!("fsqrt {0:s}, {0:s}", inout(vreg) x => result, options(pure, nomem, nostack)) }; result } + +/// `wchar_t`, which is `unsigned int` here. +pub(crate) type WChar = u32; + +// What the `long double` readers below widen, named as C names them, since +// this module is also compiled on its own (`toyos-libc-copies`). +unsafe extern "C" { + fn strtod(s: *const u8, endptr: *mut *mut u8) -> f64; + fn wcstod(s: *const WChar, endptr: *mut *mut WChar) -> f64; +} + +/// An IEEE binary128's bits, returned in `x0` and `x1`. +#[repr(C)] +struct Quad { + lo: u64, + hi: u64, +} + +/// `x` as binary128, `long double` here: every `double` is exactly one. +extern "C" fn quad(x: f64) -> Quad { + let bits = x.to_bits(); + let exponent = (bits >> 52) & 0x7ff; + let fraction = u128::from(bits & ((1 << 52) - 1)); + let (exponent, fraction) = match exponent { + 0 if fraction == 0 => (0, 0), + // A subnormal double is a normal quad: its leading one at bit `top`. + 0 => { + let top = 127 - u128::from(fraction.leading_zeros()); + (top + 16383 - 1074, (fraction << (112 - top)) & ((1 << 112) - 1)) + } + 0x7ff => (0x7fff, fraction << 60), + _ => (u128::from(exponent) + 16383 - 1023, fraction << 60), + }; + let q = u128::from(bits >> 63) << 127 | exponent << 112 | fraction; + Quad { lo: q as u64, hi: (q >> 64) as u64 } +} + +/// `strtod`'s number as `long double`, in `q0`: read to `double`'s precision +/// and widened, which is exact. +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn strtold() { + core::arch::naked_asm!( + "stp x29, x30, [sp, #-16]!", + "mov x29, sp", + "bl {strtod}", + "bl {quad}", + "fmov d0, x0", + "mov v0.d[1], x1", + "ldp x29, x30, [sp], #16", + "ret", + strtod = sym strtod, + quad = sym quad, + ); +} + +/// `strtold`, in the one locale there is. +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn strtold_l() { + core::arch::naked_asm!("b {strtold}", strtold = sym strtold); +} + +/// `wcstod`'s number as `long double`, as [`strtold`] widens `strtod`'s. +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn wcstold() { + core::arch::naked_asm!( + "stp x29, x30, [sp, #-16]!", + "mov x29, sp", + "bl {wcstod}", + "bl {quad}", + "fmov d0, x0", + "mov v0.d[1], x1", + "ldp x29, x30, [sp], #16", + "ret", + wcstod = sym wcstod, + quad = sym quad, + ); +} diff --git a/userland/libc/src/arch/x86_64.rs b/userland/libc/src/arch/x86_64.rs index f76430eca0e..f20cc8af4a2 100644 --- a/userland/libc/src/arch/x86_64.rs +++ b/userland/libc/src/arch/x86_64.rs @@ -74,3 +74,51 @@ pub(crate) fn sqrt_f32(x: f32) -> f32 { unsafe { core::arch::asm!("sqrtss {0}, {0}", inout(xmm_reg) x => result, options(pure, nomem, nostack)) }; result } + +/// `wchar_t`, which is `int` here. +pub(crate) type WChar = i32; + +// What the `long double` readers below widen, named as C names them, since +// this module is also compiled on its own (`toyos-libc-copies`). +unsafe extern "C" { + fn strtod(s: *const u8, endptr: *mut *mut u8) -> f64; + fn wcstod(s: *const WChar, endptr: *mut *mut WChar) -> f64; +} + +/// `strtod`'s number as `long double`, x87 extended precision in `st0`: read +/// to `double`'s precision and widened, which is exact. +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn strtold() { + core::arch::naked_asm!( + "sub rsp, 8", + "call {strtod}", + "movsd qword ptr [rsp], xmm0", + "fld qword ptr [rsp]", + "add rsp, 8", + "ret", + strtod = sym strtod, + ); +} + +/// `strtold`, in the one locale there is. +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn strtold_l() { + core::arch::naked_asm!("jmp {strtold}", strtold = sym strtold); +} + +/// `wcstod`'s number as `long double`, as [`strtold`] widens `strtod`'s. +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn wcstold() { + core::arch::naked_asm!( + "sub rsp, 8", + "call {wcstod}", + "movsd qword ptr [rsp], xmm0", + "fld qword ptr [rsp]", + "add rsp, 8", + "ret", + wcstod = sym wcstod, + ); +} diff --git a/userland/libc/src/ctype.rs b/userland/libc/src/ctype.rs index d4656526c69..0ebfedb620b 100644 --- a/userland/libc/src/ctype.rs +++ b/userland/libc/src/ctype.rs @@ -49,4 +49,23 @@ pub extern "C" fn toupper(c: i32) -> i32 { #[no_mangle] pub extern "C" fn tolower(c: i32) -> i32 { if isupper(c) != 0 { c + 32 } else { c } -} \ No newline at end of file +} +#[no_mangle] +pub extern "C" fn isblank(c: i32) -> i32 { + (c == b' ' as i32 || c == b'\t' as i32) as i32 +} + +#[no_mangle] +pub extern "C" fn iscntrl(c: i32) -> i32 { + ((0..0x20).contains(&c) || c == 0x7f) as i32 +} + +#[no_mangle] +pub extern "C" fn isgraph(c: i32) -> i32 { + (c > 0x20 && c <= 0x7e) as i32 +} + +#[no_mangle] +pub extern "C" fn ispunct(c: i32) -> i32 { + (isgraph(c) != 0 && isalnum(c) == 0) as i32 +} diff --git a/userland/libc/src/errno.rs b/userland/libc/src/errno.rs index 04a9ada4f77..250441f3cba 100644 --- a/userland/libc/src/errno.rs +++ b/userland/libc/src/errno.rs @@ -16,3 +16,27 @@ pub extern "C" fn __errno_location() -> *mut i32 { pub(crate) fn set(code: i32) { ERRNO.set(code); } + +pub(crate) const EPERM: i32 = 1; +pub(crate) const ENOENT: i32 = 2; +pub(crate) const ESRCH: i32 = 3; +pub(crate) const EIO: i32 = 5; +pub(crate) const EBADF: i32 = 9; +pub(crate) const ECHILD: i32 = 10; +pub(crate) const EAGAIN: i32 = 11; +pub(crate) const ENOMEM: i32 = 12; +pub(crate) const EACCES: i32 = 13; +pub(crate) const EBUSY: i32 = 16; +pub(crate) const EEXIST: i32 = 17; +pub(crate) const EINVAL: i32 = 22; +pub(crate) const EPIPE: i32 = 32; +pub(crate) const ERANGE: i32 = 34; +pub(crate) const EDEADLK: i32 = 35; +pub(crate) const ENOSYS: i32 = 38; +pub(crate) const EILSEQ: i32 = 84; +pub(crate) const EAFNOSUPPORT: i32 = 97; +pub(crate) const EADDRINUSE: i32 = 98; +pub(crate) const ECONNRESET: i32 = 104; +pub(crate) const ENOTCONN: i32 = 107; +pub(crate) const ETIMEDOUT: i32 = 110; +pub(crate) const ECONNREFUSED: i32 = 111; diff --git a/userland/libc/src/lib.rs b/userland/libc/src/lib.rs index 1052461cf53..c3943f7a935 100644 --- a/userland/libc/src/lib.rs +++ b/userland/libc/src/lib.rs @@ -1,5 +1,6 @@ #![no_std] #![feature(thread_local)] +#![cfg_attr(not(feature = "std-runtime"), feature(linkage))] extern crate alloc; @@ -7,6 +8,7 @@ mod arch; mod ctype; mod errno; mod link; +mod locale; mod math; mod memory; mod misc; @@ -16,7 +18,10 @@ mod pthread; mod socket; mod stdio; mod string; +mod strtonum; mod time; +mod utf8; +mod wchar; // C runtime: the entry `arch::_start` calls, panic handler, and global allocator. // Only for pure C programs (no Rust std). When linked into a Rust program @@ -88,14 +93,25 @@ mod runtime { toyos_abi::syscall::exit(134) // SIGABRT-like } - // Unwinding stubs — core references these symbols via .eh_frame, but with - // panic=abort the unwinding path is never taken. Provide no-op/abort stubs - // to satisfy the linker. + /// The personality `core`'s unwind tables name. This library unwinds + /// nothing of its own, and an exception that reaches one of its Rust + /// frames ends the program here, loudly. #[unsafe(no_mangle)] - extern "C" fn rust_eh_personality() {} + extern "C" fn rust_eh_personality() -> ! { + use core::fmt::Write; + let _ = write!(Stderr, "libc: an exception unwound into Rust code, which cannot catch it\n"); + toyos_abi::syscall::exit(134) + } + /// What the precompiled `alloc`'s landing pads name, so a program with no + /// unwinder links. Weak: libunwind's `UnwindLevel1.o` defines it strongly + /// beside `_Unwind_RaiseException`, the only way an unwind starts, so this + /// one stays the program's only while nothing can unwind. #[unsafe(no_mangle)] + #[linkage = "weak"] extern "C" fn _Unwind_Resume() -> ! { + use core::fmt::Write; + let _ = write!(Stderr, "libc: _Unwind_Resume with no unwinder linked\n"); toyos_abi::syscall::exit(134) } diff --git a/userland/libc/src/locale.rs b/userland/libc/src/locale.rs new file mode 100644 index 00000000000..623fc268dd9 --- /dev/null +++ b/userland/libc/src/locale.rs @@ -0,0 +1,168 @@ +//! The one locale: C's. Every name that means it (`C`, `POSIX`, `C.UTF-8` +//! and the empty default) names it, any other is refused (`ENOENT`), its +//! multibyte encoding is UTF-8 (`wchar.rs`), and every `_l` function answers +//! as its locale-free twin does. + +use core::cell::Cell; +use core::ffi::{c_char, CStr}; +use core::ptr; + +use crate::errno::{self, EINVAL, ENOENT}; + +/// `locale_t` points at this, and only ever at the one there is. +#[repr(C)] +pub struct Locale { + _only: u8, +} + +static C_LOCALE: Locale = Locale { _only: 0 }; + +/// `LC_GLOBAL_LOCALE`, `(locale_t)-1`. +const GLOBAL: *mut Locale = usize::MAX as *mut Locale; + +const LC_ALL: i32 = 6; +const LC_ALL_MASK: i32 = (1 << LC_ALL) - 1; + +#[thread_local] +static CURRENT: Cell<*mut Locale> = Cell::new(GLOBAL); + +fn c_locale() -> *mut Locale { + ptr::addr_of!(C_LOCALE).cast_mut() +} + +unsafe fn names_c(name: *const u8) -> bool { + matches!(unsafe { CStr::from_ptr(name.cast()) }.to_bytes(), b"" | b"C" | b"POSIX" | b"C.UTF-8") +} + +#[no_mangle] +pub unsafe extern "C" fn setlocale(category: i32, name: *const u8) -> *const u8 { + if !(0..=LC_ALL).contains(&category) || (!name.is_null() && !unsafe { names_c(name) }) { + return ptr::null(); + } + c"C".as_ptr().cast() +} + +#[no_mangle] +pub unsafe extern "C" fn newlocale(mask: i32, name: *const u8, _base: *mut Locale) -> *mut Locale { + if mask & !LC_ALL_MASK != 0 || name.is_null() { + errno::set(EINVAL); + return ptr::null_mut(); + } + if !unsafe { names_c(name) } { + errno::set(ENOENT); + return ptr::null_mut(); + } + c_locale() +} + +#[no_mangle] +pub unsafe extern "C" fn duplocale(_loc: *mut Locale) -> *mut Locale { + c_locale() +} + +#[no_mangle] +pub unsafe extern "C" fn freelocale(_loc: *mut Locale) {} + +#[no_mangle] +pub unsafe extern "C" fn uselocale(new: *mut Locale) -> *mut Locale { + let old = CURRENT.get(); + if !new.is_null() { + CURRENT.set(new); + } + old +} + +/// `struct lconv` as `include/locale.h` lays it out. +#[repr(C)] +pub struct Lconv { + strings: [*const c_char; 10], + chars: [c_char; 14], +} + +struct CLconv(Lconv); + +// SAFETY: every pointer is to a string literal, and nothing writes through it. +unsafe impl Sync for CLconv {} + +/// The C locale's: a decimal point, and nothing else there is to say. +static C_LCONV: CLconv = CLconv(Lconv { + strings: [c".".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr(), c"".as_ptr()], + chars: [c_char::MAX; 14], +}); + +#[no_mangle] +pub extern "C" fn localeconv() -> *mut Lconv { + ptr::addr_of!(C_LCONV.0).cast_mut() +} + +macro_rules! same_in_every_locale { + ($($name:ident => $plain:path;)*) => {$( + #[no_mangle] + pub extern "C" fn $name(c: i32, _loc: *mut Locale) -> i32 { + $plain(c) + } + )*}; +} + +same_in_every_locale! { + isalnum_l => crate::ctype::isalnum; + isalpha_l => crate::ctype::isalpha; + isblank_l => crate::ctype::isblank; + iscntrl_l => crate::ctype::iscntrl; + isdigit_l => crate::ctype::isdigit; + isgraph_l => crate::ctype::isgraph; + islower_l => crate::ctype::islower; + isprint_l => crate::ctype::isprint; + ispunct_l => crate::ctype::ispunct; + isspace_l => crate::ctype::isspace; + isupper_l => crate::ctype::isupper; + isxdigit_l => crate::ctype::isxdigit; + toupper_l => crate::ctype::toupper; + tolower_l => crate::ctype::tolower; +} + +/// C's collation is byte order. +#[no_mangle] +pub unsafe extern "C" fn strcoll(a: *const u8, b: *const u8) -> i32 { + unsafe { crate::string::strcmp(a, b) } +} + +#[no_mangle] +pub unsafe extern "C" fn strcoll_l(a: *const u8, b: *const u8, _loc: *mut Locale) -> i32 { + unsafe { strcoll(a, b) } +} + +#[no_mangle] +pub unsafe extern "C" fn strxfrm(dst: *mut u8, src: *const u8, n: usize) -> usize { + let len = unsafe { crate::string::strlen(src) }; + if len < n { + unsafe { ptr::copy_nonoverlapping(src, dst, len + 1) }; + } + len +} + +#[no_mangle] +pub unsafe extern "C" fn strxfrm_l(dst: *mut u8, src: *const u8, n: usize, _loc: *mut Locale) -> usize { + unsafe { strxfrm(dst, src, n) } +} + +#[no_mangle] +pub unsafe extern "C" fn strftime_l( + s: *mut u8, + max: usize, + fmt: *const u8, + tm: *const crate::time::Tm, + _loc: *mut Locale, +) -> usize { + unsafe { crate::time::strftime(s, max, fmt, tm) } +} + +#[no_mangle] +pub unsafe extern "C" fn strtod_l(s: *const u8, endptr: *mut *mut u8, _loc: *mut Locale) -> f64 { + unsafe { crate::misc::strtod(s, endptr) } +} + +#[no_mangle] +pub unsafe extern "C" fn strtof_l(s: *const u8, endptr: *mut *mut u8, _loc: *mut Locale) -> f32 { + unsafe { crate::misc::strtof(s, endptr) } +} diff --git a/userland/libc/src/math.rs b/userland/libc/src/math.rs index 2d97c989566..537896ae224 100644 --- a/userland/libc/src/math.rs +++ b/userland/libc/src/math.rs @@ -260,12 +260,3 @@ pub unsafe extern "C" fn frexp(x: f64, exp: *mut i32) -> f64 { *exp = biased - 1022; // exponent such that x = mantissa * 2^exp, mantissa in [0.5, 1.0) f64::from_bits((bits & 0x800FFFFFFFFFFFFF) | 0x3FE0000000000000) } - -#[no_mangle] -pub extern "C" fn isnan(x: f64) -> i32 { x.is_nan() as i32 } - -#[no_mangle] -pub extern "C" fn isinf(x: f64) -> i32 { x.is_infinite() as i32 } - -#[no_mangle] -pub extern "C" fn isfinite(x: f64) -> i32 { x.is_finite() as i32 } \ No newline at end of file diff --git a/userland/libc/src/memory.rs b/userland/libc/src/memory.rs index f3e0b101f43..906b4094a39 100644 --- a/userland/libc/src/memory.rs +++ b/userland/libc/src/memory.rs @@ -14,41 +14,66 @@ extern "C" { #[cfg(not(feature = "std-runtime"))] mod backend { use core::alloc::Layout; + use core::ptr; - const HEADER: usize = 16; // 16 for alignment - const ALIGN: usize = 16; + /// The alignment every block has at least, and the size of the two words + /// in front of it: its size and its alignment. + const MIN_ALIGN: usize = 16; - pub unsafe fn alloc(size: usize) -> *mut u8 { - let total = HEADER + size; - let layout = unsafe { Layout::from_size_align_unchecked(total, ALIGN) }; + fn layout(size: usize, align: usize) -> Option { + Layout::from_size_align(align.checked_add(size)?, align).ok() + } + + unsafe fn header(ptr: *mut u8) -> (usize, usize) { + unsafe { (*(ptr.sub(16) as *const usize), *(ptr.sub(8) as *const usize)) } + } + + /// `size` bytes aligned to `align`, a power of two. + pub unsafe fn alloc(size: usize, align: usize) -> *mut u8 { + let align = align.max(MIN_ALIGN); + let Some(layout) = layout(size, align) else { return ptr::null_mut() }; let raw = unsafe { alloc::alloc::alloc(layout) }; if raw.is_null() { return raw; } - unsafe { *(raw as *mut usize) = size; } - unsafe { raw.add(HEADER) } + unsafe { + let ptr = raw.add(align); + *(ptr.sub(16) as *mut usize) = size; + *(ptr.sub(8) as *mut usize) = align; + ptr + } } pub unsafe fn dealloc(ptr: *mut u8) { - let raw = unsafe { ptr.sub(HEADER) }; - let size = unsafe { *(raw as *const usize) }; - let total = HEADER + size; - let layout = unsafe { Layout::from_size_align_unchecked(total, ALIGN) }; - unsafe { alloc::alloc::dealloc(raw, layout) }; + let (size, align) = unsafe { header(ptr) }; + let layout = layout(size, align).expect("a block's header is the layout it was allocated with"); + unsafe { alloc::alloc::dealloc(ptr.sub(align), layout) }; } pub unsafe fn realloc(ptr: *mut u8, new_size: usize) -> *mut u8 { - let raw = unsafe { ptr.sub(HEADER) }; - let old_size = unsafe { *(raw as *const usize) }; - let old_total = HEADER + old_size; - let new_total = HEADER + new_size; - let layout = unsafe { Layout::from_size_align_unchecked(old_total, ALIGN) }; - let new_raw = unsafe { alloc::alloc::realloc(raw, layout, new_total) }; - if new_raw.is_null() { - return new_raw; + let (size, align) = unsafe { header(ptr) }; + if align > MIN_ALIGN { + let new = unsafe { alloc(new_size, align) }; + if !new.is_null() { + unsafe { + ptr::copy_nonoverlapping(ptr, new, size.min(new_size)); + dealloc(ptr); + } + } + return new; + } + let Some(new_total) = MIN_ALIGN.checked_add(new_size) else { return ptr::null_mut() }; + let layout = layout(size, MIN_ALIGN).expect("a block's header is the layout it was allocated with"); + let raw = unsafe { alloc::alloc::realloc(ptr.sub(MIN_ALIGN), layout, new_total) }; + if raw.is_null() { + return raw; + } + unsafe { + let ptr = raw.add(MIN_ALIGN); + *(ptr.sub(16) as *mut usize) = new_size; + *(ptr.sub(8) as *mut usize) = MIN_ALIGN; + ptr } - unsafe { *(new_raw as *mut usize) = new_size; } - unsafe { new_raw.add(HEADER) } } } @@ -60,7 +85,35 @@ pub unsafe extern "C" fn malloc(size: usize) -> *mut u8 { if size == 0 { return ptr::null_mut(); } - unsafe { backend::alloc(size) } + unsafe { backend::alloc(size, 16) } +} + +#[cfg(not(feature = "std-runtime"))] +#[no_mangle] +pub unsafe extern "C" fn aligned_alloc(align: usize, size: usize) -> *mut u8 { + if !align.is_power_of_two() { + crate::errno::set(crate::errno::EINVAL); + return ptr::null_mut(); + } + let p = unsafe { backend::alloc(size, align) }; + if p.is_null() { + crate::errno::set(crate::errno::ENOMEM); + } + p +} + +#[cfg(not(feature = "std-runtime"))] +#[no_mangle] +pub unsafe extern "C" fn posix_memalign(out: *mut *mut u8, align: usize, size: usize) -> i32 { + if !align.is_power_of_two() || align % core::mem::size_of::() != 0 { + return crate::errno::EINVAL; + } + let p = unsafe { backend::alloc(size, align) }; + if p.is_null() { + return crate::errno::ENOMEM; + } + unsafe { *out = p }; + 0 } #[cfg(not(feature = "std-runtime"))] diff --git a/userland/libc/src/misc.rs b/userland/libc/src/misc.rs index 4443d6cd7c8..13ad9912ab7 100644 --- a/userland/libc/src/misc.rs +++ b/userland/libc/src/misc.rs @@ -1,12 +1,13 @@ // Miscellaneous POSIX/C functions: environment, process, signals, sysconf. +use alloc::vec::Vec; use core::ptr; use core::sync::atomic::{AtomicU32, Ordering}; use toyos_abi::syscall; -const ENOSYS: i32 = 38; -const ECHILD: i32 = 10; +use crate::errno::{ECHILD, ENOSYS}; +use crate::strtonum; // Environment variables @@ -123,38 +124,52 @@ pub unsafe extern "C" fn waitpid(_pid: i32, _status: *mut i32, _options: i32) -> // Exit / abort / atexit -const MAX_ATEXIT: usize = 32; -static mut ATEXIT_FNS: [Option; MAX_ATEXIT] = [None; MAX_ATEXIT]; -static mut ATEXIT_COUNT: usize = 0; +/// A handler `exit` runs: `atexit`'s takes nothing, `__cxa_atexit`'s its object. +enum AtExit { + Plain(unsafe extern "C" fn()), + WithArg(unsafe extern "C" fn(*mut u8), *mut u8), +} + +// SAFETY: the argument is the registrant's to hand to its own handler. +unsafe impl Send for AtExit {} + +/// Every handler registered, in order; `exit` runs the last first. +static AT_EXIT: crate::pthread::Lock> = crate::pthread::Lock::new(Vec::new()); + +/// What names this image to `__cxa_atexit`: the address is the identity. +#[no_mangle] +pub static __dso_handle: usize = 0; #[no_mangle] pub unsafe extern "C" fn atexit(func: unsafe extern "C" fn()) -> i32 { - let count = ptr::addr_of!(ATEXIT_COUNT).read(); - if count >= MAX_ATEXIT { - return -1; - } - let fns = ptr::addr_of_mut!(ATEXIT_FNS).as_mut().unwrap(); - fns[count] = Some(func); - ptr::addr_of_mut!(ATEXIT_COUNT).write(count + 1); + AT_EXIT.lock().push(AtExit::Plain(func)); + 0 +} + +#[no_mangle] +pub unsafe extern "C" fn __cxa_atexit(func: unsafe extern "C" fn(*mut u8), arg: *mut u8, _dso: *mut u8) -> i32 { + AT_EXIT.lock().push(AtExit::WithArg(func, arg)); 0 } +/// Run the handlers, the last registered first, including any a handler +/// registers; none runs with the list locked. unsafe fn run_atexit() { - // Run in reverse order - let fns = ptr::addr_of_mut!(ATEXIT_FNS).as_mut().unwrap(); - let mut count = ptr::addr_of!(ATEXIT_COUNT).read(); - while count > 0 { - count -= 1; - if let Some(f) = fns[count] { - f(); + loop { + let Some(handler) = AT_EXIT.lock().pop() else { return }; + match handler { + AtExit::Plain(f) => unsafe { f() }, + AtExit::WithArg(f, arg) => unsafe { f(arg) }, } } - ptr::addr_of_mut!(ATEXIT_COUNT).write(0); } #[no_mangle] pub unsafe extern "C" fn exit(status: i32) -> ! { - run_atexit(); + unsafe { + crate::pthread::run_thread_dtors(); + run_atexit(); + } #[cfg(not(feature = "std-runtime"))] crate::runtime::fini(); super::stdio::fflush(ptr::null_mut()); @@ -296,7 +311,22 @@ pub unsafe extern "C" fn bsearch( ptr::null_mut() } -// String-to-number conversions +// String-to-number conversions: the grammar and the rounding are `strtonum`'s. + +/// What C is told of `read`, a number read from `s`: its value, returned; +/// where it ends in `endptr`, or `s` itself when there was none; and a +/// refusal in `errno`. +pub(crate) unsafe fn answer(s: *const U, read: strtonum::Read, endptr: *mut *mut U) -> T { + match read.refused { + Some(strtonum::Refusal::Base) => crate::errno::set(crate::errno::EINVAL), + Some(strtonum::Refusal::Range) => crate::errno::set(crate::errno::ERANGE), + None => {} + } + if !endptr.is_null() { + unsafe { *endptr = s.add(read.end).cast_mut() }; + } + read.value +} #[no_mangle] pub unsafe extern "C" fn atoi(s: *const u8) -> i32 { @@ -308,113 +338,97 @@ pub unsafe extern "C" fn atol(s: *const u8) -> i64 { strtol(s, ptr::null_mut(), 10) } +#[no_mangle] +pub unsafe extern "C" fn atoll(s: *const u8) -> i64 { + strtol(s, ptr::null_mut(), 10) +} + #[no_mangle] pub unsafe extern "C" fn strtol(s: *const u8, endptr: *mut *mut u8, base: i32) -> i64 { - if s.is_null() { return 0; } - let mut p = s; - // Skip whitespace - while *p == b' ' || *p == b'\t' || *p == b'\n' || *p == b'\r' { p = p.add(1); } - // Sign - let neg = *p == b'-'; - if *p == b'-' || *p == b'+' { p = p.add(1); } - // Detect base - let mut base = base as u32; - if base == 0 { - if *p == b'0' { - p = p.add(1); - if *p == b'x' || *p == b'X' { - base = 16; - p = p.add(1); - } else { - base = 8; - } - } else { - base = 10; - } - } else if base == 16 && *p == b'0' && (*p.add(1) == b'x' || *p.add(1) == b'X') { - p = p.add(2); - } - let mut val: i64 = 0; - loop { - let c = *p; - let digit = match c { - b'0'..=b'9' => c - b'0', - b'a'..=b'z' => c - b'a' + 10, - b'A'..=b'Z' => c - b'A' + 10, - _ => break, - }; - if digit as u32 >= base { break; } - val = val.wrapping_mul(base as i64).wrapping_add(digit as i64); - p = p.add(1); - } - if !endptr.is_null() { *endptr = p as *mut u8; } - if neg { -val } else { val } + unsafe { answer(s, strtonum::signed(s, base), endptr) } } #[no_mangle] pub unsafe extern "C" fn strtoul(s: *const u8, endptr: *mut *mut u8, base: i32) -> u64 { - strtol(s, endptr, base) as u64 + unsafe { answer(s, strtonum::unsigned(s, base), endptr) } } #[no_mangle] pub unsafe extern "C" fn strtoll(s: *const u8, endptr: *mut *mut u8, base: i32) -> i64 { - strtol(s, endptr, base) + unsafe { strtol(s, endptr, base) } } #[no_mangle] pub unsafe extern "C" fn strtoull(s: *const u8, endptr: *mut *mut u8, base: i32) -> u64 { - strtol(s, endptr, base) as u64 + unsafe { strtoul(s, endptr, base) } +} + +#[no_mangle] +pub unsafe extern "C" fn strtoimax(s: *const u8, endptr: *mut *mut u8, base: i32) -> i64 { + unsafe { strtol(s, endptr, base) } +} + +#[no_mangle] +pub unsafe extern "C" fn strtoumax(s: *const u8, endptr: *mut *mut u8, base: i32) -> u64 { + unsafe { strtoul(s, endptr, base) } } #[no_mangle] pub unsafe extern "C" fn strtod(s: *const u8, endptr: *mut *mut u8) -> f64 { - if s.is_null() { return 0.0; } - let mut p = s; - while *p == b' ' || *p == b'\t' || *p == b'\n' || *p == b'\r' { p = p.add(1); } - let neg = *p == b'-'; - if *p == b'-' || *p == b'+' { p = p.add(1); } - - let mut val: f64 = 0.0; - while *p >= b'0' && *p <= b'9' { - val = val * 10.0 + (*p - b'0') as f64; - p = p.add(1); - } - if *p == b'.' { - p = p.add(1); - let mut frac = 0.1; - while *p >= b'0' && *p <= b'9' { - val += (*p - b'0') as f64 * frac; - frac *= 0.1; - p = p.add(1); - } - } - if *p == b'e' || *p == b'E' { - p = p.add(1); - let exp_neg = *p == b'-'; - if *p == b'-' || *p == b'+' { p = p.add(1); } - let mut exp: i32 = 0; - while *p >= b'0' && *p <= b'9' { - exp = exp * 10 + (*p - b'0') as i32; - p = p.add(1); - } - if exp_neg { exp = -exp; } - val *= super::math::pow(10.0, exp as f64); - } + unsafe { answer(s, strtonum::float::(s), endptr) } +} - if !endptr.is_null() { *endptr = p as *mut u8; } - if neg { -val } else { val } +#[no_mangle] +pub unsafe extern "C" fn strtof(s: *const u8, endptr: *mut *mut u8) -> f32 { + unsafe { answer(s, strtonum::float::(s), endptr) } } -// abs +// abs and div #[no_mangle] pub unsafe extern "C" fn abs(j: i32) -> i32 { - if j < 0 { -j } else { j } + j.wrapping_abs() } #[no_mangle] pub unsafe extern "C" fn labs(j: i64) -> i64 { - if j < 0 { -j } else { j } + j.wrapping_abs() +} + +#[no_mangle] +pub unsafe extern "C" fn llabs(j: i64) -> i64 { + j.wrapping_abs() +} + +#[no_mangle] +pub unsafe extern "C" fn imaxabs(j: i64) -> i64 { + j.wrapping_abs() +} + +#[repr(C)] +pub struct Div { + quot: T, + rem: T, +} + +#[no_mangle] +pub unsafe extern "C" fn div(numer: i32, denom: i32) -> Div { + Div { quot: numer / denom, rem: numer % denom } +} + +#[no_mangle] +pub unsafe extern "C" fn ldiv(numer: i64, denom: i64) -> Div { + Div { quot: numer / denom, rem: numer % denom } +} + +#[no_mangle] +pub unsafe extern "C" fn lldiv(numer: i64, denom: i64) -> Div { + Div { quot: numer / denom, rem: numer % denom } +} + +#[no_mangle] +pub unsafe extern "C" fn imaxdiv(numer: i64, denom: i64) -> Div { + Div { quot: numer / denom, rem: numer % denom } } // setjmp/longjmp (minimal stub — used by some C code) @@ -474,3 +488,25 @@ pub unsafe extern "C" fn dlclose(handle: *mut u8) -> i32 { pub unsafe extern "C" fn dlerror() -> *const u8 { ptr::null() } + +/// `getentropy`: at most 256 bytes of the kernel's random source, all or an +/// error, as POSIX says. +#[no_mangle] +pub unsafe extern "C" fn getentropy(buffer: *mut u8, length: usize) -> i32 { + if length > 256 { + crate::errno::set(crate::errno::EINVAL); + return -1; + } + // Before the slice: `buffer` may be null when there is nothing to fill. + if length == 0 { + return 0; + } + let buf = unsafe { core::slice::from_raw_parts_mut(buffer, length) }; + match syscall::random(buf) { + Ok(()) => 0, + Err(_) => { + crate::errno::set(crate::errno::EIO); + -1 + } + } +} diff --git a/userland/libc/src/posix_io.rs b/userland/libc/src/posix_io.rs index d593d57696f..3d1bfb27748 100644 --- a/userland/libc/src/posix_io.rs +++ b/userland/libc/src/posix_io.rs @@ -7,6 +7,7 @@ use core::ptr; use toyos_abi::RawHandle; use toyos_abi::syscall::{self, OpenFlags, SeekFrom}; +use crate::errno::{EACCES, EAGAIN, EEXIST, EINVAL, EIO, ENOENT, EPIPE}; use crate::time::Timespec; // Constants (matching POSIX / Linux values) @@ -22,17 +23,6 @@ const SEEK_SET: i32 = 0; const SEEK_CUR: i32 = 1; const SEEK_END: i32 = 2; -// errno values -const ENOENT: i32 = 2; -const EIO: i32 = 5; -const EACCES: i32 = 13; -const EEXIST: i32 = 17; -const EINVAL: i32 = 22; -const EAGAIN: i32 = 11; -/// `include/errno.h` declares it 32; a write whose reader is gone is this and -/// not `ENOENT`, which says the file does not exist. -const EPIPE: i32 = 32; - // stat file type bits const S_IFREG: u32 = 0o100000; const S_IFIFO: u32 = 0o010000; @@ -50,6 +40,7 @@ fn set_errno(e: toyos_abi::syscall::SyscallError) -> i32 { SyscallError::AlreadyExists => EEXIST, SyscallError::InvalidArgument => EINVAL, SyscallError::WouldBlock => EAGAIN, + // A write whose reader is gone, and not `ENOENT`, which says the file does not exist. SyscallError::Gone => EPIPE, SyscallError::Io => EIO, _ => EINVAL, diff --git a/userland/libc/src/printf.rs b/userland/libc/src/printf.rs index e963db5a993..6a0603f3b55 100644 --- a/userland/libc/src/printf.rs +++ b/userland/libc/src/printf.rs @@ -2,23 +2,30 @@ use alloc::vec; use core::ffi::VaList; use core::fmt::Write; -/// Output buffer for printf family. Writes to a raw C buffer with optional capacity limit. +/// Output buffer for printf family. Writes to a raw C buffer with optional +/// capacity limit, and counts every byte it was given, as C's return value +/// does, whether or not it fit. struct BufWriter { buf: *mut u8, pos: usize, cap: usize, // usize::MAX = unlimited (sprintf) + /// A wide argument that is no Unicode scalar value: C's `EILSEQ`. + refused: bool, +} + +impl BufWriter { + fn put(&mut self, b: u8) { + if !self.buf.is_null() && self.pos + 1 < self.cap { + unsafe { *self.buf.add(self.pos) = b; } + } + self.pos += 1; + } } impl Write for BufWriter { fn write_str(&mut self, s: &str) -> core::fmt::Result { for &b in s.as_bytes() { - if self.cap > 0 && self.pos >= self.cap - 1 { - continue; // leave room for null terminator - } - if !self.buf.is_null() { - unsafe { *self.buf.add(self.pos) = b; } - } - self.pos += 1; + self.put(b); } Ok(()) } @@ -261,12 +268,12 @@ fn format_float<'a>( /// Core printf engine. Parses the format string as a byte slice. unsafe fn do_printf(buf: *mut u8, n: usize, fmt: *const u8, ap: &mut VaList<'_>) -> i32 { let fmt = core::slice::from_raw_parts(fmt, super::string::strlen(fmt)); - let mut w = BufWriter { buf, pos: 0, cap: n }; + let mut w = BufWriter { buf, pos: 0, cap: n, refused: false }; let mut i = 0; while i < fmt.len() { if fmt[i] != b'%' { - let _ = w.write_char(fmt[i] as char); + w.put(fmt[i]); i += 1; continue; } @@ -371,10 +378,37 @@ unsafe fn do_printf(buf: *mut u8, n: usize, fmt: *const u8, ap: &mut VaList<'_>) let s = format_unsigned(val, 8, false, &mut tmp); write_int_padded(&mut w, s, 0, width, pad_char, left_align, precision); } + b'c' if long => { + let mut utf8 = [0u8; 4]; + match crate::wchar::encode(ap.next_arg::() as crate::arch::WChar, &mut utf8) { + Some(n) => write_padded_bytes(&mut w, &utf8[..n], width, left_align), + None => w.refused = true, + } + } b'c' => { let c = ap.next_arg::() as u8; - let s = core::str::from_utf8_unchecked(core::slice::from_ref(&c)); - write_padded(&mut w, s, width, ' ', left_align); + write_padded_bytes(&mut w, &[c], width, left_align); + } + b's' if long => { + let p: *const crate::arch::WChar = ap.next_arg::<*const crate::arch::WChar>(); + let mut bytes = alloc::vec::Vec::new(); + if p.is_null() { + bytes.extend_from_slice(b"(null)"); + } + let mut k = 0; + while !p.is_null() && *p.add(k) != 0 { + let mut utf8 = [0u8; 4]; + let Some(n) = crate::wchar::encode(*p.add(k), &mut utf8) else { + w.refused = true; + break; + }; + if precision.is_some_and(|prec| bytes.len() + n > prec) { + break; + } + bytes.extend_from_slice(&utf8[..n]); + k += 1; + } + write_padded_bytes(&mut w, &bytes, width, left_align); } b's' => { let p: *const u8 = ap.next_arg::<*const u8>(); @@ -407,8 +441,8 @@ unsafe fn do_printf(buf: *mut u8, n: usize, fmt: *const u8, ap: &mut VaList<'_>) } b'%' => { let _ = w.write_char('%'); } other => { - let _ = w.write_char('%'); - let _ = w.write_char(other as char); + w.put(b'%'); + w.put(other); } } i += 1; @@ -418,27 +452,50 @@ unsafe fn do_printf(buf: *mut u8, n: usize, fmt: *const u8, ap: &mut VaList<'_>) *buf.add(w.pos.min(n - 1)) = 0; } + if w.refused { + crate::errno::set(crate::errno::EILSEQ); + return -1; + } w.pos as i32 } -#[no_mangle] -pub unsafe extern "C" fn printf(fmt: *const u8, mut args: ...) -> i32 { - let mut buf = [0u8; 4096]; - let n = do_printf(buf.as_mut_ptr(), buf.len(), fmt, &mut args); - if n > 0 { - super::stdio::fwrite(buf.as_ptr(), 1, n as usize, super::stdio::stdout); +/// Hand `take` the whole of what `fmt` formats to, NUL-terminated, formatted +/// into a buffer on the stack, and again on the heap for an output that +/// buffer cannot hold: its length, or -1 when the format or `take` refuses. +unsafe fn formatted(fmt: *const u8, ap: VaList<'_>, take: impl FnOnce(&[u8]) -> bool) -> i32 { + let mut stack = [0u8; 4096]; + let n = do_printf(stack.as_mut_ptr(), stack.len(), fmt, &mut ap.clone()); + if n < 0 { + return n; } - n + let len = n as usize; + let taken = if len < stack.len() { + take(&stack[..=len]) + } else { + let mut whole = vec![0u8; len + 1]; + let mut ap = ap; + do_printf(whole.as_mut_ptr(), whole.len(), fmt, &mut ap); + take(&whole) + }; + if taken { n } else { -1 } +} + +/// Write the whole of what `fmt` formats to to `f`. +unsafe fn print_to(f: *mut super::stdio::FILE, fmt: *const u8, ap: VaList<'_>) -> i32 { + formatted(fmt, ap, |bytes| { + super::stdio::fwrite(bytes.as_ptr(), 1, bytes.len() - 1, f); + true + }) } #[no_mangle] -pub unsafe extern "C" fn fprintf(f: *mut super::stdio::FILE, fmt: *const u8, mut args: ...) -> i32 { - let mut buf = [0u8; 4096]; - let n = do_printf(buf.as_mut_ptr(), buf.len(), fmt, &mut args); - if n > 0 { - super::stdio::fwrite(buf.as_ptr(), 1, n as usize, f); - } - n +pub unsafe extern "C" fn printf(fmt: *const u8, args: ...) -> i32 { + print_to(super::stdio::stdout, fmt, args) +} + +#[no_mangle] +pub unsafe extern "C" fn fprintf(f: *mut super::stdio::FILE, fmt: *const u8, args: ...) -> i32 { + print_to(f, fmt, args) } #[no_mangle] @@ -457,13 +514,8 @@ pub unsafe extern "C" fn vsnprintf(buf: *mut u8, n: usize, fmt: *const u8, mut a } #[no_mangle] -pub unsafe extern "C" fn vfprintf(f: *mut super::stdio::FILE, fmt: *const u8, mut ap: VaList<'_>) -> i32 { - let mut buf = [0u8; 4096]; - let n = do_printf(buf.as_mut_ptr(), buf.len(), fmt, &mut ap); - if n > 0 { - super::stdio::fwrite(buf.as_ptr(), 1, n as usize, f); - } - n +pub unsafe extern "C" fn vfprintf(f: *mut super::stdio::FILE, fmt: *const u8, ap: VaList<'_>) -> i32 { + print_to(f, fmt, ap) } #[no_mangle] @@ -525,4 +577,33 @@ pub unsafe extern "C" fn sscanf(input: *const u8, fmt: *const u8, mut args: ...) } } matched -} \ No newline at end of file +} + +/// `bytes`, space-padded to `width` on the side `left` says. +fn write_padded_bytes(w: &mut BufWriter, bytes: &[u8], width: usize, left: bool) { + let pad = width.saturating_sub(bytes.len()); + if !left { + (0..pad).for_each(|_| w.put(b' ')); + } + bytes.iter().for_each(|&b| w.put(b)); + if left { + (0..pad).for_each(|_| w.put(b' ')); + } +} + +#[no_mangle] +pub unsafe extern "C" fn vasprintf(out: *mut *mut u8, fmt: *const u8, ap: VaList<'_>) -> i32 { + formatted(fmt, ap, |bytes| { + let buf = super::memory::malloc(bytes.len()); + if !buf.is_null() { + core::ptr::copy_nonoverlapping(bytes.as_ptr(), buf, bytes.len()); + *out = buf; + } + !buf.is_null() + }) +} + +#[no_mangle] +pub unsafe extern "C" fn asprintf(out: *mut *mut u8, fmt: *const u8, args: ...) -> i32 { + vasprintf(out, fmt, args) +} diff --git a/userland/libc/src/pthread.rs b/userland/libc/src/pthread.rs index 625d522784b..33e45d82c96 100644 --- a/userland/libc/src/pthread.rs +++ b/userland/libc/src/pthread.rs @@ -1,189 +1,358 @@ -// POSIX threads — implemented on top of toyos-abi thread/futex syscalls. +//! POSIX threads on the kernel's thread and futex syscalls. +//! +//! **A `pthread_t` is the address of the thread's [`Thread`] block**, which +//! `pthread_create` allocates and the new thread records in [`SELF`]; a +//! thread this library did not start (the main thread, or one Rust's std +//! spawned) is named by the address of its own [`MARKER`] with [`FOREIGN`] +//! set, which no block's address has, and cannot be joined or detached. +//! +//! **No `pthread_t` exists before its thread's tid is recorded**: the new +//! thread waits for its creator to publish the tid before it runs anything, +//! so every holder of a handle can join it. +//! +//! A thread that ends detached hands its block to [`REAPED`], and the next +//! `pthread_create` joins it and frees its stack, which no thread can free +//! while standing on it. use alloc::alloc::{alloc as heap_alloc, dealloc as heap_dealloc}; +use alloc::boxed::Box; +use alloc::vec::Vec; +use core::alloc::Layout; +use core::cell::{Cell, UnsafeCell}; use core::ptr; -use core::sync::atomic::{AtomicU32, Ordering}; +use core::sync::atomic::{AtomicPtr, AtomicU32, AtomicU64, Ordering}; use toyos_abi::syscall; -// Types +use crate::errno::{EAGAIN, EBUSY, EDEADLK, EINVAL, EPERM, ESRCH, ETIMEDOUT}; -// pthread_t is a thread ID (u64 from kernel) type PthreadT = u64; +type StartRoutine = unsafe extern "C" fn(*mut u8) -> *mut u8; -// Mutex: futex-based. 0 = unlocked, 1 = locked, 2 = locked with waiters. -#[repr(C)] -pub struct PthreadMutexT { +const STACK_DEFAULT: usize = 1024 * 1024; +const STACK_MIN: usize = 16 * 1024; +const STACK_ALIGN: usize = 16; + +// `pthread_attr_t`: the stack size, with the detach state in its low bit. +const ATTR_DETACHED: u64 = 1; + +/// One thread `pthread_create` started. +struct Thread { + tid: AtomicU64, + /// Nonzero once `tid` is written; the thread waits for it before it runs. + published: AtomicU32, state: AtomicU32, + stack: *mut u8, + stack_size: usize, + start: StartRoutine, + arg: *mut u8, + result: AtomicPtr, } -// Condition variable: futex-based. -#[repr(C)] -pub struct PthreadCondT { - seq: AtomicU32, +/// The bit a `pthread_t` of a thread this library did not start carries. +const FOREIGN: PthreadT = 1; + +const JOINABLE: u32 = 0; +const DETACHED: u32 = 1; +const EXITED: u32 = 2; + +#[thread_local] +static SELF: Cell<*mut Thread> = Cell::new(ptr::null_mut()); + +#[thread_local] +static MARKER: u8 = 0; + +/// Blocks of detached threads that have ended, for the next creator to reap. +static REAPED: Lock> = Lock::new(Vec::new()); + +/// A futex lock around library state: 0 free, 1 held, 2 held with waiters. +pub(crate) struct Lock { + state: AtomicU32, + value: UnsafeCell, } -// Mutex/cond attr (unused, but needed for API compat) -type PthreadMutexattrT = u64; -type PthreadCondattrT = u64; +unsafe impl Sync for Lock {} -// Once control -#[repr(C)] -pub struct PthreadOnceT { - state: AtomicU32, // 0 = not called, 1 = in progress, 2 = done +pub(crate) struct Held<'a, T>(&'a Lock); + +impl Lock { + pub(crate) const fn new(value: T) -> Self { + Self { state: AtomicU32::new(0), value: UnsafeCell::new(value) } + } + + pub(crate) fn lock(&self) -> Held<'_, T> { + futex_lock(&self.state); + Held(self) + } } -// Thread-local storage -type PthreadKeyT = u32; +impl core::ops::Deref for Held<'_, T> { + type Target = T; + fn deref(&self) -> &T { + // SAFETY: the lock is held. + unsafe { &*self.0.value.get() } + } +} -const PTHREAD_KEYS_MAX: usize = 128; -static mut KEY_DESTRUCTORS: [Option; PTHREAD_KEYS_MAX] = - [None; PTHREAD_KEYS_MAX]; -static NEXT_KEY: AtomicU32 = AtomicU32::new(0); +impl core::ops::DerefMut for Held<'_, T> { + fn deref_mut(&mut self) -> &mut T { + // SAFETY: the lock is held. + unsafe { &mut *self.0.value.get() } + } +} -// Per-thread TLS values — stored in a global array indexed by thread ID. -// This is a simplification; real implementations use TLS segments. -const MAX_THREADS: usize = 64; -static mut TLS_VALUES: [[*mut u8; PTHREAD_KEYS_MAX]; MAX_THREADS] = - [[ptr::null_mut(); PTHREAD_KEYS_MAX]; MAX_THREADS]; +impl Drop for Held<'_, T> { + fn drop(&mut self) { + futex_unlock(&self.0.state); + } +} + +fn futex_lock(state: &AtomicU32) { + if state.compare_exchange(0, 1, Ordering::Acquire, Ordering::Relaxed).is_ok() { + return; + } + while state.swap(2, Ordering::Acquire) != 0 { + // SAFETY: `state` is a live, aligned u32. + unsafe { syscall::futex_wait(state.as_ptr(), 2, None) }; + } +} -fn thread_index() -> usize { - // Use thread ID mod MAX_THREADS as index - let tid = syscall::getpid().0 as usize; // approximate — uses pid, not tid - tid % MAX_THREADS +fn futex_unlock(state: &AtomicU32) { + if state.swap(0, Ordering::Release) == 2 { + // SAFETY: `state` is a live, aligned u32. + unsafe { syscall::futex_wake(state.as_ptr(), 1) }; + } } -// Thread create/join +fn stack_layout(size: usize) -> Layout { + Layout::from_size_align(size, STACK_ALIGN).expect("a thread stack's size is a multiple of its alignment") +} -const THREAD_STACK_SIZE: usize = 1024 * 1024; // 1 MiB +/// Wait for `thread` to be gone, free its stack and block, and answer what it +/// returned. +unsafe fn reap(thread: *mut Thread) -> *mut u8 { + let t = unsafe { &*thread }; + syscall::thread_join(t.tid.load(Ordering::Relaxed)); + let result = t.result.load(Ordering::Acquire); + unsafe { + heap_dealloc(t.stack, stack_layout(t.stack_size)); + drop(Box::from_raw(thread)); + } + result +} -// Trampoline for pthread_create: calls the user function, then exits the thread. unsafe extern "C" fn thread_entry(arg: u64) { - let info = arg as *mut ThreadStartInfo; - let start_routine = (*info).start_routine; - let user_arg = (*info).arg; - // Free the info struct - let layout = core::alloc::Layout::new::(); - heap_dealloc(info as *mut u8, layout); - // Call user function - let _retval = start_routine(user_arg); - syscall::thread_exit(0); -} - -struct ThreadStartInfo { - start_routine: unsafe extern "C" fn(*mut u8) -> *mut u8, - arg: *mut u8, + let thread = arg as *mut Thread; + let t = unsafe { &*thread }; + while t.published.load(Ordering::Acquire) == 0 { + // SAFETY: `published` is a live, aligned u32. + unsafe { syscall::futex_wait(t.published.as_ptr(), 0, None) }; + } + SELF.set(thread); + let result = unsafe { (t.start)(t.arg) }; + unsafe { exit_thread(result) } +} + +/// Run what a thread owes on its way out, record `result`, and end it. +unsafe fn exit_thread(result: *mut u8) -> ! { + unsafe { + run_thread_dtors(); + run_key_destructors(); + } + let thread = SELF.get(); + if !thread.is_null() { + let t = unsafe { &*thread }; + t.result.store(result, Ordering::Release); + if t.state.swap(EXITED, Ordering::AcqRel) == DETACHED { + REAPED.lock().push(thread as usize); + } + } + syscall::thread_exit(0) } #[no_mangle] pub unsafe extern "C" fn pthread_create( thread: *mut PthreadT, - _attr: *const u8, - start_routine: unsafe extern "C" fn(*mut u8) -> *mut u8, + attr: *const u64, + start_routine: StartRoutine, arg: *mut u8, ) -> i32 { - // Allocate thread start info on heap (freed by trampoline) - let layout = core::alloc::Layout::new::(); - let info = heap_alloc(layout) as *mut ThreadStartInfo; - if info.is_null() { return -1; } - ptr::write(info, ThreadStartInfo { start_routine, arg }); - - // Allocate stack - let stack_layout = core::alloc::Layout::from_size_align(THREAD_STACK_SIZE, 16).unwrap(); - let stack_base = heap_alloc(stack_layout); - if stack_base.is_null() { - heap_dealloc(info as *mut u8, layout); - return -1; - } - let stack_top = stack_base.add(THREAD_STACK_SIZE); - // Align stack to 16 bytes - let stack_ptr = ((stack_top as usize) & !0xF) as u64; - - // SAFETY: entry point, stack, and argument are valid; stack is freshly allocated and aligned - let tid = unsafe { syscall::thread_spawn( - thread_entry as *const () as u64, - stack_ptr, - info as u64, - stack_base as u64, - ) }; + let reaped = core::mem::take(&mut *REAPED.lock()); + for block in reaped { + unsafe { reap(block as *mut Thread) }; + } + let attr = if attr.is_null() { attr_default() } else { unsafe { *attr } }; + let stack_size = (attr & !ATTR_DETACHED) as usize; + let stack = unsafe { heap_alloc(stack_layout(stack_size)) }; + if stack.is_null() { + return EAGAIN; + } + let state = if attr & ATTR_DETACHED != 0 { DETACHED } else { JOINABLE }; + let block = Box::into_raw(Box::new(Thread { + tid: AtomicU64::new(0), + published: AtomicU32::new(0), + state: AtomicU32::new(state), + stack, + stack_size, + start: start_routine, + arg, + result: AtomicPtr::new(ptr::null_mut()), + })); + let top = (stack as usize + stack_size) & !(STACK_ALIGN - 1); + // SAFETY: the entry is a function of this library, and the stack is a + // fresh allocation of `stack_size` bytes. + let tid = unsafe { syscall::thread_spawn(thread_entry as *const () as u64, top as u64, block as u64, stack as u64) }; + if syscall::SyscallError::from_u64(tid).is_some() { + unsafe { + heap_dealloc(stack, stack_layout(stack_size)); + drop(Box::from_raw(block)); + } + return EAGAIN; + } + // Taken before the store: once `published` is set the thread may run, end + // detached and be reaped, so nothing after it reads the block. + let published = unsafe { (*block).published.as_ptr() }; + unsafe { + (*block).tid.store(tid, Ordering::Relaxed); + (*block).published.store(1, Ordering::Release); + } + // At a freed block's address this is a spurious wake, which every futex + // waiter here re-checks its word against. + unsafe { syscall::futex_wake(published, u32::MAX) }; if !thread.is_null() { - *thread = tid; + unsafe { *thread = block as PthreadT }; } 0 } #[no_mangle] pub unsafe extern "C" fn pthread_join(thread: PthreadT, retval: *mut *mut u8) -> i32 { - syscall::thread_join(thread); + if thread == pthread_self() { + return EDEADLK; + } + if thread == 0 || thread & FOREIGN != 0 { + return ESRCH; + } + let block = thread as *mut Thread; + if unsafe { &*block }.state.load(Ordering::Acquire) == DETACHED { + return EINVAL; + } + let result = unsafe { reap(block) }; if !retval.is_null() { - *retval = ptr::null_mut(); + unsafe { *retval = result }; } 0 } #[no_mangle] -pub unsafe extern "C" fn pthread_detach(_thread: PthreadT) -> i32 { - 0 // ToyOS threads are implicitly cleaned up +pub unsafe extern "C" fn pthread_detach(thread: PthreadT) -> i32 { + if thread == 0 || thread & FOREIGN != 0 { + return ESRCH; + } + let block = thread as *mut Thread; + match unsafe { &*block }.state.compare_exchange(JOINABLE, DETACHED, Ordering::AcqRel, Ordering::Acquire) { + Ok(_) => 0, + Err(EXITED) => { + unsafe { reap(block) }; + 0 + } + Err(_) => EINVAL, + } +} + +#[no_mangle] +pub unsafe extern "C" fn pthread_exit(retval: *mut u8) -> ! { + unsafe { exit_thread(retval) } } #[no_mangle] -pub unsafe extern "C" fn pthread_self() -> PthreadT { - // Return pid as a stand-in for thread ID - syscall::getpid().0 as u64 +pub extern "C" fn pthread_self() -> PthreadT { + let thread = SELF.get(); + if thread.is_null() { ptr::addr_of!(MARKER) as PthreadT | FOREIGN } else { thread as PthreadT } } #[no_mangle] -pub unsafe extern "C" fn pthread_equal(t1: PthreadT, t2: PthreadT) -> i32 { +pub extern "C" fn pthread_equal(t1: PthreadT, t2: PthreadT) -> i32 { (t1 == t2) as i32 } -// Mutex (futex-based) +#[no_mangle] +pub extern "C" fn sched_yield() -> i32 { + // No syscall gives the processor up; this is what std's `yield_now` does. + core::hint::spin_loop(); + 0 +} + +// Mutex + +const MUTEX_NORMAL: u32 = 0; +const MUTEX_RECURSIVE: u32 = 1; +const MUTEX_ERRORCHECK: u32 = 2; + +#[repr(C)] +pub struct PthreadMutexT { + state: AtomicU32, + kind: u32, + owner: AtomicU64, + count: AtomicU64, +} #[no_mangle] -pub unsafe extern "C" fn pthread_mutex_init( - mutex: *mut PthreadMutexT, _attr: *const PthreadMutexattrT, -) -> i32 { - (*mutex).state = AtomicU32::new(0); +pub unsafe extern "C" fn pthread_mutex_init(mutex: *mut PthreadMutexT, attr: *const u64) -> i32 { + let kind = if attr.is_null() { MUTEX_NORMAL } else { unsafe { *attr as u32 } }; + unsafe { + mutex.write(PthreadMutexT { state: AtomicU32::new(0), kind, owner: AtomicU64::new(0), count: AtomicU64::new(0) }); + } 0 } #[no_mangle] pub unsafe extern "C" fn pthread_mutex_lock(mutex: *mut PthreadMutexT) -> i32 { - let state = &(*mutex).state; - if state.compare_exchange(0, 1, Ordering::Acquire, Ordering::Relaxed).is_ok() { - return 0; - } - loop { - let old = state.swap(2, Ordering::Acquire); - if old == 0 { - return 0; // Got the lock + let m = unsafe { &*mutex }; + let me = pthread_self(); + if m.kind != MUTEX_NORMAL && m.owner.load(Ordering::Relaxed) == me { + if m.kind == MUTEX_ERRORCHECK { + return EDEADLK; } - let addr = state as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the mutex - unsafe { syscall::futex_wait(addr, 2, None) }; + m.count.fetch_add(1, Ordering::Relaxed); + return 0; } + futex_lock(&m.state); + m.owner.store(me, Ordering::Relaxed); + m.count.store(1, Ordering::Relaxed); + 0 } #[no_mangle] pub unsafe extern "C" fn pthread_mutex_trylock(mutex: *mut PthreadMutexT) -> i32 { - let state = &(*mutex).state; - if state.compare_exchange(0, 1, Ordering::Acquire, Ordering::Relaxed).is_ok() { - 0 - } else { - 16 // EBUSY + let m = unsafe { &*mutex }; + let me = pthread_self(); + if m.kind == MUTEX_RECURSIVE && m.owner.load(Ordering::Relaxed) == me { + m.count.fetch_add(1, Ordering::Relaxed); + return 0; + } + if m.state.compare_exchange(0, 1, Ordering::Acquire, Ordering::Relaxed).is_err() { + return EBUSY; } + m.owner.store(me, Ordering::Relaxed); + m.count.store(1, Ordering::Relaxed); + 0 } #[no_mangle] pub unsafe extern "C" fn pthread_mutex_unlock(mutex: *mut PthreadMutexT) -> i32 { - let state = &(*mutex).state; - let old = state.swap(0, Ordering::Release); - if old == 2 { - let addr = state as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the mutex - unsafe { syscall::futex_wake(addr, 1) }; + let m = unsafe { &*mutex }; + if m.kind != MUTEX_NORMAL { + if m.owner.load(Ordering::Relaxed) != pthread_self() { + return EPERM; + } + if m.count.fetch_sub(1, Ordering::Relaxed) > 1 { + return 0; + } } + m.owner.store(0, Ordering::Relaxed); + futex_unlock(&m.state); 0 } @@ -193,54 +362,107 @@ pub unsafe extern "C" fn pthread_mutex_destroy(_mutex: *mut PthreadMutexT) -> i3 } #[no_mangle] -pub unsafe extern "C" fn pthread_mutexattr_init(_attr: *mut PthreadMutexattrT) -> i32 { 0 } +pub unsafe extern "C" fn pthread_mutexattr_init(attr: *mut u64) -> i32 { + unsafe { *attr = MUTEX_NORMAL as u64 }; + 0 +} #[no_mangle] -pub unsafe extern "C" fn pthread_mutexattr_destroy(_attr: *mut PthreadMutexattrT) -> i32 { 0 } +pub unsafe extern "C" fn pthread_mutexattr_destroy(_attr: *mut u64) -> i32 { + 0 +} #[no_mangle] -pub unsafe extern "C" fn pthread_mutexattr_settype( - _attr: *mut PthreadMutexattrT, _type: i32, -) -> i32 { 0 } +pub unsafe extern "C" fn pthread_mutexattr_settype(attr: *mut u64, kind: i32) -> i32 { + match kind as u32 { + MUTEX_NORMAL | MUTEX_RECURSIVE | MUTEX_ERRORCHECK => { + unsafe { *attr = kind as u64 }; + 0 + } + _ => EINVAL, + } +} -// Condition variable (futex-based) +#[no_mangle] +pub unsafe extern "C" fn pthread_mutexattr_gettype(attr: *const u64, kind: *mut i32) -> i32 { + unsafe { *kind = *attr as i32 }; + 0 +} + +// Condition variable: a sequence number a waiter sleeps on until it moves. + +#[repr(C)] +pub struct PthreadCondT { + seq: AtomicU32, +} #[no_mangle] -pub unsafe extern "C" fn pthread_cond_init( - cond: *mut PthreadCondT, _attr: *const PthreadCondattrT, -) -> i32 { - (*cond).seq = AtomicU32::new(0); +pub unsafe extern "C" fn pthread_cond_init(cond: *mut PthreadCondT, _attr: *const u64) -> i32 { + unsafe { cond.write(PthreadCondT { seq: AtomicU32::new(0) }) }; 0 } +/// Release `mutex`, sleep until `cond` is signalled or `timeout` nanoseconds +/// pass, and take `mutex` again, as many times as it was held: 0, or +/// `ETIMEDOUT`, or `EPERM` for a recursive or error-checking mutex the caller +/// does not hold, as POSIX says. +unsafe fn cond_wait(cond: *mut PthreadCondT, mutex: *mut PthreadMutexT, timeout: Option) -> i32 { + let m = unsafe { &*mutex }; + if m.kind != MUTEX_NORMAL && m.owner.load(Ordering::Relaxed) != pthread_self() { + return EPERM; + } + let seq = unsafe { &(*cond).seq }; + let at = seq.load(Ordering::Relaxed); + let held = m.count.swap(1, Ordering::Relaxed); + unsafe { pthread_mutex_unlock(mutex) }; + // SAFETY: `seq` is a live, aligned u32. + let timed_out = unsafe { syscall::futex_wait(seq.as_ptr(), at, timeout) } == 1; + unsafe { pthread_mutex_lock(mutex) }; + m.count.store(held, Ordering::Relaxed); + if timed_out { ETIMEDOUT } else { 0 } +} + #[no_mangle] -pub unsafe extern "C" fn pthread_cond_wait( - cond: *mut PthreadCondT, mutex: *mut PthreadMutexT, +pub unsafe extern "C" fn pthread_cond_wait(cond: *mut PthreadCondT, mutex: *mut PthreadMutexT) -> i32 { + unsafe { cond_wait(cond, mutex, None) } +} + +#[no_mangle] +pub unsafe extern "C" fn pthread_cond_timedwait( + cond: *mut PthreadCondT, + mutex: *mut PthreadMutexT, + abstime: *const crate::time::Timespec, ) -> i32 { - let seq = (*cond).seq.load(Ordering::Relaxed); - pthread_mutex_unlock(mutex); - let addr = &(*cond).seq as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the condvar - unsafe { syscall::futex_wait(addr, seq, None) }; - pthread_mutex_lock(mutex); - 0 + let deadline = unsafe { &*abstime }; + if deadline.tv_nsec < 0 || deadline.tv_nsec >= 1_000_000_000 { + return EINVAL; + } + let mut now = crate::time::Timespec { tv_sec: 0, tv_nsec: 0 }; + unsafe { crate::time::clock_gettime(crate::time::CLOCK_REALTIME, &mut now) }; + let left = (deadline.tv_sec as i128 - now.tv_sec as i128) * 1_000_000_000 + + (deadline.tv_nsec as i128 - now.tv_nsec as i128); + if left <= 0 { + return ETIMEDOUT; + } + let left = u64::try_from(left).unwrap_or(u64::MAX - 1); + unsafe { cond_wait(cond, mutex, Some(left)) } } #[no_mangle] pub unsafe extern "C" fn pthread_cond_signal(cond: *mut PthreadCondT) -> i32 { - (*cond).seq.fetch_add(1, Ordering::Release); - let addr = &(*cond).seq as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the condvar - unsafe { syscall::futex_wake(addr, 1) }; + let seq = unsafe { &(*cond).seq }; + seq.fetch_add(1, Ordering::Release); + // SAFETY: `seq` is a live, aligned u32. + unsafe { syscall::futex_wake(seq.as_ptr(), 1) }; 0 } #[no_mangle] pub unsafe extern "C" fn pthread_cond_broadcast(cond: *mut PthreadCondT) -> i32 { - (*cond).seq.fetch_add(1, Ordering::Release); - let addr = &(*cond).seq as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the condvar - unsafe { syscall::futex_wake(addr, u32::MAX) }; + let seq = unsafe { &(*cond).seq }; + seq.fetch_add(1, Ordering::Release); + // SAFETY: `seq` is a live, aligned u32. + unsafe { syscall::futex_wake(seq.as_ptr(), u32::MAX) }; 0 } @@ -250,76 +472,149 @@ pub unsafe extern "C" fn pthread_cond_destroy(_cond: *mut PthreadCondT) -> i32 { } #[no_mangle] -pub unsafe extern "C" fn pthread_condattr_init(_attr: *mut PthreadCondattrT) -> i32 { 0 } +pub unsafe extern "C" fn pthread_condattr_init(attr: *mut u64) -> i32 { + unsafe { *attr = 0 }; + 0 +} #[no_mangle] -pub unsafe extern "C" fn pthread_condattr_destroy(_attr: *mut PthreadCondattrT) -> i32 { 0 } +pub unsafe extern "C" fn pthread_condattr_destroy(_attr: *mut u64) -> i32 { + 0 +} // Once +#[repr(C)] +pub struct PthreadOnceT { + state: AtomicU32, // 0 = not called, 1 = in progress, 2 = done +} + #[no_mangle] -pub unsafe extern "C" fn pthread_once( - once: *mut PthreadOnceT, init_routine: unsafe extern "C" fn(), -) -> i32 { - let state = &(*once).state; - // Already done? +pub unsafe extern "C" fn pthread_once(once: *mut PthreadOnceT, init_routine: unsafe extern "C" fn()) -> i32 { + let state = unsafe { &(*once).state }; if state.load(Ordering::Acquire) == 2 { return 0; } - // Try to be the one to run it (0 -> 1) if state.compare_exchange(0, 1, Ordering::Acquire, Ordering::Acquire).is_ok() { - init_routine(); + unsafe { init_routine() }; state.store(2, Ordering::Release); - // Wake any waiters - let addr = state as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the once control - unsafe { syscall::futex_wake(addr, u32::MAX) }; + // SAFETY: `state` is a live, aligned u32. + unsafe { syscall::futex_wake(state.as_ptr(), u32::MAX) }; return 0; } - // Someone else is running it, wait - loop { - let addr = state as *const AtomicU32 as *const u32; - // SAFETY: addr points to valid atomic state owned by the once control - unsafe { syscall::futex_wait(addr, 1, None) }; - if state.load(Ordering::Acquire) == 2 { - return 0; - } + while state.load(Ordering::Acquire) != 2 { + // SAFETY: `state` is a live, aligned u32. + unsafe { syscall::futex_wait(state.as_ptr(), 1, None) }; } + 0 } -// Thread-local storage (TLS keys) +// Keys: a process-wide table of destructors, and each thread's own values. + +const KEYS_MAX: usize = 128; +const DESTRUCTOR_ITERATIONS: usize = 4; + +type KeyDestructor = unsafe extern "C" fn(*mut u8); + +static KEY_DESTRUCTORS: [AtomicPtr<()>; KEYS_MAX] = [const { AtomicPtr::new(ptr::null_mut()) }; KEYS_MAX]; +static NEXT_KEY: AtomicU32 = AtomicU32::new(0); + +#[thread_local] +static KEY_VALUES: [Cell<*mut u8>; KEYS_MAX] = [const { Cell::new(ptr::null_mut()) }; KEYS_MAX]; #[no_mangle] -pub unsafe extern "C" fn pthread_key_create( - key: *mut PthreadKeyT, destructor: Option, -) -> i32 { +pub unsafe extern "C" fn pthread_key_create(key: *mut u32, destructor: Option) -> i32 { let k = NEXT_KEY.fetch_add(1, Ordering::Relaxed); - if k as usize >= PTHREAD_KEYS_MAX { - return -1; // EAGAIN + if k as usize >= KEYS_MAX { + return EAGAIN; } - KEY_DESTRUCTORS[k as usize] = destructor; - *key = k; + KEY_DESTRUCTORS[k as usize].store(destructor.map_or(ptr::null_mut(), |d| d as *mut ()), Ordering::Release); + unsafe { *key = k }; 0 } #[no_mangle] -pub unsafe extern "C" fn pthread_key_delete(_key: PthreadKeyT) -> i32 { +pub unsafe extern "C" fn pthread_key_delete(key: u32) -> i32 { + if key >= NEXT_KEY.load(Ordering::Relaxed).min(KEYS_MAX as u32) { + return EINVAL; + } + KEY_DESTRUCTORS[key as usize].store(ptr::null_mut(), Ordering::Release); 0 } #[no_mangle] -pub unsafe extern "C" fn pthread_getspecific(key: PthreadKeyT) -> *mut u8 { - if key as usize >= PTHREAD_KEYS_MAX { return ptr::null_mut(); } - TLS_VALUES[thread_index()][key as usize] +pub unsafe extern "C" fn pthread_getspecific(key: u32) -> *mut u8 { + KEY_VALUES.get(key as usize).map_or(ptr::null_mut(), Cell::get) +} + +#[no_mangle] +pub unsafe extern "C" fn pthread_setspecific(key: u32, value: *const u8) -> i32 { + match KEY_VALUES.get(key as usize) { + Some(slot) => { + slot.set(value as *mut u8); + 0 + } + None => EINVAL, + } +} + +/// POSIX's destructor rounds: each non-null value is cleared and its key's +/// destructor called with it, until no value is set or the rounds run out. +unsafe fn run_key_destructors() { + for _ in 0..DESTRUCTOR_ITERATIONS { + let mut ran = false; + for (slot, destructor) in KEY_VALUES.iter().zip(&KEY_DESTRUCTORS) { + let value = slot.replace(ptr::null_mut()); + let destructor = destructor.load(Ordering::Acquire); + if !value.is_null() && !destructor.is_null() { + // SAFETY: only `pthread_key_create` stores here, and it stores a destructor. + let destructor: KeyDestructor = unsafe { core::mem::transmute(destructor) }; + unsafe { destructor(value) }; + ran = true; + } + } + if !ran { + return; + } + } +} + +// C++ `thread_local` destructors, which the C++ runtime registers here. + +struct ThreadDtor { + dtor: unsafe extern "C" fn(*mut u8), + obj: *mut u8, + next: *mut ThreadDtor, } +#[thread_local] +static THREAD_DTORS: Cell<*mut ThreadDtor> = Cell::new(ptr::null_mut()); + #[no_mangle] -pub unsafe extern "C" fn pthread_setspecific(key: PthreadKeyT, value: *const u8) -> i32 { - if key as usize >= PTHREAD_KEYS_MAX { return -1; } - TLS_VALUES[thread_index()][key as usize] = value as *mut u8; +pub unsafe extern "C" fn __cxa_thread_atexit_impl( + dtor: unsafe extern "C" fn(*mut u8), + obj: *mut u8, + _dso_symbol: *mut u8, +) -> i32 { + let node = Box::into_raw(Box::new(ThreadDtor { dtor, obj, next: THREAD_DTORS.get() })); + THREAD_DTORS.set(node); 0 } +/// Run the calling thread's `thread_local` destructors, the last registered +/// first, including any a destructor registers. +pub(crate) unsafe fn run_thread_dtors() { + loop { + let node = THREAD_DTORS.get(); + if node.is_null() { + return; + } + let node = unsafe { Box::from_raw(node) }; + THREAD_DTORS.set(node.next); + unsafe { (node.dtor)(node.obj) }; + } +} + // RWLock (simple: wraps mutex, no reader parallelism) #[repr(C)] @@ -328,48 +623,73 @@ pub struct PthreadRwlockT { } #[no_mangle] -pub unsafe extern "C" fn pthread_rwlock_init( - rwlock: *mut PthreadRwlockT, _attr: *const u8, -) -> i32 { - pthread_mutex_init(&mut (*rwlock).mutex, ptr::null()) +pub unsafe extern "C" fn pthread_rwlock_init(rwlock: *mut PthreadRwlockT, _attr: *const u8) -> i32 { + unsafe { pthread_mutex_init(ptr::addr_of_mut!((*rwlock).mutex), ptr::null()) } } #[no_mangle] pub unsafe extern "C" fn pthread_rwlock_rdlock(rwlock: *mut PthreadRwlockT) -> i32 { - pthread_mutex_lock(&mut (*rwlock).mutex) + unsafe { pthread_mutex_lock(ptr::addr_of_mut!((*rwlock).mutex)) } } #[no_mangle] pub unsafe extern "C" fn pthread_rwlock_wrlock(rwlock: *mut PthreadRwlockT) -> i32 { - pthread_mutex_lock(&mut (*rwlock).mutex) + unsafe { pthread_mutex_lock(ptr::addr_of_mut!((*rwlock).mutex)) } } #[no_mangle] pub unsafe extern "C" fn pthread_rwlock_unlock(rwlock: *mut PthreadRwlockT) -> i32 { - pthread_mutex_unlock(&mut (*rwlock).mutex) + unsafe { pthread_mutex_unlock(ptr::addr_of_mut!((*rwlock).mutex)) } } #[no_mangle] pub unsafe extern "C" fn pthread_rwlock_destroy(rwlock: *mut PthreadRwlockT) -> i32 { - pthread_mutex_destroy(&mut (*rwlock).mutex) + unsafe { pthread_mutex_destroy(ptr::addr_of_mut!((*rwlock).mutex)) } } -// Attr stubs +// Attributes + +fn attr_default() -> u64 { + STACK_DEFAULT as u64 +} #[no_mangle] -pub unsafe extern "C" fn pthread_attr_init(_attr: *mut u8) -> i32 { 0 } +pub unsafe extern "C" fn pthread_attr_init(attr: *mut u64) -> i32 { + unsafe { *attr = attr_default() }; + 0 +} #[no_mangle] -pub unsafe extern "C" fn pthread_attr_destroy(_attr: *mut u8) -> i32 { 0 } +pub unsafe extern "C" fn pthread_attr_destroy(_attr: *mut u64) -> i32 { + 0 +} #[no_mangle] -pub unsafe extern "C" fn pthread_attr_setstacksize(_attr: *mut u8, _size: usize) -> i32 { 0 } +pub unsafe extern "C" fn pthread_attr_setstacksize(attr: *mut u64, size: usize) -> i32 { + if size < STACK_MIN { + return EINVAL; + } + // A size no allocation can have is refused here, not in `pthread_create`. + let Some(size) = size.checked_next_multiple_of(STACK_ALIGN).filter(|&s| s <= isize::MAX as usize) else { + return EINVAL; + }; + unsafe { *attr = size as u64 | (*attr & ATTR_DETACHED) }; + 0 +} #[no_mangle] -pub unsafe extern "C" fn pthread_attr_getstacksize(_attr: *const u8, size: *mut usize) -> i32 { - if !size.is_null() { *size = THREAD_STACK_SIZE; } +pub unsafe extern "C" fn pthread_attr_getstacksize(attr: *const u64, size: *mut usize) -> i32 { + unsafe { *size = (*attr & !ATTR_DETACHED) as usize }; 0 } #[no_mangle] -pub unsafe extern "C" fn pthread_attr_setdetachstate(_attr: *mut u8, _state: i32) -> i32 { 0 } +pub unsafe extern "C" fn pthread_attr_setdetachstate(attr: *mut u64, state: i32) -> i32 { + let detached = match state { + 0 => 0, + 1 => ATTR_DETACHED, + _ => return EINVAL, + }; + unsafe { *attr = (*attr & !ATTR_DETACHED) | detached }; + 0 +} diff --git a/userland/libc/src/socket.rs b/userland/libc/src/socket.rs index c98a7514766..5059bb5e1c5 100644 --- a/userland/libc/src/socket.rs +++ b/userland/libc/src/socket.rs @@ -7,6 +7,8 @@ use toyos_abi::RawHandle; use toyos_abi::syscall; use toyos::net::{NetError, TcpSocketId, UdpSocketId, OPT_NODELAY}; +use crate::errno::{EADDRINUSE, EAFNOSUPPORT, EBADF, ECONNREFUSED, ECONNRESET, EINVAL, EIO, ENOMEM, ENOTCONN, ETIMEDOUT}; + // C types matching POSIX type SocklenT = u32; @@ -53,17 +55,6 @@ const SOL_SOCKET: i32 = 1; const SO_ERROR: i32 = 4; const TCP_NODELAY: i32 = 1; -const EINVAL: i32 = 22; -const EBADF: i32 = 9; -const ENOMEM: i32 = 12; -const EAFNOSUPPORT: i32 = 97; -const ECONNREFUSED: i32 = 111; -const ECONNRESET: i32 = 104; -const ETIMEDOUT: i32 = 110; -const EADDRINUSE: i32 = 98; -const ENOTCONN: i32 = 107; -const EIO: i32 = 5; - // Internal socket table #[derive(Clone, Copy)] diff --git a/userland/libc/src/stdio.rs b/userland/libc/src/stdio.rs index f2c692e7f6e..129467175d8 100644 --- a/userland/libc/src/stdio.rs +++ b/userland/libc/src/stdio.rs @@ -102,8 +102,15 @@ pub struct FILE { owned: bool, /// Next stream in the open list, for `fflush(NULL)`. next: *mut FILE, + /// Bytes `ungetc` pushed back; the last pushed is read first. + unget: [u8; UNGET_MAX], + unget_len: usize, } +/// How many bytes `ungetc` holds: one multibyte character, which is what +/// libc++'s standard input pushes back after a peek. +const UNGET_MAX: usize = 4; + const STDIN_FD: i32 = 0; const STDOUT_FD: i32 = 1; const STDERR_FD: i32 = 2; @@ -117,17 +124,17 @@ static mut STDOUT_BUF: [u8; BUFSIZ] = [0; BUFSIZ]; static mut STDOUT_FILE: FILE = FILE { fd: STDOUT_FD, eof: false, error: false, mode: MODE_UNSET, buf: (&raw mut STDOUT_BUF) as *mut u8, cap: BUFSIZ, len: 0, - owned: false, next: &raw mut STDERR_FILE, + owned: false, next: &raw mut STDERR_FILE, unget: [0; UNGET_MAX], unget_len: 0, }; static mut STDERR_FILE: FILE = FILE { fd: STDERR_FD, eof: false, error: false, mode: IONBF, buf: ptr::null_mut(), cap: 0, len: 0, - owned: false, next: ptr::null_mut(), + owned: false, next: ptr::null_mut(), unget: [0; UNGET_MAX], unget_len: 0, }; static mut STDIN_FILE: FILE = FILE { fd: STDIN_FD, eof: false, error: false, mode: IONBF, buf: ptr::null_mut(), cap: 0, len: 0, - owned: false, next: ptr::null_mut(), + owned: false, next: ptr::null_mut(), unget: [0; UNGET_MAX], unget_len: 0, }; /// Head of the open-stream list. `fflush(NULL)` walks it, which is what makes @@ -372,6 +379,8 @@ unsafe fn new_stream(fd: i32) -> FILE { len: 0, owned: !buf.is_null(), next: ptr::null_mut(), + unget: [0; UNGET_MAX], + unget_len: 0, } } @@ -403,6 +412,13 @@ pub unsafe extern "C" fn fread(buf: *mut u8, size: usize, count: usize, f: *mut unsafe { sync_before_fd_use(f); } let slice = unsafe { core::slice::from_raw_parts_mut(buf, total) }; let mut read_so_far = 0; + while read_so_far < total && unsafe { (*f).unget_len } > 0 { + unsafe { + (*f).unget_len -= 1; + slice[read_so_far] = (*f).unget[(*f).unget_len]; + } + read_so_far += 1; + } while read_so_far < total { let n = sys_read(unsafe { (*f).fd }, &mut slice[read_so_far..]); if n <= 0 { @@ -428,7 +444,7 @@ pub unsafe extern "C" fn fwrite(buf: *const u8, size: usize, count: usize, f: *m #[no_mangle] pub unsafe extern "C" fn fseek(f: *mut FILE, offset: i64, whence: i32) -> i32 { if f.is_null() { return -1; } - unsafe { sync_before_fd_use(f); (*f).eof = false; } + unsafe { sync_before_fd_use(f); (*f).eof = false; (*f).unget_len = 0; } if sys_seek(unsafe { (*f).fd }, offset, whence) >= 0 { 0 } else { -1 } } @@ -438,7 +454,8 @@ pub unsafe extern "C" fn ftell(f: *mut FILE) -> i64 { // Pending bytes have not moved the fd's offset yet, so reporting it now // would be short by exactly what is still in the buffer. unsafe { sync_before_fd_use(f); } - sys_seek(unsafe { (*f).fd }, 0, 1) // SEEK_CUR + let at = sys_seek(unsafe { (*f).fd }, 0, 1); // SEEK_CUR + if at < 0 { at } else { at - unsafe { (*f).unget_len } as i64 } } #[no_mangle] @@ -568,7 +585,28 @@ pub unsafe extern "C" fn puts(s: *const u8) -> i32 { } #[no_mangle] -pub unsafe extern "C" fn ungetc(_c: i32, _f: *mut FILE) -> i32 { -1 } +pub unsafe extern "C" fn ungetc(c: i32, f: *mut FILE) -> i32 { + if c == -1 || f.is_null() || !unsafe { unget(f, &[c as u8]) } { + return -1; + } + i32::from(c as u8) +} + +/// Push `bytes` back onto `f`, to be read in their order: all of them, or +/// none when they do not fit. +pub(crate) unsafe fn unget(f: *mut FILE, bytes: &[u8]) -> bool { + unsafe { + if (*f).unget_len + bytes.len() > UNGET_MAX { + return false; + } + for &b in bytes.iter().rev() { + (*f).unget[(*f).unget_len] = b; + (*f).unget_len += 1; + } + (*f).eof = false; + } + true +} // File operations diff --git a/userland/libc/src/string.rs b/userland/libc/src/string.rs index 4d204d64327..72a21972ab0 100644 --- a/userland/libc/src/string.rs +++ b/userland/libc/src/string.rs @@ -176,6 +176,19 @@ pub unsafe extern "C" fn strerror(_errnum: i32) -> *const u8 { b"unknown error\0".as_ptr() } +/// POSIX's `strerror_r`: `strerror`'s text in `buf`, or `ERANGE` when it does +/// not fit with its terminator. +#[no_mangle] +pub unsafe extern "C" fn strerror_r(errnum: i32, buf: *mut u8, buflen: usize) -> i32 { + let text = strerror(errnum); + let len = strlen(text); + if len >= buflen { + return crate::errno::ERANGE; + } + ptr::copy_nonoverlapping(text, buf, len + 1); + 0 +} + #[no_mangle] pub unsafe extern "C" fn strspn(s: *const u8, accept: *const u8) -> usize { let mut count = 0; diff --git a/userland/libc/src/strtonum.rs b/userland/libc/src/strtonum.rs new file mode 100644 index 00000000000..29d13daa62f --- /dev/null +++ b/userland/libc/src/strtonum.rs @@ -0,0 +1,368 @@ +//! The one reader of numbers out of C strings, narrow and wide: the grammar of +//! `strtol` and `strtod`, their `endptr`, and their `ERANGE`. A decimal +//! floating-point number is rounded by `core`'s parser, correctly. It reads +//! and sets nothing but what it is handed, so the host tests it +//! (`toyos-libc-copies`). + +use alloc::vec::Vec; + +/// A number read out of a C string. +pub(crate) struct Read { + pub(crate) value: T, + /// How many code units it spans: 0 when there is no number. + pub(crate) end: usize, + pub(crate) refused: Option, +} + +/// What C's readers tell `errno`. +pub(crate) enum Refusal { + /// A base C does not define: `EINVAL`. + Base, + /// A value outside the type: `ERANGE`. + Range, +} + +/// A code unit of a C string: `char` or `wchar_t`. +pub(crate) trait Unit: Copy { + fn code(self) -> u32; +} + +impl Unit for u8 { + fn code(self) -> u32 { + u32::from(self) + } +} + +impl Unit for crate::arch::WChar { + // `WChar` is `i32` on x86-64 and `u32` on AArch64. + #[allow(clippy::unnecessary_cast)] + fn code(self) -> u32 { + self as u32 + } +} + +fn is_space(c: u32) -> bool { + c == 0x20 || (0x09..=0x0d).contains(&c) +} + +fn digit(c: u32) -> Option { + match c { + 0x30..=0x39 => Some(c - 0x30), + 0x41..=0x5a => Some(c - 0x41 + 10), + 0x61..=0x7a => Some(c - 0x61 + 10), + _ => None, + } +} + +fn lower(c: u32) -> u32 { + if (0x41..=0x5a).contains(&c) { c + 0x20 } else { c } +} + +/// The code unit `i` places into `s`. +unsafe fn at(s: *const U, i: usize) -> u32 { + unsafe { *s.add(i) }.code() +} + +/// Whether `s + i` begins with `word`, ignoring ASCII case. +unsafe fn starts(s: *const U, i: usize, word: &[u8]) -> bool { + word.iter().enumerate().all(|(k, &b)| unsafe { lower(at(s, i + k)) } == u32::from(b)) +} + +/// Past the whitespace and the sign: where the number starts, and whether it +/// is negative. +unsafe fn lead(s: *const U) -> (usize, bool) { + let mut i = 0; + while is_space(unsafe { at(s, i) }) { + i += 1; + } + match unsafe { at(s, i) } { + 0x2d => (i + 1, true), + 0x2b => (i + 1, false), + _ => (i, false), + } +} + +/// An integer as `strtol`'s grammar reads it. +pub(crate) struct Int { + pub(crate) negative: bool, + /// Its magnitude, `None` when it does not fit in 64 bits. + pub(crate) magnitude: Option, + /// How many code units it spans: 0 when there is no number. + pub(crate) end: usize, +} + +/// Read an integer in `base` (0 for C's prefixes) from `s`. `None` is a base +/// C refuses. +pub(crate) unsafe fn int(s: *const U, base: i32) -> Option { + if base < 0 || base == 1 || base > 36 { + return None; + } + let (mut i, negative) = unsafe { lead(s) }; + let mut base = base as u32; + let hex_prefix = unsafe { at(s, i) == 0x30 && lower(at(s, i + 1)) == 0x78 && digit(at(s, i + 2)).is_some_and(|d| d < 16) }; + if (base == 0 || base == 16) && hex_prefix { + base = 16; + i += 2; + } else if base == 0 { + base = if unsafe { at(s, i) } == 0x30 { 8 } else { 10 }; + } + let start = i; + let mut magnitude = Some(0u64); + while let Some(d) = digit(unsafe { at(s, i) }).filter(|&d| d < base) { + magnitude = magnitude.and_then(|m| m.checked_mul(u64::from(base))).and_then(|m| m.checked_add(u64::from(d))); + i += 1; + } + let end = if i == start { 0 } else { i }; + Some(Int { negative, magnitude, end }) +} + +/// `strtol`'s answer. +pub(crate) unsafe fn signed(s: *const U, base: i32) -> Read { + let Some(n) = (unsafe { int(s, base) }) else { + return Read { value: 0, end: 0, refused: Some(Refusal::Base) }; + }; + let (value, refused) = match (n.magnitude, n.negative) { + (Some(m), false) if i64::try_from(m).is_ok() => (m as i64, None), + (Some(m), true) if m <= i64::MIN.unsigned_abs() => ((m as i64).wrapping_neg(), None), + (_, negative) => (if negative { i64::MIN } else { i64::MAX }, Some(Refusal::Range)), + }; + Read { value, end: n.end, refused } +} + +/// `strtoul`'s answer: a negative number is negated in the unsigned type, as +/// C says. +pub(crate) unsafe fn unsigned(s: *const U, base: i32) -> Read { + let Some(n) = (unsafe { int(s, base) }) else { + return Read { value: 0, end: 0, refused: Some(Refusal::Base) }; + }; + let (value, refused) = match n.magnitude { + Some(m) if n.negative => (m.wrapping_neg(), None), + Some(m) => (m, None), + None => (u64::MAX, Some(Refusal::Range)), + }; + Read { value, end: n.end, refused } +} + +/// The IEEE formats `strtof` and `strtod` round to. +pub(crate) trait Float: Copy + core::str::FromStr + core::ops::Neg { + const INFINITY: Self; + const NAN: Self; + /// Significand bits, the leading one included. + const PRECISION: u32; + const MIN_EXP: i64; + const MAX_EXP: i64; + fn from_parts(biased_exponent: u64, fraction: u64) -> Self; + fn is_zero_or_subnormal(self) -> bool; + fn is_infinite(self) -> bool; +} + +impl Float for f64 { + const INFINITY: Self = f64::INFINITY; + const NAN: Self = f64::NAN; + const PRECISION: u32 = 53; + const MIN_EXP: i64 = -1022; + const MAX_EXP: i64 = 1023; + fn from_parts(biased_exponent: u64, fraction: u64) -> Self { + f64::from_bits(biased_exponent << 52 | fraction) + } + fn is_zero_or_subnormal(self) -> bool { + !self.is_normal() && self.is_finite() + } + fn is_infinite(self) -> bool { + f64::is_infinite(self) + } +} + +impl Float for f32 { + const INFINITY: Self = f32::INFINITY; + const NAN: Self = f32::NAN; + const PRECISION: u32 = 24; + const MIN_EXP: i64 = -126; + const MAX_EXP: i64 = 127; + fn from_parts(biased_exponent: u64, fraction: u64) -> Self { + f32::from_bits((biased_exponent << 23 | fraction) as u32) + } + fn is_zero_or_subnormal(self) -> bool { + !self.is_normal() && self.is_finite() + } + fn is_infinite(self) -> bool { + f32::is_infinite(self) + } +} + +/// Read a floating-point number from `s` as `strtod` does, rounded to `F`. +pub(crate) unsafe fn float(s: *const U) -> Read { + let (i, negative) = unsafe { lead(s) }; + let sign = |x: F| if negative { -x } else { x }; + let read = |value: F, end: usize| Read { value, end, refused: None }; + if unsafe { starts(s, i, b"infinity") } { + return read(sign(F::INFINITY), i + 8); + } + if unsafe { starts(s, i, b"inf") } { + return read(sign(F::INFINITY), i + 3); + } + if unsafe { starts(s, i, b"nan") } { + let mut end = i + 3; + if unsafe { at(s, end) } == 0x28 { + let mut k = end + 1; + while unsafe { at(s, k) }.try_into().is_ok_and(|c: u8| c.is_ascii_alphanumeric() || c == b'_') { + k += 1; + } + if unsafe { at(s, k) } == 0x29 { + end = k + 1; + } + } + return read(sign(F::NAN), end); + } + let hex = unsafe { + at(s, i) == 0x30 + && lower(at(s, i + 1)) == 0x78 + && (digit(at(s, i + 2)).is_some_and(|d| d < 16) + || (at(s, i + 2) == 0x2e && digit(at(s, i + 3)).is_some_and(|d| d < 16))) + }; + let (value, end, nonzero) = if hex { unsafe { hex_float::(s, i + 2) } } else { unsafe { decimal::(s, i) } }; + if end == 0 { + return read(F::from_parts(0, 0), 0); + } + let range = value.is_infinite() || (nonzero && value.is_zero_or_subnormal()); + Read { value: sign(value), end, refused: range.then_some(Refusal::Range) } +} + +/// A decimal number at `s + i`: its value, its end (0 if none), and whether +/// any digit of it is nonzero. +unsafe fn decimal(s: *const U, mut i: usize) -> (F, usize, bool) { + let mut text = Vec::new(); + let mut digits = 0; + let mut nonzero = false; + let mut take = |text: &mut Vec, c: u32| { + text.push(c as u8); + nonzero |= c != 0x30; + }; + while digit(unsafe { at(s, i) }).is_some_and(|d| d < 10) { + take(&mut text, unsafe { at(s, i) }); + digits += 1; + i += 1; + } + if unsafe { at(s, i) } == 0x2e { + text.push(b'.'); + i += 1; + while digit(unsafe { at(s, i) }).is_some_and(|d| d < 10) { + take(&mut text, unsafe { at(s, i) }); + digits += 1; + i += 1; + } + } + if digits == 0 { + return (F::NAN, 0, false); + } + if lower(unsafe { at(s, i) }) == 0x65 { + let mut k = i + 1; + let sign = unsafe { at(s, k) }; + if sign == 0x2b || sign == 0x2d { + k += 1; + } + if digit(unsafe { at(s, k) }).is_some_and(|d| d < 10) { + text.push(b'e'); + if sign == 0x2d { + text.push(b'-'); + } + while digit(unsafe { at(s, k) }).is_some_and(|d| d < 10) { + text.push(unsafe { at(s, k) } as u8); + k += 1; + } + i = k; + } + } + let text = core::str::from_utf8(&text).expect("only ASCII digits, '.', 'e' and '-' were taken"); + let value = text.parse::().unwrap_or_else(|_| unreachable!("{text} is decimal-float syntax")); + (value, i, nonzero) +} + +/// A hexadecimal number whose digits start at `s + i`, rounded to nearest, +/// ties to even: its value, its end, and whether any digit is nonzero. +unsafe fn hex_float(s: *const U, mut i: usize) -> (F, usize, bool) { + let mut mantissa = 0u64; + let mut exponent = 0i64; + let mut sticky = false; + let mut add = |d: u32, after_point: bool, mantissa: &mut u64, exponent: &mut i64| { + if *mantissa >> 60 == 0 { + *mantissa = *mantissa << 4 | u64::from(d); + if after_point { + *exponent -= 4; + } + } else { + sticky |= d != 0; + if !after_point { + *exponent += 4; + } + } + }; + while let Some(d) = digit(unsafe { at(s, i) }).filter(|&d| d < 16) { + add(d, false, &mut mantissa, &mut exponent); + i += 1; + } + if unsafe { at(s, i) } == 0x2e { + i += 1; + while let Some(d) = digit(unsafe { at(s, i) }).filter(|&d| d < 16) { + add(d, true, &mut mantissa, &mut exponent); + i += 1; + } + } + if lower(unsafe { at(s, i) }) == 0x70 { + let mut k = i + 1; + let negative = unsafe { at(s, k) } == 0x2d; + if negative || unsafe { at(s, k) } == 0x2b { + k += 1; + } + if digit(unsafe { at(s, k) }).is_some_and(|d| d < 10) { + let mut power = 0i64; + while let Some(d) = digit(unsafe { at(s, k) }).filter(|&d| d < 10) { + power = power.saturating_mul(10).saturating_add(i64::from(d)); + k += 1; + } + exponent = exponent.saturating_add(if negative { -power } else { power }); + i = k; + } + } + let nonzero = mantissa != 0 || sticky; + (round::(mantissa, exponent, sticky), i, nonzero) +} + +/// `mantissa * 2^exponent`, plus less than one unit of `mantissa` when +/// `sticky`, rounded to `F`. +fn round(mantissa: u64, exponent: i64, sticky: bool) -> F { + if mantissa == 0 { + return F::from_parts(0, 0); + } + let shift = mantissa.leading_zeros(); + let mantissa = u128::from(mantissa << shift) << 64; + // The value is 1.f * 2^top, with the leading one at bit 127 of `mantissa`. + let top = exponent.saturating_add(63 - i64::from(shift)); + if top > F::MAX_EXP { + return F::INFINITY; + } + // Bits below the kept significand: the precision's, and as many more as a + // subnormal gives up. + let dropped = 128 - i64::from(F::PRECISION) + (F::MIN_EXP - top).max(0); + if dropped > 128 { + return F::from_parts(0, 0); + } + let dropped = dropped as u32; + let kept = if dropped == 128 { 0 } else { mantissa >> dropped }; + let rest = if dropped == 128 { mantissa } else { mantissa & ((1u128 << dropped) - 1) }; + // `sticky` bits lie below every bit of `rest`, which ends at bit 64. + let half = 1u128 << (dropped - 1); + let up = rest > half || (rest == half && (sticky || kept & 1 == 1)); + let kept = kept as u64 + u64::from(up); + let fraction_bits = F::PRECISION - 1; + if top < F::MIN_EXP { + // Subnormal, or rounded up into the smallest normal. + return F::from_parts(kept >> fraction_bits, kept & ((1 << fraction_bits) - 1)); + } + let (kept, top) = if kept >> F::PRECISION != 0 { (kept >> 1, top + 1) } else { (kept, top) }; + if top > F::MAX_EXP { + return F::INFINITY; + } + let biased = (top - F::MIN_EXP + 1) as u64; + F::from_parts(biased, kept & ((1 << fraction_bits) - 1)) +} diff --git a/userland/libc/src/time.rs b/userland/libc/src/time.rs index 48fc0e38125..202bacffcfe 100644 --- a/userland/libc/src/time.rs +++ b/userland/libc/src/time.rs @@ -39,7 +39,7 @@ pub struct Timezone { pub tz_dsttime: i32, } -const CLOCK_REALTIME: i32 = 0; +pub(crate) const CLOCK_REALTIME: i32 = 0; const CLOCK_MONOTONIC: i32 = 1; /// POSIX's own answer for a machine that cannot tell the time: `time()` returns diff --git a/userland/libc/src/utf8.rs b/userland/libc/src/utf8.rs new file mode 100644 index 00000000000..d49542b0e4a --- /dev/null +++ b/userland/libc/src/utf8.rs @@ -0,0 +1,75 @@ +//! UTF-8 read one byte at a time, the one locale's multibyte encoding +//! (`wchar.rs`). **A sequence is refused at the first byte after which no +//! continuation makes it a Unicode scalar value in its shortest form**, as C's +//! `(size_t)-1` says; a prefix some continuation completes is `(size_t)-2`'s. + +/// `mbstate_t`: the bits of a code point so far, the continuation bytes it +/// still needs in the low byte of `needed`, and its whole length above it. +#[repr(C)] +pub struct MbState { + bits: u32, + needed: u32, +} + +/// What one byte does to the character being read. +pub(crate) enum Byte { + /// It ends a character: its code point. + Ends(u32), + /// It leaves a prefix some continuation completes. + Continues, + /// No continuation completes what was read; the state is initial again. + Refused, +} + +/// The smallest code point a sequence of `len` bytes may encode. +fn shortest(len: u32) -> u32 { + match len { + 2 => 0x80, + 3 => 0x800, + _ => 0x10000, + } +} + +impl MbState { + pub(crate) const INITIAL: MbState = MbState { bits: 0, needed: 0 }; + + /// Whether a character has been begun and not ended. + pub(crate) fn is_partial(&self) -> bool { + self.needed != 0 + } + + pub(crate) fn feed(&mut self, b: u8) -> Byte { + let b = u32::from(b); + let (bits, len, left) = if self.needed == 0 { + match b { + 0x00..=0x7f => return Byte::Ends(b), + 0xc2..=0xdf => (b & 0x1f, 2, 1), + 0xe0..=0xef => (b & 0x0f, 3, 2), + 0xf0..=0xf4 => (b & 0x07, 4, 3), + _ => return self.refuse(), + } + } else if b & 0xc0 != 0x80 { + return self.refuse(); + } else { + (self.bits << 6 | (b & 0x3f), self.needed >> 8, (self.needed & 0xff) - 1) + }; + // Every code point the bytes so far can still become: `low` to `high`. + let span = 6 * left; + let low = bits << span; + let high = low | ((1 << span) - 1); + if high < shortest(len) || low > 0x10ffff || (low >= 0xd800 && high <= 0xdfff) { + return self.refuse(); + } + if left == 0 { + *self = MbState::INITIAL; + return Byte::Ends(bits); + } + *self = MbState { bits, needed: len << 8 | left }; + Byte::Continues + } + + fn refuse(&mut self) -> Byte { + *self = MbState::INITIAL; + Byte::Refused + } +} diff --git a/userland/libc/src/wchar.rs b/userland/libc/src/wchar.rs new file mode 100644 index 00000000000..60a64a4520e --- /dev/null +++ b/userland/libc/src/wchar.rs @@ -0,0 +1,734 @@ +//! Wide characters, and the UTF-8 the one locale (`locale.rs`) encodes them +//! in. A multibyte conversion refuses what is not UTF-8 (`EILSEQ`): an +//! overlong form, a surrogate, a code point above U+10FFFF (`utf8.rs`). Only +//! reading UTF-8 has a state; writing it has none. The character classes are +//! the C locale's, which are ASCII's. + +use core::ffi::VaList; +use core::ptr; + +use crate::arch::WChar; +use crate::errno::{self, EILSEQ}; +use crate::locale::Locale; +use crate::strtonum; +use crate::utf8::{Byte, MbState}; + +type WInt = i32; +const WEOF: WInt = -1; +const EOF: i32 = -1; + +/// `(size_t)-1`: a conversion refused. +const INVALID: usize = usize::MAX; +/// `(size_t)-2`: a character cut short, its bytes so far in the state. +const INCOMPLETE: usize = usize::MAX - 1; + +/// The state a caller that passes none uses, one per function as C says. +struct Internal(core::cell::UnsafeCell); +unsafe impl Sync for Internal {} + +static MBRTOWC_STATE: Internal = Internal(core::cell::UnsafeCell::new(MbState::INITIAL)); +static MBRLEN_STATE: Internal = Internal(core::cell::UnsafeCell::new(MbState::INITIAL)); +static MBSRTOWCS_STATE: Internal = Internal(core::cell::UnsafeCell::new(MbState::INITIAL)); + +fn state_or(ps: *mut MbState, internal: &'static Internal) -> *mut MbState { + if ps.is_null() { internal.0.get() } else { ps } +} + +#[no_mangle] +pub unsafe extern "C" fn mbrtowc(pwc: *mut WChar, s: *const u8, n: usize, ps: *mut MbState) -> usize { + let st = unsafe { &mut *state_or(ps, &MBRTOWC_STATE) }; + if s.is_null() { + if st.is_partial() { + *st = MbState::INITIAL; + errno::set(EILSEQ); + return INVALID; + } + return 0; + } + for i in 0..n { + match st.feed(unsafe { *s.add(i) }) { + Byte::Continues => {} + Byte::Refused => { + errno::set(EILSEQ); + return INVALID; + } + Byte::Ends(cp) => { + if !pwc.is_null() { + unsafe { *pwc = cp as WChar }; + } + return if cp == 0 { 0 } else { i + 1 }; + } + } + } + INCOMPLETE +} + +/// `wc`'s UTF-8 bytes into `out`, and how many; `None` for what is not a +/// Unicode scalar value. +pub(crate) fn encode(wc: WChar, out: &mut [u8; 4]) -> Option { + Some(char::from_u32(wc as u32)?.encode_utf8(out).len()) +} + +#[no_mangle] +pub unsafe extern "C" fn wcrtomb(s: *mut u8, wc: WChar, _ps: *mut MbState) -> usize { + let mut buf = [0u8; 4]; + let wc = if s.is_null() { 0 } else { wc }; + let Some(len) = encode(wc, &mut buf) else { + errno::set(EILSEQ); + return INVALID; + }; + if !s.is_null() { + unsafe { ptr::copy_nonoverlapping(buf.as_ptr(), s, len) }; + } + len +} + +#[no_mangle] +pub unsafe extern "C" fn mbrlen(s: *const u8, n: usize, ps: *mut MbState) -> usize { + unsafe { mbrtowc(ptr::null_mut(), s, n, state_or(ps, &MBRLEN_STATE)) } +} + +#[no_mangle] +pub unsafe extern "C" fn mbsinit(ps: *const MbState) -> i32 { + (ps.is_null() || !unsafe { &*ps }.is_partial()) as i32 +} + +#[no_mangle] +pub unsafe extern "C" fn btowc(c: i32) -> WInt { + if (0..0x80).contains(&c) { c } else { WEOF } +} + +#[no_mangle] +pub unsafe extern "C" fn wctob(c: WInt) -> i32 { + if (0..0x80).contains(&c) { c } else { EOF } +} + +#[no_mangle] +pub unsafe extern "C" fn mbtowc(pwc: *mut WChar, s: *const u8, n: usize) -> i32 { + if s.is_null() { + return 0; + } + let mut st = MbState::INITIAL; + match unsafe { mbrtowc(pwc, s, n, &mut st) } { + INVALID | INCOMPLETE => { + errno::set(EILSEQ); + -1 + } + len => len as i32, + } +} + +#[no_mangle] +pub unsafe extern "C" fn mblen(s: *const u8, n: usize) -> i32 { + unsafe { mbtowc(ptr::null_mut(), s, n) } +} + +#[no_mangle] +pub unsafe extern "C" fn wctomb(s: *mut u8, wc: WChar) -> i32 { + if s.is_null() { + return 0; + } + let mut st = MbState::INITIAL; + match unsafe { wcrtomb(s, wc, &mut st) } { + INVALID => -1, + len => len as i32, + } +} + +#[no_mangle] +pub unsafe extern "C" fn mbsnrtowcs( + dst: *mut WChar, + src: *mut *const u8, + nms: usize, + len: usize, + ps: *mut MbState, +) -> usize { + let st = state_or(ps, &MBSRTOWCS_STATE); + let mut p = unsafe { *src }; + let mut left = nms; + let mut count = 0; + while dst.is_null() || count < len { + if left == 0 { + break; + } + let mut wc: WChar = 0; + match unsafe { mbrtowc(&mut wc, p, left, st) } { + 0 => { + if !dst.is_null() { + unsafe { + *dst.add(count) = 0; + *src = ptr::null(); + } + } + return count; + } + INVALID => { + if !dst.is_null() { + unsafe { *src = p }; + } + return INVALID; + } + INCOMPLETE => { + p = unsafe { p.add(left) }; + break; + } + used => { + if !dst.is_null() { + unsafe { *dst.add(count) = wc }; + } + count += 1; + p = unsafe { p.add(used) }; + left -= used; + } + } + } + if !dst.is_null() { + unsafe { *src = p }; + } + count +} + +#[no_mangle] +pub unsafe extern "C" fn mbsrtowcs(dst: *mut WChar, src: *mut *const u8, len: usize, ps: *mut MbState) -> usize { + unsafe { mbsnrtowcs(dst, src, usize::MAX, len, ps) } +} + +#[no_mangle] +pub unsafe extern "C" fn mbstowcs(dst: *mut WChar, src: *const u8, n: usize) -> usize { + let mut src = src; + let mut st = MbState::INITIAL; + unsafe { mbsnrtowcs(dst, &mut src, usize::MAX, n, &mut st) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsnrtombs( + dst: *mut u8, + src: *mut *const WChar, + nwc: usize, + len: usize, + _ps: *mut MbState, +) -> usize { + let mut p = unsafe { *src }; + let mut count = 0; + for _ in 0..nwc { + let wc = unsafe { *p }; + let mut buf = [0u8; 4]; + let Some(n) = encode(wc, &mut buf) else { + if !dst.is_null() { + unsafe { *src = p }; + } + errno::set(EILSEQ); + return INVALID; + }; + if !dst.is_null() { + if count + n > len { + break; + } + unsafe { ptr::copy_nonoverlapping(buf.as_ptr(), dst.add(count), n) }; + } + if wc == 0 { + if !dst.is_null() { + unsafe { *src = ptr::null() }; + } + return count; + } + count += n; + p = unsafe { p.add(1) }; + } + if !dst.is_null() { + unsafe { *src = p }; + } + count +} + +#[no_mangle] +pub unsafe extern "C" fn wcsrtombs(dst: *mut u8, src: *mut *const WChar, len: usize, ps: *mut MbState) -> usize { + unsafe { wcsnrtombs(dst, src, usize::MAX, len, ps) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstombs(dst: *mut u8, src: *const WChar, n: usize) -> usize { + let mut src = src; + let mut st = MbState::INITIAL; + unsafe { wcsnrtombs(dst, &mut src, usize::MAX, n, &mut st) } +} + +// Wide strings + +#[no_mangle] +pub unsafe extern "C" fn wcslen(s: *const WChar) -> usize { + let mut n = 0; + while unsafe { *s.add(n) } != 0 { + n += 1; + } + n +} + +#[no_mangle] +pub unsafe extern "C" fn wcscpy(dst: *mut WChar, src: *const WChar) -> *mut WChar { + unsafe { wmemcpy(dst, src, wcslen(src) + 1) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsncpy(dst: *mut WChar, src: *const WChar, n: usize) -> *mut WChar { + let mut i = 0; + while i < n && unsafe { *src.add(i) } != 0 { + unsafe { *dst.add(i) = *src.add(i) }; + i += 1; + } + unsafe { wmemset(dst.add(i), 0, n - i) }; + dst +} + +#[no_mangle] +pub unsafe extern "C" fn wcscat(dst: *mut WChar, src: *const WChar) -> *mut WChar { + unsafe { wcscpy(dst.add(wcslen(dst)), src) }; + dst +} + +#[no_mangle] +pub unsafe extern "C" fn wcsncat(dst: *mut WChar, src: *const WChar, n: usize) -> *mut WChar { + let end = unsafe { dst.add(wcslen(dst)) }; + let mut i = 0; + while i < n && unsafe { *src.add(i) } != 0 { + unsafe { *end.add(i) = *src.add(i) }; + i += 1; + } + unsafe { *end.add(i) = 0 }; + dst +} + +#[no_mangle] +pub unsafe extern "C" fn wcscmp(a: *const WChar, b: *const WChar) -> i32 { + unsafe { wcsncmp(a, b, usize::MAX) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsncmp(a: *const WChar, b: *const WChar, n: usize) -> i32 { + for i in 0..n { + let (x, y) = unsafe { (*a.add(i), *b.add(i)) }; + if x != y { + return if x < y { -1 } else { 1 }; + } + if x == 0 { + break; + } + } + 0 +} + +#[no_mangle] +pub unsafe extern "C" fn wcscoll(a: *const WChar, b: *const WChar) -> i32 { + unsafe { wcscmp(a, b) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcscoll_l(a: *const WChar, b: *const WChar, _loc: *mut Locale) -> i32 { + unsafe { wcscmp(a, b) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsxfrm(dst: *mut WChar, src: *const WChar, n: usize) -> usize { + let len = unsafe { wcslen(src) }; + if len < n { + unsafe { wmemcpy(dst, src, len + 1) }; + } + len +} + +#[no_mangle] +pub unsafe extern "C" fn wcsxfrm_l(dst: *mut WChar, src: *const WChar, n: usize, _loc: *mut Locale) -> usize { + unsafe { wcsxfrm(dst, src, n) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcschr(s: *const WChar, c: WChar) -> *mut WChar { + let mut p = s; + loop { + let x = unsafe { *p }; + if x == c { + return p.cast_mut(); + } + if x == 0 { + return ptr::null_mut(); + } + p = unsafe { p.add(1) }; + } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsrchr(s: *const WChar, c: WChar) -> *mut WChar { + let mut found = ptr::null_mut(); + let mut p = s; + loop { + let x = unsafe { *p }; + if x == c { + found = p.cast_mut(); + } + if x == 0 { + return found; + } + p = unsafe { p.add(1) }; + } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsspn(s: *const WChar, accept: *const WChar) -> usize { + let mut n = 0; + while unsafe { *s.add(n) } != 0 && !unsafe { wcschr(accept, *s.add(n)) }.is_null() { + n += 1; + } + n +} + +#[no_mangle] +pub unsafe extern "C" fn wcscspn(s: *const WChar, reject: *const WChar) -> usize { + let mut n = 0; + while unsafe { *s.add(n) } != 0 && unsafe { wcschr(reject, *s.add(n)) }.is_null() { + n += 1; + } + n +} + +#[no_mangle] +pub unsafe extern "C" fn wcspbrk(s: *const WChar, accept: *const WChar) -> *mut WChar { + let p = unsafe { s.add(wcscspn(s, accept)) }; + if unsafe { *p } == 0 { ptr::null_mut() } else { p.cast_mut() } +} + +#[no_mangle] +pub unsafe extern "C" fn wcsstr(haystack: *const WChar, needle: *const WChar) -> *mut WChar { + let n = unsafe { wcslen(needle) }; + let mut p = haystack; + loop { + if unsafe { wmemcmp(p, needle, n) } == 0 { + return p.cast_mut(); + } + if unsafe { *p } == 0 { + return ptr::null_mut(); + } + p = unsafe { p.add(1) }; + } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstok(s: *mut WChar, delim: *const WChar, save: *mut *mut WChar) -> *mut WChar { + let mut p = if s.is_null() { unsafe { *save } } else { s }; + if p.is_null() { + return ptr::null_mut(); + } + p = unsafe { p.add(wcsspn(p, delim)) }; + if unsafe { *p } == 0 { + unsafe { *save = ptr::null_mut() }; + return ptr::null_mut(); + } + let end = unsafe { p.add(wcscspn(p, delim)) }; + unsafe { + if *end == 0 { + *save = ptr::null_mut(); + } else { + *end = 0; + *save = end.add(1); + } + } + p +} + +#[no_mangle] +pub unsafe extern "C" fn wmemchr(s: *const WChar, c: WChar, n: usize) -> *mut WChar { + (0..n).map(|i| unsafe { s.add(i) }).find(|&p| unsafe { *p } == c).map_or(ptr::null_mut(), <*const WChar>::cast_mut) +} + +#[no_mangle] +pub unsafe extern "C" fn wmemcmp(a: *const WChar, b: *const WChar, n: usize) -> i32 { + for i in 0..n { + let (x, y) = unsafe { (*a.add(i), *b.add(i)) }; + if x != y { + return if x < y { -1 } else { 1 }; + } + } + 0 +} + +#[no_mangle] +pub unsafe extern "C" fn wmemcpy(dst: *mut WChar, src: *const WChar, n: usize) -> *mut WChar { + unsafe { ptr::copy_nonoverlapping(src, dst, n) }; + dst +} + +#[no_mangle] +pub unsafe extern "C" fn wmemmove(dst: *mut WChar, src: *const WChar, n: usize) -> *mut WChar { + unsafe { ptr::copy(src, dst, n) }; + dst +} + +#[no_mangle] +pub unsafe extern "C" fn wmemset(dst: *mut WChar, c: WChar, n: usize) -> *mut WChar { + for i in 0..n { + unsafe { *dst.add(i) = c }; + } + dst +} + +// Numbers + +#[no_mangle] +pub unsafe extern "C" fn wcstol(s: *const WChar, endptr: *mut *mut WChar, base: i32) -> i64 { + unsafe { crate::misc::answer(s, strtonum::signed(s, base), endptr) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstoul(s: *const WChar, endptr: *mut *mut WChar, base: i32) -> u64 { + unsafe { crate::misc::answer(s, strtonum::unsigned(s, base), endptr) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstoll(s: *const WChar, endptr: *mut *mut WChar, base: i32) -> i64 { + unsafe { wcstol(s, endptr, base) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstoull(s: *const WChar, endptr: *mut *mut WChar, base: i32) -> u64 { + unsafe { wcstoul(s, endptr, base) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstoimax(s: *const WChar, endptr: *mut *mut WChar, base: i32) -> i64 { + unsafe { wcstol(s, endptr, base) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstoumax(s: *const WChar, endptr: *mut *mut WChar, base: i32) -> u64 { + unsafe { wcstoul(s, endptr, base) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstod(s: *const WChar, endptr: *mut *mut WChar) -> f64 { + unsafe { crate::misc::answer(s, strtonum::float::(s), endptr) } +} + +#[no_mangle] +pub unsafe extern "C" fn wcstof(s: *const WChar, endptr: *mut *mut WChar) -> f32 { + unsafe { crate::misc::answer(s, strtonum::float::(s), endptr) } +} + +// Formatted output: the narrow engine's, read back as UTF-8. + +#[no_mangle] +pub unsafe extern "C" fn swprintf(s: *mut WChar, n: usize, fmt: *const WChar, args: ...) -> i32 { + unsafe { vswprintf(s, n, fmt, args) } +} + +/// `vswprintf`: the format's wide characters as UTF-8 for `vsnprintf`, whose +/// output is decoded back into `s`. Like C's, it refuses (-1) an output that +/// does not fit in `n` wide characters with its terminator. +#[no_mangle] +pub unsafe extern "C" fn vswprintf(s: *mut WChar, n: usize, fmt: *const WChar, ap: VaList<'_>) -> i32 { + let mut narrow = alloc::vec::Vec::new(); + let len = unsafe { wcslen(fmt) }; + for i in 0..=len { + let mut buf = [0u8; 4]; + let Some(k) = encode(unsafe { *fmt.add(i) }, &mut buf) else { + errno::set(EILSEQ); + return -1; + }; + narrow.extend_from_slice(&buf[..k]); + } + let bytes = unsafe { crate::printf::vsnprintf(ptr::null_mut(), 0, narrow.as_ptr(), ap.clone()) }; + if bytes < 0 { + return -1; + } + let mut out = alloc::vec![0u8; bytes as usize + 1]; + unsafe { crate::printf::vsnprintf(out.as_mut_ptr(), out.len(), narrow.as_ptr(), ap) }; + let mut src = out.as_ptr(); + let mut st = MbState::INITIAL; + let wide = unsafe { mbsnrtowcs(ptr::null_mut(), &mut src, bytes as usize, 0, &mut st) }; + if wide == INVALID || wide >= n { + return -1; + } + let mut src = out.as_ptr(); + let mut st = MbState::INITIAL; + unsafe { mbsnrtowcs(s, &mut src, bytes as usize, wide, &mut st) }; + unsafe { *s.add(wide) = 0 }; + wide as i32 +} + +// Classes and cases: the C locale's, ASCII's. + +fn ascii(wc: WInt) -> Option { + (0..0x80).contains(&wc).then_some(wc) +} + +macro_rules! wide_class { + ($($name:ident, $name_l:ident => $narrow:path;)*) => {$( + #[no_mangle] + pub extern "C" fn $name(wc: WInt) -> i32 { + ascii(wc).map_or(0, |c| $narrow(c)) + } + + #[no_mangle] + pub extern "C" fn $name_l(wc: WInt, _loc: *mut Locale) -> i32 { + $name(wc) + } + )*}; +} + +wide_class! { + iswalnum, iswalnum_l => crate::ctype::isalnum; + iswalpha, iswalpha_l => crate::ctype::isalpha; + iswblank, iswblank_l => crate::ctype::isblank; + iswcntrl, iswcntrl_l => crate::ctype::iscntrl; + iswdigit, iswdigit_l => crate::ctype::isdigit; + iswgraph, iswgraph_l => crate::ctype::isgraph; + iswlower, iswlower_l => crate::ctype::islower; + iswprint, iswprint_l => crate::ctype::isprint; + iswpunct, iswpunct_l => crate::ctype::ispunct; + iswspace, iswspace_l => crate::ctype::isspace; + iswupper, iswupper_l => crate::ctype::isupper; + iswxdigit, iswxdigit_l => crate::ctype::isxdigit; +} + +/// `wctype`'s names, in the order its nonzero answers number them. +const CLASSES: [&[u8]; 12] = + [b"alnum", b"alpha", b"blank", b"cntrl", b"digit", b"graph", b"lower", b"print", b"punct", b"space", b"upper", b"xdigit"]; + +#[no_mangle] +pub unsafe extern "C" fn wctype(name: *const u8) -> u64 { + let name = unsafe { core::ffi::CStr::from_ptr(name.cast()) }.to_bytes(); + CLASSES.iter().position(|&c| c == name).map_or(0, |i| i as u64 + 1) +} + +#[no_mangle] +pub unsafe extern "C" fn wctype_l(name: *const u8, _loc: *mut Locale) -> u64 { + unsafe { wctype(name) } +} + +#[no_mangle] +pub extern "C" fn iswctype(wc: WInt, desc: u64) -> i32 { + let class: fn(WInt) -> i32 = match desc { + 1 => |c| iswalnum(c), + 2 => |c| iswalpha(c), + 3 => |c| iswblank(c), + 4 => |c| iswcntrl(c), + 5 => |c| iswdigit(c), + 6 => |c| iswgraph(c), + 7 => |c| iswlower(c), + 8 => |c| iswprint(c), + 9 => |c| iswpunct(c), + 10 => |c| iswspace(c), + 11 => |c| iswupper(c), + 12 => |c| iswxdigit(c), + _ => return 0, + }; + class(wc) +} + +#[no_mangle] +pub extern "C" fn iswctype_l(wc: WInt, desc: u64, _loc: *mut Locale) -> i32 { + iswctype(wc, desc) +} + +#[no_mangle] +pub extern "C" fn towupper(wc: WInt) -> WInt { + ascii(wc).map_or(wc, |c| crate::ctype::toupper(c)) +} + +#[no_mangle] +pub extern "C" fn towlower(wc: WInt) -> WInt { + ascii(wc).map_or(wc, |c| crate::ctype::tolower(c)) +} + +#[no_mangle] +pub extern "C" fn towupper_l(wc: WInt, _loc: *mut Locale) -> WInt { + towupper(wc) +} + +#[no_mangle] +pub extern "C" fn towlower_l(wc: WInt, _loc: *mut Locale) -> WInt { + towlower(wc) +} + +const TRANS_TOLOWER: u64 = 1; +const TRANS_TOUPPER: u64 = 2; + +#[no_mangle] +pub unsafe extern "C" fn wctrans(name: *const u8) -> u64 { + match unsafe { core::ffi::CStr::from_ptr(name.cast()) }.to_bytes() { + b"tolower" => TRANS_TOLOWER, + b"toupper" => TRANS_TOUPPER, + _ => 0, + } +} + +#[no_mangle] +pub unsafe extern "C" fn wctrans_l(name: *const u8, _loc: *mut Locale) -> u64 { + unsafe { wctrans(name) } +} + +#[no_mangle] +pub extern "C" fn towctrans(wc: WInt, desc: u64) -> WInt { + match desc { + TRANS_TOLOWER => towlower(wc), + TRANS_TOUPPER => towupper(wc), + _ => wc, + } +} + +#[no_mangle] +pub extern "C" fn towctrans_l(wc: WInt, desc: u64, _loc: *mut Locale) -> WInt { + towctrans(wc, desc) +} + +// Wide streams: UTF-8 on the byte stream underneath. + +use crate::stdio::FILE; + +#[no_mangle] +pub unsafe extern "C" fn fgetwc(f: *mut FILE) -> WInt { + let mut st = MbState::INITIAL; + loop { + let c = unsafe { crate::stdio::fgetc(f) }; + if c == EOF { + if st.is_partial() { + errno::set(EILSEQ); + } + return WEOF; + } + let byte = c as u8; + let mut wc: WChar = 0; + match unsafe { mbrtowc(&mut wc, &byte, 1, &mut st) } { + INCOMPLETE => continue, + INVALID => return WEOF, + _ => return wc as WInt, + } + } +} + +#[no_mangle] +pub unsafe extern "C" fn getwc(f: *mut FILE) -> WInt { + unsafe { fgetwc(f) } +} + +#[no_mangle] +pub unsafe extern "C" fn ungetwc(wc: WInt, f: *mut FILE) -> WInt { + let mut buf = [0u8; 4]; + match encode(wc as WChar, &mut buf) { + Some(n) if !f.is_null() && unsafe { crate::stdio::unget(f, &buf[..n]) } => wc, + _ => WEOF, + } +} + +#[no_mangle] +pub unsafe extern "C" fn fputwc(wc: WChar, f: *mut FILE) -> WInt { + let mut buf = [0u8; 4]; + let Some(n) = encode(wc, &mut buf) else { + errno::set(EILSEQ); + return WEOF; + }; + if unsafe { crate::stdio::fwrite(buf.as_ptr(), 1, n, f) } == n { wc as WInt } else { WEOF } +} + +#[no_mangle] +pub unsafe extern "C" fn putwc(wc: WChar, f: *mut FILE) -> WInt { + unsafe { fputwc(wc, f) } +}