diff --git a/include/sys/mman.h b/include/sys/mman.h new file mode 100644 index 0000000..2914841 --- /dev/null +++ b/include/sys/mman.h @@ -0,0 +1,216 @@ +#ifndef VLIBC_SYS_MMAN_H +#define VLIBC_SYS_MMAN_H + +/* + * vlibc — . + * + * Memory mapping, protection, and synchronization (POSIX.1-2008). Every + * function here is an unbuffered pass-through to the kernel: failures are + * reported as -1 (or MAP_FAILED for mmap) with errno set by the syscall + * layer. + * + * Level 1 (onlyposix): mmap, munmap, mprotect, msync, mlock, munlock, + * mlockall, munlockall, posix_madvise. + * Level 2 (muslmimic): madvise (XSI — posix_madvise is the POSIX base + * form), shm_open, shm_unlink. + * + * All constant values are Linux x86_64 kernel-UAPI facts (asm-generic/ + * mman-common.h and mman.h), transcribed, not invented. PROT_*, MAP_*, + * MAP_FAILED, MS_* and MCL_* exist at level 1. The POSIX_MADV_* advice + * values match the Linux MADV_* values for the same advice (0-4), which is + * what lets posix_madvise pass its advice straight to SYS_madvise; the + * Linux MADV_* names themselves are gated at level 2 with madvise, the only + * function that takes them. MAP_ANONYMOUS is provided as a source + * compatibility alias of the primary spelling MAP_ANON (the BSD/POSIX + * draft name glibc also accepts). + * + * POSIX shared memory (shm_open/shm_unlink, level 2) is backed by a real + * named file under /dev/shm (tmpfs), NOT by memfd_create: the object is a + * path in the shared-memory filesystem, visible there and re-openable by + * name until shm_unlink removes it, which matches POSIX's named-object + * semantics. vlibc renders name as "/dev/shm/" + name and requires that + * name contain no '/' — the object is a single directory entry, so a slash + * would escape the namespace (EINVAL). This differs from the POSIX wording + * that a name begin with a slash: vlibc names carry no leading slash and + * the prefix is added internally. Otherwise shm_open behaves as open(2) on + * the rendered path (the oflag access modes plus O_CREAT/O_EXCL/O_TRUNC and + * mode bits are honored; O_WRONLY alone is undefined by POSIX and passed + * through to the kernel) and shm_unlink as unlink(2) on it. + * + * None of these declarations carry an intent attribute: every function + * changes the caller's address space or performs I/O with side effects and + * reports failures through errno, so const/pure would be unsound (the same + * rationale unistd.h documents for its I/O family). + */ + +#include + +#include + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +/* ---- memory protection ---- */ + +#define PROT_NONE 0x0 /* page cannot be accessed */ +#define PROT_READ 0x1 /* page can be read */ +#define PROT_WRITE 0x2 /* page can be written */ +#define PROT_EXEC 0x4 /* page can be executed */ + +/* ---- mapping types and flags ---- */ + +#define MAP_SHARED 0x01 /* share changes with the file and other mappers */ +#define MAP_PRIVATE 0x02 /* private copy-on-write mapping */ +#define MAP_FIXED 0x10 /* interpret addr exactly; fail if it cannot be used */ +#define MAP_ANON 0x20 /* anonymous mapping, not file-backed (primary name) */ +#define MAP_ANONYMOUS MAP_ANON /* source-compatibility alias of MAP_ANON */ + +/* Value returned by mmap on failure (never a valid mapping address). */ +#define MAP_FAILED ((void *)-1) + +/* ---- msync synchronization flags ---- */ + +#define MS_ASYNC 0x1 /* return before the writes take effect */ +#define MS_INVALIDATE 0x2 /* invalidate other mappings of the file */ +#define MS_SYNC 0x4 /* perform synchronous writes */ + +/* ---- mlockall flags ---- */ + +#define MCL_CURRENT 0x1 /* lock all currently mapped pages */ +#define MCL_FUTURE 0x2 /* lock all pages mapped in the future */ + +/* ---- posix_madvise advice (Linux MADV_* values, see the header note) ---- */ + +#define POSIX_MADV_NORMAL 0 /* no special treatment */ +#define POSIX_MADV_RANDOM 1 /* pages will be accessed randomly */ +#define POSIX_MADV_SEQUENTIAL 2 /* pages will be accessed sequentially */ +#define POSIX_MADV_WILLNEED 3 /* pages will be needed soon */ +#define POSIX_MADV_DONTNEED 4 /* pages are not needed soon */ + +/* + * Map len bytes starting at offset off of the object open on fildes into + * the process address space and return the mapping address, or MAP_FAILED + * with errno set. addr is a hint for the placement (0 for "anywhere") + * unless flags contains MAP_FIXED; prot is a combination of PROT_* (PROT_NONE + * alone forbids all access). With MAP_ANON fildes is ignored and must be -1 + * and off 0; the mapping is zero-filled. The mapping is shared or private + * per MAP_SHARED/MAP_PRIVATE. len is rounded up to whole pages. + */ +void * +mmap(void *addr, size_t len, int prot, int flags, int fildes, off_t off); + +/* + * Remove the mapping at addr (a page boundary) covering len bytes; return + * 0, or -1 with errno set. The address range becomes invalid. + */ +int +munmap(void *addr, size_t len); + +/* + * Change the protection of the mapped pages at addr covering len bytes to + * prot; return 0, or -1 with errno set. addr must be page-aligned. + */ +int +mprotect(void *addr, size_t len, int prot); + +/* + * Flush the mapped pages at addr covering len bytes of a MAP_SHARED mapping + * to (or invalidate them from) the underlying object per flags (MS_ASYNC, + * MS_SYNC, MS_INVALIDATE); return 0, or -1 with errno set. + */ +int +msync(void *addr, size_t len, int flags); + +/* + * Lock the pages at addr covering len bytes into memory so they are never + * paged out; return 0, or -1 with errno set. munlock unlocks them again. + */ +int +mlock(const void *addr, size_t len); + +int +munlock(const void *addr, size_t len); + +/* + * Lock all pages mapped by the process into memory per flags (MCL_CURRENT, + * MCL_FUTURE); return 0, or -1 with errno set. munlockall unlocks them. + */ +int +mlockall(int flags); + +int +munlockall(void); + +/* + * Give the kernel advice (one of the POSIX_MADV_* values) about how the + * pages at addr covering len bytes will be used; return 0, or -1 with errno + * set. The advice may be ignored; the mapping is unchanged. + */ +int +posix_madvise(void *addr, size_t len, int advice); + +#if VLIBC_LEVEL_GE(2) +/* Level 2 (muslmimic): XSI + POSIX shared memory. */ + +/* ---- madvise advice (Linux x86_64 kernel-UAPI values) ---- */ + +#define MADV_NORMAL 0 /* no special treatment */ +#define MADV_RANDOM 1 /* pages will be accessed randomly */ +#define MADV_SEQUENTIAL 2 /* pages will be accessed sequentially */ +#define MADV_WILLNEED 3 /* pages will be needed soon */ +#define MADV_DONTNEED 4 /* pages are not needed soon; free them */ +#define MADV_FREE 8 /* free pages; contents lost on later write */ +#define MADV_REMOVE 9 /* free the pages and punch a hole in the file */ +#define MADV_DONTFORK 10 /* do not inherit the pages across fork */ +#define MADV_DOFORK 11 /* revert MADV_DONTFORK */ +#define MADV_MERGEABLE 12 /* enable KSM merging of the pages */ +#define MADV_UNMERGEABLE 13 /* revert MADV_MERGEABLE */ +#define MADV_HUGEPAGE 14 /* prefer transparent huge pages */ +#define MADV_NOHUGEPAGE 15 /* revert MADV_HUGEPAGE */ +#define MADV_DONTDUMP 16 /* exclude the pages from core dumps */ +#define MADV_DODUMP 17 /* revert MADV_DONTDUMP */ +#define MADV_WIPEONFORK 18 /* zero the pages in the child after fork */ +#define MADV_KEEPONFORK 19 /* revert MADV_WIPEONFORK */ +#define MADV_COLD 20 /* pages will be accessed less soon */ +#define MADV_PAGEOUT 21 /* reclaim the pages immediately */ +#define MADV_POPULATE_READ 22 /* fault the pages in for reading */ +#define MADV_POPULATE_WRITE 23 /* fault the pages in for writing */ +#define MADV_DONTNEED_LOCKED 24 /* like MADV_DONTNEED on locked pages */ +#define MADV_COLLAPSE 25 /* collapse the range into a THP */ + +/* + * Linux madvise: give the kernel advice (one of the MADV_* values above) + * about the pages at addr covering len bytes; return 0, or -1 with errno + * set. XSI — posix_madvise (level 1) is the POSIX base form; this is the + * raw Linux interface exposing the full MADV_* set. + */ +int +madvise(void *addr, size_t len, int advice); + +/* + * Open (creating with mode if oflag contains O_CREAT) the POSIX shared + * memory object name and return a descriptor, or -1 with errno set. The + * object is a file under /dev/shm; name must not contain '/', and the + * object persists (re-openable by name) until shm_unlink removes it. See + * the header note on the naming deviation. oflag behaves as for open(2). + */ +int +shm_open(const char *name, int oflag, mode_t mode); + +/* + * Remove the shared memory object name from /dev/shm; return 0, or -1 with + * errno set. Open descriptors of the object stay valid until closed. See + * the header note on the naming deviation. + */ +int +shm_unlink(const char *name); +#endif /* VLIBC_LEVEL_GE(2) */ + +#ifdef __cplusplus +} +#endif + +#endif /* VLIBC_SYS_MMAN_H */ diff --git a/src/mman/mmap.c b/src/mman/mmap.c new file mode 100644 index 0000000..4950be2 --- /dev/null +++ b/src/mman/mmap.c @@ -0,0 +1,227 @@ +#ifdef HAVE_CONFIG_H +#include +#endif + +#include + +#include +#include + +#include + +#include "../internal/syscall.h" + +/* + * The mmap family over the raw x86_64 syscalls. Every wrapper is a thin + * pass-through: SYS_mmap, SYS_munmap, SYS_mprotect, SYS_msync, SYS_mlock, + * SYS_munlock, SYS_mlockall, SYS_munlockall, SYS_madvise. Arguments are + * cast to long because that is the shape of the syscall register file (the + * wrappers in arch/x86_64/syscall_arch.h take long arguments). + * + * mmap is the one member that cannot use syscall_ret(): its success value + * is an address, which syscall_ret() would truncate to int. The kernel + * result is therefore checked against the -4095 MAX_ERRNO bound directly + * (the same test malloc.c applies to its private mmap) and the full pointer + * is preserved on success. The other members funnel their result through + * syscall_ret(), which returns the value on success and -1 with errno set + * on error. + * + * Level 2 adds madvise (the raw Linux interface behind posix_madvise) and + * the POSIX shared-memory pair. shm_open/shm_unlink live on real files + * under /dev/shm (see for the naming rationale and the + * deviation from POSIX's leading-slash requirement): the object name is + * validated, rendered as "/dev/shm/" + name into a local buffer, and handed + * to SYS_openat / SYS_unlinkat directly — no inter-object dependency on + * the unistd wrappers, the same self-containment creat.c uses. + */ + +/* Length of the "/dev/shm/" prefix (VLIBC_SHM_PREFIX_LEN bytes + NUL). */ +#define VLIBC_SHM_PREFIX "/dev/shm/" +#define VLIBC_SHM_PREFIX_LEN (sizeof(VLIBC_SHM_PREFIX) - 1) + +/* Longest object name: /dev/shm is a tmpfs, whose NAME_MAX is 255. */ +#define VLIBC_SHM_NAME_MAX 255 +#define VLIBC_SHM_PATH_MAX (VLIBC_SHM_PREFIX_LEN + VLIBC_SHM_NAME_MAX + 1) + +/* + * Map len bytes starting at offset off of the file descriptor fildes into + * the process address space and return the mapping address. On failure + * MAP_FAILED is returned with errno set. len is rounded up to whole pages + * by the kernel; addr is a hint unless MAP_FIXED is given. + */ +void * +mmap(void *addr, size_t len, int prot, int flags, int fildes, off_t off) +{ + long r = __syscall6(SYS_mmap, (long)addr, (long)len, prot, flags, fildes, (long)off); + + if ((uintptr_t)r > (uintptr_t)-4096) + { + errno = (int)-r; /* the kernel returned -errno */ + return MAP_FAILED; + } + return (void *)r; +} + +/* + * Remove the mapping at addr covering len bytes; return 0, or -1 with errno + * set. addr must be page-aligned. + */ +int +munmap(void *addr, size_t len) +{ + return syscall_ret(__syscall2(SYS_munmap, (long)addr, (long)len)); +} + +/* + * Change the protection of the pages at addr covering len bytes to prot; + * return 0, or -1 with errno set. + */ +int +mprotect(void *addr, size_t len, int prot) +{ + return syscall_ret(__syscall3(SYS_mprotect, (long)addr, (long)len, prot)); +} + +/* + * Flush the pages at addr covering len bytes of a MAP_SHARED mapping to + * (or invalidate them from) the underlying object per flags; return 0, or + * -1 with errno set. + */ +int +msync(void *addr, size_t len, int flags) +{ + return syscall_ret(__syscall3(SYS_msync, (long)addr, (long)len, flags)); +} + +/* + * Lock the pages at addr covering len bytes into memory so they are never + * paged out; return 0, or -1 with errno set. munlock undoes the lock. + */ +int +mlock(const void *addr, size_t len) +{ + return syscall_ret(__syscall2(SYS_mlock, (long)addr, (long)len)); +} + +int +munlock(const void *addr, size_t len) +{ + return syscall_ret(__syscall2(SYS_munlock, (long)addr, (long)len)); +} + +/* + * Lock all pages mapped by the process into memory per flags (MCL_CURRENT, + * MCL_FUTURE); return 0, or -1 with errno set. munlockall undoes the lock. + */ +int +mlockall(int flags) +{ + return syscall_ret(__syscall1(SYS_mlockall, flags)); +} + +int +munlockall(void) +{ + return syscall_ret(__syscall0(SYS_munlockall)); +} + +/* + * Give the kernel the POSIX_MADV_* advice about the pages at addr covering + * len bytes; return 0, or -1 with errno set. The advice values (0-4) match + * the Linux MADV_* values, so the call is a direct SYS_madvise. + */ +int +posix_madvise(void *addr, size_t len, int advice) +{ + return syscall_ret(__syscall3(SYS_madvise, (long)addr, (long)len, advice)); +} + +#if VLIBC_LEVEL_GE(2) + +/* + * Linux madvise: give the kernel the MADV_* advice about the pages at addr + * covering len bytes; return 0, or -1 with errno set. XSI — posix_madvise + * is the POSIX base form; this exposes the full Linux MADV_* set. + */ +int +madvise(void *addr, size_t len, int advice) +{ + return syscall_ret(__syscall3(SYS_madvise, (long)addr, (long)len, advice)); +} + +/* + * Validate name (no empty names, no '/', tmpfs NAME_MAX bound) and render + * the /dev/shm path into buf (capacity VLIBC_SHM_PATH_MAX). Returns 0 and + * fills buf on success; -1 with errno set (EINVAL, ENAMETOOLONG) on error. + */ +static int +vlibc_shm_path(char *buf, const char *name) +{ + size_t i; + size_t n; + + for (n = 0; name[n] != '\0'; n++) + { + if (name[n] == '/') + { + errno = EINVAL; + return -1; + } + } + if (n == 0) + { + errno = EINVAL; + return -1; + } + if (n > VLIBC_SHM_NAME_MAX) + { + errno = ENAMETOOLONG; + return -1; + } + for (i = 0; i < VLIBC_SHM_PREFIX_LEN; i++) + { + buf[i] = VLIBC_SHM_PREFIX[i]; + } + for (i = 0; i < n; i++) + { + buf[VLIBC_SHM_PREFIX_LEN + i] = name[i]; + } + buf[VLIBC_SHM_PREFIX_LEN + n] = '\0'; + return 0; +} + +/* + * Open (creating with mode when oflag contains O_CREAT) the POSIX shared + * memory object name and return a descriptor, or -1 with errno set. The + * object is a real file under /dev/shm and persists until shm_unlink. See + * on the naming deviation. + */ +int +shm_open(const char *name, int oflag, mode_t mode) +{ + char path[VLIBC_SHM_PATH_MAX]; + + if (vlibc_shm_path(path, name) != 0) + { + return -1; + } + return syscall_ret(__syscall4(SYS_openat, AT_FDCWD, (long)path, oflag, (long)mode)); +} + +/* + * Remove the shared memory object name from /dev/shm; return 0, or -1 with + * errno set. Open descriptors of the object stay valid until closed. + */ +int +shm_unlink(const char *name) +{ + char path[VLIBC_SHM_PATH_MAX]; + + if (vlibc_shm_path(path, name) != 0) + { + return -1; + } + return syscall_ret(__syscall3(SYS_unlinkat, AT_FDCWD, (long)path, 0)); +} + +#endif /* VLIBC_LEVEL_GE(2) */ diff --git a/tests/test_mman.c b/tests/test_mman.c new file mode 100644 index 0000000..f97ba0e --- /dev/null +++ b/tests/test_mman.c @@ -0,0 +1,353 @@ +/* + * vlibc — sys/mman.h test (task 33). + * + * Exercises the mmap family end to end over the raw syscall wrappers: + * + * 1. Anonymous mmap of one page → readable/writable, write + read-back + * round-trip, munmap releases it. + * 2. mmap PROT_NONE succeeds; mprotect to PROT_READ|PROT_WRITE makes the + * page writable (the write succeeds and reads back). + * 3. msync MS_SYNC on a MAP_SHARED file mapping flushes the pointer + * writes to the backing file: a read() of the same fd afterwards sees + * the data written through the mapping. + * 4. mlock/munlock of an anonymous page, and mlockall/munlockall, + * return 0. + * 5. posix_madvise (POSIX_MADV_DONTNEED) returns 0. + * 6. MAP_SHARED file mapping round-trip is proven by scenario 3 (write + * via the mapping, read back through the fd). + * + * Level-2 gated section: madvise(MADV_DONTNEED) returns 0, and POSIX + * shared memory round-trips: shm_open creates a /dev/shm object, a write + * survives close, a second shm_open (no O_CREAT) re-opens the same object + * and reads the data back (proving the real-file backing), and shm_unlink + * removes it. + * + * The -f mode runs the failure scenarios — mmap MAP_FIXED at an unaligned + * address → MAP_FAILED, munmap of that address → -1, shm_open of a name + * containing '/' → -1. These make the LIBRARY write errno (syscall_ret), + * which under a host-linked binary targets glibc's private dtv slot at + * %fs:0+8; the process therefore exits via raw SYS_exit_group without + * running host cleanup (house pattern, tests/test_unistd_file.c). In the + * default mode every exercised call succeeds, so no library errno write + * happens; the one host-TCB slot-1 save/restore at main brackets the whole + * run anyway (task 13 technique), keeping host state intact for the return + * through __libc_start_main. The test NEVER reads errno; every negative is + * asserted on the return value only. + * + * All diagnostics go through raw SYS_write (no stdio): under -Iinclude the + * vlibc public headers shadow GCC's internal ones, so a host header would + * not compile. Not part of the library proper; compiled manually for this + * todo (the tests/ + make check wiring is owned by a later todo). + */ + +#include "../include/fcntl.h" +#include "../include/sys/mman.h" +#include "../include/unistd.h" + +#include "../src/internal/syscall.h" + +static int failures; + +/* Write a NUL-terminated string to fd via the raw syscall layer. The + * optimize attribute keeps GCC from lowering the length loop into a + * strlen call, which would leave a vlibc-owned symbol undefined in this + * host-linked standalone binary (house idiom, see src/string). */ +static __attribute__((optimize("no-tree-loop-distribute-patterns"))) void +say(int fd, const char *s) +{ + long n = 0; + + while (s[n] != '\0') + { + n++; + } + __syscall3(SYS_write, fd, (long)s, n); +} + +static void +check(int cond, const char *what) +{ + if (cond) + { + say(1, "PASS: "); + say(1, what); + say(1, "\n"); + } + else + { + say(2, "FAIL: "); + say(2, what); + say(2, "\n"); + failures++; + } +} + +/* + * Host-TCB slot-1 bracket: the library's errno write on a negative path + * lands at %fs:0+8, glibc's dtv pointer. Save and restore it around the + * whole default-mode run; only vlibc/raw-syscall code runs in between + * (task 13 technique). + */ +static unsigned long +tcb_slot1(void) +{ + return *(unsigned long *)((char *)__builtin_thread_pointer() + 8); +} + +static void +tcb_slot1_set(unsigned long value) +{ + *(unsigned long *)((char *)__builtin_thread_pointer() + 8) = value; +} + +/* Fill buf with the repeating pattern i * 7 + 1 (unsigned char). */ +static void +fill_pattern(unsigned char *buf, unsigned long n) +{ + unsigned long i; + + for (i = 0; i < n; i++) + { + buf[i] = (unsigned char)(i * 7 + 1); + } +} + +/* 1+2: anonymous mappings, protection change. */ +static void +anon_scenarios(void) +{ + static unsigned char pattern[4096]; + unsigned char *p; + unsigned long i; + int ok = 1; + + fill_pattern(pattern, sizeof(pattern)); + p = mmap(NULL, 4096, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANON, -1, 0); + check(p != MAP_FAILED && p != NULL, "mmap 4096 anonymous returns a mapping"); + if (p != MAP_FAILED && p != NULL) + { + for (i = 0; i < 4096; i++) + { + p[i] = pattern[i]; + } + for (i = 0; i < 4096; i++) + { + if (p[i] != pattern[i]) + { + ok = 0; + } + } + check(ok, "write + read-back round-trips on the anonymous mapping"); + check(munmap(p, 4096) == 0, "munmap of the anonymous mapping returns 0"); + } + + p = mmap(NULL, 4096, PROT_NONE, MAP_PRIVATE | MAP_ANON, -1, 0); + check(p != MAP_FAILED && p != NULL, "mmap 4096 PROT_NONE anonymous returns a mapping"); + if (p != MAP_FAILED && p != NULL) + { + check(mprotect(p, 4096, PROT_READ | PROT_WRITE) == 0, + "mprotect(PROT_NONE -> PROT_READ|PROT_WRITE) returns 0"); + p[0] = 'P'; + p[1] = 'W'; + check(p[0] == 'P' && p[1] == 'W', "the page is writable after mprotect"); + check(munmap(p, 4096) == 0, "munmap of the mprotect page returns 0"); + } +} + +/* 3+6: msync + MAP_SHARED file visibility, plus the mlock family (4+5). */ +static void +file_and_lock_scenarios(const char *path) +{ + static unsigned char pattern[4096]; + unsigned char *p; + unsigned char rbuf[16]; + int fd; + + fill_pattern(pattern, sizeof(pattern)); + fd = open(path, O_RDWR | O_CREAT | O_TRUNC, 0600); + check(fd >= 0, "open creates the msync backing file"); + if (fd < 0) + { + return; + } + check(write(fd, pattern, sizeof(pattern)) == (ssize_t)sizeof(pattern), + "write of 4096 bytes to the backing file returns 4096"); + p = mmap(NULL, 4096, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); + check(p != MAP_FAILED && p != NULL, "mmap MAP_SHARED of the file returns a mapping"); + if (p != MAP_FAILED && p != NULL) + { + p[0] = 'M'; + p[1] = 'S'; + p[2] = 'Y'; + p[3] = 'N'; + p[4] = 'C'; + p[5] = '!'; + check(msync(p, 4096, MS_SYNC) == 0, "msync MS_SYNC returns 0"); + check(lseek(fd, 0, SEEK_SET) == 0, "lseek back to offset 0 returns 0"); + check(read(fd, rbuf, sizeof(rbuf)) == (ssize_t)sizeof(rbuf), + "read of the backing file returns 16 bytes"); + check(rbuf[0] == 'M' && rbuf[1] == 'S' && rbuf[2] == 'Y' && rbuf[3] == 'N' && + rbuf[4] == 'C' && rbuf[5] == '!', + "the msync'd mapping writes are visible through the fd"); + check(munmap(p, 4096) == 0, "munmap of the shared mapping returns 0"); + } + check(close(fd) == 0, "close of the backing file returns 0"); + + { + unsigned char *q; + + q = mmap(NULL, 4096, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANON, -1, 0); + check(q != MAP_FAILED && q != NULL, "mmap a page for the mlock family"); + if (q != MAP_FAILED && q != NULL) + { + q[0] = 'L'; + check(mlock(q, 4096) == 0, "mlock of the page returns 0"); + check(munlock(q, 4096) == 0, "munlock of the page returns 0"); + check(munmap(q, 4096) == 0, "munmap of the mlock page returns 0"); + } + } + check(mlockall(MCL_CURRENT | MCL_FUTURE) == 0, "mlockall(MCL_CURRENT|MCL_FUTURE) returns 0"); + check(munlockall() == 0, "munlockall returns 0"); + + { + unsigned char *r; + + r = mmap(NULL, 4096, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANON, -1, 0); + check(r != MAP_FAILED && r != NULL, "mmap a page for posix_madvise"); + if (r != MAP_FAILED && r != NULL) + { + check(posix_madvise(r, 4096, POSIX_MADV_DONTNEED) == 0, + "posix_madvise(POSIX_MADV_DONTNEED) returns 0"); + check(munmap(r, 4096) == 0, "munmap of the posix_madvise page returns 0"); + } + } +} + +#if VLIBC_LEVEL_GE(2) + +/* Level 2: madvise and POSIX shared memory (real /dev/shm objects). */ +static void +level2_scenarios(void) +{ + const char *name = "vlibc-t33-shm"; + unsigned char *p; + char rbuf[16]; + int fd; + int fd2; + + p = mmap(NULL, 4096, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANON, -1, 0); + check(p != MAP_FAILED && p != NULL, "mmap a page for madvise"); + if (p != MAP_FAILED && p != NULL) + { + p[0] = 'D'; + check(madvise(p, 4096, MADV_DONTNEED) == 0, "madvise(MADV_DONTNEED) returns 0"); + check(munmap(p, 4096) == 0, "munmap of the madvise page returns 0"); + } + + fd = shm_open(name, O_RDWR | O_CREAT, 0600); + check(fd >= 0, "shm_open creates the shared memory object"); + if (fd < 0) + { + return; + } + check(write(fd, "shmdata", 7) == 7, "write to the shm object returns 7"); + check(close(fd) == 0, "close of the shm object returns 0"); + fd2 = shm_open(name, O_RDWR, 0); + check(fd2 >= 0, "shm_open re-opens the object without O_CREAT"); + if (fd2 >= 0) + { + check(read(fd2, rbuf, sizeof(rbuf)) == 7, "read of the re-opened object returns 7"); + check(rbuf[0] == 's' && rbuf[1] == 'h' && rbuf[2] == 'm' && rbuf[3] == 'd' && + rbuf[4] == 'a' && rbuf[5] == 't' && rbuf[6] == 'a', + "the re-opened object holds the earlier write (file-backed)"); + check(close(fd2) == 0, "close of the re-opened object returns 0"); + } + check(shm_unlink(name) == 0, "shm_unlink removes the object"); +} + +#endif /* VLIBC_LEVEL_GE(2) */ + +/* + * Failure scenarios (-f): every assertion is on the return value only, and + * the process exits through raw SYS_exit_group because the library writes + * errno on these paths (host-TCB hazard). + */ +static int +failure_scenarios(void) +{ + int rc = 0; + + if (mmap((void *)0x1, 4096, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANON | MAP_FIXED, -1, + 0) != MAP_FAILED) + { + say(2, "FAIL: mmap MAP_FIXED at an unaligned address did not return MAP_FAILED\n"); + rc = 1; + } + else + { + say(1, "PASS: mmap MAP_FIXED at 0x1 -> MAP_FAILED\n"); + } + if (munmap((void *)0x1, 4096) != -1) + { + say(2, "FAIL: munmap of an unaligned address did not return -1\n"); + rc = 1; + } + else + { + say(1, "PASS: munmap((void *)0x1, 4096) -> -1\n"); + } +#if VLIBC_LEVEL_GE(2) + if (shm_open("bad/name", O_RDWR | O_CREAT, 0600) != -1) + { + say(2, "FAIL: shm_open of a name containing '/' did not return -1\n"); + rc = 1; + } + else + { + say(1, "PASS: shm_open(\"bad/name\") -> -1\n"); + } +#endif + return rc; +} + +int +main(int argc, char **argv) +{ + const char *path = "/tmp/vlibc-t33-shmfile"; + unsigned long saved; + int rc; + + if (argc == 2 && argv[1][0] == '-' && argv[1][1] == 'f') + { + /* + * The failure scenarios write errno inside the library; under the + * host libc that slot is glibc's private TLS state, so leave via + * the raw syscall without running host cleanup. + */ + rc = failure_scenarios(); + __syscall1(SYS_exit_group, rc); + return rc; /* not reached */ + } + + /* Bracket the whole default-mode run against the host-TCB hazard. */ + saved = tcb_slot1(); + anon_scenarios(); + file_and_lock_scenarios(path); +#if VLIBC_LEVEL_GE(2) + level2_scenarios(); +#endif + __syscall3(SYS_unlinkat, AT_FDCWD, (long)path, 0); + + if (failures > 0) + { + say(2, "FAILED: one or more mman checks failed\n"); + rc = 1; + } + else + { + say(1, "all mman tests passed\n"); + rc = 0; + } + tcb_slot1_set(saved); + return rc; +}