From 79f160534b8cbb390024a209e73fa086daf6ac34 Mon Sep 17 00:00:00 2001 From: Lucas Holt Date: Fri, 25 Sep 2026 09:33:36 -0400 Subject: [PATCH 1/5] linux: implement memory protection key syscalls Backport Linuxulator pkey_alloc, pkey_free, and pkey_mprotect support for Chromium V8 sandboxing. Add the required XSAVE layout helpers and Linux-compatible PKRU process lifecycle handling. AI-Assisted-by: OpenAI Codex (GPT-5) Signed-off-by: Lucas Holt --- LINUX_BRAVE_SANDBOX_BACKPORT.md | 52 +++++++ sys/amd64/amd64/fpu.c | 96 +++++++++++- sys/amd64/amd64/sys_machdep.c | 26 ++++ sys/amd64/linux/linux_emul_md.h | 35 +++++ sys/amd64/linux/linux_machdep.c | 19 +++ sys/amd64/linux/linux_pkru.c | 227 ++++++++++++++++++++++++++++ sys/amd64/linux/linux_sysvec.c | 3 + sys/amd64/linux32/linux32_machdep.c | 19 +++ sys/amd64/linux32/linux32_sysvec.c | 3 + sys/arm64/linux/linux_emul_md.c | 52 +++++++ sys/arm64/linux/linux_emul_md.h | 18 +++ sys/compat/linux/linux_dummy.c | 3 - sys/compat/linux/linux_emul.c | 2 + sys/compat/linux/linux_emul.h | 6 + sys/compat/linux/linux_mmap.c | 38 +++++ sys/compat/linux/linux_mmap.h | 12 ++ sys/i386/linux/linux_emul_md.c | 52 +++++++ sys/i386/linux/linux_emul_md.h | 18 +++ sys/modules/linux/Makefile | 1 + sys/modules/linux_common/Makefile | 10 +- sys/x86/include/fpu.h | 9 ++ sys/x86/include/specialreg.h | 7 + sys/x86/include/sysarch.h | 1 + 23 files changed, 703 insertions(+), 6 deletions(-) create mode 100644 LINUX_BRAVE_SANDBOX_BACKPORT.md create mode 100644 sys/amd64/linux/linux_emul_md.h create mode 100644 sys/amd64/linux/linux_pkru.c create mode 100644 sys/arm64/linux/linux_emul_md.c create mode 100644 sys/arm64/linux/linux_emul_md.h create mode 100644 sys/i386/linux/linux_emul_md.c create mode 100644 sys/i386/linux/linux_emul_md.h diff --git a/LINUX_BRAVE_SANDBOX_BACKPORT.md b/LINUX_BRAVE_SANDBOX_BACKPORT.md new file mode 100644 index 00000000000..cec937dcb7c --- /dev/null +++ b/LINUX_BRAVE_SANDBOX_BACKPORT.md @@ -0,0 +1,52 @@ +# Linux Brave Sandbox Backport Plan + +## Goal + +Backport FreeBSD's Linuxulator memory protection key support so +Chromium-based Linux applications, including Brave, can use the V8 heap and +JIT sandbox on MidnightBSD/amd64. + +## Upstream changes + +1. Backport FreeBSD commit `7bcaff05223e`, which exposes XSAVE feature and + save-area layout information. +2. Backport FreeBSD commit `b9951017bab3`, which extends the XSAVE helpers to + account for supervisor-state components. +3. Adapt FreeBSD commit `bdb561843e86`, which implements Linux + `pkey_alloc(2)`, `pkey_free(2)`, and `pkey_mprotect(2)` using native amd64 + PKU support. +4. Treat FreeBSD commit `34718e01869b`, which maps `IFF_LOWER_UP` through + Linux `NETLINK_ROUTE`, as a separate follow-up because it fixes Brave + network detection rather than sandboxing. + +## Integration approach + +- Preserve the upstream split between common Linux syscall validation and + machine-dependent PKU operations. +- Keep protection-key allocation state in Linux per-process emulation data, + inherited on fork and reset on exec. +- Initialize PKRU to Linux's `0x55555554` default during Linux exec. +- Retain Linux-compatible no-PKU behavior on unsupported architectures. +- Adapt source and module Makefiles to MidnightBSD's current tree instead of + applying conflicting upstream hunks mechanically. +- Preserve existing syscall numbers and replace only their ENOSYS stubs. + +## Validation + +1. Run the repository C static-analysis scripts on staged C and header files. +2. Build the affected `linux_common`, Linux ABI modules, and amd64 kernel. +3. Exercise allocation, protection changes, access-right changes, fork + inheritance, exec reset, key exhaustion, and protection-key faults with a + small Linux test program. +4. Confirm protection-key faults translate to Linux `SEGV_PKUERR`. +5. Start Linux Brave on PKU-capable amd64 hardware and inspect its sandbox + status. +6. Test the no-PKU fallback where suitable hardware is available. + +## Commit structure + +- XSAVE query helpers. +- Linuxulator protection-key syscall support. +- Tests, if kept separate by the existing test layout. +- `UPDATING` entry as an independently reviewable commit, after approval. +- Optional `IFF_LOWER_UP` compatibility fix as a separate change. diff --git a/sys/amd64/amd64/fpu.c b/sys/amd64/amd64/fpu.c index 256bd83e720..df3c3c040ad 100644 --- a/sys/amd64/amd64/fpu.c +++ b/sys/amd64/amd64/fpu.c @@ -191,12 +191,15 @@ SYSCTL_INT(_hw, HW_FLOATINGPT, floatingpoint, CTLFLAG_RD, int use_xsave; /* non-static for cpu_switch.S */ uint64_t xsave_mask; /* the same */ +static uint64_t xsave_mask_supervisor; +static uint64_t xsave_extensions; static uma_zone_t fpu_save_area_zone; static struct savefpu *fpu_initialstate; static struct xsave_area_elm_descr { u_int offset; u_int size; + u_int flags; } *xsave_area_desc; static void @@ -349,6 +352,7 @@ fpuinit_bsp1(void) ctx_switch_xsave[3] |= 0x10; restore_wp(old_wp); } + xsave_mask_supervisor = ((uint64_t)cp[3] << 32) | cp[2]; } /* @@ -444,7 +448,7 @@ fpuinitstate(void *arg __unused) XSAVE_AREA_ALIGN - 1, 0); fpu_initialstate = uma_zalloc(fpu_save_area_zone, M_WAITOK | M_ZERO); if (use_xsave) { - max_ext_n = flsl(xsave_mask); + max_ext_n = flsl(xsave_mask | xsave_mask_supervisor); xsave_area_desc = malloc(max_ext_n * sizeof(struct xsave_area_elm_descr), M_DEVBUF, M_WAITOK | M_ZERO); } @@ -477,6 +481,9 @@ fpuinitstate(void *arg __unused) * Region of an XSAVE Area" for the source of offsets/sizes. */ if (use_xsave) { + cpuid_count(0xd, 1, cp); + xsave_extensions = cp[0]; + xstate_bv = (uint64_t *)((char *)(fpu_initialstate + 1) + offsetof(struct xstate_hdr, xstate_bv)); *xstate_bv = XFEATURE_ENABLED_X87 | XFEATURE_ENABLED_SSE; @@ -492,6 +499,7 @@ fpuinitstate(void *arg __unused) cpuid_count(0xd, i, cp); xsave_area_desc[i].offset = cp[1]; xsave_area_desc[i].size = cp[0]; + xsave_area_desc[i].flags = cp[2]; } } @@ -1312,3 +1320,89 @@ fpu_save_area_reset(struct savefpu *fsa) bcopy(fpu_initialstate, fsa, cpu_max_ext_state_size); } + +static __inline void +xsave_extfeature_check(uint64_t feature, bool supervisor) +{ + KASSERT((feature & (feature - 1)) == 0, + ("%s: invalid XFEATURE 0x%lx", __func__, feature)); + KASSERT(flsl(feature) <= flsl(supervisor ? xsave_mask_supervisor : + xsave_mask), + ("%s: unsupported %s XFEATURE 0x%lx", __func__, + supervisor ? "supervisor" : "user", feature)); +} + +static __inline void +xsave_extstate_bv_check(uint64_t xstate_bv, bool supervisor) +{ + KASSERT(xstate_bv != 0 && flsl(xstate_bv) <= + flsl(supervisor ? xsave_mask_supervisor : xsave_mask), + ("%s: invalid XSTATE_BV 0x%lx", __func__, xstate_bv)); +} + +bool +xsave_extfeature_supported(uint64_t feature, bool supervisor) +{ + uint64_t mask; + int idx; + + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + xsave_extfeature_check(feature, supervisor); + mask = supervisor ? xsave_mask_supervisor : xsave_mask; + if ((mask & feature) == 0) + return (false); + idx = flsl(feature) - 1; + return (((xsave_area_desc[idx].flags & + CPUID_EXTSTATE_SUPERVISOR) != 0) == supervisor); +} + +bool +xsave_extension_supported(uint64_t extension) +{ + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + return ((xsave_extensions & extension) != 0); +} + +size_t +xsave_area_offset(uint64_t xstate_bv, uint64_t feature, bool compact, + bool supervisor) +{ + struct xsave_area_elm_descr *xep; + size_t offs; + int i, idx; + + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + xsave_extstate_bv_check(xstate_bv, supervisor); + xsave_extfeature_check(feature, supervisor); + idx = flsl(feature) - 1; + if (!compact) + return (xsave_area_desc[idx].offset); + offs = sizeof(struct savefpu) + sizeof(struct xstate_hdr); + xstate_bv &= ~(XFEATURE_ENABLED_X87 | XFEATURE_ENABLED_SSE); + while ((i = ffs(xstate_bv) - 1) > 0 && i < idx) { + xep = &xsave_area_desc[i]; + if ((xep->flags & CPUID_EXTSTATE_ALIGNED) != 0) + offs = roundup2(offs, 64); + offs += xep->size; + xstate_bv &= ~((uint64_t)1 << i); + } + return (offs); +} + +size_t +xsave_area_size(uint64_t xstate_bv, bool compact, bool supervisor) +{ + int last_idx; + + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + xsave_extstate_bv_check(xstate_bv, supervisor); + last_idx = flsl(xstate_bv) - 1; + return (xsave_area_offset(xstate_bv, (uint64_t)1 << last_idx, compact, + supervisor) + xsave_area_desc[last_idx].size); +} + +size_t +xsave_area_hdr_offset(void) +{ + return (sizeof(struct savefpu)); +} diff --git a/sys/amd64/amd64/sys_machdep.c b/sys/amd64/amd64/sys_machdep.c index 3fbf44d9e48..9dc0c6da513 100644 --- a/sys/amd64/amd64/sys_machdep.c +++ b/sys/amd64/amd64/sys_machdep.c @@ -168,6 +168,32 @@ update_gdt_fsbase(struct thread *td, uint32_t base) critical_exit(); } +int +amd64_pkru_update(struct thread *td, uintptr_t addr, size_t len, u_int keyidx, + int flags, bool clear) +{ + struct vm_map *map; + vm_offset_t start, end; + int error; + + MPASS(td == curthread); + map = &td->td_proc->p_vmspace->vm_map; + vm_map_lock_read(map); + if (len == 0 || !vm_map_check_boundary(map, addr, addr + len)) { + vm_map_unlock_read(map); + return (EINVAL); + } + start = trunc_page(addr); + end = round_page(addr + len); + if (clear) + error = pmap_pkru_clear(PCPU_GET(curpmap), start, end); + else + error = pmap_pkru_set(PCPU_GET(curpmap), start, end, keyidx, + flags); + vm_map_unlock_read(map); + return (error); +} + int sysarch(struct thread *td, struct sysarch_args *uap) { diff --git a/sys/amd64/linux/linux_emul_md.h b/sys/amd64/linux/linux_emul_md.h new file mode 100644 index 00000000000..b594d0eba66 --- /dev/null +++ b/sys/amd64/linux/linux_emul_md.h @@ -0,0 +1,35 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#ifndef _AMD64_LINUX_EMUL_MD_H_ +#define _AMD64_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + uint32_t md_pkey_allocation_map; /* x86 protection keys */ +}; + +/* + * Initial protection key allocation map: key 0 is the default key, + * implicitly allocated on Linux (mm_pkey_allocation_map is initialized + * to 0x1). Inherited on fork, reset on exec. + */ +#define LINUX_PKEY_INITIAL_MAP 0x1 + +/* + * Initial PKRU at exec: access disabled for keys 1..15, key 0 open; + * the Linux init_pkru default. + */ +#define LINUX_PKRU_INIT 0x55555554 + +struct thread; + +void linux_pkru_exec_init(struct thread *); + +#endif /* !_AMD64_LINUX_EMUL_MD_H_ */ diff --git a/sys/amd64/linux/linux_machdep.c b/sys/amd64/linux/linux_machdep.c index 54dbad76d6d..b3169872159 100644 --- a/sys/amd64/linux/linux_machdep.c +++ b/sys/amd64/linux/linux_machdep.c @@ -105,6 +105,25 @@ linux_mprotect(struct thread *td, struct linux_mprotect_args *uap) return (linux_mprotect_common(td, uap->addr, uap->len, uap->prot)); } +int +linux_pkey_mprotect(struct thread *td, struct linux_pkey_mprotect_args *uap) +{ + return (linux_pkey_mprotect_common(td, uap->start, uap->len, + uap->prot, uap->pkey)); +} + +int +linux_pkey_alloc(struct thread *td, struct linux_pkey_alloc_args *uap) +{ + return (linux_pkey_alloc_common(td, uap->flags, uap->init_val)); +} + +int +linux_pkey_free(struct thread *td, struct linux_pkey_free_args *uap) +{ + return (linux_pkey_free_common(td, uap->pkey)); +} + int linux_madvise(struct thread *td, struct linux_madvise_args *uap) { diff --git a/sys/amd64/linux/linux_pkru.c b/sys/amd64/linux/linux_pkru.c new file mode 100644 index 00000000000..405a9eb93d3 --- /dev/null +++ b/sys/amd64/linux/linux_pkru.c @@ -0,0 +1,227 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +/* + * x86 memory protection keys (PKU) for the Linuxulator, serving both + * the 64-bit and 32-bit Linux ABIs. + * + * The PKRU register is directly user-visible: Linux programs read and + * write it with RDPKRU/WRPKRU, which execute natively. The kernel's + * part is key allocation bookkeeping (per address space: inherited on + * fork, reset on exec, as with Linux mm->context.pkey_allocation_map), + * tagging pages (pkey_mprotect), and applying the initial access + * rights of pkey_alloc() to the calling thread's PKRU. + */ + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include + +#include +#include + +static bool +linux_pkey_supported(void) +{ + + return ((cpu_stdext_feature2 & CPUID_STDEXT2_OSPKE) != 0); +} + +/* + * Update the calling thread's PKRU: new value is (PKRU & keep) | set. + */ +static void +linux_pkru_write(struct thread *td, uint32_t keep, uint32_t set) +{ + struct pcb *pcb; + struct xstate_hdr *hdr; + char *sa; + uint32_t *pkru; + + MPASS(td == curthread); + pcb = td->td_pcb; + + /* + * The critical section is held across the save area update to + * exclude preemption: a context switch could otherwise load the + * xsave area back into the CPU after fpugetregs(), and the + * stores below would then be lost to the next save. + */ + critical_enter(); + if ((pcb->pcb_flags & PCB_USERFPUINITDONE) != 0 && + td == PCPU_GET(fpcurthread) && PCB_USER_FPU(pcb)) { + wrpkru((rdpkru() & keep) | set); + critical_exit(); + return; + } + + /* + * The user FPU state is in the PCB save area, or is not yet + * initialized, in which case fpugetregs() installs the initial + * state there. + */ + (void)fpugetregs(td); + sa = (char *)get_pcb_user_save_td(td); + hdr = (struct xstate_hdr *)(sa + xsave_area_hdr_offset()); + pkru = (uint32_t *)(sa + + xsave_area_offset(xsave_mask, XFEATURE_ENABLED_PKRU, false, false)); + if ((hdr->xstate_bv & XFEATURE_ENABLED_PKRU) == 0) { + hdr->xstate_bv |= XFEATURE_ENABLED_PKRU; + *pkru = 0; + } + *pkru = (*pkru & keep) | set; + critical_exit(); +} + +/* + * Set the calling thread's PKRU access rights for the given key. + */ +static void +linux_pkru_set_perm(struct thread *td, u_int keyidx, uint32_t rights) +{ + + linux_pkru_write(td, ~(LINUX_PKEY_ACCESS_MASK << (keyidx * 2)), + rights << (keyidx * 2)); +} + +/* + * Called from the Linux sysvecs' exec_setregs. Linux initializes + * PKRU at exec to deny access to all keys but key 0 + * (arch/x86/mm/pkeys.c init_pkru_value), so memory tagged with a not + * yet allocated key is inaccessible; FreeBSD's initial PKRU is 0. + * This initializes the user FPU state slightly earlier than the lazy + * first-use path; the state would be initialized moments later in + * rtld/libc startup regardless. + */ +void +linux_pkru_exec_init(struct thread *td) +{ + + if (!linux_pkey_supported()) + return; + linux_pkru_write(td, 0, LINUX_PKRU_INIT); +} + +/* + * Protection keys are a property of the address space: inherit the + * allocation map on fork, as Linux does. When a FreeBSD process is + * switching to the Linux ABI there is no parent emuldata; start from + * the initial map. The unlocked read is atomic on the aligned word; + * a pkey_alloc() racing the fork in another thread yields a valid + * serialization either way. + */ +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ + struct linux_pemuldata *ppem; + + ppem = pem_find(td->td_proc); + if (ppem != NULL) + pem->pem_md.md_pkey_allocation_map = + ppem->pem_md.md_pkey_allocation_map; + else + pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP; +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ + + pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP; +} + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + struct linux_pemuldata *pem; + uint32_t free_keys; + int key; + + if (!linux_pkey_supported()) + return (ENOSPC); + + pem = pem_find(td->td_proc); + LINUX_PEM_XLOCK(pem); + free_keys = ~pem->pem_md.md_pkey_allocation_map & + ((1u << LINUX_PKEY_MAX) - 1) & ~LINUX_PKEY_INITIAL_MAP; + if (free_keys == 0) { + LINUX_PEM_XUNLOCK(pem); + return (ENOSPC); + } + key = ffs(free_keys) - 1; + pem->pem_md.md_pkey_allocation_map |= 1u << key; + LINUX_PEM_XUNLOCK(pem); + + linux_pkru_set_perm(td, key, init_val); + td->td_retval[0] = key; + return (0); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + struct linux_pemuldata *pem; + + if (!linux_pkey_supported()) + return (EINVAL); + + pem = pem_find(td->td_proc); + LINUX_PEM_XLOCK(pem); + if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) { + LINUX_PEM_XUNLOCK(pem); + return (EINVAL); + } + pem->pem_md.md_pkey_allocation_map &= ~(1u << pkey); + LINUX_PEM_XUNLOCK(pem); + + /* + * As on Linux, freeing a key neither untags pages nor updates + * PKRU; that is the application's responsibility. + */ + return (0); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + struct linux_pemuldata *pem; + int error; + + if (!linux_pkey_supported()) + return (EINVAL); + + pem = pem_find(td->td_proc); + LINUX_PEM_SLOCK(pem); + if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) { + LINUX_PEM_SUNLOCK(pem); + return (EINVAL); + } + LINUX_PEM_SUNLOCK(pem); + + error = linux_mprotect_common(td, addr, len, prot); + if (error != 0 || len == 0) + return (error); + + /* + * Tag the range; a pkey of 0 untags it. The tag is not + * persistent: it dies with the mapping, matching Linux VMA + * semantics. + */ + return (amd64_pkru_update(td, addr, len, pkey, 0, pkey == 0)); +} diff --git a/sys/amd64/linux/linux_sysvec.c b/sys/amd64/linux/linux_sysvec.c index 92993c4548e..0eb175a0199 100644 --- a/sys/amd64/linux/linux_sysvec.c +++ b/sys/amd64/linux/linux_sysvec.c @@ -63,6 +63,7 @@ #include #include +#include #include #include #include @@ -277,6 +278,8 @@ linux_exec_setregs(struct thread *td, struct image_params *imgp, * clean FP state if it uses the FPU again. */ fpstate_drop(td); + + linux_pkru_exec_init(td); } static int diff --git a/sys/amd64/linux32/linux32_machdep.c b/sys/amd64/linux32/linux32_machdep.c index 6f3b3e9fd49..79132aec3df 100644 --- a/sys/amd64/linux32/linux32_machdep.c +++ b/sys/amd64/linux32/linux32_machdep.c @@ -460,6 +460,25 @@ linux_mprotect(struct thread *td, struct linux_mprotect_args *uap) return (linux_mprotect_common(td, PTROUT(uap->addr), uap->len, uap->prot)); } +int +linux_pkey_mprotect(struct thread *td, struct linux_pkey_mprotect_args *uap) +{ + return (linux_pkey_mprotect_common(td, PTROUT(uap->start), uap->len, + uap->prot, uap->pkey)); +} + +int +linux_pkey_alloc(struct thread *td, struct linux_pkey_alloc_args *uap) +{ + return (linux_pkey_alloc_common(td, uap->flags, uap->init_val)); +} + +int +linux_pkey_free(struct thread *td, struct linux_pkey_free_args *uap) +{ + return (linux_pkey_free_common(td, uap->pkey)); +} + int linux_madvise(struct thread *td, struct linux_madvise_args *uap) { diff --git a/sys/amd64/linux32/linux32_sysvec.c b/sys/amd64/linux32/linux32_sysvec.c index b4fb9fd1b1e..8dbdbc151ff 100644 --- a/sys/amd64/linux32/linux32_sysvec.c +++ b/sys/amd64/linux32/linux32_sysvec.c @@ -67,6 +67,7 @@ #include #include +#include #include #include #include @@ -611,6 +612,8 @@ linux_exec_setregs(struct thread *td, struct image_params *imgp, fpstate_drop(td); + linux_pkru_exec_init(td); + /* Do full restore on return so that we can change to a different %cs */ set_pcb_flags(pcb, PCB_32BIT | PCB_FULL_IRET); } diff --git a/sys/arm64/linux/linux_emul_md.c b/sys/arm64/linux/linux_emul_md.c new file mode 100644 index 00000000000..9dd507ad4f4 --- /dev/null +++ b/sys/arm64/linux/linux_emul_md.c @@ -0,0 +1,52 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#include +#include +#include + +#include +#include + +/* No machine-dependent emuldata state yet. */ + +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ +} + +/* + * Protection key back ends: behave as Linux does on hardware without + * protection keys. pkey_alloc() reports no free keys and only the + * default key semantics remain. + */ + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + + return (ENOSPC); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + + return (EINVAL); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + + return (EINVAL); +} diff --git a/sys/arm64/linux/linux_emul_md.h b/sys/arm64/linux/linux_emul_md.h new file mode 100644 index 00000000000..0353f853167 --- /dev/null +++ b/sys/arm64/linux/linux_emul_md.h @@ -0,0 +1,18 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#ifndef _ARM64_LINUX_EMUL_MD_H_ +#define _ARM64_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + int md_dummy; /* no machine-dependent state yet */ +}; + +#endif /* !_ARM64_LINUX_EMUL_MD_H_ */ diff --git a/sys/compat/linux/linux_dummy.c b/sys/compat/linux/linux_dummy.c index b7dc490c9ff..566e9a4e879 100644 --- a/sys/compat/linux/linux_dummy.c +++ b/sys/compat/linux/linux_dummy.c @@ -123,9 +123,6 @@ DUMMY(mlock2); DUMMY(preadv2); DUMMY(pwritev2); /* Linux 4.8: */ -DUMMY(pkey_mprotect); -DUMMY(pkey_alloc); -DUMMY(pkey_free); DUMMY(open_tree); DUMMY(move_mount); DUMMY(fsopen); diff --git a/sys/compat/linux/linux_emul.c b/sys/compat/linux/linux_emul.c index 0c9527408a7..22dd9df4337 100644 --- a/sys/compat/linux/linux_emul.c +++ b/sys/compat/linux/linux_emul.c @@ -158,6 +158,7 @@ linux_proc_init(struct thread *td, struct thread *newtd, bool init_thread) pem = malloc(sizeof(*pem), M_LINUX, M_WAITOK | M_ZERO); sx_init(&pem->pem_sx, "lpemlk"); + linux_pemuldata_init_md(td, pem); p->p_emuldata = pem; } newtd->td_emuldata = em; @@ -184,6 +185,7 @@ linux_proc_init(struct thread *td, struct thread *newtd, bool init_thread) KASSERT(pem != NULL, ("proc_init: proc emuldata not found.\n")); pem->persona = 0; pem->oom_score_adj = 0; + linux_pemuldata_exec_md(pem); } } diff --git a/sys/compat/linux/linux_emul.h b/sys/compat/linux/linux_emul.h index 52a3cffe8f7..2a3cf5a6c42 100644 --- a/sys/compat/linux/linux_emul.h +++ b/sys/compat/linux/linux_emul.h @@ -30,6 +30,8 @@ #ifndef _LINUX_EMUL_H_ #define _LINUX_EMUL_H_ +#include + struct image_params; /* @@ -70,6 +72,7 @@ struct linux_pemuldata { uint32_t oom_score_adj; /* /proc/self/oom_score_adj */ uint32_t so_timestamp; /* requested timeval */ uint32_t so_timestampns; /* requested timespec */ + struct linux_pemuldata_md pem_md; /* machine-dependent state */ }; #define LINUX_PEM_XLOCK(p) sx_xlock(&(p)->pem_sx) @@ -79,4 +82,7 @@ struct linux_pemuldata { struct linux_pemuldata *pem_find(struct proc *); +void linux_pemuldata_init_md(struct thread *, struct linux_pemuldata *); +void linux_pemuldata_exec_md(struct linux_pemuldata *); + #endif /* !_LINUX_EMUL_H_ */ diff --git a/sys/compat/linux/linux_mmap.c b/sys/compat/linux/linux_mmap.c index d371c1b0935..d806d439dbb 100644 --- a/sys/compat/linux/linux_mmap.c +++ b/sys/compat/linux/linux_mmap.c @@ -243,6 +243,44 @@ linux_mprotect_common(struct thread *td, uintptr_t addr, size_t len, int prot) return (kern_mprotect(td, addr, len, prot, flags)); } +/* + * x86 memory protection keys. The common entry points perform the + * parameter validation Linux applies regardless of hardware support, + * then defer to the machine-dependent back end. + */ + +int +linux_pkey_alloc_common(struct thread *td, uint64_t flags, uint64_t init_val) +{ + + if (flags != 0) + return (EINVAL); + if ((init_val & ~(uint64_t)LINUX_PKEY_ACCESS_MASK) != 0) + return (EINVAL); + return (linux_pkey_alloc_machdep(td, init_val)); +} + +int +linux_pkey_free_common(struct thread *td, int pkey) +{ + + if (pkey < 0 || pkey >= LINUX_PKEY_MAX) + return (EINVAL); + return (linux_pkey_free_machdep(td, pkey)); +} + +int +linux_pkey_mprotect_common(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + + if (pkey < -1 || pkey >= LINUX_PKEY_MAX) + return (EINVAL); + if (pkey == -1) + return (linux_mprotect_common(td, addr, len, prot)); + return (linux_pkey_mprotect_machdep(td, addr, len, prot, pkey)); +} + /* * Implement Linux madvise(MADV_DONTNEED), which has unusual semantics: for * anonymous memory, pages in the range are immediately discarded. diff --git a/sys/compat/linux/linux_mmap.h b/sys/compat/linux/linux_mmap.h index 043dec9d40b..f6267a7eb1a 100644 --- a/sys/compat/linux/linux_mmap.h +++ b/sys/compat/linux/linux_mmap.h @@ -66,6 +66,18 @@ int linux_mmap_common(struct thread *, uintptr_t, size_t, int, int, int, off_t); int linux_mprotect_common(struct thread *, uintptr_t, size_t, int); +int linux_pkey_alloc_common(struct thread *, uint64_t, uint64_t); +int linux_pkey_free_common(struct thread *, int); +int linux_pkey_mprotect_common(struct thread *, uintptr_t, size_t, int, int); +int linux_pkey_alloc_machdep(struct thread *, uint64_t); +int linux_pkey_free_machdep(struct thread *, int); +int linux_pkey_mprotect_machdep(struct thread *, uintptr_t, size_t, int, int); + +#define LINUX_PKEY_DISABLE_ACCESS 0x1 +#define LINUX_PKEY_DISABLE_WRITE 0x2 +#define LINUX_PKEY_ACCESS_MASK (LINUX_PKEY_DISABLE_ACCESS | \ + LINUX_PKEY_DISABLE_WRITE) +#define LINUX_PKEY_MAX 16 int linux_madvise_common(struct thread *, uintptr_t, size_t, int); #endif /* _LINUX_MMAP_H_ */ diff --git a/sys/i386/linux/linux_emul_md.c b/sys/i386/linux/linux_emul_md.c new file mode 100644 index 00000000000..9dd507ad4f4 --- /dev/null +++ b/sys/i386/linux/linux_emul_md.c @@ -0,0 +1,52 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#include +#include +#include + +#include +#include + +/* No machine-dependent emuldata state yet. */ + +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ +} + +/* + * Protection key back ends: behave as Linux does on hardware without + * protection keys. pkey_alloc() reports no free keys and only the + * default key semantics remain. + */ + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + + return (ENOSPC); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + + return (EINVAL); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + + return (EINVAL); +} diff --git a/sys/i386/linux/linux_emul_md.h b/sys/i386/linux/linux_emul_md.h new file mode 100644 index 00000000000..9cfe6363d02 --- /dev/null +++ b/sys/i386/linux/linux_emul_md.h @@ -0,0 +1,18 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#ifndef _I386_LINUX_EMUL_MD_H_ +#define _I386_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + int md_dummy; /* no machine-dependent state yet */ +}; + +#endif /* !_I386_LINUX_EMUL_MD_H_ */ diff --git a/sys/modules/linux/Makefile b/sys/modules/linux/Makefile index 431db42ba0b..0b9d2a51719 100644 --- a/sys/modules/linux/Makefile +++ b/sys/modules/linux/Makefile @@ -66,6 +66,7 @@ SRCS+= imgact_linux.c \ linux.c \ linux_dummy.c \ linux_emul.c \ + linux_emul_md.c \ linux_errno.c \ linux_mib.c \ linux_mmap.c \ diff --git a/sys/modules/linux_common/Makefile b/sys/modules/linux_common/Makefile index 63d90e64a89..68492424fdc 100644 --- a/sys/modules/linux_common/Makefile +++ b/sys/modules/linux_common/Makefile @@ -1,7 +1,10 @@ .PATH: ${SRCTOP}/sys/compat/linux .if ${MACHINE_CPUARCH} == "amd64" -.PATH: ${SRCTOP}/sys/x86/linux +.PATH: ${SRCTOP}/sys/amd64/linux ${SRCTOP}/sys/x86/linux +.endif +.if ${MACHINE_CPUARCH} == "aarch64" +.PATH: ${SRCTOP}/sys/arm64/linux .endif KMOD= linux_common @@ -10,7 +13,10 @@ SRCS= linux_common.c linux_mib.c linux_mmap.c linux_util.c linux_emul.c \ linux.c device_if.h vnode_if.h bus_if.h opt_inet6.h opt_inet.h .if ${MACHINE_CPUARCH} == "amd64" -SRCS+= linux_x86.c linux_vdso_selector_x86.c +SRCS+= linux_pkru.c linux_x86.c linux_vdso_selector_x86.c +.endif +.if ${MACHINE_CPUARCH} == "aarch64" +SRCS+= linux_emul_md.c .endif EXPORT_SYMS= diff --git a/sys/x86/include/fpu.h b/sys/x86/include/fpu.h index e1ec6a592d2..a7cb3453065 100644 --- a/sys/x86/include/fpu.h +++ b/sys/x86/include/fpu.h @@ -213,4 +213,13 @@ struct savefpu_ymm { */ #define X86_XSTATE_XCR0_OFFSET 464 +#ifdef _KERNEL +bool xsave_extfeature_supported(uint64_t feature, bool supervisor); +bool xsave_extension_supported(uint64_t extension); +size_t xsave_area_hdr_offset(void); +size_t xsave_area_offset(uint64_t xstate_bv, uint64_t feature, bool compact, + bool supervisor); +size_t xsave_area_size(uint64_t xstate_bv, bool compact, bool supervisor); +#endif + #endif /* !_X86_FPU_H_ */ diff --git a/sys/x86/include/specialreg.h b/sys/x86/include/specialreg.h index 02cc23b562c..53e2643034b 100644 --- a/sys/x86/include/specialreg.h +++ b/sys/x86/include/specialreg.h @@ -370,6 +370,13 @@ #define CPUID_EXTSTATE_XINUSE 0x00000004 #define CPUID_EXTSTATE_XSAVES 0x00000008 +/* + * CPUID instruction 0xd Processor Extended State Enumeration, + * sub-leaf greater than 1, ECX information. + */ +#define CPUID_EXTSTATE_SUPERVISOR 0x00000001 +#define CPUID_EXTSTATE_ALIGNED 0x00000002 + /* * AMD extended function 8000_0007h ebx info */ diff --git a/sys/x86/include/sysarch.h b/sys/x86/include/sysarch.h index 3226f3b9d93..654354c5b97 100644 --- a/sys/x86/include/sysarch.h +++ b/sys/x86/include/sysarch.h @@ -160,6 +160,7 @@ int amd64_set_ldt(struct thread *, struct i386_ldt_args *, struct user_segment_descriptor *); int amd64_get_ioperm(struct thread *, struct i386_ioperm_args *); int amd64_set_ioperm(struct thread *, struct i386_ioperm_args *); +int amd64_pkru_update(struct thread *, uintptr_t, size_t, u_int, int, bool); #endif #endif /* !_MACHINE_SYSARCH_H_ */ From ab3dd6cfb8210ddfb1dd4e39854e878ad057d912 Mon Sep 17 00:00:00 2001 From: Lucas Holt Date: Fri, 25 Sep 2026 09:33:43 -0400 Subject: [PATCH 2/5] UPDATING: note Linuxulator protection key support AI-Assisted-by: OpenAI Codex (GPT-5) Signed-off-by: Lucas Holt --- UPDATING | 3 +++ 1 file changed, 3 insertions(+) diff --git a/UPDATING b/UPDATING index e9774aca1cc..3d099f60226 100644 --- a/UPDATING +++ b/UPDATING @@ -1,5 +1,8 @@ Updating Information for MidnightBSD users. +20260925: + linuxulator: add protection-key syscalls for Chromium V8 sandboxing + 20260922: ncurses 6.6 (from 6.2). Among many bug fixes this closes CVE-2022-29458 and CVE-2023-29491 (memory corruption from a From 2c13119cedaf6f6c2f56e3644693cda01fe8d8e7 Mon Sep 17 00:00:00 2001 From: Lucas Holt Date: Fri, 25 Sep 2026 11:53:48 -0400 Subject: [PATCH 3/5] linux: fix pkey syscall portability and feature checks Define the pkey syscall wrappers in shared Linux compatibility code so arm64 and i386 modules resolve their syscall entries. Require active PKU, OSPKE, and PKRU XSAVE state before accessing PKRU. AI-Assisted-by: OpenAI Codex (GPT-5) Signed-off-by: Lucas Holt --- sys/amd64/linux/linux_machdep.c | 19 ------------------- sys/amd64/linux/linux_pkru.c | 5 ++++- sys/amd64/linux32/linux32_machdep.c | 19 ------------------- sys/compat/linux/linux_misc.c | 23 +++++++++++++++++++++++ 4 files changed, 27 insertions(+), 39 deletions(-) diff --git a/sys/amd64/linux/linux_machdep.c b/sys/amd64/linux/linux_machdep.c index b3169872159..54dbad76d6d 100644 --- a/sys/amd64/linux/linux_machdep.c +++ b/sys/amd64/linux/linux_machdep.c @@ -105,25 +105,6 @@ linux_mprotect(struct thread *td, struct linux_mprotect_args *uap) return (linux_mprotect_common(td, uap->addr, uap->len, uap->prot)); } -int -linux_pkey_mprotect(struct thread *td, struct linux_pkey_mprotect_args *uap) -{ - return (linux_pkey_mprotect_common(td, uap->start, uap->len, - uap->prot, uap->pkey)); -} - -int -linux_pkey_alloc(struct thread *td, struct linux_pkey_alloc_args *uap) -{ - return (linux_pkey_alloc_common(td, uap->flags, uap->init_val)); -} - -int -linux_pkey_free(struct thread *td, struct linux_pkey_free_args *uap) -{ - return (linux_pkey_free_common(td, uap->pkey)); -} - int linux_madvise(struct thread *td, struct linux_madvise_args *uap) { diff --git a/sys/amd64/linux/linux_pkru.c b/sys/amd64/linux/linux_pkru.c index 405a9eb93d3..0807d1ee61f 100644 --- a/sys/amd64/linux/linux_pkru.c +++ b/sys/amd64/linux/linux_pkru.c @@ -39,7 +39,10 @@ static bool linux_pkey_supported(void) { - return ((cpu_stdext_feature2 & CPUID_STDEXT2_OSPKE) != 0); + return ((cpu_stdext_feature2 & + (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE)) == + (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE) && + (xsave_mask & XFEATURE_ENABLED_PKRU) != 0); } /* diff --git a/sys/amd64/linux32/linux32_machdep.c b/sys/amd64/linux32/linux32_machdep.c index 79132aec3df..6f3b3e9fd49 100644 --- a/sys/amd64/linux32/linux32_machdep.c +++ b/sys/amd64/linux32/linux32_machdep.c @@ -460,25 +460,6 @@ linux_mprotect(struct thread *td, struct linux_mprotect_args *uap) return (linux_mprotect_common(td, PTROUT(uap->addr), uap->len, uap->prot)); } -int -linux_pkey_mprotect(struct thread *td, struct linux_pkey_mprotect_args *uap) -{ - return (linux_pkey_mprotect_common(td, PTROUT(uap->start), uap->len, - uap->prot, uap->pkey)); -} - -int -linux_pkey_alloc(struct thread *td, struct linux_pkey_alloc_args *uap) -{ - return (linux_pkey_alloc_common(td, uap->flags, uap->init_val)); -} - -int -linux_pkey_free(struct thread *td, struct linux_pkey_free_args *uap) -{ - return (linux_pkey_free_common(td, uap->pkey)); -} - int linux_madvise(struct thread *td, struct linux_madvise_args *uap) { diff --git a/sys/compat/linux/linux_misc.c b/sys/compat/linux/linux_misc.c index bb3888a6325..02cb11aa920 100644 --- a/sys/compat/linux/linux_misc.c +++ b/sys/compat/linux/linux_misc.c @@ -75,6 +75,7 @@ #include #include #include +#include #include #include #include @@ -348,6 +349,28 @@ linux_msync(struct thread *td, struct linux_msync_args *args) args->fl & ~LINUX_MS_SYNC)); } +int +linux_pkey_mprotect(struct thread *td, struct linux_pkey_mprotect_args *uap) +{ + + return (linux_pkey_mprotect_common(td, uap->start, uap->len, + uap->prot, uap->pkey)); +} + +int +linux_pkey_alloc(struct thread *td, struct linux_pkey_alloc_args *uap) +{ + + return (linux_pkey_alloc_common(td, uap->flags, uap->init_val)); +} + +int +linux_pkey_free(struct thread *td, struct linux_pkey_free_args *uap) +{ + + return (linux_pkey_free_common(td, uap->pkey)); +} + #ifdef LINUX_LEGACY_SYSCALLS int linux_time(struct thread *td, struct linux_time_args *args) From 3a32191eaf18f07a14b9aaab8f32139a06525044 Mon Sep 17 00:00:00 2001 From: Lucas Holt Date: Fri, 25 Sep 2026 13:23:05 -0400 Subject: [PATCH 4/5] linux: include image activation definitions in emul MD files linux_emul.h declares linux_common_execve with struct image_args. Include sys/imgact.h so i386 and arm64 builds see the complete declaration and do not fail -Wvisibility. AI-Assisted-by: Codex GPT-5 Signed-off-by: Lucas Holt --- sys/arm64/linux/linux_emul_md.c | 1 + sys/i386/linux/linux_emul_md.c | 1 + 2 files changed, 2 insertions(+) diff --git a/sys/arm64/linux/linux_emul_md.c b/sys/arm64/linux/linux_emul_md.c index 9dd507ad4f4..53bcf88e407 100644 --- a/sys/arm64/linux/linux_emul_md.c +++ b/sys/arm64/linux/linux_emul_md.c @@ -6,6 +6,7 @@ #include #include +#include #include #include diff --git a/sys/i386/linux/linux_emul_md.c b/sys/i386/linux/linux_emul_md.c index 9dd507ad4f4..53bcf88e407 100644 --- a/sys/i386/linux/linux_emul_md.c +++ b/sys/i386/linux/linux_emul_md.c @@ -6,6 +6,7 @@ #include #include +#include #include #include From a09d98dc245d85944ef4dbe5231e3d33a80a5ab6 Mon Sep 17 00:00:00 2001 From: Lucas Holt Date: Fri, 25 Sep 2026 13:30:55 -0400 Subject: [PATCH 5/5] linux: prevalidate pkey_mprotect ranges Check fixed-boundary VM entries before changing ordinary page protections so a predictable PKRU tagging rejection cannot leave a partial update. Keep the validation in amd64_pkru_update to cover concurrent map changes. AI-Assisted-by: Codex GPT-5 Signed-off-by: Lucas Holt --- sys/amd64/linux/linux_pkru.c | 31 ++++++++++++++++++++++++++++--- 1 file changed, 28 insertions(+), 3 deletions(-) diff --git a/sys/amd64/linux/linux_pkru.c b/sys/amd64/linux/linux_pkru.c index 0807d1ee61f..79d86f4e7ce 100644 --- a/sys/amd64/linux/linux_pkru.c +++ b/sys/amd64/linux/linux_pkru.c @@ -23,6 +23,10 @@ #include #include +#include +#include +#include + #include #include #include @@ -39,12 +43,25 @@ static bool linux_pkey_supported(void) { - return ((cpu_stdext_feature2 & - (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE)) == - (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE) && + return ( + (cpu_stdext_feature2 & (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE)) == + (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE) && (xsave_mask & XFEATURE_ENABLED_PKRU) != 0); } +static bool +linux_pkey_range_valid(struct thread *td, uintptr_t addr, size_t len) +{ + struct vm_map *map; + bool valid; + + map = &td->td_proc->p_vmspace->vm_map; + vm_map_lock_read(map); + valid = vm_map_check_boundary(map, addr, addr + len); + vm_map_unlock_read(map); + return (valid); +} + /* * Update the calling thread's PKRU: new value is (PKRU & keep) | set. */ @@ -217,6 +234,14 @@ linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, } LINUX_PEM_SUNLOCK(pem); + /* + * Validate the PKRU tagging range before changing the ordinary page + * protections. amd64_pkru_update() repeats this check while applying + * the tag to synchronize with concurrent map changes. + */ + if (len != 0 && !linux_pkey_range_valid(td, addr, len)) + return (EINVAL); + error = linux_mprotect_common(td, addr, len, prot); if (error != 0 || len == 0) return (error);