diff --git a/LINUX_BRAVE_SANDBOX_BACKPORT.md b/LINUX_BRAVE_SANDBOX_BACKPORT.md new file mode 100644 index 00000000000..cec937dcb7c --- /dev/null +++ b/LINUX_BRAVE_SANDBOX_BACKPORT.md @@ -0,0 +1,52 @@ +# Linux Brave Sandbox Backport Plan + +## Goal + +Backport FreeBSD's Linuxulator memory protection key support so +Chromium-based Linux applications, including Brave, can use the V8 heap and +JIT sandbox on MidnightBSD/amd64. + +## Upstream changes + +1. Backport FreeBSD commit `7bcaff05223e`, which exposes XSAVE feature and + save-area layout information. +2. Backport FreeBSD commit `b9951017bab3`, which extends the XSAVE helpers to + account for supervisor-state components. +3. Adapt FreeBSD commit `bdb561843e86`, which implements Linux + `pkey_alloc(2)`, `pkey_free(2)`, and `pkey_mprotect(2)` using native amd64 + PKU support. +4. Treat FreeBSD commit `34718e01869b`, which maps `IFF_LOWER_UP` through + Linux `NETLINK_ROUTE`, as a separate follow-up because it fixes Brave + network detection rather than sandboxing. + +## Integration approach + +- Preserve the upstream split between common Linux syscall validation and + machine-dependent PKU operations. +- Keep protection-key allocation state in Linux per-process emulation data, + inherited on fork and reset on exec. +- Initialize PKRU to Linux's `0x55555554` default during Linux exec. +- Retain Linux-compatible no-PKU behavior on unsupported architectures. +- Adapt source and module Makefiles to MidnightBSD's current tree instead of + applying conflicting upstream hunks mechanically. +- Preserve existing syscall numbers and replace only their ENOSYS stubs. + +## Validation + +1. Run the repository C static-analysis scripts on staged C and header files. +2. Build the affected `linux_common`, Linux ABI modules, and amd64 kernel. +3. Exercise allocation, protection changes, access-right changes, fork + inheritance, exec reset, key exhaustion, and protection-key faults with a + small Linux test program. +4. Confirm protection-key faults translate to Linux `SEGV_PKUERR`. +5. Start Linux Brave on PKU-capable amd64 hardware and inspect its sandbox + status. +6. Test the no-PKU fallback where suitable hardware is available. + +## Commit structure + +- XSAVE query helpers. +- Linuxulator protection-key syscall support. +- Tests, if kept separate by the existing test layout. +- `UPDATING` entry as an independently reviewable commit, after approval. +- Optional `IFF_LOWER_UP` compatibility fix as a separate change. diff --git a/UPDATING b/UPDATING index e9774aca1cc..3d099f60226 100644 --- a/UPDATING +++ b/UPDATING @@ -1,5 +1,8 @@ Updating Information for MidnightBSD users. +20260925: + linuxulator: add protection-key syscalls for Chromium V8 sandboxing + 20260922: ncurses 6.6 (from 6.2). Among many bug fixes this closes CVE-2022-29458 and CVE-2023-29491 (memory corruption from a diff --git a/sys/amd64/amd64/fpu.c b/sys/amd64/amd64/fpu.c index 256bd83e720..df3c3c040ad 100644 --- a/sys/amd64/amd64/fpu.c +++ b/sys/amd64/amd64/fpu.c @@ -191,12 +191,15 @@ SYSCTL_INT(_hw, HW_FLOATINGPT, floatingpoint, CTLFLAG_RD, int use_xsave; /* non-static for cpu_switch.S */ uint64_t xsave_mask; /* the same */ +static uint64_t xsave_mask_supervisor; +static uint64_t xsave_extensions; static uma_zone_t fpu_save_area_zone; static struct savefpu *fpu_initialstate; static struct xsave_area_elm_descr { u_int offset; u_int size; + u_int flags; } *xsave_area_desc; static void @@ -349,6 +352,7 @@ fpuinit_bsp1(void) ctx_switch_xsave[3] |= 0x10; restore_wp(old_wp); } + xsave_mask_supervisor = ((uint64_t)cp[3] << 32) | cp[2]; } /* @@ -444,7 +448,7 @@ fpuinitstate(void *arg __unused) XSAVE_AREA_ALIGN - 1, 0); fpu_initialstate = uma_zalloc(fpu_save_area_zone, M_WAITOK | M_ZERO); if (use_xsave) { - max_ext_n = flsl(xsave_mask); + max_ext_n = flsl(xsave_mask | xsave_mask_supervisor); xsave_area_desc = malloc(max_ext_n * sizeof(struct xsave_area_elm_descr), M_DEVBUF, M_WAITOK | M_ZERO); } @@ -477,6 +481,9 @@ fpuinitstate(void *arg __unused) * Region of an XSAVE Area" for the source of offsets/sizes. */ if (use_xsave) { + cpuid_count(0xd, 1, cp); + xsave_extensions = cp[0]; + xstate_bv = (uint64_t *)((char *)(fpu_initialstate + 1) + offsetof(struct xstate_hdr, xstate_bv)); *xstate_bv = XFEATURE_ENABLED_X87 | XFEATURE_ENABLED_SSE; @@ -492,6 +499,7 @@ fpuinitstate(void *arg __unused) cpuid_count(0xd, i, cp); xsave_area_desc[i].offset = cp[1]; xsave_area_desc[i].size = cp[0]; + xsave_area_desc[i].flags = cp[2]; } } @@ -1312,3 +1320,89 @@ fpu_save_area_reset(struct savefpu *fsa) bcopy(fpu_initialstate, fsa, cpu_max_ext_state_size); } + +static __inline void +xsave_extfeature_check(uint64_t feature, bool supervisor) +{ + KASSERT((feature & (feature - 1)) == 0, + ("%s: invalid XFEATURE 0x%lx", __func__, feature)); + KASSERT(flsl(feature) <= flsl(supervisor ? xsave_mask_supervisor : + xsave_mask), + ("%s: unsupported %s XFEATURE 0x%lx", __func__, + supervisor ? "supervisor" : "user", feature)); +} + +static __inline void +xsave_extstate_bv_check(uint64_t xstate_bv, bool supervisor) +{ + KASSERT(xstate_bv != 0 && flsl(xstate_bv) <= + flsl(supervisor ? xsave_mask_supervisor : xsave_mask), + ("%s: invalid XSTATE_BV 0x%lx", __func__, xstate_bv)); +} + +bool +xsave_extfeature_supported(uint64_t feature, bool supervisor) +{ + uint64_t mask; + int idx; + + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + xsave_extfeature_check(feature, supervisor); + mask = supervisor ? xsave_mask_supervisor : xsave_mask; + if ((mask & feature) == 0) + return (false); + idx = flsl(feature) - 1; + return (((xsave_area_desc[idx].flags & + CPUID_EXTSTATE_SUPERVISOR) != 0) == supervisor); +} + +bool +xsave_extension_supported(uint64_t extension) +{ + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + return ((xsave_extensions & extension) != 0); +} + +size_t +xsave_area_offset(uint64_t xstate_bv, uint64_t feature, bool compact, + bool supervisor) +{ + struct xsave_area_elm_descr *xep; + size_t offs; + int i, idx; + + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + xsave_extstate_bv_check(xstate_bv, supervisor); + xsave_extfeature_check(feature, supervisor); + idx = flsl(feature) - 1; + if (!compact) + return (xsave_area_desc[idx].offset); + offs = sizeof(struct savefpu) + sizeof(struct xstate_hdr); + xstate_bv &= ~(XFEATURE_ENABLED_X87 | XFEATURE_ENABLED_SSE); + while ((i = ffs(xstate_bv) - 1) > 0 && i < idx) { + xep = &xsave_area_desc[i]; + if ((xep->flags & CPUID_EXTSTATE_ALIGNED) != 0) + offs = roundup2(offs, 64); + offs += xep->size; + xstate_bv &= ~((uint64_t)1 << i); + } + return (offs); +} + +size_t +xsave_area_size(uint64_t xstate_bv, bool compact, bool supervisor) +{ + int last_idx; + + KASSERT(use_xsave, ("%s: XSAVE not supported", __func__)); + xsave_extstate_bv_check(xstate_bv, supervisor); + last_idx = flsl(xstate_bv) - 1; + return (xsave_area_offset(xstate_bv, (uint64_t)1 << last_idx, compact, + supervisor) + xsave_area_desc[last_idx].size); +} + +size_t +xsave_area_hdr_offset(void) +{ + return (sizeof(struct savefpu)); +} diff --git a/sys/amd64/amd64/sys_machdep.c b/sys/amd64/amd64/sys_machdep.c index 3fbf44d9e48..9dc0c6da513 100644 --- a/sys/amd64/amd64/sys_machdep.c +++ b/sys/amd64/amd64/sys_machdep.c @@ -168,6 +168,32 @@ update_gdt_fsbase(struct thread *td, uint32_t base) critical_exit(); } +int +amd64_pkru_update(struct thread *td, uintptr_t addr, size_t len, u_int keyidx, + int flags, bool clear) +{ + struct vm_map *map; + vm_offset_t start, end; + int error; + + MPASS(td == curthread); + map = &td->td_proc->p_vmspace->vm_map; + vm_map_lock_read(map); + if (len == 0 || !vm_map_check_boundary(map, addr, addr + len)) { + vm_map_unlock_read(map); + return (EINVAL); + } + start = trunc_page(addr); + end = round_page(addr + len); + if (clear) + error = pmap_pkru_clear(PCPU_GET(curpmap), start, end); + else + error = pmap_pkru_set(PCPU_GET(curpmap), start, end, keyidx, + flags); + vm_map_unlock_read(map); + return (error); +} + int sysarch(struct thread *td, struct sysarch_args *uap) { diff --git a/sys/amd64/linux/linux_emul_md.h b/sys/amd64/linux/linux_emul_md.h new file mode 100644 index 00000000000..b594d0eba66 --- /dev/null +++ b/sys/amd64/linux/linux_emul_md.h @@ -0,0 +1,35 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#ifndef _AMD64_LINUX_EMUL_MD_H_ +#define _AMD64_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + uint32_t md_pkey_allocation_map; /* x86 protection keys */ +}; + +/* + * Initial protection key allocation map: key 0 is the default key, + * implicitly allocated on Linux (mm_pkey_allocation_map is initialized + * to 0x1). Inherited on fork, reset on exec. + */ +#define LINUX_PKEY_INITIAL_MAP 0x1 + +/* + * Initial PKRU at exec: access disabled for keys 1..15, key 0 open; + * the Linux init_pkru default. + */ +#define LINUX_PKRU_INIT 0x55555554 + +struct thread; + +void linux_pkru_exec_init(struct thread *); + +#endif /* !_AMD64_LINUX_EMUL_MD_H_ */ diff --git a/sys/amd64/linux/linux_pkru.c b/sys/amd64/linux/linux_pkru.c new file mode 100644 index 00000000000..79d86f4e7ce --- /dev/null +++ b/sys/amd64/linux/linux_pkru.c @@ -0,0 +1,255 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +/* + * x86 memory protection keys (PKU) for the Linuxulator, serving both + * the 64-bit and 32-bit Linux ABIs. + * + * The PKRU register is directly user-visible: Linux programs read and + * write it with RDPKRU/WRPKRU, which execute natively. The kernel's + * part is key allocation bookkeeping (per address space: inherited on + * fork, reset on exec, as with Linux mm->context.pkey_allocation_map), + * tagging pages (pkey_mprotect), and applying the initial access + * rights of pkey_alloc() to the calling thread's PKRU. + */ + +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include + +#include +#include + +static bool +linux_pkey_supported(void) +{ + + return ( + (cpu_stdext_feature2 & (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE)) == + (CPUID_STDEXT2_PKU | CPUID_STDEXT2_OSPKE) && + (xsave_mask & XFEATURE_ENABLED_PKRU) != 0); +} + +static bool +linux_pkey_range_valid(struct thread *td, uintptr_t addr, size_t len) +{ + struct vm_map *map; + bool valid; + + map = &td->td_proc->p_vmspace->vm_map; + vm_map_lock_read(map); + valid = vm_map_check_boundary(map, addr, addr + len); + vm_map_unlock_read(map); + return (valid); +} + +/* + * Update the calling thread's PKRU: new value is (PKRU & keep) | set. + */ +static void +linux_pkru_write(struct thread *td, uint32_t keep, uint32_t set) +{ + struct pcb *pcb; + struct xstate_hdr *hdr; + char *sa; + uint32_t *pkru; + + MPASS(td == curthread); + pcb = td->td_pcb; + + /* + * The critical section is held across the save area update to + * exclude preemption: a context switch could otherwise load the + * xsave area back into the CPU after fpugetregs(), and the + * stores below would then be lost to the next save. + */ + critical_enter(); + if ((pcb->pcb_flags & PCB_USERFPUINITDONE) != 0 && + td == PCPU_GET(fpcurthread) && PCB_USER_FPU(pcb)) { + wrpkru((rdpkru() & keep) | set); + critical_exit(); + return; + } + + /* + * The user FPU state is in the PCB save area, or is not yet + * initialized, in which case fpugetregs() installs the initial + * state there. + */ + (void)fpugetregs(td); + sa = (char *)get_pcb_user_save_td(td); + hdr = (struct xstate_hdr *)(sa + xsave_area_hdr_offset()); + pkru = (uint32_t *)(sa + + xsave_area_offset(xsave_mask, XFEATURE_ENABLED_PKRU, false, false)); + if ((hdr->xstate_bv & XFEATURE_ENABLED_PKRU) == 0) { + hdr->xstate_bv |= XFEATURE_ENABLED_PKRU; + *pkru = 0; + } + *pkru = (*pkru & keep) | set; + critical_exit(); +} + +/* + * Set the calling thread's PKRU access rights for the given key. + */ +static void +linux_pkru_set_perm(struct thread *td, u_int keyidx, uint32_t rights) +{ + + linux_pkru_write(td, ~(LINUX_PKEY_ACCESS_MASK << (keyidx * 2)), + rights << (keyidx * 2)); +} + +/* + * Called from the Linux sysvecs' exec_setregs. Linux initializes + * PKRU at exec to deny access to all keys but key 0 + * (arch/x86/mm/pkeys.c init_pkru_value), so memory tagged with a not + * yet allocated key is inaccessible; FreeBSD's initial PKRU is 0. + * This initializes the user FPU state slightly earlier than the lazy + * first-use path; the state would be initialized moments later in + * rtld/libc startup regardless. + */ +void +linux_pkru_exec_init(struct thread *td) +{ + + if (!linux_pkey_supported()) + return; + linux_pkru_write(td, 0, LINUX_PKRU_INIT); +} + +/* + * Protection keys are a property of the address space: inherit the + * allocation map on fork, as Linux does. When a FreeBSD process is + * switching to the Linux ABI there is no parent emuldata; start from + * the initial map. The unlocked read is atomic on the aligned word; + * a pkey_alloc() racing the fork in another thread yields a valid + * serialization either way. + */ +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ + struct linux_pemuldata *ppem; + + ppem = pem_find(td->td_proc); + if (ppem != NULL) + pem->pem_md.md_pkey_allocation_map = + ppem->pem_md.md_pkey_allocation_map; + else + pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP; +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ + + pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP; +} + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + struct linux_pemuldata *pem; + uint32_t free_keys; + int key; + + if (!linux_pkey_supported()) + return (ENOSPC); + + pem = pem_find(td->td_proc); + LINUX_PEM_XLOCK(pem); + free_keys = ~pem->pem_md.md_pkey_allocation_map & + ((1u << LINUX_PKEY_MAX) - 1) & ~LINUX_PKEY_INITIAL_MAP; + if (free_keys == 0) { + LINUX_PEM_XUNLOCK(pem); + return (ENOSPC); + } + key = ffs(free_keys) - 1; + pem->pem_md.md_pkey_allocation_map |= 1u << key; + LINUX_PEM_XUNLOCK(pem); + + linux_pkru_set_perm(td, key, init_val); + td->td_retval[0] = key; + return (0); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + struct linux_pemuldata *pem; + + if (!linux_pkey_supported()) + return (EINVAL); + + pem = pem_find(td->td_proc); + LINUX_PEM_XLOCK(pem); + if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) { + LINUX_PEM_XUNLOCK(pem); + return (EINVAL); + } + pem->pem_md.md_pkey_allocation_map &= ~(1u << pkey); + LINUX_PEM_XUNLOCK(pem); + + /* + * As on Linux, freeing a key neither untags pages nor updates + * PKRU; that is the application's responsibility. + */ + return (0); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + struct linux_pemuldata *pem; + int error; + + if (!linux_pkey_supported()) + return (EINVAL); + + pem = pem_find(td->td_proc); + LINUX_PEM_SLOCK(pem); + if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) { + LINUX_PEM_SUNLOCK(pem); + return (EINVAL); + } + LINUX_PEM_SUNLOCK(pem); + + /* + * Validate the PKRU tagging range before changing the ordinary page + * protections. amd64_pkru_update() repeats this check while applying + * the tag to synchronize with concurrent map changes. + */ + if (len != 0 && !linux_pkey_range_valid(td, addr, len)) + return (EINVAL); + + error = linux_mprotect_common(td, addr, len, prot); + if (error != 0 || len == 0) + return (error); + + /* + * Tag the range; a pkey of 0 untags it. The tag is not + * persistent: it dies with the mapping, matching Linux VMA + * semantics. + */ + return (amd64_pkru_update(td, addr, len, pkey, 0, pkey == 0)); +} diff --git a/sys/amd64/linux/linux_sysvec.c b/sys/amd64/linux/linux_sysvec.c index 92993c4548e..0eb175a0199 100644 --- a/sys/amd64/linux/linux_sysvec.c +++ b/sys/amd64/linux/linux_sysvec.c @@ -63,6 +63,7 @@ #include #include +#include #include #include #include @@ -277,6 +278,8 @@ linux_exec_setregs(struct thread *td, struct image_params *imgp, * clean FP state if it uses the FPU again. */ fpstate_drop(td); + + linux_pkru_exec_init(td); } static int diff --git a/sys/amd64/linux32/linux32_sysvec.c b/sys/amd64/linux32/linux32_sysvec.c index b4fb9fd1b1e..8dbdbc151ff 100644 --- a/sys/amd64/linux32/linux32_sysvec.c +++ b/sys/amd64/linux32/linux32_sysvec.c @@ -67,6 +67,7 @@ #include #include +#include #include #include #include @@ -611,6 +612,8 @@ linux_exec_setregs(struct thread *td, struct image_params *imgp, fpstate_drop(td); + linux_pkru_exec_init(td); + /* Do full restore on return so that we can change to a different %cs */ set_pcb_flags(pcb, PCB_32BIT | PCB_FULL_IRET); } diff --git a/sys/arm64/linux/linux_emul_md.c b/sys/arm64/linux/linux_emul_md.c new file mode 100644 index 00000000000..53bcf88e407 --- /dev/null +++ b/sys/arm64/linux/linux_emul_md.c @@ -0,0 +1,53 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#include +#include +#include +#include + +#include +#include + +/* No machine-dependent emuldata state yet. */ + +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ +} + +/* + * Protection key back ends: behave as Linux does on hardware without + * protection keys. pkey_alloc() reports no free keys and only the + * default key semantics remain. + */ + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + + return (ENOSPC); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + + return (EINVAL); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + + return (EINVAL); +} diff --git a/sys/arm64/linux/linux_emul_md.h b/sys/arm64/linux/linux_emul_md.h new file mode 100644 index 00000000000..0353f853167 --- /dev/null +++ b/sys/arm64/linux/linux_emul_md.h @@ -0,0 +1,18 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#ifndef _ARM64_LINUX_EMUL_MD_H_ +#define _ARM64_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + int md_dummy; /* no machine-dependent state yet */ +}; + +#endif /* !_ARM64_LINUX_EMUL_MD_H_ */ diff --git a/sys/compat/linux/linux_dummy.c b/sys/compat/linux/linux_dummy.c index b7dc490c9ff..566e9a4e879 100644 --- a/sys/compat/linux/linux_dummy.c +++ b/sys/compat/linux/linux_dummy.c @@ -123,9 +123,6 @@ DUMMY(mlock2); DUMMY(preadv2); DUMMY(pwritev2); /* Linux 4.8: */ -DUMMY(pkey_mprotect); -DUMMY(pkey_alloc); -DUMMY(pkey_free); DUMMY(open_tree); DUMMY(move_mount); DUMMY(fsopen); diff --git a/sys/compat/linux/linux_emul.c b/sys/compat/linux/linux_emul.c index 0c9527408a7..22dd9df4337 100644 --- a/sys/compat/linux/linux_emul.c +++ b/sys/compat/linux/linux_emul.c @@ -158,6 +158,7 @@ linux_proc_init(struct thread *td, struct thread *newtd, bool init_thread) pem = malloc(sizeof(*pem), M_LINUX, M_WAITOK | M_ZERO); sx_init(&pem->pem_sx, "lpemlk"); + linux_pemuldata_init_md(td, pem); p->p_emuldata = pem; } newtd->td_emuldata = em; @@ -184,6 +185,7 @@ linux_proc_init(struct thread *td, struct thread *newtd, bool init_thread) KASSERT(pem != NULL, ("proc_init: proc emuldata not found.\n")); pem->persona = 0; pem->oom_score_adj = 0; + linux_pemuldata_exec_md(pem); } } diff --git a/sys/compat/linux/linux_emul.h b/sys/compat/linux/linux_emul.h index 52a3cffe8f7..2a3cf5a6c42 100644 --- a/sys/compat/linux/linux_emul.h +++ b/sys/compat/linux/linux_emul.h @@ -30,6 +30,8 @@ #ifndef _LINUX_EMUL_H_ #define _LINUX_EMUL_H_ +#include + struct image_params; /* @@ -70,6 +72,7 @@ struct linux_pemuldata { uint32_t oom_score_adj; /* /proc/self/oom_score_adj */ uint32_t so_timestamp; /* requested timeval */ uint32_t so_timestampns; /* requested timespec */ + struct linux_pemuldata_md pem_md; /* machine-dependent state */ }; #define LINUX_PEM_XLOCK(p) sx_xlock(&(p)->pem_sx) @@ -79,4 +82,7 @@ struct linux_pemuldata { struct linux_pemuldata *pem_find(struct proc *); +void linux_pemuldata_init_md(struct thread *, struct linux_pemuldata *); +void linux_pemuldata_exec_md(struct linux_pemuldata *); + #endif /* !_LINUX_EMUL_H_ */ diff --git a/sys/compat/linux/linux_misc.c b/sys/compat/linux/linux_misc.c index bb3888a6325..02cb11aa920 100644 --- a/sys/compat/linux/linux_misc.c +++ b/sys/compat/linux/linux_misc.c @@ -75,6 +75,7 @@ #include #include #include +#include #include #include #include @@ -348,6 +349,28 @@ linux_msync(struct thread *td, struct linux_msync_args *args) args->fl & ~LINUX_MS_SYNC)); } +int +linux_pkey_mprotect(struct thread *td, struct linux_pkey_mprotect_args *uap) +{ + + return (linux_pkey_mprotect_common(td, uap->start, uap->len, + uap->prot, uap->pkey)); +} + +int +linux_pkey_alloc(struct thread *td, struct linux_pkey_alloc_args *uap) +{ + + return (linux_pkey_alloc_common(td, uap->flags, uap->init_val)); +} + +int +linux_pkey_free(struct thread *td, struct linux_pkey_free_args *uap) +{ + + return (linux_pkey_free_common(td, uap->pkey)); +} + #ifdef LINUX_LEGACY_SYSCALLS int linux_time(struct thread *td, struct linux_time_args *args) diff --git a/sys/compat/linux/linux_mmap.c b/sys/compat/linux/linux_mmap.c index d371c1b0935..d806d439dbb 100644 --- a/sys/compat/linux/linux_mmap.c +++ b/sys/compat/linux/linux_mmap.c @@ -243,6 +243,44 @@ linux_mprotect_common(struct thread *td, uintptr_t addr, size_t len, int prot) return (kern_mprotect(td, addr, len, prot, flags)); } +/* + * x86 memory protection keys. The common entry points perform the + * parameter validation Linux applies regardless of hardware support, + * then defer to the machine-dependent back end. + */ + +int +linux_pkey_alloc_common(struct thread *td, uint64_t flags, uint64_t init_val) +{ + + if (flags != 0) + return (EINVAL); + if ((init_val & ~(uint64_t)LINUX_PKEY_ACCESS_MASK) != 0) + return (EINVAL); + return (linux_pkey_alloc_machdep(td, init_val)); +} + +int +linux_pkey_free_common(struct thread *td, int pkey) +{ + + if (pkey < 0 || pkey >= LINUX_PKEY_MAX) + return (EINVAL); + return (linux_pkey_free_machdep(td, pkey)); +} + +int +linux_pkey_mprotect_common(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + + if (pkey < -1 || pkey >= LINUX_PKEY_MAX) + return (EINVAL); + if (pkey == -1) + return (linux_mprotect_common(td, addr, len, prot)); + return (linux_pkey_mprotect_machdep(td, addr, len, prot, pkey)); +} + /* * Implement Linux madvise(MADV_DONTNEED), which has unusual semantics: for * anonymous memory, pages in the range are immediately discarded. diff --git a/sys/compat/linux/linux_mmap.h b/sys/compat/linux/linux_mmap.h index 043dec9d40b..f6267a7eb1a 100644 --- a/sys/compat/linux/linux_mmap.h +++ b/sys/compat/linux/linux_mmap.h @@ -66,6 +66,18 @@ int linux_mmap_common(struct thread *, uintptr_t, size_t, int, int, int, off_t); int linux_mprotect_common(struct thread *, uintptr_t, size_t, int); +int linux_pkey_alloc_common(struct thread *, uint64_t, uint64_t); +int linux_pkey_free_common(struct thread *, int); +int linux_pkey_mprotect_common(struct thread *, uintptr_t, size_t, int, int); +int linux_pkey_alloc_machdep(struct thread *, uint64_t); +int linux_pkey_free_machdep(struct thread *, int); +int linux_pkey_mprotect_machdep(struct thread *, uintptr_t, size_t, int, int); + +#define LINUX_PKEY_DISABLE_ACCESS 0x1 +#define LINUX_PKEY_DISABLE_WRITE 0x2 +#define LINUX_PKEY_ACCESS_MASK (LINUX_PKEY_DISABLE_ACCESS | \ + LINUX_PKEY_DISABLE_WRITE) +#define LINUX_PKEY_MAX 16 int linux_madvise_common(struct thread *, uintptr_t, size_t, int); #endif /* _LINUX_MMAP_H_ */ diff --git a/sys/i386/linux/linux_emul_md.c b/sys/i386/linux/linux_emul_md.c new file mode 100644 index 00000000000..53bcf88e407 --- /dev/null +++ b/sys/i386/linux/linux_emul_md.c @@ -0,0 +1,53 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#include +#include +#include +#include + +#include +#include + +/* No machine-dependent emuldata state yet. */ + +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ +} + +/* + * Protection key back ends: behave as Linux does on hardware without + * protection keys. pkey_alloc() reports no free keys and only the + * default key semantics remain. + */ + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + + return (ENOSPC); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + + return (EINVAL); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + + return (EINVAL); +} diff --git a/sys/i386/linux/linux_emul_md.h b/sys/i386/linux/linux_emul_md.h new file mode 100644 index 00000000000..9cfe6363d02 --- /dev/null +++ b/sys/i386/linux/linux_emul_md.h @@ -0,0 +1,18 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske + */ + +#ifndef _I386_LINUX_EMUL_MD_H_ +#define _I386_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + int md_dummy; /* no machine-dependent state yet */ +}; + +#endif /* !_I386_LINUX_EMUL_MD_H_ */ diff --git a/sys/modules/linux/Makefile b/sys/modules/linux/Makefile index 431db42ba0b..0b9d2a51719 100644 --- a/sys/modules/linux/Makefile +++ b/sys/modules/linux/Makefile @@ -66,6 +66,7 @@ SRCS+= imgact_linux.c \ linux.c \ linux_dummy.c \ linux_emul.c \ + linux_emul_md.c \ linux_errno.c \ linux_mib.c \ linux_mmap.c \ diff --git a/sys/modules/linux_common/Makefile b/sys/modules/linux_common/Makefile index 63d90e64a89..68492424fdc 100644 --- a/sys/modules/linux_common/Makefile +++ b/sys/modules/linux_common/Makefile @@ -1,7 +1,10 @@ .PATH: ${SRCTOP}/sys/compat/linux .if ${MACHINE_CPUARCH} == "amd64" -.PATH: ${SRCTOP}/sys/x86/linux +.PATH: ${SRCTOP}/sys/amd64/linux ${SRCTOP}/sys/x86/linux +.endif +.if ${MACHINE_CPUARCH} == "aarch64" +.PATH: ${SRCTOP}/sys/arm64/linux .endif KMOD= linux_common @@ -10,7 +13,10 @@ SRCS= linux_common.c linux_mib.c linux_mmap.c linux_util.c linux_emul.c \ linux.c device_if.h vnode_if.h bus_if.h opt_inet6.h opt_inet.h .if ${MACHINE_CPUARCH} == "amd64" -SRCS+= linux_x86.c linux_vdso_selector_x86.c +SRCS+= linux_pkru.c linux_x86.c linux_vdso_selector_x86.c +.endif +.if ${MACHINE_CPUARCH} == "aarch64" +SRCS+= linux_emul_md.c .endif EXPORT_SYMS= diff --git a/sys/x86/include/fpu.h b/sys/x86/include/fpu.h index e1ec6a592d2..a7cb3453065 100644 --- a/sys/x86/include/fpu.h +++ b/sys/x86/include/fpu.h @@ -213,4 +213,13 @@ struct savefpu_ymm { */ #define X86_XSTATE_XCR0_OFFSET 464 +#ifdef _KERNEL +bool xsave_extfeature_supported(uint64_t feature, bool supervisor); +bool xsave_extension_supported(uint64_t extension); +size_t xsave_area_hdr_offset(void); +size_t xsave_area_offset(uint64_t xstate_bv, uint64_t feature, bool compact, + bool supervisor); +size_t xsave_area_size(uint64_t xstate_bv, bool compact, bool supervisor); +#endif + #endif /* !_X86_FPU_H_ */ diff --git a/sys/x86/include/specialreg.h b/sys/x86/include/specialreg.h index 02cc23b562c..53e2643034b 100644 --- a/sys/x86/include/specialreg.h +++ b/sys/x86/include/specialreg.h @@ -370,6 +370,13 @@ #define CPUID_EXTSTATE_XINUSE 0x00000004 #define CPUID_EXTSTATE_XSAVES 0x00000008 +/* + * CPUID instruction 0xd Processor Extended State Enumeration, + * sub-leaf greater than 1, ECX information. + */ +#define CPUID_EXTSTATE_SUPERVISOR 0x00000001 +#define CPUID_EXTSTATE_ALIGNED 0x00000002 + /* * AMD extended function 8000_0007h ebx info */ diff --git a/sys/x86/include/sysarch.h b/sys/x86/include/sysarch.h index 3226f3b9d93..654354c5b97 100644 --- a/sys/x86/include/sysarch.h +++ b/sys/x86/include/sysarch.h @@ -160,6 +160,7 @@ int amd64_set_ldt(struct thread *, struct i386_ldt_args *, struct user_segment_descriptor *); int amd64_get_ioperm(struct thread *, struct i386_ioperm_args *); int amd64_set_ioperm(struct thread *, struct i386_ioperm_args *); +int amd64_pkru_update(struct thread *, uintptr_t, size_t, u_int, int, bool); #endif #endif /* !_MACHINE_SYSARCH_H_ */