Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 52 additions & 0 deletions LINUX_BRAVE_SANDBOX_BACKPORT.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
# Linux Brave Sandbox Backport Plan

## Goal

Backport FreeBSD's Linuxulator memory protection key support so
Chromium-based Linux applications, including Brave, can use the V8 heap and
JIT sandbox on MidnightBSD/amd64.

## Upstream changes

1. Backport FreeBSD commit `7bcaff05223e`, which exposes XSAVE feature and
save-area layout information.
2. Backport FreeBSD commit `b9951017bab3`, which extends the XSAVE helpers to
account for supervisor-state components.
3. Adapt FreeBSD commit `bdb561843e86`, which implements Linux
`pkey_alloc(2)`, `pkey_free(2)`, and `pkey_mprotect(2)` using native amd64
PKU support.
4. Treat FreeBSD commit `34718e01869b`, which maps `IFF_LOWER_UP` through
Linux `NETLINK_ROUTE`, as a separate follow-up because it fixes Brave
network detection rather than sandboxing.

## Integration approach

- Preserve the upstream split between common Linux syscall validation and
machine-dependent PKU operations.
- Keep protection-key allocation state in Linux per-process emulation data,
inherited on fork and reset on exec.
- Initialize PKRU to Linux's `0x55555554` default during Linux exec.
- Retain Linux-compatible no-PKU behavior on unsupported architectures.
- Adapt source and module Makefiles to MidnightBSD's current tree instead of
applying conflicting upstream hunks mechanically.
- Preserve existing syscall numbers and replace only their ENOSYS stubs.

## Validation

1. Run the repository C static-analysis scripts on staged C and header files.
2. Build the affected `linux_common`, Linux ABI modules, and amd64 kernel.
3. Exercise allocation, protection changes, access-right changes, fork
inheritance, exec reset, key exhaustion, and protection-key faults with a
small Linux test program.
4. Confirm protection-key faults translate to Linux `SEGV_PKUERR`.
5. Start Linux Brave on PKU-capable amd64 hardware and inspect its sandbox
status.
6. Test the no-PKU fallback where suitable hardware is available.

## Commit structure

- XSAVE query helpers.
- Linuxulator protection-key syscall support.
- Tests, if kept separate by the existing test layout.
- `UPDATING` entry as an independently reviewable commit, after approval.
- Optional `IFF_LOWER_UP` compatibility fix as a separate change.
3 changes: 3 additions & 0 deletions UPDATING
Original file line number Diff line number Diff line change
@@ -1,5 +1,8 @@
Updating Information for MidnightBSD users.

20260925:
linuxulator: add protection-key syscalls for Chromium V8 sandboxing

20260922:
ncurses 6.6 (from 6.2). Among many bug fixes this closes
CVE-2022-29458 and CVE-2023-29491 (memory corruption from a
Expand Down
96 changes: 95 additions & 1 deletion sys/amd64/amd64/fpu.c
Original file line number Diff line number Diff line change
Expand Up @@ -191,12 +191,15 @@ SYSCTL_INT(_hw, HW_FLOATINGPT, floatingpoint, CTLFLAG_RD,

int use_xsave; /* non-static for cpu_switch.S */
uint64_t xsave_mask; /* the same */
static uint64_t xsave_mask_supervisor;
static uint64_t xsave_extensions;
static uma_zone_t fpu_save_area_zone;
static struct savefpu *fpu_initialstate;

static struct xsave_area_elm_descr {
u_int offset;
u_int size;
u_int flags;
} *xsave_area_desc;

static void
Expand Down Expand Up @@ -349,6 +352,7 @@ fpuinit_bsp1(void)
ctx_switch_xsave[3] |= 0x10;
restore_wp(old_wp);
}
xsave_mask_supervisor = ((uint64_t)cp[3] << 32) | cp[2];
}

/*
Expand Down Expand Up @@ -444,7 +448,7 @@ fpuinitstate(void *arg __unused)
XSAVE_AREA_ALIGN - 1, 0);
fpu_initialstate = uma_zalloc(fpu_save_area_zone, M_WAITOK | M_ZERO);
if (use_xsave) {
max_ext_n = flsl(xsave_mask);
max_ext_n = flsl(xsave_mask | xsave_mask_supervisor);
xsave_area_desc = malloc(max_ext_n * sizeof(struct
xsave_area_elm_descr), M_DEVBUF, M_WAITOK | M_ZERO);
}
Expand Down Expand Up @@ -477,6 +481,9 @@ fpuinitstate(void *arg __unused)
* Region of an XSAVE Area" for the source of offsets/sizes.
*/
if (use_xsave) {
cpuid_count(0xd, 1, cp);
xsave_extensions = cp[0];

xstate_bv = (uint64_t *)((char *)(fpu_initialstate + 1) +
offsetof(struct xstate_hdr, xstate_bv));
*xstate_bv = XFEATURE_ENABLED_X87 | XFEATURE_ENABLED_SSE;
Expand All @@ -492,6 +499,7 @@ fpuinitstate(void *arg __unused)
cpuid_count(0xd, i, cp);
xsave_area_desc[i].offset = cp[1];
xsave_area_desc[i].size = cp[0];
xsave_area_desc[i].flags = cp[2];
}
}

Expand Down Expand Up @@ -1312,3 +1320,89 @@ fpu_save_area_reset(struct savefpu *fsa)

bcopy(fpu_initialstate, fsa, cpu_max_ext_state_size);
}

static __inline void
xsave_extfeature_check(uint64_t feature, bool supervisor)
{
KASSERT((feature & (feature - 1)) == 0,
("%s: invalid XFEATURE 0x%lx", __func__, feature));
KASSERT(flsl(feature) <= flsl(supervisor ? xsave_mask_supervisor :
xsave_mask),
("%s: unsupported %s XFEATURE 0x%lx", __func__,
supervisor ? "supervisor" : "user", feature));
}

static __inline void
xsave_extstate_bv_check(uint64_t xstate_bv, bool supervisor)
{
KASSERT(xstate_bv != 0 && flsl(xstate_bv) <=
flsl(supervisor ? xsave_mask_supervisor : xsave_mask),
("%s: invalid XSTATE_BV 0x%lx", __func__, xstate_bv));
}

bool
xsave_extfeature_supported(uint64_t feature, bool supervisor)
{
uint64_t mask;
int idx;

KASSERT(use_xsave, ("%s: XSAVE not supported", __func__));
xsave_extfeature_check(feature, supervisor);
mask = supervisor ? xsave_mask_supervisor : xsave_mask;
if ((mask & feature) == 0)
return (false);
idx = flsl(feature) - 1;
return (((xsave_area_desc[idx].flags &
CPUID_EXTSTATE_SUPERVISOR) != 0) == supervisor);
}

bool
xsave_extension_supported(uint64_t extension)
{
KASSERT(use_xsave, ("%s: XSAVE not supported", __func__));
return ((xsave_extensions & extension) != 0);
}

size_t
xsave_area_offset(uint64_t xstate_bv, uint64_t feature, bool compact,
bool supervisor)
{
struct xsave_area_elm_descr *xep;
size_t offs;
int i, idx;

KASSERT(use_xsave, ("%s: XSAVE not supported", __func__));
xsave_extstate_bv_check(xstate_bv, supervisor);
xsave_extfeature_check(feature, supervisor);
idx = flsl(feature) - 1;
if (!compact)
return (xsave_area_desc[idx].offset);
offs = sizeof(struct savefpu) + sizeof(struct xstate_hdr);
xstate_bv &= ~(XFEATURE_ENABLED_X87 | XFEATURE_ENABLED_SSE);
while ((i = ffs(xstate_bv) - 1) > 0 && i < idx) {

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Low: xstate_bv is uint64_t but ffs() takes int, so any component bit >= 32 is silently dropped from the compact-offset walk. Not reachable today (the only caller passes compact=false, and no enabled component is above bit 31), but the loop is wrong as written: with a high component set in xstate_bv the loop terminates early and returns a too-small offset. ffsl() (or ffsll()) is the intended primitive.

xep = &xsave_area_desc[i];
if ((xep->flags & CPUID_EXTSTATE_ALIGNED) != 0)
offs = roundup2(offs, 64);
offs += xep->size;
xstate_bv &= ~((uint64_t)1 << i);
}
return (offs);
}

size_t
xsave_area_size(uint64_t xstate_bv, bool compact, bool supervisor)
{
int last_idx;

KASSERT(use_xsave, ("%s: XSAVE not supported", __func__));
xsave_extstate_bv_check(xstate_bv, supervisor);
last_idx = flsl(xstate_bv) - 1;
return (xsave_area_offset(xstate_bv, (uint64_t)1 << last_idx, compact,
supervisor) + xsave_area_desc[last_idx].size);
}

size_t
xsave_area_hdr_offset(void)
{
return (sizeof(struct savefpu));
}
26 changes: 26 additions & 0 deletions sys/amd64/amd64/sys_machdep.c
Original file line number Diff line number Diff line change
Expand Up @@ -168,6 +168,32 @@ update_gdt_fsbase(struct thread *td, uint32_t base)
critical_exit();
}

int
amd64_pkru_update(struct thread *td, uintptr_t addr, size_t len, u_int keyidx,
int flags, bool clear)
{
struct vm_map *map;
vm_offset_t start, end;
int error;

MPASS(td == curthread);
map = &td->td_proc->p_vmspace->vm_map;
vm_map_lock_read(map);
if (len == 0 || !vm_map_check_boundary(map, addr, addr + len)) {
vm_map_unlock_read(map);
return (EINVAL);
}
start = trunc_page(addr);
end = round_page(addr + len);
if (clear)
error = pmap_pkru_clear(PCPU_GET(curpmap), start, end);
else
error = pmap_pkru_set(PCPU_GET(curpmap), start, end, keyidx,
flags);
vm_map_unlock_read(map);
return (error);
}

int
sysarch(struct thread *td, struct sysarch_args *uap)
{
Expand Down
35 changes: 35 additions & 0 deletions sys/amd64/linux/linux_emul_md.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
/*
* SPDX-License-Identifier: BSD-2-Clause
*
* Copyright (c) 2026 Devin Teske <dteske@FreeBSD.org>
*/

#ifndef _AMD64_LINUX_EMUL_MD_H_
#define _AMD64_LINUX_EMUL_MD_H_

/*
* Machine-dependent part of the Linux process emuldata, embedded in
* struct linux_pemuldata as pem_md.
*/
struct linux_pemuldata_md {
uint32_t md_pkey_allocation_map; /* x86 protection keys */
};

/*
* Initial protection key allocation map: key 0 is the default key,
* implicitly allocated on Linux (mm_pkey_allocation_map is initialized
* to 0x1). Inherited on fork, reset on exec.
*/
#define LINUX_PKEY_INITIAL_MAP 0x1

/*
* Initial PKRU at exec: access disabled for keys 1..15, key 0 open;
* the Linux init_pkru default.
Comment thread
sourcery-ai[bot] marked this conversation as resolved.
*/
#define LINUX_PKRU_INIT 0x55555554

struct thread;

void linux_pkru_exec_init(struct thread *);

#endif /* !_AMD64_LINUX_EMUL_MD_H_ */
Loading
Loading