diff options
| author | Devin Teske <dteske@FreeBSD.org> | 2026-08-16 00:58:19 +0000 |
|---|---|---|
| committer | Devin Teske <dteske@FreeBSD.org> | 2026-08-16 00:58:43 +0000 |
| commit | bdb561843e865eaa5bbdc5394ed9d9c91136240c (patch) | |
| tree | 7fe259ff3acd58946f51470a14653efde87650ca /sys/amd64/linux | |
| parent | 5b48968c1a57bd1a7f086d7e09add59afa158340 (diff) | |
linux: implement pkey_alloc, pkey_free and pkey_mprotect
Bridge the Linux memory protection key syscalls to FreeBSD's native
MPK support instead of returning ENOSYS. Modern Linux software
probes these at startup: Chromium-based browsers (found via
www/linux-brave) use protection keys for V8's heap and JIT
sandboxing, and glibc >= 2.27 exposes the full API.
pkey_alloc() allocates from a per-process bitmap kept in the process
emuldata (key 0 implicitly allocated, matching Linux's
mm_pkey_allocation_map; ENOSPC once keys 1..15 are exhausted or when
PKU is absent, as Linux returns on such hardware) and applies the
requested initial access rights to the calling thread's PKRU, located
in the XSAVE area via xsave_area_offset(). pkey_free() is
bookkeeping only: as on Linux, freeing neither untags pages nor
updates PKRU. pkey_mprotect() performs the protection change and
tags the range through amd64_pkru_update(), factored out of
sysarch(2)'s AMD64_SET_PKRU/AMD64_CLEAR_PKRU implementation so that
both share the same argument checking and map read lock
synchronization with a parallel pmap_vmspace_copy() on fork; tags die
with the mapping, matching Linux VMA semantics. A pkey of -1
degrades to plain mprotect.
The allocation map is inherited on fork and reset on exec. At exec
the Linux sysvecs initialize PKRU to 0x55555554, Linux's init_pkru
default (access disabled for keys 1..15), so memory tagged with a
not yet allocated key is inaccessible to threads that were never
granted rights -- the property V8's thread isolation relies on.
Setting PKRU at exec initializes the user FPU state slightly earlier
than the lazy first-use path; the state would be initialized moments
later in rtld/libc startup regardless. Protection key faults
already deliver SEGV_PKUERR through the existing siginfo
translation.
The common code carries no architecture ifdefs. Machine-dependent
state lives in struct linux_pemuldata_md, embedded in the process
emuldata in the manner of struct mdthread, and common code calls
per-arch lifecycle hooks (linux_pemuldata_init_md/_exec_md) and pkey
back ends after performing the parameter validation Linux applies
regardless of hardware support. On amd64 the implementation lives
in sys/amd64/linux/linux_pkru.c, compiled into linux_common and
serving both the 64-bit and 32-bit Linux ABIs. Elsewhere (arm64,
i386) linux_emul_md.c provides stubs returning what Linux returns on
hardware without protection keys (ENOSPC from pkey_alloc;
pkey_mprotect with a pkey of -1 acts as plain mprotect), so
applications take their normal no-PKU fallback instead of the ENOSYS
path.
PR: 297427
MFC after: 1 month
Reviewed by: kib
Differential Revision: https://reviews.freebsd.org/D58782
Diffstat (limited to 'sys/amd64/linux')
| -rw-r--r-- | sys/amd64/linux/linux_emul_md.h | 35 | ||||
| -rw-r--r-- | sys/amd64/linux/linux_pkru.c | 226 | ||||
| -rw-r--r-- | sys/amd64/linux/linux_sysvec.c | 4 |
3 files changed, 265 insertions, 0 deletions
diff --git a/sys/amd64/linux/linux_emul_md.h b/sys/amd64/linux/linux_emul_md.h new file mode 100644 index 000000000000..a5ea9c20e20a --- /dev/null +++ b/sys/amd64/linux/linux_emul_md.h @@ -0,0 +1,35 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske <dteske@FreeBSD.org> + */ + +#ifndef _AMD64_LINUX_EMUL_MD_H_ +#define _AMD64_LINUX_EMUL_MD_H_ + +/* + * Machine-dependent part of the Linux process emuldata, embedded in + * struct linux_pemuldata as pem_md. + */ +struct linux_pemuldata_md { + uint32_t md_pkey_allocation_map; /* x86 protection keys */ +}; + +/* + * Initial protection key allocation map: key 0 is the default key, + * implicitly allocated on Linux (mm_pkey_allocation_map is initialized + * to 0x1). Inherited on fork, reset on exec. + */ +#define LINUX_PKEY_INITIAL_MAP 0x1 + +/* + * Initial PKRU at exec: access disabled for keys 1..15, key 0 open; + * the Linux init_pkru default. + */ +#define LINUX_PKRU_INIT 0x55555554 + +struct thread; + +void linux_pkru_exec_init(struct thread *); + +#endif /* !_AMD64_LINUX_EMUL_MD_H_ */ diff --git a/sys/amd64/linux/linux_pkru.c b/sys/amd64/linux/linux_pkru.c new file mode 100644 index 000000000000..159f8492ff35 --- /dev/null +++ b/sys/amd64/linux/linux_pkru.c @@ -0,0 +1,226 @@ +/* + * SPDX-License-Identifier: BSD-2-Clause + * + * Copyright (c) 2026 Devin Teske <dteske@FreeBSD.org> + */ + +/* + * x86 memory protection keys (PKU) for the Linuxulator, serving both + * the 64-bit and 32-bit Linux ABIs. + * + * The PKRU register is directly user-visible: Linux programs read and + * write it with RDPKRU/WRPKRU, which execute natively. The kernel's + * part is key allocation bookkeeping (per address space: inherited on + * fork, reset on exec, as with Linux mm->context.pkey_allocation_map), + * tagging pages (pkey_mprotect), and applying the initial access + * rights of pkey_alloc() to the calling thread's PKRU. + */ + +#include <sys/systm.h> +#include <sys/imgact.h> +#include <sys/lock.h> +#include <sys/pcpu.h> +#include <sys/proc.h> +#include <sys/sx.h> + +#include <machine/cpufunc.h> +#include <machine/fpu.h> +#include <machine/md_var.h> +#include <machine/pcb.h> +#include <machine/specialreg.h> +#include <machine/sysarch.h> +#include <x86/x86_var.h> + +#include <compat/linux/linux_emul.h> +#include <compat/linux/linux_mmap.h> + +static bool +linux_pkey_supported(void) +{ + + return ((cpu_stdext_feature2 & CPUID_STDEXT2_OSPKE) != 0); +} + +/* + * Update the calling thread's PKRU: new value is (PKRU & keep) | set. + */ +static void +linux_pkru_write(struct thread *td, uint32_t keep, uint32_t set) +{ + struct pcb *pcb; + struct xstate_hdr *hdr; + char *sa; + uint32_t *pkru; + + MPASS(td == curthread); + pcb = td->td_pcb; + + /* + * The critical section is held across the save area update to + * exclude preemption: a context switch could otherwise load the + * xsave area back into the CPU after fpugetregs(), and the + * stores below would then be lost to the next save. + */ + critical_enter(); + if ((pcb->pcb_flags & PCB_USERFPUINITDONE) != 0 && + td == PCPU_GET(fpcurthread) && PCB_USER_FPU(pcb)) { + wrpkru((rdpkru() & keep) | set); + critical_exit(); + return; + } + + /* + * The user FPU state is in the PCB save area, or is not yet + * initialized, in which case fpugetregs() installs the initial + * state there. + */ + (void)fpugetregs(td); + sa = (char *)get_pcb_user_save_td(td); + hdr = (struct xstate_hdr *)(sa + xsave_area_hdr_offset()); + pkru = (uint32_t *)(sa + xsave_area_offset(xsave_mask, + XFEATURE_ENABLED_PKRU, false, false)); + if ((hdr->xstate_bv & XFEATURE_ENABLED_PKRU) == 0) { + hdr->xstate_bv |= XFEATURE_ENABLED_PKRU; + *pkru = 0; + } + *pkru = (*pkru & keep) | set; + critical_exit(); +} + +/* + * Set the calling thread's PKRU access rights for the given key. + */ +static void +linux_pkru_set_perm(struct thread *td, u_int keyidx, uint32_t rights) +{ + + linux_pkru_write(td, ~(LINUX_PKEY_ACCESS_MASK << (keyidx * 2)), + rights << (keyidx * 2)); +} + +/* + * Called from the Linux sysvecs' exec_setregs. Linux initializes + * PKRU at exec to deny access to all keys but key 0 + * (arch/x86/mm/pkeys.c init_pkru_value), so memory tagged with a not + * yet allocated key is inaccessible; FreeBSD's initial PKRU is 0. + * This initializes the user FPU state slightly earlier than the lazy + * first-use path; the state would be initialized moments later in + * rtld/libc startup regardless. + */ +void +linux_pkru_exec_init(struct thread *td) +{ + + if (!linux_pkey_supported()) + return; + linux_pkru_write(td, 0, LINUX_PKRU_INIT); +} + +/* + * Protection keys are a property of the address space: inherit the + * allocation map on fork, as Linux does. When a FreeBSD process is + * switching to the Linux ABI there is no parent emuldata; start from + * the initial map. The unlocked read is atomic on the aligned word; + * a pkey_alloc() racing the fork in another thread yields a valid + * serialization either way. + */ +void +linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem) +{ + struct linux_pemuldata *ppem; + + ppem = pem_find(td->td_proc); + if (ppem != NULL) + pem->pem_md.md_pkey_allocation_map = + ppem->pem_md.md_pkey_allocation_map; + else + pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP; +} + +void +linux_pemuldata_exec_md(struct linux_pemuldata *pem) +{ + + pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP; +} + +int +linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val) +{ + struct linux_pemuldata *pem; + uint32_t free_keys; + int key; + + if (!linux_pkey_supported()) + return (ENOSPC); + + pem = pem_find(td->td_proc); + LINUX_PEM_XLOCK(pem); + free_keys = ~pem->pem_md.md_pkey_allocation_map & + ((1u << LINUX_PKEY_MAX) - 1) & ~LINUX_PKEY_INITIAL_MAP; + if (free_keys == 0) { + LINUX_PEM_XUNLOCK(pem); + return (ENOSPC); + } + key = ffs(free_keys) - 1; + pem->pem_md.md_pkey_allocation_map |= 1u << key; + LINUX_PEM_XUNLOCK(pem); + + linux_pkru_set_perm(td, key, init_val); + td->td_retval[0] = key; + return (0); +} + +int +linux_pkey_free_machdep(struct thread *td, int pkey) +{ + struct linux_pemuldata *pem; + + if (!linux_pkey_supported()) + return (EINVAL); + + pem = pem_find(td->td_proc); + LINUX_PEM_XLOCK(pem); + if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) { + LINUX_PEM_XUNLOCK(pem); + return (EINVAL); + } + pem->pem_md.md_pkey_allocation_map &= ~(1u << pkey); + LINUX_PEM_XUNLOCK(pem); + + /* + * As on Linux, freeing a key neither untags pages nor updates + * PKRU; that is the application's responsibility. + */ + return (0); +} + +int +linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len, + int prot, int pkey) +{ + struct linux_pemuldata *pem; + int error; + + if (!linux_pkey_supported()) + return (EINVAL); + + pem = pem_find(td->td_proc); + LINUX_PEM_SLOCK(pem); + if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) { + LINUX_PEM_SUNLOCK(pem); + return (EINVAL); + } + LINUX_PEM_SUNLOCK(pem); + + error = linux_mprotect_common(td, addr, len, prot); + if (error != 0 || len == 0) + return (error); + + /* + * Tag the range; a pkey of 0 untags it. The tag is not + * persistent: it dies with the mapping, matching Linux VMA + * semantics. + */ + return (amd64_pkru_update(td, addr, len, pkey, 0, pkey == 0)); +} diff --git a/sys/amd64/linux/linux_sysvec.c b/sys/amd64/linux/linux_sysvec.c index 890cf01c46a0..ecb497c61a1a 100644 --- a/sys/amd64/linux/linux_sysvec.c +++ b/sys/amd64/linux/linux_sysvec.c @@ -57,6 +57,7 @@ #include <x86/linux/linux_x86.h> #include <amd64/linux/linux.h> +#include <amd64/linux/linux_emul_md.h> #include <amd64/linux/linux_proto.h> #include <compat/linux/linux_elf.h> #include <compat/linux/linux_emul.h> @@ -271,6 +272,9 @@ linux_exec_setregs(struct thread *td, struct image_params *imgp, * clean FP state if it uses the FPU again. */ fpstate_drop(td); + + /* Linux processes start with PKRU denying unallocated keys. */ + linux_pkru_exec_init(td); } static int |
