aboutsummaryrefslogtreecommitdiff
path: root/sys/amd64/linux
diff options
context:
space:
mode:
authorDevin Teske <dteske@FreeBSD.org>2026-08-16 00:58:19 +0000
committerDevin Teske <dteske@FreeBSD.org>2026-08-16 00:58:43 +0000
commitbdb561843e865eaa5bbdc5394ed9d9c91136240c (patch)
tree7fe259ff3acd58946f51470a14653efde87650ca /sys/amd64/linux
parent5b48968c1a57bd1a7f086d7e09add59afa158340 (diff)
linux: implement pkey_alloc, pkey_free and pkey_mprotect
Bridge the Linux memory protection key syscalls to FreeBSD's native MPK support instead of returning ENOSYS. Modern Linux software probes these at startup: Chromium-based browsers (found via www/linux-brave) use protection keys for V8's heap and JIT sandboxing, and glibc >= 2.27 exposes the full API. pkey_alloc() allocates from a per-process bitmap kept in the process emuldata (key 0 implicitly allocated, matching Linux's mm_pkey_allocation_map; ENOSPC once keys 1..15 are exhausted or when PKU is absent, as Linux returns on such hardware) and applies the requested initial access rights to the calling thread's PKRU, located in the XSAVE area via xsave_area_offset(). pkey_free() is bookkeeping only: as on Linux, freeing neither untags pages nor updates PKRU. pkey_mprotect() performs the protection change and tags the range through amd64_pkru_update(), factored out of sysarch(2)'s AMD64_SET_PKRU/AMD64_CLEAR_PKRU implementation so that both share the same argument checking and map read lock synchronization with a parallel pmap_vmspace_copy() on fork; tags die with the mapping, matching Linux VMA semantics. A pkey of -1 degrades to plain mprotect. The allocation map is inherited on fork and reset on exec. At exec the Linux sysvecs initialize PKRU to 0x55555554, Linux's init_pkru default (access disabled for keys 1..15), so memory tagged with a not yet allocated key is inaccessible to threads that were never granted rights -- the property V8's thread isolation relies on. Setting PKRU at exec initializes the user FPU state slightly earlier than the lazy first-use path; the state would be initialized moments later in rtld/libc startup regardless. Protection key faults already deliver SEGV_PKUERR through the existing siginfo translation. The common code carries no architecture ifdefs. Machine-dependent state lives in struct linux_pemuldata_md, embedded in the process emuldata in the manner of struct mdthread, and common code calls per-arch lifecycle hooks (linux_pemuldata_init_md/_exec_md) and pkey back ends after performing the parameter validation Linux applies regardless of hardware support. On amd64 the implementation lives in sys/amd64/linux/linux_pkru.c, compiled into linux_common and serving both the 64-bit and 32-bit Linux ABIs. Elsewhere (arm64, i386) linux_emul_md.c provides stubs returning what Linux returns on hardware without protection keys (ENOSPC from pkey_alloc; pkey_mprotect with a pkey of -1 acts as plain mprotect), so applications take their normal no-PKU fallback instead of the ENOSYS path. PR: 297427 MFC after: 1 month Reviewed by: kib Differential Revision: https://reviews.freebsd.org/D58782
Diffstat (limited to 'sys/amd64/linux')
-rw-r--r--sys/amd64/linux/linux_emul_md.h35
-rw-r--r--sys/amd64/linux/linux_pkru.c226
-rw-r--r--sys/amd64/linux/linux_sysvec.c4
3 files changed, 265 insertions, 0 deletions
diff --git a/sys/amd64/linux/linux_emul_md.h b/sys/amd64/linux/linux_emul_md.h
new file mode 100644
index 000000000000..a5ea9c20e20a
--- /dev/null
+++ b/sys/amd64/linux/linux_emul_md.h
@@ -0,0 +1,35 @@
+/*
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * Copyright (c) 2026 Devin Teske <dteske@FreeBSD.org>
+ */
+
+#ifndef _AMD64_LINUX_EMUL_MD_H_
+#define _AMD64_LINUX_EMUL_MD_H_
+
+/*
+ * Machine-dependent part of the Linux process emuldata, embedded in
+ * struct linux_pemuldata as pem_md.
+ */
+struct linux_pemuldata_md {
+ uint32_t md_pkey_allocation_map; /* x86 protection keys */
+};
+
+/*
+ * Initial protection key allocation map: key 0 is the default key,
+ * implicitly allocated on Linux (mm_pkey_allocation_map is initialized
+ * to 0x1). Inherited on fork, reset on exec.
+ */
+#define LINUX_PKEY_INITIAL_MAP 0x1
+
+/*
+ * Initial PKRU at exec: access disabled for keys 1..15, key 0 open;
+ * the Linux init_pkru default.
+ */
+#define LINUX_PKRU_INIT 0x55555554
+
+struct thread;
+
+void linux_pkru_exec_init(struct thread *);
+
+#endif /* !_AMD64_LINUX_EMUL_MD_H_ */
diff --git a/sys/amd64/linux/linux_pkru.c b/sys/amd64/linux/linux_pkru.c
new file mode 100644
index 000000000000..159f8492ff35
--- /dev/null
+++ b/sys/amd64/linux/linux_pkru.c
@@ -0,0 +1,226 @@
+/*
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * Copyright (c) 2026 Devin Teske <dteske@FreeBSD.org>
+ */
+
+/*
+ * x86 memory protection keys (PKU) for the Linuxulator, serving both
+ * the 64-bit and 32-bit Linux ABIs.
+ *
+ * The PKRU register is directly user-visible: Linux programs read and
+ * write it with RDPKRU/WRPKRU, which execute natively. The kernel's
+ * part is key allocation bookkeeping (per address space: inherited on
+ * fork, reset on exec, as with Linux mm->context.pkey_allocation_map),
+ * tagging pages (pkey_mprotect), and applying the initial access
+ * rights of pkey_alloc() to the calling thread's PKRU.
+ */
+
+#include <sys/systm.h>
+#include <sys/imgact.h>
+#include <sys/lock.h>
+#include <sys/pcpu.h>
+#include <sys/proc.h>
+#include <sys/sx.h>
+
+#include <machine/cpufunc.h>
+#include <machine/fpu.h>
+#include <machine/md_var.h>
+#include <machine/pcb.h>
+#include <machine/specialreg.h>
+#include <machine/sysarch.h>
+#include <x86/x86_var.h>
+
+#include <compat/linux/linux_emul.h>
+#include <compat/linux/linux_mmap.h>
+
+static bool
+linux_pkey_supported(void)
+{
+
+ return ((cpu_stdext_feature2 & CPUID_STDEXT2_OSPKE) != 0);
+}
+
+/*
+ * Update the calling thread's PKRU: new value is (PKRU & keep) | set.
+ */
+static void
+linux_pkru_write(struct thread *td, uint32_t keep, uint32_t set)
+{
+ struct pcb *pcb;
+ struct xstate_hdr *hdr;
+ char *sa;
+ uint32_t *pkru;
+
+ MPASS(td == curthread);
+ pcb = td->td_pcb;
+
+ /*
+ * The critical section is held across the save area update to
+ * exclude preemption: a context switch could otherwise load the
+ * xsave area back into the CPU after fpugetregs(), and the
+ * stores below would then be lost to the next save.
+ */
+ critical_enter();
+ if ((pcb->pcb_flags & PCB_USERFPUINITDONE) != 0 &&
+ td == PCPU_GET(fpcurthread) && PCB_USER_FPU(pcb)) {
+ wrpkru((rdpkru() & keep) | set);
+ critical_exit();
+ return;
+ }
+
+ /*
+ * The user FPU state is in the PCB save area, or is not yet
+ * initialized, in which case fpugetregs() installs the initial
+ * state there.
+ */
+ (void)fpugetregs(td);
+ sa = (char *)get_pcb_user_save_td(td);
+ hdr = (struct xstate_hdr *)(sa + xsave_area_hdr_offset());
+ pkru = (uint32_t *)(sa + xsave_area_offset(xsave_mask,
+ XFEATURE_ENABLED_PKRU, false, false));
+ if ((hdr->xstate_bv & XFEATURE_ENABLED_PKRU) == 0) {
+ hdr->xstate_bv |= XFEATURE_ENABLED_PKRU;
+ *pkru = 0;
+ }
+ *pkru = (*pkru & keep) | set;
+ critical_exit();
+}
+
+/*
+ * Set the calling thread's PKRU access rights for the given key.
+ */
+static void
+linux_pkru_set_perm(struct thread *td, u_int keyidx, uint32_t rights)
+{
+
+ linux_pkru_write(td, ~(LINUX_PKEY_ACCESS_MASK << (keyidx * 2)),
+ rights << (keyidx * 2));
+}
+
+/*
+ * Called from the Linux sysvecs' exec_setregs. Linux initializes
+ * PKRU at exec to deny access to all keys but key 0
+ * (arch/x86/mm/pkeys.c init_pkru_value), so memory tagged with a not
+ * yet allocated key is inaccessible; FreeBSD's initial PKRU is 0.
+ * This initializes the user FPU state slightly earlier than the lazy
+ * first-use path; the state would be initialized moments later in
+ * rtld/libc startup regardless.
+ */
+void
+linux_pkru_exec_init(struct thread *td)
+{
+
+ if (!linux_pkey_supported())
+ return;
+ linux_pkru_write(td, 0, LINUX_PKRU_INIT);
+}
+
+/*
+ * Protection keys are a property of the address space: inherit the
+ * allocation map on fork, as Linux does. When a FreeBSD process is
+ * switching to the Linux ABI there is no parent emuldata; start from
+ * the initial map. The unlocked read is atomic on the aligned word;
+ * a pkey_alloc() racing the fork in another thread yields a valid
+ * serialization either way.
+ */
+void
+linux_pemuldata_init_md(struct thread *td, struct linux_pemuldata *pem)
+{
+ struct linux_pemuldata *ppem;
+
+ ppem = pem_find(td->td_proc);
+ if (ppem != NULL)
+ pem->pem_md.md_pkey_allocation_map =
+ ppem->pem_md.md_pkey_allocation_map;
+ else
+ pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP;
+}
+
+void
+linux_pemuldata_exec_md(struct linux_pemuldata *pem)
+{
+
+ pem->pem_md.md_pkey_allocation_map = LINUX_PKEY_INITIAL_MAP;
+}
+
+int
+linux_pkey_alloc_machdep(struct thread *td, uint64_t init_val)
+{
+ struct linux_pemuldata *pem;
+ uint32_t free_keys;
+ int key;
+
+ if (!linux_pkey_supported())
+ return (ENOSPC);
+
+ pem = pem_find(td->td_proc);
+ LINUX_PEM_XLOCK(pem);
+ free_keys = ~pem->pem_md.md_pkey_allocation_map &
+ ((1u << LINUX_PKEY_MAX) - 1) & ~LINUX_PKEY_INITIAL_MAP;
+ if (free_keys == 0) {
+ LINUX_PEM_XUNLOCK(pem);
+ return (ENOSPC);
+ }
+ key = ffs(free_keys) - 1;
+ pem->pem_md.md_pkey_allocation_map |= 1u << key;
+ LINUX_PEM_XUNLOCK(pem);
+
+ linux_pkru_set_perm(td, key, init_val);
+ td->td_retval[0] = key;
+ return (0);
+}
+
+int
+linux_pkey_free_machdep(struct thread *td, int pkey)
+{
+ struct linux_pemuldata *pem;
+
+ if (!linux_pkey_supported())
+ return (EINVAL);
+
+ pem = pem_find(td->td_proc);
+ LINUX_PEM_XLOCK(pem);
+ if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) {
+ LINUX_PEM_XUNLOCK(pem);
+ return (EINVAL);
+ }
+ pem->pem_md.md_pkey_allocation_map &= ~(1u << pkey);
+ LINUX_PEM_XUNLOCK(pem);
+
+ /*
+ * As on Linux, freeing a key neither untags pages nor updates
+ * PKRU; that is the application's responsibility.
+ */
+ return (0);
+}
+
+int
+linux_pkey_mprotect_machdep(struct thread *td, uintptr_t addr, size_t len,
+ int prot, int pkey)
+{
+ struct linux_pemuldata *pem;
+ int error;
+
+ if (!linux_pkey_supported())
+ return (EINVAL);
+
+ pem = pem_find(td->td_proc);
+ LINUX_PEM_SLOCK(pem);
+ if ((pem->pem_md.md_pkey_allocation_map & (1u << pkey)) == 0) {
+ LINUX_PEM_SUNLOCK(pem);
+ return (EINVAL);
+ }
+ LINUX_PEM_SUNLOCK(pem);
+
+ error = linux_mprotect_common(td, addr, len, prot);
+ if (error != 0 || len == 0)
+ return (error);
+
+ /*
+ * Tag the range; a pkey of 0 untags it. The tag is not
+ * persistent: it dies with the mapping, matching Linux VMA
+ * semantics.
+ */
+ return (amd64_pkru_update(td, addr, len, pkey, 0, pkey == 0));
+}
diff --git a/sys/amd64/linux/linux_sysvec.c b/sys/amd64/linux/linux_sysvec.c
index 890cf01c46a0..ecb497c61a1a 100644
--- a/sys/amd64/linux/linux_sysvec.c
+++ b/sys/amd64/linux/linux_sysvec.c
@@ -57,6 +57,7 @@
#include <x86/linux/linux_x86.h>
#include <amd64/linux/linux.h>
+#include <amd64/linux/linux_emul_md.h>
#include <amd64/linux/linux_proto.h>
#include <compat/linux/linux_elf.h>
#include <compat/linux/linux_emul.h>
@@ -271,6 +272,9 @@ linux_exec_setregs(struct thread *td, struct image_params *imgp,
* clean FP state if it uses the FPU again.
*/
fpstate_drop(td);
+
+ /* Linux processes start with PKRU denying unallocated keys. */
+ linux_pkru_exec_init(td);
}
static int