aboutsummaryrefslogtreecommitdiff
path: root/sys/dev
diff options
context:
space:
mode:
authorMark Johnston <markj@FreeBSD.org>2019-09-09 21:32:42 +0000
committerMark Johnston <markj@FreeBSD.org>2019-09-09 21:32:42 +0000
commitfee2a2fa39834d8d5eaa981298fce9d2ed31546d (patch)
tree290b84257a055cb0fbd4eb498ca16690ad749aa3 /sys/dev
parent58a11be1cf32a2ad832d7167bd41819cb105851e (diff)
Change synchonization rules for vm_page reference counting.
There are several mechanisms by which a vm_page reference is held, preventing the page from being freed back to the page allocator. In particular, holding the page's object lock is sufficient to prevent the page from being freed; holding the busy lock or a wiring is sufficent as well. These references are protected by the page lock, which must therefore be acquired for many per-page operations. This results in false sharing since the page locks are external to the vm_page structures themselves and each lock protects multiple structures. Transition to using an atomically updated per-page reference counter. The object's reference is counted using a flag bit in the counter. A second flag bit is used to atomically block new references via pmap_extract_and_hold() while removing managed mappings of a page. Thus, the reference count of a page is guaranteed not to increase if the page is unbusied, unmapped, and the object's write lock is held. As a consequence of this, the page lock no longer protects a page's identity; operations which move pages between objects are now synchronized solely by the objects' locks. The vm_page_wire() and vm_page_unwire() KPIs are changed. The former requires that either the object lock or the busy lock is held. The latter no longer has a return value and may free the page if it releases the last reference to that page. vm_page_unwire_noq() behaves the same as before; the caller is responsible for checking its return value and freeing or enqueuing the page as appropriate. vm_page_wire_mapped() is introduced for use in pmap_extract_and_hold(). It fails if the page is concurrently being unmapped, typically triggering a fallback to the fault handler. vm_page_wire() no longer requires the page lock and vm_page_unwire() now internally acquires the page lock when releasing the last wiring of a page (since the page lock still protects a page's queue state). In particular, synchronization details are no longer leaked into the caller. The change excises the page lock from several frequently executed code paths. In particular, vm_object_terminate() no longer bounces between page locks as it releases an object's pages, and direct I/O and sendfile(SF_NOCACHE) completions no longer require the page lock. In these latter cases we now get linear scalability in the common scenario where different threads are operating on different files. __FreeBSD_version is bumped. The DRM ports have been updated to accomodate the KPI changes. Reviewed by: jeff (earlier version) Tested by: gallatin (earlier version), pho Sponsored by: Netflix Differential Revision: https://reviews.freebsd.org/D20486
Notes
svn path=/head/; revision=352110
Diffstat (limited to 'sys/dev')
-rw-r--r--sys/dev/agp/agp.c6
-rw-r--r--sys/dev/agp/agp_i810.c2
-rw-r--r--sys/dev/cxgbe/tom/t4_cpl_io.c5
-rw-r--r--sys/dev/cxgbe/tom/t4_ddp.c2
-rw-r--r--sys/dev/drm2/ttm/ttm_bo_vm.c4
-rw-r--r--sys/dev/drm2/ttm/ttm_page_alloc.c2
-rw-r--r--sys/dev/drm2/ttm/ttm_tt.c2
-rw-r--r--sys/dev/md/md.c2
-rw-r--r--sys/dev/netmap/netmap_freebsd.c2
-rw-r--r--sys/dev/xen/gntdev/gntdev.c6
-rw-r--r--sys/dev/xen/privcmd/privcmd.c6
11 files changed, 6 insertions, 33 deletions
diff --git a/sys/dev/agp/agp.c b/sys/dev/agp/agp.c
index 011c89afeb37..62664e8dfe87 100644
--- a/sys/dev/agp/agp.c
+++ b/sys/dev/agp/agp.c
@@ -616,9 +616,7 @@ bad:
m = vm_page_lookup(mem->am_obj, OFF_TO_IDX(k));
if (k >= i)
vm_page_xunbusy(m);
- vm_page_lock(m);
vm_page_unwire(m, PQ_INACTIVE);
- vm_page_unlock(m);
}
VM_OBJECT_WUNLOCK(mem->am_obj);
@@ -653,9 +651,7 @@ agp_generic_unbind_memory(device_t dev, struct agp_memory *mem)
VM_OBJECT_WLOCK(mem->am_obj);
for (i = 0; i < mem->am_size; i += PAGE_SIZE) {
m = vm_page_lookup(mem->am_obj, atop(i));
- vm_page_lock(m);
vm_page_unwire(m, PQ_INACTIVE);
- vm_page_unlock(m);
}
VM_OBJECT_WUNLOCK(mem->am_obj);
@@ -1003,7 +999,7 @@ agp_bind_pages(device_t dev, vm_page_t *pages, vm_size_t size,
mtx_lock(&sc->as_lock);
for (i = 0; i < size; i += PAGE_SIZE) {
m = pages[OFF_TO_IDX(i)];
- KASSERT(m->wire_count > 0,
+ KASSERT(vm_page_wired(m),
("agp_bind_pages: page %p hasn't been wired", m));
/*
diff --git a/sys/dev/agp/agp_i810.c b/sys/dev/agp/agp_i810.c
index 27d7f1114a08..501f78ca0a32 100644
--- a/sys/dev/agp/agp_i810.c
+++ b/sys/dev/agp/agp_i810.c
@@ -1795,9 +1795,7 @@ agp_i810_free_memory(device_t dev, struct agp_memory *mem)
*/
VM_OBJECT_WLOCK(mem->am_obj);
m = vm_page_lookup(mem->am_obj, 0);
- vm_page_lock(m);
vm_page_unwire(m, PQ_INACTIVE);
- vm_page_unlock(m);
VM_OBJECT_WUNLOCK(mem->am_obj);
} else {
contigfree(sc->argb_cursor, mem->am_size, M_AGP);
diff --git a/sys/dev/cxgbe/tom/t4_cpl_io.c b/sys/dev/cxgbe/tom/t4_cpl_io.c
index c698f0f7d69e..5269cf3ad0fa 100644
--- a/sys/dev/cxgbe/tom/t4_cpl_io.c
+++ b/sys/dev/cxgbe/tom/t4_cpl_io.c
@@ -1910,7 +1910,6 @@ aiotx_free_pgs(struct mbuf *m)
{
struct mbuf_ext_pgs *ext_pgs;
struct kaiocb *job;
- struct mtx *mtx;
vm_page_t pg;
MBUF_EXT_PGS_ASSERT(m);
@@ -1921,14 +1920,10 @@ aiotx_free_pgs(struct mbuf *m)
m->m_len, jobtotid(job));
#endif
- mtx = NULL;
for (int i = 0; i < ext_pgs->npgs; i++) {
pg = PHYS_TO_VM_PAGE(ext_pgs->pa[i]);
- vm_page_change_lock(pg, &mtx);
vm_page_unwire(pg, PQ_ACTIVE);
}
- if (mtx != NULL)
- mtx_unlock(mtx);
aiotx_free_job(job);
}
diff --git a/sys/dev/cxgbe/tom/t4_ddp.c b/sys/dev/cxgbe/tom/t4_ddp.c
index e460d2cb6a7d..0d42a0289edf 100644
--- a/sys/dev/cxgbe/tom/t4_ddp.c
+++ b/sys/dev/cxgbe/tom/t4_ddp.c
@@ -114,9 +114,7 @@ free_pageset(struct tom_data *td, struct pageset *ps)
for (i = 0; i < ps->npages; i++) {
p = ps->pages[i];
- vm_page_lock(p);
vm_page_unwire(p, PQ_INACTIVE);
- vm_page_unlock(p);
}
mtx_lock(&ddp_orphan_pagesets_lock);
TAILQ_INSERT_TAIL(&ddp_orphan_pagesets, ps, link);
diff --git a/sys/dev/drm2/ttm/ttm_bo_vm.c b/sys/dev/drm2/ttm/ttm_bo_vm.c
index 43d027fc5cd9..6dc1fabab28c 100644
--- a/sys/dev/drm2/ttm/ttm_bo_vm.c
+++ b/sys/dev/drm2/ttm/ttm_bo_vm.c
@@ -114,9 +114,7 @@ ttm_bo_vm_fault(vm_object_t vm_obj, vm_ooffset_t offset,
vm_object_pip_add(vm_obj, 1);
if (*mres != NULL) {
- vm_page_lock(*mres);
(void)vm_page_remove(*mres);
- vm_page_unlock(*mres);
}
retry:
VM_OBJECT_WUNLOCK(vm_obj);
@@ -261,9 +259,7 @@ reserve:
vm_page_xbusy(m);
if (*mres != NULL) {
KASSERT(*mres != m, ("losing %p %p", *mres, m));
- vm_page_lock(*mres);
vm_page_free(*mres);
- vm_page_unlock(*mres);
}
*mres = m;
diff --git a/sys/dev/drm2/ttm/ttm_page_alloc.c b/sys/dev/drm2/ttm/ttm_page_alloc.c
index 1e9055175449..fbb830405de0 100644
--- a/sys/dev/drm2/ttm/ttm_page_alloc.c
+++ b/sys/dev/drm2/ttm/ttm_page_alloc.c
@@ -132,7 +132,7 @@ ttm_vm_page_free(vm_page_t m)
{
KASSERT(m->object == NULL, ("ttm page %p is owned", m));
- KASSERT(m->wire_count == 1, ("ttm lost wire %p", m));
+ KASSERT(vm_page_wired(m), ("ttm lost wire %p", m));
KASSERT((m->flags & PG_FICTITIOUS) != 0, ("ttm lost fictitious %p", m));
KASSERT((m->oflags & VPO_UNMANAGED) == 0, ("ttm got unmanaged %p", m));
m->flags &= ~PG_FICTITIOUS;
diff --git a/sys/dev/drm2/ttm/ttm_tt.c b/sys/dev/drm2/ttm/ttm_tt.c
index 1e2db3cd8755..82aaddf4b1d4 100644
--- a/sys/dev/drm2/ttm/ttm_tt.c
+++ b/sys/dev/drm2/ttm/ttm_tt.c
@@ -294,9 +294,7 @@ int ttm_tt_swapin(struct ttm_tt *ttm)
rv = vm_pager_get_pages(obj, &from_page, 1,
NULL, NULL);
if (rv != VM_PAGER_OK) {
- vm_page_lock(from_page);
vm_page_free(from_page);
- vm_page_unlock(from_page);
ret = -EIO;
goto err_ret;
}
diff --git a/sys/dev/md/md.c b/sys/dev/md/md.c
index c9cd5a6e95a0..110cbfdecc87 100644
--- a/sys/dev/md/md.c
+++ b/sys/dev/md/md.c
@@ -1029,9 +1029,7 @@ md_swap_page_free(vm_page_t m)
{
vm_page_xunbusy(m);
- vm_page_lock(m);
vm_page_free(m);
- vm_page_unlock(m);
}
static int
diff --git a/sys/dev/netmap/netmap_freebsd.c b/sys/dev/netmap/netmap_freebsd.c
index 59837840eed9..42551df09c2a 100644
--- a/sys/dev/netmap/netmap_freebsd.c
+++ b/sys/dev/netmap/netmap_freebsd.c
@@ -1052,9 +1052,7 @@ netmap_dev_pager_fault(vm_object_t object, vm_ooffset_t offset,
VM_OBJECT_WUNLOCK(object);
page = vm_page_getfake(paddr, memattr);
VM_OBJECT_WLOCK(object);
- vm_page_lock(*mres);
vm_page_free(*mres);
- vm_page_unlock(*mres);
*mres = page;
vm_page_insert(page, object, pidx);
}
diff --git a/sys/dev/xen/gntdev/gntdev.c b/sys/dev/xen/gntdev/gntdev.c
index ed42e177b860..667d46f333b3 100644
--- a/sys/dev/xen/gntdev/gntdev.c
+++ b/sys/dev/xen/gntdev/gntdev.c
@@ -826,14 +826,12 @@ gntdev_gmap_pg_fault(vm_object_t object, vm_ooffset_t offset, int prot,
KASSERT((page->flags & PG_FICTITIOUS) != 0,
("not fictitious %p", page));
- KASSERT(page->wire_count == 1, ("wire_count not 1 %p", page));
- KASSERT(vm_page_busied(page) == 0, ("page %p is busy", page));
+ KASSERT(vm_page_wired(page), ("page %p is not wired", page));
+ KASSERT(!vm_page_busied(page), ("page %p is busy", page));
if (*mres != NULL) {
oldm = *mres;
- vm_page_lock(oldm);
vm_page_free(oldm);
- vm_page_unlock(oldm);
*mres = NULL;
}
diff --git a/sys/dev/xen/privcmd/privcmd.c b/sys/dev/xen/privcmd/privcmd.c
index e09886f42adb..3b6b2033e80f 100644
--- a/sys/dev/xen/privcmd/privcmd.c
+++ b/sys/dev/xen/privcmd/privcmd.c
@@ -169,14 +169,12 @@ privcmd_pg_fault(vm_object_t object, vm_ooffset_t offset,
KASSERT((page->flags & PG_FICTITIOUS) != 0,
("not fictitious %p", page));
- KASSERT(page->wire_count == 1, ("wire_count not 1 %p", page));
- KASSERT(vm_page_busied(page) == 0, ("page %p is busy", page));
+ KASSERT(vm_page_wired(page), ("page %p not wired", page));
+ KASSERT(!vm_page_busied(page), ("page %p is busy", page));
if (*mres != NULL) {
oldm = *mres;
- vm_page_lock(oldm);
vm_page_free(oldm);
- vm_page_unlock(oldm);
*mres = NULL;
}