Commit 7ca43d7ad0 for qemu.org
commit 7ca43d7ad02fc1aee56866caf3fb9b1b851872ce
Author: Peter Xu <peterx@redhat.com>
Date: Tue Sep 29 18:50:16 2026 -0500
hostmem: Support fully shared guest memfd to back a VM
Host backends supports guest-memfd now by detecting whether it's a
confidential VM. There's no way to choose it yet from the memory level to
use it fully shared. If we use guest-memfd, it so far always implies we
need two layers of memory backends, while the guest-memfd only provides the
private set of pages.
This patch introduces a way so that QEMU can consume guest memfd as the
only source of memory to back the object (aka, fully shared).
To use the fully shared guest-memfd, one can add a memfd object with:
-object memory-backend-memfd,x-guest-memfd=on,share=on
Note that share=on is required with fully shared guest_memfd.
Since guest-memfd is not yet a drop-in replacement for memfd (e.g. lack
of hugepage support), the option is tagged as experimental for now so
that it can be made available for early enablement/testing without
causing any surprises for other users.
PS: there's a trivial touch-up on fd<0 check, because the stub to create
guest-memfd may return negative but not -1.
Reviewed-by: Xiaoyao Li <xiaoyao.li@intel.com>
Reviewed-by: Fabiano Rosas <farosas@suse.de>
Reviewed-by: David Hildenbrand <david@kernel.org>
Co-developed-by: Michael Roth <michael.roth@amd.com>
Signed-off-by: Michael Roth <michael.roth@amd.com>
Acked-by: Markus Armbruster <armbru@redhat.com>
Link: https://lore.kernel.org/r/20260929235109.1398509-11-michael.roth@amd.com
Signed-off-by: Peter Xu <peterx@redhat.com>
diff --git a/backends/hostmem-memfd.c b/backends/hostmem-memfd.c
index 7a7855aaab..666593ff6c 100644
--- a/backends/hostmem-memfd.c
+++ b/backends/hostmem-memfd.c
@@ -18,6 +18,9 @@
#include "qapi/error.h"
#include "qom/object.h"
#include "migration/cpr.h"
+#include "system/kvm.h"
+#include <linux/kvm.h>
+#include "qapi/qapi-visit-common.h"
OBJECT_DECLARE_SIMPLE_TYPE(HostMemoryBackendMemfd, MEMORY_BACKEND_MEMFD)
@@ -28,6 +31,13 @@ struct HostMemoryBackendMemfd {
bool hugetlb;
uint64_t hugetlbsize;
bool seal;
+ /*
+ * NOTE: this differs from HostMemoryBackend's guest_memfd_private,
+ * which represents an internally private guest-memfd that only backs
+ * private pages. Instead, this flag marks the memory backend will
+ * 100% use the guest-memfd pages in-place.
+ */
+ OnOffAuto guest_memfd;
};
static bool
@@ -47,11 +57,31 @@ memfd_backend_memory_alloc(HostMemoryBackend *backend, Error **errp)
goto have_fd;
}
- fd = qemu_memfd_create(TYPE_MEMORY_BACKEND_MEMFD, backend->size,
- m->hugetlb, m->hugetlbsize, m->seal ?
- F_SEAL_GROW | F_SEAL_SHRINK | F_SEAL_SEAL : 0,
- errp);
- if (fd == -1) {
+ if (m->guest_memfd == ON_OFF_AUTO_ON) {
+ /*
+ * NOTE: guest-memfd ignores seal=on/off because it always
+ * implicitly seals the FD by definition.
+ */
+ if (!backend->share) {
+ error_setg(errp, "guest-memfd=on must be used with share=on");
+ return false;
+ } else if (m->hugetlb) {
+ error_setg(errp, "guest-memfd=on doesn't support hugetlb=on");
+ return false;
+ }
+
+ fd = kvm_create_guest_memfd(backend->size,
+ GUEST_MEMFD_FLAG_MMAP |
+ GUEST_MEMFD_FLAG_INIT_SHARED,
+ errp);
+ } else {
+ fd = qemu_memfd_create(TYPE_MEMORY_BACKEND_MEMFD, backend->size,
+ m->hugetlb, m->hugetlbsize, m->seal ?
+ F_SEAL_GROW | F_SEAL_SHRINK | F_SEAL_SEAL : 0,
+ errp);
+ }
+
+ if (fd < 0) {
return false;
}
cpr_save_fd(name, 0, fd);
@@ -65,6 +95,26 @@ have_fd:
backend->size, ram_flags, fd, 0, errp);
}
+static void
+memfd_backend_get_guest_memfd(Object *o, Visitor *v,
+ const char *value, void *opaque,
+ Error **errp)
+{
+ HostMemoryBackendMemfd *m = MEMORY_BACKEND_MEMFD(o);
+
+ visit_type_OnOffAuto(v, value, &m->guest_memfd, errp);
+}
+
+static void
+memfd_backend_set_guest_memfd(Object *o, Visitor *v,
+ const char *value, void *opaque,
+ Error **errp)
+{
+ HostMemoryBackendMemfd *m = MEMORY_BACKEND_MEMFD(o);
+
+ visit_type_OnOffAuto(v, value, &m->guest_memfd, errp);
+}
+
static bool
memfd_backend_get_hugetlb(Object *o, Error **errp)
{
@@ -152,6 +202,13 @@ memfd_backend_class_init(ObjectClass *oc, const void *data)
object_class_property_set_description(oc, "hugetlbsize",
"Huge pages size (ex: 2M, 1G)");
}
+
+ object_class_property_add(oc, "guest-memfd", "OnOffAuto",
+ memfd_backend_get_guest_memfd,
+ memfd_backend_set_guest_memfd, NULL, NULL);
+ object_class_property_set_description(oc, "guest-memfd",
+ "Use guest memfd");
+
object_class_property_add_bool(oc, "seal",
memfd_backend_get_seal,
memfd_backend_set_seal);
diff --git a/docs/system/confidential-guest-support.rst b/docs/system/confidential-guest-support.rst
index 562a7c3c28..58732d658f 100644
--- a/docs/system/confidential-guest-support.rst
+++ b/docs/system/confidential-guest-support.rst
@@ -1,3 +1,5 @@
+.. _confidential-guest-support:
+
Confidential Guest Support
==========================
diff --git a/docs/system/guest-memfd.rst b/docs/system/guest-memfd.rst
new file mode 100644
index 0000000000..d62a214a91
--- /dev/null
+++ b/docs/system/guest-memfd.rst
@@ -0,0 +1,60 @@
+.. _guest-memfd:
+
+guest-memfd support
+===================
+
+Recent kernels allow for the creation of a guest-memfd file
+descriptor, which can be used to back VMs in a similar manner as a
+memfd file descriptor, but is intended specifically for this purpose
+and allows for closer coordination between KVM and the management of
+this memory to enable more advanced/VM-specific use-cases.
+
+Initially this additional functionality centered around providing a
+common/centralized place for managing kernel-side memory handing
+requirements for various Confidential Guest architectures. (For more
+on Confidential Guests, see :ref:`confidential-guest-support`).
+
+guest_memfd has since evolved to become a more general-purpose way to
+allocate/manage guest memory and potentially allow for things like
+providing additional memory isolation within the kernel[1] and support
+for persisting a guest's state across kexec to allow for live-updating
+the host kernel with minimal guest downtime[2].
+
+Usage
+-----
+
+For Confidential Guests, guest-memfd is currently utilized internally
+by QEMU to handle private guest memory, independently of whatever
+memory backend the user has configured for normal/non-private/shared
+guest memory. To avoid doubling memory, QEMU discards memory in
+response to the guest converting GPA ranges between shared/private.
+(e.g. if GPA X is converted from private to shared, the guest-memfd FD
+offset corresponding to GPA x will be truncated since the memory will
+be provided by the memory backend the user configured for shared
+memory, and vice-versa). This is handled automatically/internally for
+Confidential Guests that rely on this handling and is not directly
+exposed by QEMU command-line options.
+
+For non-Confidential guests, guest-memfd can be used in a manner that
+is somewhat interchangeable with a normal memfd. Currently, this is
+handled by using the same memory-backend implementation as memfd, but
+with an additional 'guest-memfd=on' option. E.g.::
+
+ qemu ... \
+ -object memory-backend-memfd,id=ID,size=SIZE,share=on,guest-memfd=on
+
+Note that the share=on option is required for guest-memfd, since it
+does not support anonymous memory allocations or COW-like semantics.
+
+Also note that there are a couple of limitations in using guest_memfd
+in this way compared to a normal memfd, which is why the option is
+exposed as an experimental for the time being:
+
+ * guest_memfd does not currently support the hugetlb=on option
+ * guest_memfd does not currently support Transparent Huge Pages
+
+References
+----------
+
+- `[1] directmap removal <https://lore.kernel.org/kvm/20260317141031.514-1-kalyazin@amazon.com/>`__
+- `[2] LUO <https://lore.kernel.org/kvm/20260728121138.1103610-1-tarunsahu@google.com/>`__
diff --git a/docs/system/index.rst b/docs/system/index.rst
index 4509630fa4..060850a739 100644
--- a/docs/system/index.rst
+++ b/docs/system/index.rst
@@ -44,3 +44,4 @@ or Hypervisor.Framework.
vm-templating
sriov
qemu-colo
+ guest-memfd
diff --git a/qapi/qom.json b/qapi/qom.json
index 5895c97750..b61bfdf386 100644
--- a/qapi/qom.json
+++ b/qapi/qom.json
@@ -780,13 +780,26 @@
# @seal: if true, create a sealed-file, which will block further
# resizing of the memory (default: true)
#
+# @guest-memfd: if 'on', use guest-memfd to back the memory region.
+# See the :doc:`/system/guest-memfd` documentation for more
+# details. If 'auto', the option will default to 'off'
+# currently, but in the future it may result in the option being
+# set to 'on' for QEMU configurations that explicitly require it.
+# (default: auto, since: 11.2)
+#
+# Features:
+#
+# @unstable: Member @guest-memfd is experimental.
+#
# Since: 2.12
##
{ 'struct': 'MemoryBackendMemfdProperties',
'base': 'MemoryBackendProperties',
'data': { '*hugetlb': 'bool',
'*hugetlbsize': 'size',
- '*seal': 'bool' },
+ '*seal': 'bool',
+ '*guest-memfd': { 'type': 'OnOffAuto',
+ 'features': [ 'unstable' ] } },
'if': 'CONFIG_LINUX' }
##