public inbox for io-uring@vger.kernel.org
 help / color / mirror / Atom feed
* [PATCH] io_uring: keep memlock accounting while regions are mapped
@ 2026-08-16  7:07 Marina via B4 Relay
  2026-08-25 17:52 ` Jens Axboe
  0 siblings, 1 reply; 2+ messages in thread
From: Marina via B4 Relay @ 2026-08-16  7:07 UTC (permalink / raw)
  To: Jens Axboe; +Cc: io-uring, linux-kernel

From: Marina <dejamarina@proton.me>

io_uring unaccounts kernel-allocated SQ/CQ and SQE regions when resize
replaces them, even if existing VMAs still retain their pages. Repeating
mmap and resize can therefore retain memory beyond RLIMIT_MEMLOCK.

Move each charge to a refcounted record owned by the active region and its
MMU or NOMMU mappings. Release it after both the region and final VMA are
gone. Keep direct accounting for user-provided regions, which cannot be
mapped through the io_uring file.

Accounting stays region-granular, so a partial mapping retains the complete
region charge.

Fixes: 79cfe9e59c2a ("io_uring/register: add IORING_REGISTER_RESIZE_RINGS")
Signed-off-by: Marina <dejamarina@proton.me>
---
 include/linux/io_uring_types.h |   3 ++
 io_uring/memmap.c              | 108 ++++++++++++++++++++++++++++++++++++++---
 2 files changed, 105 insertions(+), 6 deletions(-)

diff --git a/include/linux/io_uring_types.h b/include/linux/io_uring_types.h
index 87151a5b62c1..6feb7ee3440f 100644
--- a/include/linux/io_uring_types.h
+++ b/include/linux/io_uring_types.h
@@ -94,11 +94,14 @@ struct io_hash_table {
 	unsigned		hash_bits;
 };
 
+struct io_region_account;
+
 struct io_mapped_region {
 	struct page		**pages;
 	void			*ptr;
 	unsigned		nr_pages;
 	unsigned		flags;
+	struct io_region_account	*account;
 };
 
 /*
diff --git a/io_uring/memmap.c b/io_uring/memmap.c
index 23e8a85111bc..ca28a9429c16 100644
--- a/io_uring/memmap.c
+++ b/io_uring/memmap.c
@@ -4,6 +4,8 @@
 #include <linux/errno.h>
 #include <linux/mm.h>
 #include <linux/mman.h>
+#include <linux/refcount.h>
+#include <linux/sched/user.h>
 #include <linux/slab.h>
 #include <linux/vmalloc.h>
 #include <linux/io_uring.h>
@@ -88,6 +90,52 @@ enum {
 	IO_REGION_F_SINGLE_REF			= 4,
 };
 
+struct io_region_account {
+	refcount_t refs;
+	struct user_struct *user;
+	unsigned long nr_pages;
+};
+
+static struct io_region_account *
+io_region_account_alloc(struct user_struct *user, unsigned long nr_pages)
+{
+	struct io_region_account *account;
+	int ret;
+
+	if (!user)
+		return NULL;
+
+	account = kmalloc_obj(*account, GFP_KERNEL_ACCOUNT);
+	if (!account)
+		return ERR_PTR(-ENOMEM);
+
+	ret = __io_account_mem(user, nr_pages);
+	if (ret) {
+		kfree(account);
+		return ERR_PTR(ret);
+	}
+
+	refcount_set(&account->refs, 1);
+	account->user = get_uid(user);
+	account->nr_pages = nr_pages;
+	return account;
+}
+
+static void io_region_account_get(struct io_region_account *account)
+{
+	if (account)
+		refcount_inc(&account->refs);
+}
+
+static void io_region_account_put(struct io_region_account *account)
+{
+	if (account && refcount_dec_and_test(&account->refs)) {
+		__io_unaccount_mem(account->user, account->nr_pages);
+		free_uid(account->user);
+		kfree(account);
+	}
+}
+
 void io_free_region(struct user_struct *user, struct io_mapped_region *mr)
 {
 	if (mr->pages) {
@@ -105,8 +153,12 @@ void io_free_region(struct user_struct *user, struct io_mapped_region *mr)
 	}
 	if ((mr->flags & IO_REGION_F_VMAP) && mr->ptr)
 		vunmap(mr->ptr);
-	if (mr->nr_pages && user)
+	if (mr->account) {
+		WARN_ON_ONCE(mr->account->user != user);
+		io_region_account_put(mr->account);
+	} else if (mr->nr_pages && user) {
 		__io_unaccount_mem(user, mr->nr_pages);
+	}
 
 	memset(mr, 0, sizeof(*mr));
 }
@@ -188,7 +240,8 @@ int io_create_region(struct io_ring_ctx *ctx, struct io_mapped_region *mr,
 	int nr_pages, ret;
 	u64 end;
 
-	if (WARN_ON_ONCE(mr->pages || mr->ptr || mr->nr_pages))
+	if (WARN_ON_ONCE(mr->pages || mr->ptr || mr->nr_pages ||
+			 mr->account))
 		return -EFAULT;
 	if (memchr_inv(&reg->__resv, 0, sizeof(reg->__resv)))
 		return -EINVAL;
@@ -207,10 +260,23 @@ int io_create_region(struct io_ring_ctx *ctx, struct io_mapped_region *mr,
 		return -EOVERFLOW;
 
 	nr_pages = reg->size >> PAGE_SHIFT;
-	if (ctx->user) {
-		ret = __io_account_mem(ctx->user, nr_pages);
-		if (ret)
+	if (reg->flags & IORING_MEM_REGION_TYPE_USER) {
+		if (ctx->user) {
+			ret = __io_account_mem(ctx->user, nr_pages);
+			if (ret)
+				return ret;
+		}
+	} else {
+		/*
+		 * Kernel-allocated pages can outlive their active region through
+		 * userspace mappings.
+		 */
+		mr->account = io_region_account_alloc(ctx->user, nr_pages);
+		if (IS_ERR(mr->account)) {
+			ret = PTR_ERR(mr->account);
+			mr->account = NULL;
 			return ret;
+		}
 	}
 	mr->nr_pages = nr_pages;
 
@@ -281,15 +347,41 @@ static void *io_uring_validate_mmap_request(struct file *file, loff_t pgoff)
 
 #ifdef CONFIG_MMU
 
+static void io_region_vm_open(struct vm_area_struct *vma)
+{
+	io_region_account_get(vma->vm_private_data);
+}
+
+static void io_region_vm_close(struct vm_area_struct *vma)
+{
+	io_region_account_put(vma->vm_private_data);
+}
+
+static const struct vm_operations_struct io_region_vm_ops = {
+	.open = io_region_vm_open,
+	.close = io_region_vm_close,
+};
+
 static int io_region_mmap(struct io_ring_ctx *ctx,
 			  struct io_mapped_region *mr,
 			  struct vm_area_struct *vma,
 			  unsigned max_pages)
 {
 	unsigned long nr_pages = min(mr->nr_pages, max_pages);
+	int ret;
 
 	vm_flags_set(vma, VM_DONTEXPAND);
-	return vm_insert_pages(vma, vma->vm_start, mr->pages, &nr_pages);
+	ret = vm_insert_pages(vma, vma->vm_start, mr->pages, &nr_pages);
+	if (!ret && mr->account) {
+		/*
+		 * Accounting deliberately remains at region granularity when
+		 * this VMA maps only part of the region.
+		 */
+		vma->vm_private_data = mr->account;
+		vma->vm_ops = &io_region_vm_ops;
+		vma->vm_ops->open(vma);
+	}
+	return ret;
 }
 
 __cold int io_uring_mmap(struct file *file, struct vm_area_struct *vma)
@@ -379,6 +471,8 @@ static void io_uring_nommu_vm_close(struct vm_area_struct *vma)
 
 	for (index = vma->vm_start; index < vma->vm_end; index += PAGE_SIZE)
 		put_page(virt_to_page((void *) index));
+
+	io_region_account_put(vma->vm_private_data);
 }
 
 static const struct vm_operations_struct io_uring_nommu_vm_ops = {
@@ -411,6 +505,8 @@ int io_uring_mmap(struct file *file, struct vm_area_struct *vma)
 	for (i = 0; i < region->nr_pages; i++)
 		get_page(region->pages[i]);
 
+	vma->vm_private_data = region->account;
+	io_region_account_get(region->account);
 	vma->vm_ops = &io_uring_nommu_vm_ops;
 	return 0;
 }

---
base-commit: a5161661ae99f497affa83a5b8654e457cda6267
change-id: 20260815-io-uring-retained-mmap-accounting-bfbc7d71143a

Best regards,
-- 
Marina <dejamarina@proton.me>



^ permalink raw reply related	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-08-25 17:52 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-16  7:07 [PATCH] io_uring: keep memlock accounting while regions are mapped Marina via B4 Relay
2026-08-25 17:52 ` Jens Axboe

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox