From 4b7579e46a866641e9f083532056a9dfc45cb7b0 Mon Sep 17 00:00:00 2001 From: Tomer Gilad Date: Mon, 28 Sep 2026 14:21:41 +0200 Subject: [PATCH] UCT/CUDA_IPC: Publish the chunk layout of multi-allocation VMM ranges A VMM range backed by several physical allocations cannot be described by the one allocation handle a key carries today. Add the exporter side of a multi-chunk key: discover the allocations behind a range, write their descriptors to a GPU buffer shared by fabric handle, and keep that buffer on the local key until deregistration. Nothing packs such a key yet; the pack path is wired up separately. --- src/uct/cuda/Makefile.am | 2 + src/uct/cuda/cuda_ipc/cuda_ipc.inl | 10 + src/uct/cuda/cuda_ipc/cuda_ipc_cache.c | 8 - src/uct/cuda/cuda_ipc/cuda_ipc_md.c | 8 + src/uct/cuda/cuda_ipc/cuda_ipc_md.h | 18 ++ src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.c | 314 +++++++++++++++++++++ src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.h | 73 +++++ test/gtest/uct/cuda/cuda_vmm_mem_buffer.h | 13 +- test/gtest/uct/cuda/test_cuda_ipc_md.cc | 87 +++++- 9 files changed, 521 insertions(+), 12 deletions(-) create mode 100644 src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.c create mode 100644 src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.h diff --git a/src/uct/cuda/Makefile.am b/src/uct/cuda/Makefile.am index bd11d39fd09..9d5abc1cc50 100644 --- a/src/uct/cuda/Makefile.am +++ b/src/uct/cuda/Makefile.am @@ -31,6 +31,7 @@ noinst_HEADERS = \ cuda_copy/cuda_copy_iface.h \ cuda_copy/cuda_copy_ep.h \ cuda_ipc/cuda_ipc_md.h \ + cuda_ipc/cuda_ipc_vmm_multi.h \ cuda_ipc/cuda_ipc_iface_address.h \ cuda_ipc/cuda_ipc_iface.h \ cuda_ipc/cuda_ipc_ep.h \ @@ -47,6 +48,7 @@ libuct_cuda_la_SOURCES = \ cuda_copy/cuda_copy_iface.c \ cuda_copy/cuda_copy_ep.c \ cuda_ipc/cuda_ipc_md.c \ + cuda_ipc/cuda_ipc_vmm_multi.c \ cuda_ipc/cuda_ipc_iface_address.c \ cuda_ipc/cuda_ipc_iface.c \ cuda_ipc/cuda_ipc_ep.c \ diff --git a/src/uct/cuda/cuda_ipc/cuda_ipc.inl b/src/uct/cuda/cuda_ipc/cuda_ipc.inl index 4f2dd8d176e..042ca280b8d 100644 --- a/src/uct/cuda/cuda_ipc/cuda_ipc.inl +++ b/src/uct/cuda/cuda_ipc/cuda_ipc.inl @@ -82,6 +82,16 @@ uct_cuda_ipc_check_and_pop_ctx(int is_ctx_pushed) } } +#if HAVE_CUDA_FABRIC || HAVE_DECL_SYS_PIDFD_GETFD +static UCS_F_ALWAYS_INLINE void +uct_cuda_ipc_init_access_desc(CUmemAccessDesc *access_desc, CUdevice cu_dev) +{ + access_desc->location.type = CU_MEM_LOCATION_TYPE_DEVICE; + access_desc->flags = CU_MEM_ACCESS_FLAGS_PROT_READWRITE; + access_desc->location.id = cu_dev; +} +#endif + static UCS_F_ALWAYS_INLINE int uct_cuda_ipc_is_rkey_local(pid_t rkey_pid, ucs_sys_ns_t rkey_pid_ns) { diff --git a/src/uct/cuda/cuda_ipc/cuda_ipc_cache.c b/src/uct/cuda/cuda_ipc/cuda_ipc_cache.c index bafd8048132..8cb3f5fd501 100644 --- a/src/uct/cuda/cuda_ipc/cuda_ipc_cache.c +++ b/src/uct/cuda/cuda_ipc/cuda_ipc_cache.c @@ -300,14 +300,6 @@ uct_cuda_ipc_open_memhandle_legacy(CUipcMemHandle memh, CUdevice cu_dev, } #if HAVE_CUDA_FABRIC || HAVE_DECL_SYS_PIDFD_GETFD -static void -uct_cuda_ipc_init_access_desc(CUmemAccessDesc *access_desc, CUdevice cu_dev) -{ - access_desc->location.type = CU_MEM_LOCATION_TYPE_DEVICE; - access_desc->flags = CU_MEM_ACCESS_FLAGS_PROT_READWRITE; - access_desc->location.id = cu_dev; -} - static ucs_status_t uct_cuda_ipc_open_memhandle_vmm(const uct_cuda_ipc_rkey_t *key, CUdevice cu_dev, CUdeviceptr *mapped_addr, diff --git a/src/uct/cuda/cuda_ipc/cuda_ipc_md.c b/src/uct/cuda/cuda_ipc/cuda_ipc_md.c index ba7e1e649aa..1327d77a0d5 100644 --- a/src/uct/cuda/cuda_ipc/cuda_ipc_md.c +++ b/src/uct/cuda/cuda_ipc/cuda_ipc_md.c @@ -10,6 +10,7 @@ #include "cuda_ipc.inl" #include "cuda_ipc_cache.h" #include "cuda_ipc_md.h" +#include "cuda_ipc_vmm_multi.h" #include #include @@ -333,6 +334,10 @@ uct_cuda_ipc_mem_add_reg(void *addr, uct_cuda_ipc_memh_t *memh, return UCS_ERR_NO_MEMORY; } +#if HAVE_CUDA_FABRIC + ucs_list_head_init(&key->vmm_multi_list); +#endif + status = uct_cuda_ipc_check_and_push_ctx((CUdeviceptr)addr, &cuda_device, &is_ctx_pushed); if (status != UCS_OK) { @@ -687,6 +692,9 @@ uct_cuda_ipc_mem_dereg(uct_md_h md, const uct_md_mem_dereg_params_t *params) UCT_MD_MEM_DEREG_CHECK_PARAMS(params, 0); ucs_list_for_each_safe(key, tmp, &memh->list, link) { +#if HAVE_CUDA_FABRIC + uct_cuda_ipc_vmm_multi_meta_cleanup(key); +#endif if (key->ph.handle_type == UCT_CUDA_IPC_KEY_HANDLE_TYPE_POSIX_FD) { close(key->ph.handle.posix_fd.fd); } diff --git a/src/uct/cuda/cuda_ipc/cuda_ipc_md.h b/src/uct/cuda/cuda_ipc/cuda_ipc_md.h index 94dc6416b47..d3e15b22757 100644 --- a/src/uct/cuda/cuda_ipc/cuda_ipc_md.h +++ b/src/uct/cuda/cuda_ipc/cuda_ipc_md.h @@ -23,6 +23,21 @@ typedef enum { } uct_cuda_ipc_key_handle_t; +#if HAVE_CUDA_FABRIC +/** + * @brief Inline description of VMM multi-chunk metadata + */ +typedef struct { + uint8_t version; + uint8_t reserved; + uint16_t num_chunks; + uint16_t info_size; + uint16_t chunk_desc_size; + size_t alloc_size; +} uct_cuda_ipc_vmm_multi_info_t; +#endif + + typedef struct uct_cuda_ipc_md_handle { uct_cuda_ipc_key_handle_t handle_type; union { @@ -138,6 +153,9 @@ typedef struct { CUdeviceptr d_bptr; /* Allocation base address */ size_t b_len; /* Allocation size */ ucs_list_link_t link; +#if HAVE_CUDA_FABRIC + ucs_list_link_t vmm_multi_list; /* Published VMM metadata */ +#endif } uct_cuda_ipc_lkey_t; diff --git a/src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.c b/src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.c new file mode 100644 index 00000000000..7db02da169c --- /dev/null +++ b/src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.c @@ -0,0 +1,314 @@ +/** + * Copyright (c) NVIDIA CORPORATION & AFFILIATES, 2026. ALL RIGHTS RESERVED. + * See file LICENSE for terms. + */ + +#ifdef HAVE_CONFIG_H +# include "config.h" +#endif + +#include "cuda_ipc_vmm_multi.h" +#include "cuda_ipc.inl" + +#include +#include +#include +#include + +#if HAVE_CUDA_FABRIC + + +void uct_cuda_ipc_vmm_multi_meta_cleanup(uct_cuda_ipc_lkey_t *key) +{ + uct_cuda_ipc_vmm_multi_meta_t *meta, *tmp; + + ucs_list_for_each_safe(meta, tmp, &key->vmm_multi_list, link) { + UCT_CUDADRV_FUNC_LOG_WARN( + cuMemUnmap(meta->dev_ptr, meta->info.alloc_size)); + UCT_CUDADRV_FUNC_LOG_WARN( + cuMemAddressFree(meta->dev_ptr, meta->info.alloc_size)); + + ucs_list_del(&meta->link); + ucs_free(meta); + } +} + +UCS_ARRAY_DECLARE_TYPE(uct_cuda_ipc_vmm_chunk_array_t, uint32_t, + uct_cuda_ipc_vmm_chunk_desc_t); + +static ucs_status_t +uct_cuda_ipc_vmm_multi_discover_chunks(CUdeviceptr va_base, size_t va_len, + uct_cuda_ipc_vmm_chunk_desc_t **chunks_p, + uint16_t *num_chunks_p) +{ + uct_cuda_ipc_vmm_chunk_array_t chunks; + uct_cuda_ipc_vmm_chunk_desc_t *elem; + CUmemGenericAllocationHandle handle; + CUdeviceptr pos, chunk_base; + unsigned long long chunk_buffer_id; + uint64_t allowed_handle_types; + CUpointer_attribute attr_type[2]; + void *attr_data[2]; + size_t chunk_size; + ucs_status_t status; + + attr_type[0] = CU_POINTER_ATTRIBUTE_BUFFER_ID; + attr_data[0] = &chunk_buffer_id; + attr_type[1] = CU_POINTER_ATTRIBUTE_ALLOWED_HANDLE_TYPES; + attr_data[1] = &allowed_handle_types; + + ucs_array_init_dynamic(&chunks); + + for (pos = va_base; pos < va_base + va_len; pos = chunk_base + chunk_size) { + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemGetAddressRange(&chunk_base, &chunk_size, pos)); + if (status != UCS_OK) { + goto err; + } + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuPointerGetAttributes(ucs_static_array_size(attr_data), + attr_type, attr_data, chunk_base)); + if (status != UCS_OK) { + goto err; + } + + if (!(allowed_handle_types & CU_MEM_HANDLE_TYPE_FABRIC)) { + ucs_debug("VMM chunk 0x%llx does not allow fabric handles", + chunk_base); + status = UCS_ERR_UNSUPPORTED; + goto err; + } + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemRetainAllocationHandle(&handle, (void*)pos)); + if (status != UCS_OK) { + goto err; + } + + elem = ucs_array_append(&chunks, { + UCT_CUDADRV_FUNC_LOG_WARN(cuMemRelease(handle)); + status = UCS_ERR_NO_MEMORY; + goto err; + }); + + status = UCT_CUDADRV_FUNC_LOG_ERR(cuMemExportToShareableHandle( + &elem->fabric_handle, handle, CU_MEM_HANDLE_TYPE_FABRIC, 0)); + UCT_CUDADRV_FUNC_LOG_WARN(cuMemRelease(handle)); + if (status != UCS_OK) { + goto err; + } + + elem->d_bptr = chunk_base; + elem->b_len = chunk_size; + elem->buffer_id = chunk_buffer_id; + } + + if (ucs_array_length(&chunks) > UINT16_MAX) { + ucs_error("VMM region has %zu chunks, exceeding maximum of %u", + (size_t)ucs_array_length(&chunks), UINT16_MAX); + status = UCS_ERR_EXCEEDS_LIMIT; + goto err; + } + + *num_chunks_p = (uint16_t)ucs_array_length(&chunks); + *chunks_p = ucs_array_extract_buffer(&chunks); + return UCS_OK; + +err: + ucs_array_cleanup_dynamic(&chunks); + return status; +} + +static ucs_status_t uct_cuda_ipc_vmm_multi_meta_alloc_buffer( + CUdeviceptr *dev_ptr_p, size_t *alloc_size_p, + CUmemFabricHandle *fabric_handle_p, size_t data_size, + const CUmemAllocationProp *prop) +{ + CUmemAccessDesc access = {}; + CUmemGenericAllocationHandle alloc_handle; + CUdeviceptr dev_ptr; + ucs_status_t status; + size_t alloc_size, alloc_granularity; + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemGetAllocationGranularity(&alloc_granularity, prop, + CU_MEM_ALLOC_GRANULARITY_MINIMUM)); + if (status != UCS_OK) { + return status; + } + + alloc_size = ucs_align_up(data_size, alloc_granularity); + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemCreate(&alloc_handle, alloc_size, prop, 0)); + if (status != UCS_OK) { + return status; + } + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemAddressReserve(&dev_ptr, alloc_size, 0, 0, 0)); + if (status != UCS_OK) { + goto err_release; + } + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemMap(dev_ptr, alloc_size, 0, alloc_handle, 0)); + if (status != UCS_OK) { + goto err_free_va; + } + + uct_cuda_ipc_init_access_desc(&access, prop->location.id); + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemSetAccess(dev_ptr, alloc_size, &access, 1)); + if (status != UCS_OK) { + goto err_unmap; + } + + status = UCT_CUDADRV_FUNC_LOG_ERR(cuMemExportToShareableHandle( + fabric_handle_p, alloc_handle, CU_MEM_HANDLE_TYPE_FABRIC, 0)); + if (status != UCS_OK) { + goto err_unmap; + } + + UCT_CUDADRV_FUNC_LOG_WARN(cuMemRelease(alloc_handle)); + + *dev_ptr_p = dev_ptr; + *alloc_size_p = alloc_size; + return UCS_OK; + +err_unmap: + UCT_CUDADRV_FUNC_LOG_WARN(cuMemUnmap(dev_ptr, alloc_size)); +err_free_va: + UCT_CUDADRV_FUNC_LOG_WARN(cuMemAddressFree(dev_ptr, alloc_size)); +err_release: + UCT_CUDADRV_FUNC_LOG_WARN(cuMemRelease(alloc_handle)); + return status; +} + +static ucs_status_t +uct_cuda_ipc_vmm_multi_create_meta_buffer(uct_cuda_ipc_vmm_multi_meta_t *meta, + int dev_num) +{ + uct_cuda_ipc_vmm_chunk_desc_t *host_chunks = NULL; + uint16_t num_chunks = 0; + CUmemAllocationProp prop = {}; + CUdeviceptr dev_ptr; + size_t alloc_size; + ucs_status_t status; + size_t chunks_data_size; + + status = uct_cuda_ipc_vmm_multi_discover_chunks(meta->d_bptr, meta->b_len, + &host_chunks, &num_chunks); + if (status != UCS_OK) { + return status; + } + + chunks_data_size = num_chunks * sizeof(*host_chunks); + + prop.type = CU_MEM_ALLOCATION_TYPE_PINNED; + prop.location.type = CU_MEM_LOCATION_TYPE_DEVICE; + prop.location.id = dev_num; + prop.requestedHandleTypes = CU_MEM_HANDLE_TYPE_FABRIC; + + status = uct_cuda_ipc_vmm_multi_meta_alloc_buffer( + &dev_ptr, &alloc_size, &meta->fabric_handle, chunks_data_size, + &prop); + if (status != UCS_OK) { + goto err_free_host; + } + + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemcpyHtoD(dev_ptr, host_chunks, chunks_data_size)); + if (status != UCS_OK) { + goto err_cleanup; + } + + meta->dev_ptr = dev_ptr; + meta->info.version = UCT_CUDA_IPC_VMM_MULTI_VERSION; + meta->info.num_chunks = num_chunks; + meta->info.info_size = sizeof(meta->info); + meta->info.chunk_desc_size = sizeof(*host_chunks); + meta->info.alloc_size = alloc_size; + + ucs_trace("created VMM metadata: %u chunks, allocation size %zu on GPU", + num_chunks, alloc_size); + ucs_free(host_chunks); + return UCS_OK; + +err_cleanup: + UCT_CUDADRV_FUNC_LOG_WARN(cuMemUnmap(dev_ptr, alloc_size)); + UCT_CUDADRV_FUNC_LOG_WARN(cuMemAddressFree(dev_ptr, alloc_size)); +err_free_host: + ucs_free(host_chunks); + return status; +} + +ucs_status_t uct_cuda_ipc_mkey_pack_vmm_multi_chunk( + uct_cuda_ipc_lkey_t *key, void *address, size_t length, + const uct_cuda_ipc_vmm_multi_meta_t **meta_p) +{ + uct_cuda_ipc_vmm_multi_meta_t *meta; + CUdeviceptr last_base; + size_t last_size; + CUdevice cuda_device; + int is_ctx_pushed; + ucs_status_t status; + + /* Contained in the lkey's own allocation, so it cannot span chunks */ + if (((CUdeviceptr)address + length) <= (key->d_bptr + key->b_len)) { + return UCS_ERR_UNSUPPORTED; + } + + ucs_list_for_each(meta, &key->vmm_multi_list, link) { + if (((CUdeviceptr)address >= meta->d_bptr) && + (((CUdeviceptr)address + length) <= + (meta->d_bptr + meta->b_len))) { + *meta_p = meta; + return UCS_OK; + } + } + + status = uct_cuda_ipc_check_and_push_ctx((CUdeviceptr)address, &cuda_device, + &is_ctx_pushed); + if (status != UCS_OK) { + return status; + } + + /* The range starts in the lkey's allocation and ends past it */ + status = UCT_CUDADRV_FUNC_LOG_ERR( + cuMemGetAddressRange(&last_base, &last_size, + (CUdeviceptr)address + length - 1)); + if (status != UCS_OK) { + goto out_pop; + } + + meta = ucs_calloc(1, sizeof(*meta), "cuda_ipc_vmm_multi_meta"); + if (meta == NULL) { + ucs_error("failed to allocate VMM metadata record"); + status = UCS_ERR_NO_MEMORY; + goto out_pop; + } + + meta->d_bptr = key->d_bptr; + meta->b_len = (last_base + last_size) - key->d_bptr; + + status = uct_cuda_ipc_vmm_multi_create_meta_buffer(meta, cuda_device); + if (status != UCS_OK) { + goto err_free_meta; + } + + /* Peers may unpack any published metadata until memh deregistration. */ + ucs_list_add_tail(&key->vmm_multi_list, &meta->link); + *meta_p = meta; + goto out_pop; + +err_free_meta: + ucs_free(meta); + +out_pop: + uct_cuda_ipc_check_and_pop_ctx(is_ctx_pushed); + return status; +} + +#endif diff --git a/src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.h b/src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.h new file mode 100644 index 00000000000..0e52678efcd --- /dev/null +++ b/src/uct/cuda/cuda_ipc/cuda_ipc_vmm_multi.h @@ -0,0 +1,73 @@ +/** + * Copyright (c) NVIDIA CORPORATION & AFFILIATES, 2026. ALL RIGHTS RESERVED. + * See file LICENSE for terms. + */ + +#ifndef UCT_CUDA_IPC_VMM_MULTI_H +#define UCT_CUDA_IPC_VMM_MULTI_H + +#include "cuda_ipc_md.h" + +#if HAVE_CUDA_FABRIC + +/** + * @brief Descriptor of one chunk in a multi-chunk VMM region + */ +typedef struct uct_cuda_ipc_vmm_chunk_desc { + CUmemFabricHandle fabric_handle; + CUdeviceptr d_bptr; + size_t b_len; + /* Identity of the backing allocation. Remote addresses are recycled, so + * this is what distinguishes "the same chunk" from "a different allocation + * that happens to sit at the same address". */ + uint64_t buffer_id; +} uct_cuda_ipc_vmm_chunk_desc_t; + + +/** + * @brief Wire format version of the multi-chunk VMM metadata + * + * Bump when the meaning of the inline metadata information or chunk descriptor + * fields changes. + */ +#define UCT_CUDA_IPC_VMM_MULTI_VERSION 1 + + +/** + * @brief multi-chunk VMM registration metadata + */ +typedef struct { + ucs_list_link_t link; /* Entry in metadata list */ + CUdeviceptr d_bptr; /* Base of all chunks */ + size_t b_len; /* Length of all chunks */ + CUdeviceptr dev_ptr; /* Metadata buffer VA */ + CUmemFabricHandle fabric_handle; /* Metadata fabric handle */ + uct_cuda_ipc_vmm_multi_info_t info; /* Inline metadata layout */ +} uct_cuda_ipc_vmm_multi_meta_t; + + +/** + * @brief Release all exporter metadata associated with a local key + */ +void uct_cuda_ipc_vmm_multi_meta_cleanup(uct_cuda_ipc_lkey_t *key); + + +/** + * @brief Pack a VMM range spanning multiple physical allocations + * + * @param key Local key whose allocation contains @a address + * @param address Requested range start + * @param length Requested range length + * @param meta_p Metadata record used to pack the key + * + * @return UCS_OK on success, UCS_ERR_UNSUPPORTED for a single allocation, or + * another error status on failure + */ +ucs_status_t uct_cuda_ipc_mkey_pack_vmm_multi_chunk( + uct_cuda_ipc_lkey_t *key, void *address, size_t length, + const uct_cuda_ipc_vmm_multi_meta_t **meta_p); + + +#endif + +#endif diff --git a/test/gtest/uct/cuda/cuda_vmm_mem_buffer.h b/test/gtest/uct/cuda/cuda_vmm_mem_buffer.h index 73a22f93786..307ebc4f56e 100644 --- a/test/gtest/uct/cuda/cuda_vmm_mem_buffer.h +++ b/test/gtest/uct/cuda/cuda_vmm_mem_buffer.h @@ -56,6 +56,11 @@ class cuda_vmm_mem_buffer { return m_size; } + size_t chunk_size() const + { + return m_chunk_size; + } + CUresult alloc(size_t size, unsigned handle_type, CUmemLocationType location_type = CU_MEM_LOCATION_TYPE_DEVICE, size_t num_chunks = 1, @@ -163,7 +168,7 @@ class cuda_vmm_mem_buffer { } for (auto alloc_handle : m_alloc_handles) { - cuMemRelease(alloc_handle); + EXPECT_EQ(CUDA_SUCCESS, cuMemRelease(alloc_handle)); } if (m_ptr != 0) { @@ -185,9 +190,11 @@ class cuda_vmm_mem_buffer { #if HAVE_CUDA_FABRIC class cuda_fabric_mem_buffer : public cuda_vmm_mem_buffer { public: - cuda_fabric_mem_buffer(size_t size, ucs_memory_type_t mem_type) + cuda_fabric_mem_buffer(size_t size, ucs_memory_type_t mem_type, + size_t num_chunks = 1) { - skip_unless_ok(alloc(size, CU_MEM_HANDLE_TYPE_FABRIC)); + skip_unless_ok(alloc(size, CU_MEM_HANDLE_TYPE_FABRIC, + CU_MEM_LOCATION_TYPE_DEVICE, num_chunks)); } }; #endif diff --git a/test/gtest/uct/cuda/test_cuda_ipc_md.cc b/test/gtest/uct/cuda/test_cuda_ipc_md.cc index 0cfbf5f8d84..0dded12cafb 100644 --- a/test/gtest/uct/cuda/test_cuda_ipc_md.cc +++ b/test/gtest/uct/cuda/test_cuda_ipc_md.cc @@ -1,5 +1,5 @@ /** - * Copyright (c) NVIDIA CORPORATION & AFFILIATES, 2024. ALL RIGHTS RESERVED. + * Copyright (c) NVIDIA CORPORATION & AFFILIATES, 2024-2026. ALL RIGHTS RESERVED. * * See file LICENSE for terms. */ @@ -19,6 +19,7 @@ extern "C" { #include #include +#include #include #include #include @@ -114,6 +115,32 @@ class test_cuda_ipc_md : public test_md { free_mempool(&ptr, &mpool, &cu_stream); return rkey; } + + /* Publish the chunk metadata of [address, address + length), a range + * spanning allocations. A plain pack of 'alloc_len' bytes at 'address', + * within the first allocation, creates the local key it is attached to. */ + const uct_cuda_ipc_vmm_multi_meta_t * + vmm_multi_publish(uct_mem_h memh, void *address, size_t alloc_len, + size_t length) + { + uct_cuda_ipc_memh_t *cuda_memh = static_cast( + memh); + const uct_cuda_ipc_vmm_multi_meta_t *meta = NULL; + uct_cuda_ipc_extended_rkey_t rkey; + uct_cuda_ipc_lkey_t *key; + + EXPECT_UCS_OK(md()->ops->mkey_pack(md(), memh, address, alloc_len, + NULL, &rkey)); + ucs_list_for_each(key, &cuda_memh->list, link) { + if (((uintptr_t)address >= key->d_bptr) && + ((uintptr_t)address < (key->d_bptr + key->b_len))) { + EXPECT_UCS_OK(uct_cuda_ipc_mkey_pack_vmm_multi_chunk( + key, address, length, &meta)); + break; + } + } + return meta; + } #endif void test_mkey_pack_on_thread(void *ptr, size_t size) @@ -320,6 +347,64 @@ UCS_TEST_P(test_cuda_ipc_md, mnnvl_disabled) EXPECT_FALSE(cuda_ipc_md->enable_mnnvl); } +UCS_TEST_P(test_cuda_ipc_md, vmm_multi_publish_chunks) +{ +#if HAVE_CUDA_FABRIC + constexpr unsigned num_chunks = 4; + cuda_fabric_mem_buffer alloc(1, UCS_MEMORY_TYPE_CUDA, num_chunks); + std::vector descs(num_chunks); + const uct_cuda_ipc_vmm_multi_meta_t *meta; + uct_md_mem_dereg_params_t params; + unsigned long long buffer_id; + uct_mem_h memh; + CUdeviceptr chunk; + + ASSERT_UCS_OK(md()->ops->mem_reg(md(), alloc.ptr(), alloc.size(), NULL, + &memh)); + + /* A range starting and ending inside allocations covers them whole */ + meta = vmm_multi_publish(memh, + UCS_PTR_BYTE_OFFSET(alloc.ptr(), + alloc.chunk_size() / 2), + alloc.chunk_size() / 2, + alloc.size() - alloc.chunk_size()); + ASSERT_TRUE(meta != NULL); + EXPECT_EQ(meta, vmm_multi_publish(memh, alloc.ptr(), alloc.chunk_size(), + alloc.size())); + + EXPECT_EQ((CUdeviceptr)alloc.ptr(), meta->d_bptr); + EXPECT_EQ(alloc.size(), meta->b_len); + EXPECT_EQ(UCT_CUDA_IPC_VMM_MULTI_VERSION, meta->info.version); + EXPECT_EQ(num_chunks, meta->info.num_chunks); + EXPECT_EQ(sizeof(uct_cuda_ipc_vmm_chunk_desc_t), + meta->info.chunk_desc_size); + + /* The published descriptors name every allocation, in order */ + ASSERT_EQ(CUDA_SUCCESS, + cuMemcpyDtoH(descs.data(), meta->dev_ptr, + num_chunks * sizeof(descs[0]))); + for (unsigned i = 0; i < num_chunks; ++i) { + chunk = (CUdeviceptr)alloc.ptr() + (i * alloc.chunk_size()); + ASSERT_EQ(CUDA_SUCCESS, + cuPointerGetAttribute(&buffer_id, + CU_POINTER_ATTRIBUTE_BUFFER_ID, chunk)); + EXPECT_EQ(chunk, descs[i].d_bptr); + EXPECT_EQ(alloc.chunk_size(), descs[i].b_len); + EXPECT_EQ(buffer_id, descs[i].buffer_id); + } + + /* A range inside published metadata reuses it */ + EXPECT_EQ(meta, vmm_multi_publish(memh, alloc.ptr(), alloc.chunk_size(), + 2 * alloc.chunk_size())); + + params.field_mask = UCT_MD_MEM_DEREG_FIELD_MEMH; + params.memh = memh; + EXPECT_UCS_OK(md()->ops->mem_dereg(md(), ¶ms)); +#else + UCS_TEST_SKIP_R("built without fabric support"); +#endif +} + UCS_TEST_P(test_cuda_ipc_md, posix_fd_same_node_ipc) { #if HAVE_DECL_SYS_PIDFD_GETFD