From 8568050e2caead9ebf2c87df997cbf0ce70fe426 Mon Sep 17 00:00:00 2001 From: Matthew Wittwer Date: Thu, 1 Oct 2026 05:12:27 +0000 Subject: [PATCH] fix: Handle CPU_PINNED output buffers when GPU tensors fall back to host memory --- src/pb_memory.cc | 24 +++++++++++------------- src/python_be.cc | 12 ++++++++++-- 2 files changed, 21 insertions(+), 15 deletions(-) diff --git a/src/pb_memory.cc b/src/pb_memory.cc index 5b678f1a..704c0d94 100644 --- a/src/pb_memory.cc +++ b/src/pb_memory.cc @@ -52,7 +52,8 @@ PbMemory::Create( shm_pool->GetCUDAMemoryPoolManager(), memory_type, memory_type_id, byte_size, data, memory_shm.data_.get(), memory_shm.handle_, copy_gpu); - if (memory_type == TRITONSERVER_MEMORY_CPU) { + if (memory_type != TRITONSERVER_MEMORY_GPU) { + // CPU and CPU_PINNED intermediates both live in shared memory. data = memory_shm.data_.get() + sizeof(MemoryShm); } @@ -123,25 +124,22 @@ PbMemory::CopyBuffer( " != " + std::to_string(src->ByteSize())); } - if (src->MemoryType() == TRITONSERVER_MEMORY_CPU && - dst->MemoryType() == TRITONSERVER_MEMORY_CPU) { + // CPU and CPU_PINNED are both host memory for the purpose of copies. + const bool src_is_host = (src->MemoryType() != TRITONSERVER_MEMORY_GPU); + const bool dst_is_host = (dst->MemoryType() != TRITONSERVER_MEMORY_GPU); + + if (src_is_host && dst_is_host) { std::memcpy(dst->DataPtr(), src->DataPtr(), dst->ByteSize()); return; } #ifdef TRITON_ENABLE_GPU - cudaMemcpyKind kind = cudaMemcpyHostToDevice; - - if (src->MemoryType() == TRITONSERVER_MEMORY_CPU && - dst->MemoryType() == TRITONSERVER_MEMORY_GPU) { + cudaMemcpyKind kind; + if (src_is_host && !dst_is_host) { kind = cudaMemcpyHostToDevice; - } else if ( - src->MemoryType() == TRITONSERVER_MEMORY_GPU && - dst->MemoryType() == TRITONSERVER_MEMORY_CPU) { + } else if (!src_is_host && dst_is_host) { kind = cudaMemcpyDeviceToHost; - } else if ( - src->MemoryType() == TRITONSERVER_MEMORY_GPU && - dst->MemoryType() == TRITONSERVER_MEMORY_GPU) { + } else { kind = cudaMemcpyDeviceToDevice; } diff --git a/src/python_be.cc b/src/python_be.cc index 4c9ca49e..83b828fe 100644 --- a/src/python_be.cc +++ b/src/python_be.cc @@ -1529,7 +1529,11 @@ ModelInstanceState::ResponseSendDecoupled( bool cuda_used; try { - if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU) { + if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU || + pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU_PINNED) { + // The GPU tensor was staged into a shared-memory intermediate by + // the stub (Triton fell back to host / pinned output memory), so + // it still has to be copied into the Triton-provided buffer. THROW_IF_TRITON_ERROR(CopyBuffer( "Failed to copy the CPU output tensor to buffer.", TRITONSERVER_MEMORY_CPU, 0, TRITONSERVER_MEMORY_CPU, 0, @@ -1838,7 +1842,11 @@ ModelInstanceState::ProcessRequests( void* pointer = buffer_memory_pair.second; bool cuda_used = false; - if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU) { + if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU || + pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU_PINNED) { + // The GPU tensor was staged into a shared-memory intermediate by + // the stub (Triton fell back to host / pinned output memory), so + // it still has to be copied into the Triton-provided buffer. GUARDED_RESPOND_IF_ERROR( responses, response_index, CopyBuffer(