Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 11 additions & 13 deletions src/pb_memory.cc
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,8 @@ PbMemory::Create(
shm_pool->GetCUDAMemoryPoolManager(), memory_type, memory_type_id,
byte_size, data, memory_shm.data_.get(), memory_shm.handle_, copy_gpu);

if (memory_type == TRITONSERVER_MEMORY_CPU) {
if (memory_type != TRITONSERVER_MEMORY_GPU) {
// CPU and CPU_PINNED intermediates both live in shared memory.
data = memory_shm.data_.get() + sizeof(MemoryShm);
}

Expand Down Expand Up @@ -123,25 +124,22 @@ PbMemory::CopyBuffer(
" != " + std::to_string(src->ByteSize()));
}

if (src->MemoryType() == TRITONSERVER_MEMORY_CPU &&
dst->MemoryType() == TRITONSERVER_MEMORY_CPU) {
// CPU and CPU_PINNED are both host memory for the purpose of copies.
const bool src_is_host = (src->MemoryType() != TRITONSERVER_MEMORY_GPU);
const bool dst_is_host = (dst->MemoryType() != TRITONSERVER_MEMORY_GPU);

if (src_is_host && dst_is_host) {
std::memcpy(dst->DataPtr(), src->DataPtr(), dst->ByteSize());
return;
}

#ifdef TRITON_ENABLE_GPU
cudaMemcpyKind kind = cudaMemcpyHostToDevice;

if (src->MemoryType() == TRITONSERVER_MEMORY_CPU &&
dst->MemoryType() == TRITONSERVER_MEMORY_GPU) {
cudaMemcpyKind kind;
if (src_is_host && !dst_is_host) {
kind = cudaMemcpyHostToDevice;
} else if (
src->MemoryType() == TRITONSERVER_MEMORY_GPU &&
dst->MemoryType() == TRITONSERVER_MEMORY_CPU) {
} else if (!src_is_host && dst_is_host) {
kind = cudaMemcpyDeviceToHost;
} else if (
src->MemoryType() == TRITONSERVER_MEMORY_GPU &&
dst->MemoryType() == TRITONSERVER_MEMORY_GPU) {
} else {
kind = cudaMemcpyDeviceToDevice;
}

Expand Down
12 changes: 10 additions & 2 deletions src/python_be.cc
Original file line number Diff line number Diff line change
Expand Up @@ -1529,7 +1529,11 @@ ModelInstanceState::ResponseSendDecoupled(
bool cuda_used;

try {
if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU) {
if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU ||
pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU_PINNED) {
// The GPU tensor was staged into a shared-memory intermediate by
// the stub (Triton fell back to host / pinned output memory), so
// it still has to be copied into the Triton-provided buffer.
THROW_IF_TRITON_ERROR(CopyBuffer(
"Failed to copy the CPU output tensor to buffer.",
TRITONSERVER_MEMORY_CPU, 0, TRITONSERVER_MEMORY_CPU, 0,
Expand Down Expand Up @@ -1838,7 +1842,11 @@ ModelInstanceState::ProcessRequests(
void* pointer = buffer_memory_pair.second;
bool cuda_used = false;

if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU) {
if (pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU ||
pb_memory->MemoryType() == TRITONSERVER_MEMORY_CPU_PINNED) {
Comment on lines +1845 to +1846

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Pinned fallback lacks tests The new CPU_PINNED branch copies GPU output through a shared-memory buffer before filling Triton’s output buffer, but there is no regression test for that transfer in either ordinary or decoupled responses. A test that forces a pinned-host output and checks the returned bytes in both modes would help catch a future change that returns incorrect output.

Knowledge Base Used:

Note: If this suggestion doesn't match your team's coding style, reply to this and let me know. I'll remember it for next time!

// The GPU tensor was staged into a shared-memory intermediate by
// the stub (Triton fell back to host / pinned output memory), so
// it still has to be copied into the Triton-provided buffer.
GUARDED_RESPOND_IF_ERROR(
responses, response_index,
CopyBuffer(
Expand Down
Loading