Clean memcpy async

0c24b3f9 · Yu Yang · bfbbe19f · 0c24b3f9 · 0c24b3f9
隐藏空白更改
内联并排

Showing with 8 addition and 33 deletion

paddle/fluid/framework/details/fetch_op_handle.cc paddle/fluid/framework/details/fetch_op_handle.cc +0 -1

paddle/fluid/pybind/tensor_py.h paddle/fluid/pybind/tensor_py.h +8 -32

未找到文件。
--- a/paddle/fluid/framework/details/fetch_op_handle.cc
+++ b/paddle/fluid/framework/details/fetch_op_handle.cc
@@ -67,7 +67,6 @@ void FetchOpHandle::RunImpl() {
    if (platform::is_gpu_place(t.place())) {
 #ifdef PADDLE_WITH_CUDA
      TensorCopy(t, cpu, *dev_ctxes_[t.place()], &tensors_[i], true);
-      dev_ctxes_.at(t.place())->Wait();
 #endif
    } else {
      tensors_[i].ShareDataWith(t);

--- a/paddle/fluid/pybind/tensor_py.h
+++ b/paddle/fluid/pybind/tensor_py.h
@@ -63,15 +63,9 @@ struct CastToPyBufferImpl<true, I, ARGS...> {
        auto *dst_ptr = static_cast<void *>(dst_tensor.mutable_data<CUR_TYPE>(
            tensor.dims(), platform::CPUPlace()));

-        platform::DeviceContextPool &pool =
-            platform::DeviceContextPool::Instance();
-        auto dev_ctx = static_cast<const platform::CUDADeviceContext *>(
-            pool.Get(tensor.place()));
-
-        paddle::platform::GpuMemcpyAsync(
-            dst_ptr, src_ptr, sizeof(CUR_TYPE) * tensor.numel(),
-            cudaMemcpyDeviceToHost, dev_ctx->stream());
-        dev_ctx->Wait();
+        paddle::platform::GpuMemcpySync(dst_ptr, src_ptr,
+                                        sizeof(CUR_TYPE) * tensor.numel(),
+                                        cudaMemcpyDeviceToHost);
 #else
        PADDLE_THROW("'CUDAPlace' is not supported in CPU only device.");
 #endif
@@ -184,17 +178,8 @@ void PyCUDATensorSetFromArray(

  self->Resize(framework::make_ddim(dims));
  auto *dst = self->mutable_data<T>(place);
-
-  platform::DeviceContextPool &pool = platform::DeviceContextPool::Instance();
-  auto dev_ctx =
-      static_cast<const platform::CUDADeviceContext *>(pool.Get(place));
-  paddle::platform::GpuMemcpyAsync(dst, array.data(), sizeof(T) * array.size(),
-                                   cudaMemcpyHostToDevice, dev_ctx->stream());
-  // NOTE: For safety, here wait the copy complete.
-  // It because the CPU array.data() could be destroyed after this method.
-  // If we make this method async, it could be copied data from a memory buffer
-  // that has been freed.
-  dev_ctx->Wait();
+  paddle::platform::GpuMemcpySync(dst, array.data(), sizeof(T) * array.size(),
+                                  cudaMemcpyHostToDevice);
 }

 template <>
@@ -214,18 +199,9 @@ void PyCUDATensorSetFromArray(

  self->Resize(framework::make_ddim(dims));
  auto *dst = self->mutable_data<platform::float16>(place);
-
-  platform::DeviceContextPool &pool = platform::DeviceContextPool::Instance();
-  auto dev_ctx =
-      static_cast<const platform::CUDADeviceContext *>(pool.Get(place));
-  paddle::platform::GpuMemcpyAsync(dst, array.data(),
-                                   sizeof(uint16_t) * array.size(),
-                                   cudaMemcpyHostToDevice, dev_ctx->stream());
-  // NOTE: For safety, here wait the copy complete.
-  // It because the CPU array.data() could be destroyed after this method.
-  // If we make this method async, it could be copied data from a memory buffer
-  // that has been freed.
-  dev_ctx->Wait();
+  paddle::platform::GpuMemcpySync(dst, array.data(),
+                                  sizeof(uint16_t) * array.size(),
+                                  cudaMemcpyHostToDevice);
 }

 template <typename T>