nccl_context.cc 7.6 KB
Newer Older
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15
//   Copyright (c) 2019 PaddlePaddle Authors. All Rights Reserved.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
//     http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

#include "paddle/fluid/imperative/nccl_context.h"
16

17 18
namespace paddle {
namespace imperative {
19
#if defined(PADDLE_WITH_NCCL)
20 21 22
void NCCLParallelContext::RecvNCCLID(const std::string &ep,
                                     ncclUniqueId *nccl_id) {
  auto addr = paddle::string::Split(ep, ':');
23 24 25 26
  PADDLE_ENFORCE_EQ(
      addr.size(), 2UL,
      platform::errors::InvalidArgument(
          "The endpoint should contain host and port, but got %s.", ep));
27 28 29 30 31 32 33 34 35
  std::string host = addr[0];
  int port = std::stoi(addr[1]);

  int server_fd, new_socket;
  struct sockaddr_in address;
  int addrlen = sizeof(address);
  char buffer[1024] = {0};
  int opt = 0;
  // creating socket fd
36 37 38 39 40 41 42 43
  if ((server_fd = socket(AF_INET, SOCK_STREAM, 0)) == 0) {
    PADDLE_THROW(
        platform::errors::Unavailable("Create server file descriptor failed."));
  }

  if (setsockopt(server_fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt))) {
    PADDLE_THROW(platform::errors::Unavailable("Set socket options failed."));
  }
44 45 46 47 48

  address.sin_family = AF_INET;
  address.sin_addr.s_addr = INADDR_ANY;
  address.sin_port = htons(port);

49
  int try_times = 0;
50
  int retry_time = 0;
51 52
  while (true) {
    if (bind(server_fd, (struct sockaddr *)&address, sizeof(address)) < 0) {
53
      retry_time = 3 * (try_times + 1);
54
      LOG(WARNING) << "Socket bind worker " << ep
55 56 57 58 59 60 61 62 63
                   << (try_times < 9
                           ? " failed, try again after " +
                                 std::to_string(retry_time) + " seconds."
                           : " failed, try again after " +
                                 std::to_string(retry_time) +
                                 " seconds. Bind on endpoint " + ep +
                                 " failed. Please confirm whether the "
                                 "communication port or GPU card is occupied.");
      std::this_thread::sleep_for(std::chrono::seconds(retry_time));
64 65 66 67
      ++try_times;
      continue;
    }
    break;
68 69
  }

70
  VLOG(3) << "listening on: " << ep;
71 72 73 74
  if (listen(server_fd, 3) < 0) {
    PADDLE_THROW(platform::errors::Unavailable(
        "Listen on server file descriptor failed."));
  }
75 76 77

  if ((new_socket =
           accept(server_fd, reinterpret_cast<struct sockaddr *>(&address),
78 79 80 81 82 83 84 85
                  reinterpret_cast<socklen_t *>(&addrlen))) < 0) {
    PADDLE_THROW(platform::errors::Unavailable(
        "Accept the new socket file descriptor failed."));
  }

  if (read(new_socket, buffer, 1024) < 0) {
    PADDLE_THROW(platform::errors::Unavailable("Read from socket failed."));
  }
86 87 88 89 90 91 92 93 94 95 96

  VLOG(3) << "recevived the ncclUniqueId";
  memcpy(nccl_id, buffer, NCCL_UNIQUE_ID_BYTES);

  VLOG(3) << "closing the socket server: " << ep;
  close(server_fd);
}

void NCCLParallelContext::SendNCCLID(const std::string &ep,
                                     ncclUniqueId *nccl_id) {
  auto addr = paddle::string::Split(ep, ':');
97 98 99 100
  PADDLE_ENFORCE_EQ(
      addr.size(), 2UL,
      platform::errors::InvalidArgument(
          "The endpoint should contain host and port, but got %s.", ep));
101 102 103 104 105 106 107 108
  std::string host = addr[0];
  int port = std::stoi(addr[1]);
  // struct sockaddr_in address;
  int sock = 0;
  struct sockaddr_in serv_addr;
  char buffer[1024] = {0};

  memcpy(buffer, nccl_id, NCCL_UNIQUE_ID_BYTES);
109 110 111
  if ((sock = socket(AF_INET, SOCK_STREAM, 0)) < 0) {
    PADDLE_THROW(platform::errors::Unavailable("Create socket failed."));
  }
112 113 114 115 116

  memset(&serv_addr, '0', sizeof(serv_addr));
  serv_addr.sin_family = AF_INET;
  serv_addr.sin_port = htons(port);

117 118 119 120 121 122 123 124 125 126 127 128 129
  char *ip = NULL;
  struct hostent *hp;
  if ((hp = gethostbyname(host.c_str())) == NULL) {
    PADDLE_THROW(platform::errors::InvalidArgument(
        "Fail to get host by name %s.", host));
  }
  int i = 0;
  while (hp->h_addr_list[i] != NULL) {
    ip = inet_ntoa(*(struct in_addr *)hp->h_addr_list[i]);
    VLOG(3) << "gethostbyname  host:" << host << "  ->ip: " << ip;
    break;
  }
  if (inet_pton(AF_INET, ip, &serv_addr.sin_addr) <= 0) {
130 131
    PADDLE_THROW(platform::errors::Unavailable("Open address %s failed.", ep));
  }
132

133
  int try_times = 0;
134
  int retry_time = 0;
135 136
  while (true) {
    if (connect(sock, (struct sockaddr *)&serv_addr, sizeof(serv_addr)) < 0) {
137
      retry_time = 3 * (try_times + 1);
138 139
      LOG(WARNING)
          << "Socket connect worker " << ep
140 141 142 143 144 145 146 147
          << (try_times < 9
                  ? " failed, try again after " + std::to_string(retry_time) +
                        " seconds."
                  : " failed, try again after " + std::to_string(retry_time) +
                        " seconds. Maybe that some process is occupied the "
                        "GPUs of this node now, and you should kill those "
                        "process manually.");
      std::this_thread::sleep_for(std::chrono::seconds(retry_time));
148
      ++try_times;
149 150 151 152 153 154
      continue;
    }
    VLOG(3) << "sending the ncclUniqueId to " << ep;
    send(sock, buffer, NCCL_UNIQUE_ID_BYTES, 0);
    break;
  }
C
chengduo 已提交
155
  close(sock);
156 157 158 159 160 161 162 163 164 165 166 167 168
}

void NCCLParallelContext::BcastNCCLId(ncclUniqueId *nccl_id, int root) {
  if (strategy_.local_rank_ == root) {
    for (auto ep : strategy_.trainer_endpoints_) {
      if (ep != strategy_.current_endpoint_) SendNCCLID(ep, nccl_id);
    }
  } else {
    RecvNCCLID(strategy_.current_endpoint_, nccl_id);
  }
}

void NCCLParallelContext::Init() {
169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200
  for (int ring_id = 0; ring_id < strategy_.nrings_; ring_id++) {
    ncclUniqueId nccl_id;
    if (strategy_.local_rank_ == 0) {
      // generate the unique ncclid on the root worker
      platform::dynload::ncclGetUniqueId(&nccl_id);
      BcastNCCLId(&nccl_id, 0);
    } else {
      BcastNCCLId(&nccl_id, 0);
    }
    int gpu_id = BOOST_GET_CONST(platform::CUDAPlace, place_).device;
    VLOG(0) << "init nccl context nranks: " << strategy_.nranks_
            << " local rank: " << strategy_.local_rank_ << " gpu id: " << gpu_id
            << " ring id: " << ring_id;

    // it will assign nccl_comm in CUDADeviceContext within ring_id
    platform::NCCLCommContext::Instance().CreateNCCLComm(
        &nccl_id, strategy_.nranks_, strategy_.local_rank_, gpu_id, ring_id);
  }
}

void NCCLParallelContext::AllReduceByStream(const framework::Variable &src,
                                            framework::Variable *dst,
                                            int ring_id, bool use_calc_stream) {
  PADDLE_ENFORCE_EQ(
      platform::is_gpu_place(place_), true,
      platform::errors::Unimplemented(
          "Dynamic graph mode does not support multi-CPU training yet."));
  auto comm = platform::NCCLCommContext::Instance().Get(ring_id, place_);
  cudaStream_t stream = nullptr;
  if (use_calc_stream) {
    auto dev_ctx = platform::DeviceContextPool::Instance().Get(place_);
    stream = static_cast<platform::CUDADeviceContext *>(dev_ctx)->stream();
201
  } else {
202
    stream = comm->stream();
203
  }
204 205
  AllReduce(src, dst, strategy_, stream);
}
206

207 208 209 210 211
paddle::platform::CUDADeviceContext *NCCLParallelContext::GetDeviceContext(
    int ring_id) {
  return platform::NCCLCommContext::Instance()
      .Get(ring_id, place_)
      ->dev_context();
212
}
213

214 215 216 217
#endif

}  //  namespace imperative
}  //  namespace paddle