api_impl_tester.cc 12.4 KB
Newer Older
X
Xin Pan 已提交
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17
/* Copyright (c) 2018 PaddlePaddle Authors. All Rights Reserved.

Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at

http://www.apache.org/licenses/LICENSE-2.0

Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License. */

#include <glog/logging.h>
#include <gtest/gtest.h>

L
Luo Tao 已提交
18
#include <thread>  // NOLINT
T
tensor-tang 已提交
19

X
Xin Pan 已提交
20
#include "gflags/gflags.h"
L
Luo Tao 已提交
21
#include "paddle/fluid/inference/api/api_impl.h"
X
Xin Pan 已提交
22 23
#include "paddle/fluid/inference/tests/test_helper.h"

J
JiabinYang 已提交
24
#ifdef __clang__
25
#define ACC_DIFF 4e-3
J
JiabinYang 已提交
26
#else
27
#define ACC_DIFF 1e-3
J
JiabinYang 已提交
28 29
#endif

30 31 32
DEFINE_string(word2vec_dirname, "",
              "Directory of the word2vec inference model.");
DEFINE_string(book_dirname, "", "Directory of the book inference model.");
X
Xin Pan 已提交
33 34 35 36 37 38

namespace paddle {

PaddleTensor LodTensorToPaddleTensor(framework::LoDTensor* t) {
  PaddleTensor pt;

Y
Yu Yang 已提交
39
  if (t->type() == framework::proto::VarType::INT64) {
40
    pt.data.Reset(t->data<void>(), t->numel() * sizeof(int64_t));
X
Xin Pan 已提交
41
    pt.dtype = PaddleDType::INT64;
Y
Fix ut  
Yu Yang 已提交
42
  } else if (t->type() == framework::proto::VarType::FP32) {
43
    pt.data.Reset(t->data<void>(), t->numel() * sizeof(float));
X
Xin Pan 已提交
44
    pt.dtype = PaddleDType::FLOAT32;
45 46 47
  } else if (t->type() == framework::proto::VarType::INT32) {
    pt.data.Reset(t->data<void>(), t->numel() * sizeof(int32_t));
    pt.dtype = PaddleDType::INT32;
X
Xin Pan 已提交
48
  } else {
49 50
    PADDLE_THROW(platform::errors::Unimplemented(
        "Unsupported tensor date type. Now only supports INT64, FP32, INT32."));
X
Xin Pan 已提交
51
  }
52
  pt.shape = framework::vectorize<int>(t->dims());
X
Xin Pan 已提交
53 54 55
  return pt;
}

Y
Yan Chunwei 已提交
56 57
NativeConfig GetConfig() {
  NativeConfig config;
58
  config.model_dir = FLAGS_word2vec_dirname;
X
Xin Pan 已提交
59
  LOG(INFO) << "dirname  " << config.model_dir;
X
Xin Pan 已提交
60
  config.fraction_of_gpu_memory = 0.15;
X
Xin Pan 已提交
61
  config.device = 0;
62 63
  return config;
}
X
Xin Pan 已提交
64

65
void MainWord2Vec(const paddle::PaddlePlace& place) {
Y
Yan Chunwei 已提交
66 67
  NativeConfig config = GetConfig();
  auto predictor = CreatePaddlePredictor<NativeConfig>(config);
68 69
  config.use_gpu = paddle::gpu_place_used(place);
  config.use_xpu = paddle::xpu_place_used(place);
W
Wilber 已提交
70
  config.use_npu = paddle::npu_place_used(place);
X
Xin Pan 已提交
71 72 73 74 75 76 77 78 79 80

  framework::LoDTensor first_word, second_word, third_word, fourth_word;
  framework::LoD lod{{0, 1}};
  int64_t dict_size = 2073;  // The size of dictionary

  SetupLoDTensor(&first_word, lod, static_cast<int64_t>(0), dict_size - 1);
  SetupLoDTensor(&second_word, lod, static_cast<int64_t>(0), dict_size - 1);
  SetupLoDTensor(&third_word, lod, static_cast<int64_t>(0), dict_size - 1);
  SetupLoDTensor(&fourth_word, lod, static_cast<int64_t>(0), dict_size - 1);

81 82 83 84 85 86 87 88 89
  std::vector<PaddleTensor> paddle_tensor_feeds;
  paddle_tensor_feeds.push_back(LodTensorToPaddleTensor(&first_word));
  paddle_tensor_feeds.push_back(LodTensorToPaddleTensor(&second_word));
  paddle_tensor_feeds.push_back(LodTensorToPaddleTensor(&third_word));
  paddle_tensor_feeds.push_back(LodTensorToPaddleTensor(&fourth_word));

  std::vector<PaddleTensor> outputs;
  ASSERT_TRUE(predictor->Run(paddle_tensor_feeds, &outputs));
  ASSERT_EQ(outputs.size(), 1UL);
90 91
  size_t len = outputs[0].data.length();
  float* data = static_cast<float*>(outputs[0].data.data());
92
  for (size_t j = 0; j < len / sizeof(float); ++j) {
93 94 95 96 97 98 99 100 101 102
    ASSERT_LT(data[j], 1.0);
    ASSERT_GT(data[j], -1.0);
  }

  std::vector<paddle::framework::LoDTensor*> cpu_feeds;
  cpu_feeds.push_back(&first_word);
  cpu_feeds.push_back(&second_word);
  cpu_feeds.push_back(&third_word);
  cpu_feeds.push_back(&fourth_word);

103 104
  framework::FetchType output1;
  std::vector<paddle::framework::FetchType*> cpu_fetchs1;
105 106 107 108
  cpu_fetchs1.push_back(&output1);

  TestInference<platform::CPUPlace>(config.model_dir, cpu_feeds, cpu_fetchs1);

109
  auto output1_tensor = BOOST_GET(paddle::framework::LoDTensor, output1);
110 111
  float* lod_data = output1_tensor.data<float>();
  for (int i = 0; i < output1_tensor.numel(); ++i) {
J
JiabinYang 已提交
112 113
    EXPECT_LT(lod_data[i] - data[i], ACC_DIFF);
    EXPECT_GT(lod_data[i] - data[i], -ACC_DIFF);
114 115 116
  }
}

117
void MainImageClassification(const paddle::PaddlePlace& place) {
118 119
  int batch_size = 2;
  bool repeat = false;
Y
Yan Chunwei 已提交
120
  NativeConfig config = GetConfig();
121 122
  config.use_gpu = paddle::gpu_place_used(place);
  config.use_xpu = paddle::xpu_place_used(place);
W
Wilber 已提交
123
  config.use_npu = paddle::npu_place_used(place);
124
  config.model_dir =
125
      FLAGS_book_dirname + "/image_classification_resnet.inference.model";
126 127 128 129 130 131 132 133 134 135

  const bool is_combined = false;
  std::vector<std::vector<int64_t>> feed_target_shapes =
      GetFeedTargetShapes(config.model_dir, is_combined);

  framework::LoDTensor input;
  // Use normilized image pixels as input data,
  // which should be in the range [0.0, 1.0].
  feed_target_shapes[0][0] = batch_size;
  framework::DDim input_dims = framework::make_ddim(feed_target_shapes[0]);
136 137
  SetupTensor<float>(&input, input_dims, static_cast<float>(0),
                     static_cast<float>(1));
138 139 140
  std::vector<framework::LoDTensor*> cpu_feeds;
  cpu_feeds.push_back(&input);

141 142
  framework::FetchType output1;
  std::vector<framework::FetchType*> cpu_fetchs1;
143 144
  cpu_fetchs1.push_back(&output1);

L
Luo Tao 已提交
145 146
  TestInference<platform::CPUPlace, false, true>(
      config.model_dir, cpu_feeds, cpu_fetchs1, repeat, is_combined);
147

Y
Yan Chunwei 已提交
148
  auto predictor = CreatePaddlePredictor(config);
149 150
  std::vector<PaddleTensor> paddle_tensor_feeds;
  paddle_tensor_feeds.push_back(LodTensorToPaddleTensor(&input));
X
Xin Pan 已提交
151 152

  std::vector<PaddleTensor> outputs;
153
  ASSERT_TRUE(predictor->Run(paddle_tensor_feeds, &outputs));
154
  ASSERT_EQ(outputs.size(), 1UL);
155 156
  size_t len = outputs[0].data.length();
  float* data = static_cast<float*>(outputs[0].data.data());
157
  float* lod_data =
158
      BOOST_GET(paddle::framework::LoDTensor, output1).data<float>();
159
  for (size_t j = 0; j < len / sizeof(float); ++j) {
J
JiabinYang 已提交
160
    EXPECT_NEAR(lod_data[j], data[j], ACC_DIFF);
X
Xin Pan 已提交
161 162 163
  }
}

164
void MainThreadsWord2Vec(const paddle::PaddlePlace& place) {
T
tensor-tang 已提交
165
  NativeConfig config = GetConfig();
166 167
  config.use_gpu = paddle::gpu_place_used(place);
  config.use_xpu = paddle::xpu_place_used(place);
W
Wilber 已提交
168
  config.use_npu = paddle::npu_place_used(place);
T
tensor-tang 已提交
169 170
  auto main_predictor = CreatePaddlePredictor<NativeConfig>(config);

171
  // prepare inputs data and reference results
T
tensor-tang 已提交
172 173 174
  constexpr int num_jobs = 3;
  std::vector<std::vector<framework::LoDTensor>> jobs(num_jobs);
  std::vector<std::vector<PaddleTensor>> paddle_tensor_feeds(num_jobs);
175
  std::vector<framework::FetchType> refs(num_jobs);
T
tensor-tang 已提交
176 177 178 179 180 181 182 183 184 185 186 187
  for (size_t i = 0; i < jobs.size(); ++i) {
    // each job has 4 words
    jobs[i].resize(4);
    for (size_t j = 0; j < 4; ++j) {
      framework::LoD lod{{0, 1}};
      int64_t dict_size = 2073;  // The size of dictionary
      SetupLoDTensor(&jobs[i][j], lod, static_cast<int64_t>(0), dict_size - 1);
      paddle_tensor_feeds[i].push_back(LodTensorToPaddleTensor(&jobs[i][j]));
    }

    // get reference result of each job
    std::vector<paddle::framework::LoDTensor*> ref_feeds;
188
    std::vector<paddle::framework::FetchType*> ref_fetches(1, &refs[i]);
T
tensor-tang 已提交
189 190 191 192 193 194 195 196 197 198
    for (auto& word : jobs[i]) {
      ref_feeds.push_back(&word);
    }
    TestInference<platform::CPUPlace>(config.model_dir, ref_feeds, ref_fetches);
  }

  // create threads and each thread run 1 job
  std::vector<std::thread> threads;
  for (int tid = 0; tid < num_jobs; ++tid) {
    threads.emplace_back([&, tid]() {
Y
Yan Chunwei 已提交
199
      auto predictor = CreatePaddlePredictor(config);
T
tensor-tang 已提交
200 201 202 203 204 205
      auto& local_inputs = paddle_tensor_feeds[tid];
      std::vector<PaddleTensor> local_outputs;
      ASSERT_TRUE(predictor->Run(local_inputs, &local_outputs));

      // check outputs range
      ASSERT_EQ(local_outputs.size(), 1UL);
206 207
      const size_t len = local_outputs[0].data.length();
      float* data = static_cast<float*>(local_outputs[0].data.data());
T
tensor-tang 已提交
208 209 210 211 212 213
      for (size_t j = 0; j < len / sizeof(float); ++j) {
        ASSERT_LT(data[j], 1.0);
        ASSERT_GT(data[j], -1.0);
      }

      // check outputs correctness
214
      auto ref_tensor = BOOST_GET(paddle::framework::LoDTensor, refs[tid]);
215 216 217
      float* ref_data = ref_tensor.data<float>();
      EXPECT_EQ(ref_tensor.numel(), static_cast<int64_t>(len / sizeof(float)));
      for (int i = 0; i < ref_tensor.numel(); ++i) {
S
update  
superjomn 已提交
218
        EXPECT_NEAR(ref_data[i], data[i], 2e-3);
T
tensor-tang 已提交
219
      }
220 221 222 223 224 225 226
    });
  }
  for (int i = 0; i < num_jobs; ++i) {
    threads[i].join();
  }
}

227
void MainThreadsImageClassification(const paddle::PaddlePlace& place) {
228 229 230
  constexpr int num_jobs = 4;  // each job run 1 batch
  constexpr int batch_size = 1;
  NativeConfig config = GetConfig();
231 232
  config.use_gpu = paddle::gpu_place_used(place);
  config.use_xpu = paddle::xpu_place_used(place);
W
Wilber 已提交
233
  config.use_npu = paddle::npu_place_used(place);
234
  config.model_dir =
235
      FLAGS_book_dirname + "/image_classification_resnet.inference.model";
236 237 238 239

  auto main_predictor = CreatePaddlePredictor<NativeConfig>(config);
  std::vector<framework::LoDTensor> jobs(num_jobs);
  std::vector<std::vector<PaddleTensor>> paddle_tensor_feeds(num_jobs);
240
  std::vector<framework::FetchType> refs(num_jobs);
241 242 243 244 245 246 247 248 249 250 251
  for (size_t i = 0; i < jobs.size(); ++i) {
    // prepare inputs
    std::vector<std::vector<int64_t>> feed_target_shapes =
        GetFeedTargetShapes(config.model_dir, /*is_combined*/ false);
    feed_target_shapes[0][0] = batch_size;
    framework::DDim input_dims = framework::make_ddim(feed_target_shapes[0]);
    SetupTensor<float>(&jobs[i], input_dims, 0.f, 1.f);
    paddle_tensor_feeds[i].push_back(LodTensorToPaddleTensor(&jobs[i]));

    // get reference result of each job
    std::vector<framework::LoDTensor*> ref_feeds(1, &jobs[i]);
252
    std::vector<framework::FetchType*> ref_fetches(1, &refs[i]);
253 254
    TestInference<platform::CPUPlace>(config.model_dir, ref_feeds, ref_fetches);
  }
T
tensor-tang 已提交
255

256 257 258 259
  // create threads and each thread run 1 job
  std::vector<std::thread> threads;
  for (int tid = 0; tid < num_jobs; ++tid) {
    threads.emplace_back([&, tid]() {
Y
Yan Chunwei 已提交
260
      auto predictor = CreatePaddlePredictor(config);
261 262 263 264 265 266
      auto& local_inputs = paddle_tensor_feeds[tid];
      std::vector<PaddleTensor> local_outputs;
      ASSERT_TRUE(predictor->Run(local_inputs, &local_outputs));

      // check outputs correctness
      ASSERT_EQ(local_outputs.size(), 1UL);
267 268
      const size_t len = local_outputs[0].data.length();
      float* data = static_cast<float*>(local_outputs[0].data.data());
269
      auto ref_tensor = BOOST_GET(paddle::framework::LoDTensor, refs[tid]);
270 271 272
      float* ref_data = ref_tensor.data<float>();
      EXPECT_EQ((size_t)ref_tensor.numel(), len / sizeof(float));
      for (int i = 0; i < ref_tensor.numel(); ++i) {
J
JiabinYang 已提交
273
        EXPECT_NEAR(ref_data[i], data[i], ACC_DIFF);
274
      }
T
tensor-tang 已提交
275 276 277 278 279 280 281
    });
  }
  for (int i = 0; i < num_jobs; ++i) {
    threads[i].join();
  }
}

282 283 284
TEST(inference_api_native, word2vec_cpu) {
  MainWord2Vec(paddle::PaddlePlace::kCPU);
}
T
tensor-tang 已提交
285
TEST(inference_api_native, word2vec_cpu_threads) {
286
  MainThreadsWord2Vec(paddle::PaddlePlace::kCPU);
T
tensor-tang 已提交
287 288
}
TEST(inference_api_native, image_classification_cpu) {
289
  MainImageClassification(paddle::PaddlePlace::kCPU);
T
tensor-tang 已提交
290 291
}
TEST(inference_api_native, image_classification_cpu_threads) {
292
  MainThreadsImageClassification(paddle::PaddlePlace::kCPU);
T
tensor-tang 已提交
293 294
}

295 296 297 298 299 300 301 302 303
#ifdef PADDLE_WITH_XPU
TEST(inference_api_native, word2vec_xpu) {
  MainWord2Vec(paddle::PaddlePlace::kXPU);
}
TEST(inference_api_native, image_classification_xpu) {
  MainImageClassification(paddle::PaddlePlace::kXPU);
}
#endif

W
Wilber 已提交
304 305 306 307 308 309 310 311 312
#ifdef PADDLE_WITH_ASCEND_CL
TEST(inference_api_native, word2vec_npu) {
  MainWord2Vec(paddle::PaddlePlace::kNPU);
}
// TEST(inference_api_native, image_classification_npu) {
//   MainImageClassification(paddle::PaddlePlace::kNPU);
// }
#endif

313
#if defined(PADDLE_WITH_CUDA) || defined(PADDLE_WITH_HIP)
314 315 316
TEST(inference_api_native, word2vec_gpu) {
  MainWord2Vec(paddle::PaddlePlace::kGPU);
}
S
superjomn 已提交
317 318
// Turn off temporarily for the unstable result.
// TEST(inference_api_native, word2vec_gpu_threads) {
319
//   MainThreadsWord2Vec(paddle::PaddlePlace::kGPU);
S
superjomn 已提交
320
// }
T
tensor-tang 已提交
321
TEST(inference_api_native, image_classification_gpu) {
322
  MainImageClassification(paddle::PaddlePlace::kGPU);
T
tensor-tang 已提交
323
}
S
superjomn 已提交
324 325
// Turn off temporarily for the unstable result.
// TEST(inference_api_native, image_classification_gpu_threads) {
326
//   MainThreadsImageClassification(paddle::PaddlePlace::kGPU);
S
superjomn 已提交
327
// }
T
tensor-tang 已提交
328 329
#endif

330
TEST(PassBuilder, Delete) {
331
  AnalysisConfig config;
332
  config.DisableGpu();
333 334 335 336 337 338
  config.pass_builder()->DeletePass("attention_lstm_fuse_pass");
  const auto& passes = config.pass_builder()->AllPasses();
  auto it = std::find(passes.begin(), passes.end(), "attention_lstm_fuse_pass");
  ASSERT_EQ(it, passes.end());
}

X
Xin Pan 已提交
339
}  // namespace paddle