op_registry.cc 6.5 KB
Newer Older
Y
Yan Chunwei 已提交
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56
// Copyright (c) 2019 PaddlePaddle Authors. All Rights Reserved.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
//     http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

#include "lite/core/op_registry.h"
#include <list>
#include <set>

namespace paddle {
namespace lite {

std::list<std::unique_ptr<KernelBase>> KernelRegistry::Create(
    const std::string &op_type,
    TargetType target,
    PrecisionType precision,
    DataLayoutType layout) {
  Place place{target, precision, layout};
  VLOG(5) << "creating " << op_type << " kernel for " << place.DebugString();
#define CREATE_KERNEL1(target__, precision__)                                \
  switch (layout) {                                                          \
    case DATALAYOUT(kNCHW):                                                  \
      return Create<TARGET(target__),                                        \
                    PRECISION(precision__),                                  \
                    DATALAYOUT(kNCHW)>(op_type);                             \
    case DATALAYOUT(kAny):                                                   \
      return Create<TARGET(target__),                                        \
                    PRECISION(precision__),                                  \
                    DATALAYOUT(kAny)>(op_type);                              \
    case DATALAYOUT(kNHWC):                                                  \
      return Create<TARGET(target__),                                        \
                    PRECISION(precision__),                                  \
                    DATALAYOUT(kNHWC)>(op_type);                             \
    default:                                                                 \
      LOG(FATAL) << "unsupported kernel layout " << DataLayoutToStr(layout); \
  }

#define CREATE_KERNEL(target__)                         \
  switch (precision) {                                  \
    case PRECISION(kFloat):                             \
      CREATE_KERNEL1(target__, kFloat);                 \
    case PRECISION(kInt8):                              \
      CREATE_KERNEL1(target__, kInt8);                  \
    case PRECISION(kFP16):                              \
      CREATE_KERNEL1(target__, kFP16);                  \
    case PRECISION(kAny):                               \
      CREATE_KERNEL1(target__, kAny);                   \
J
juncaipeng 已提交
57 58
    case PRECISION(kInt32):                             \
      CREATE_KERNEL1(target__, kInt32);                 \
59 60
    case PRECISION(kInt64):                             \
      CREATE_KERNEL1(target__, kInt64);                 \
Y
Yan Chunwei 已提交
61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84
    default:                                            \
      CHECK(false) << "not supported kernel precision " \
                   << PrecisionToStr(precision);        \
  }

  switch (target) {
    case TARGET(kHost): {
      CREATE_KERNEL(kHost);
    } break;
    case TARGET(kX86): {
      CREATE_KERNEL(kX86);
    } break;
    case TARGET(kCUDA): {
      CREATE_KERNEL(kCUDA);
    } break;
    case TARGET(kARM): {
      CREATE_KERNEL(kARM);
    } break;
    case TARGET(kOpenCL): {
      CREATE_KERNEL(kOpenCL);
    } break;
    case TARGET(kNPU): {
      CREATE_KERNEL(kNPU);
    } break;
85 86 87
    case TARGET(kXPU): {
      CREATE_KERNEL(kXPU);
    } break;
Y
Yan Chunwei 已提交
88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114
    case TARGET(kFPGA): {
      CREATE_KERNEL(kFPGA);
    } break;
    default:
      CHECK(false) << "not supported kernel target " << TargetToStr(target);
  }

#undef CREATE_KERNEL
  return std::list<std::unique_ptr<KernelBase>>();
}

KernelRegistry::KernelRegistry()
    : registries_(static_cast<int>(TARGET(NUM)) *
                  static_cast<int>(PRECISION(NUM)) *
                  static_cast<int>(DATALAYOUT(NUM))) {
#define INIT_FOR(target__, precision__, layout__)                      \
  registries_[KernelRegistry::GetKernelOffset<TARGET(target__),        \
                                              PRECISION(precision__),  \
                                              DATALAYOUT(layout__)>()] \
      .set<KernelRegistryForTarget<TARGET(target__),                   \
                                   PRECISION(precision__),             \
                                   DATALAYOUT(layout__)> *>(           \
          &KernelRegistryForTarget<TARGET(target__),                   \
                                   PRECISION(precision__),             \
                                   DATALAYOUT(layout__)>::Global());
  // Currently, just register 2 kernel targets.
  INIT_FOR(kCUDA, kFloat, kNCHW);
115
  INIT_FOR(kCUDA, kFloat, kNHWC);
Z
Zhen Wang 已提交
116
  INIT_FOR(kCUDA, kInt8, kNCHW);
Y
Yan Chunwei 已提交
117 118
  INIT_FOR(kCUDA, kAny, kNCHW);
  INIT_FOR(kCUDA, kAny, kAny);
119
  INIT_FOR(kCUDA, kInt8, kNHWC);
120 121
  INIT_FOR(kCUDA, kInt64, kNCHW);
  INIT_FOR(kCUDA, kInt64, kNHWC);
Y
Yan Chunwei 已提交
122 123 124 125 126 127 128 129 130 131 132 133 134

  INIT_FOR(kHost, kFloat, kNCHW);
  INIT_FOR(kHost, kAny, kNCHW);
  INIT_FOR(kHost, kFloat, kNHWC);
  INIT_FOR(kHost, kFloat, kAny);
  INIT_FOR(kHost, kAny, kNHWC);
  INIT_FOR(kHost, kAny, kAny);
  INIT_FOR(kHost, kAny, kNHWC);
  INIT_FOR(kHost, kAny, kAny);

  INIT_FOR(kX86, kFloat, kNCHW);
  INIT_FOR(kX86, kAny, kNCHW);
  INIT_FOR(kX86, kAny, kAny);
135
  INIT_FOR(kX86, kInt64, kNCHW);
Y
Yan Chunwei 已提交
136 137 138 139 140

  INIT_FOR(kARM, kFloat, kNCHW);
  INIT_FOR(kARM, kInt8, kNCHW);
  INIT_FOR(kARM, kAny, kNCHW);
  INIT_FOR(kARM, kAny, kAny);
J
juncaipeng 已提交
141
  INIT_FOR(kARM, kInt32, kNCHW);
Y
Yan Chunwei 已提交
142 143

  INIT_FOR(kOpenCL, kFloat, kNCHW);
144
  INIT_FOR(kOpenCL, kFloat, kNHWC);
Y
Yan Chunwei 已提交
145
  INIT_FOR(kOpenCL, kAny, kNCHW);
146 147 148
  INIT_FOR(kOpenCL, kAny, kNHWC);
  INIT_FOR(kOpenCL, kFloat, kAny);
  INIT_FOR(kOpenCL, kInt8, kNCHW);
Y
Yan Chunwei 已提交
149 150 151 152 153 154 155
  INIT_FOR(kOpenCL, kAny, kAny);

  INIT_FOR(kNPU, kFloat, kNCHW);
  INIT_FOR(kNPU, kInt8, kNCHW);
  INIT_FOR(kNPU, kAny, kNCHW);
  INIT_FOR(kNPU, kAny, kAny);

156 157 158 159 160
  INIT_FOR(kXPU, kFloat, kNCHW);
  INIT_FOR(kXPU, kInt8, kNCHW);
  INIT_FOR(kXPU, kAny, kNCHW);
  INIT_FOR(kXPU, kAny, kAny);

Y
Yan Chunwei 已提交
161 162 163 164 165 166 167 168 169 170 171 172 173 174 175
  INIT_FOR(kFPGA, kFP16, kNHWC);
  INIT_FOR(kFPGA, kFP16, kAny);
  INIT_FOR(kFPGA, kFloat, kNHWC);
  INIT_FOR(kFPGA, kAny, kNHWC);
  INIT_FOR(kFPGA, kAny, kAny);
#undef INIT_FOR
}

KernelRegistry &KernelRegistry::Global() {
  static auto *x = new KernelRegistry;
  return *x;
}

}  // namespace lite
}  // namespace paddle