op_meta_info.h 32.8 KB
Newer Older
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16
/* Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.

Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at

    http://www.apache.org/licenses/LICENSE-2.0

Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License. */

#pragma once

17
#include <iostream>
18 19
#include <string>
#include <unordered_map>
20
#include <utility>
21 22
#include <vector>

23
#include "paddle/phi/api/ext/exception.h"
24
#include "paddle/phi/api/include/dll_decl.h"
25
#include "paddle/phi/api/include/tensor.h"
26
#include "paddle/utils/any.h"
27 28
#include "paddle/utils/none.h"
#include "paddle/utils/optional.h"
29 30 31 32 33 34 35 36 37 38

/**
 * Op Meta Info Related Define.
 *
 * Used to maintain operator core information.
 *
 */

namespace paddle {

39
class PADDLE_API OpMetaInfoHelper;
40
using Tensor = paddle::Tensor;
41

42 43
///////////////// Util Marco Define ////////////////

44 45 46 47 48 49 50
#define PD_DISABLE_COPY_AND_ASSIGN(classname)      \
 private:                                          \
  classname(const classname&) = delete;            \
  classname(classname&&) = delete;                 \
  classname& operator=(const classname&) = delete; \
  classname& operator=(classname&&) = delete

51 52 53 54 55 56
#define STATIC_ASSERT_GLOBAL_NAMESPACE(uniq_name, msg)                        \
  struct __test_global_namespace_##uniq_name##__ {};                          \
  static_assert(std::is_same<::__test_global_namespace_##uniq_name##__,       \
                             __test_global_namespace_##uniq_name##__>::value, \
                msg)

57 58
///////////////// Util Define and Function ////////////////

59 60
constexpr char kGradTensorSuffix[] = "@GRAD";
constexpr char kTensorVectorSuffix[] = "@VECTOR";
61
constexpr char kDoubleGradNewOutSuffix[] = "@NEW";
62
constexpr char kOptionalSuffix[] = "@OPTIONAL";
63 64 65 66 67 68 69 70 71 72 73 74

// Used for Construct Grad Tensor name
inline std::string Grad(const std::string& t_name) {
  std::string result;
  result.reserve(t_name.size() + 5U);
  result += t_name;
  result += kGradTensorSuffix;
  return result;
}

// Used for Construct std::vector<Tensor> name
inline std::string Vec(const std::string& t_name) {
75
  std::string result;
76 77 78
  result.reserve(t_name.size() + 7U);
  result += t_name;
  result += kTensorVectorSuffix;
79 80 81
  return result;
}

82 83 84 85 86 87 88 89 90
// Used for Construct double grad output name
inline std::string New(const std::string& t_name) {
  std::string result;
  result.reserve(t_name.size() + 4U);
  result += t_name;
  result += kDoubleGradNewOutSuffix;
  return result;
}

91 92 93 94 95 96 97 98 99
// Used for Construct paddle::optional name
inline std::string Optional(const std::string& t_name) {
  std::string result;
  result.reserve(t_name.size() + 9U);
  result += t_name;
  result += kOptionalSuffix;
  return result;
}

100 101 102 103 104 105 106 107 108
PADDLE_API void AssignTensorImpl(const Tensor& src, Tensor* dst);

////////////////////// Kernel Context ////////////////////////

class PADDLE_API CustomOpKernelContext {
 public:
  CustomOpKernelContext() = default;

  void EmplaceBackInput(Tensor&& input);
109
  void EmplaceBackInputs(const std::vector<Tensor>& inputs);
110
  void EmplaceBackOutput(Tensor&& output);
111
  void EmplaceBackOutputs(const std::vector<Tensor>& outputs);
112
  void EmplaceBackAttr(paddle::any attr);
113 114 115
  void EmplaceBackAttrs(const std::vector<paddle::any>& attrs) {
    attrs_ = std::move(attrs);
  }
116 117 118 119 120
  const std::pair<size_t, size_t>& InputRangeAt(size_t idx) const;
  const std::pair<size_t, size_t>& OutputRangeAt(size_t idx) const;

  const Tensor& InputAt(size_t idx) const;
  std::vector<Tensor> InputsBetween(size_t start, size_t end) const;
121
  Tensor& MutableInputAt(size_t idx);
122 123 124 125 126 127 128
  const std::vector<paddle::any>& Attrs() const { return attrs_; }
  const std::vector<std::pair<size_t, size_t>>& InputRange() {
    return input_range_;
  }
  const std::vector<std::pair<size_t, size_t>>& OutputRange() {
    return output_range_;
  }
129 130
  Tensor* MutableOutputAt(size_t idx);
  std::vector<Tensor*> MutableOutputBetweeen(size_t start, size_t end);
131
  std::vector<Tensor> OutputsBetweeen(size_t start, size_t end);
132 133 134 135 136 137 138 139 140 141 142
  std::vector<Tensor>* AllMutableOutput();

  template <typename AttrType>
  AttrType AttrAt(size_t idx) const {
    try {
      return paddle::any_cast<AttrType>(attrs_.at(idx));
    } catch (paddle::bad_any_cast&) {
      PD_THROW("Attribute cast error in Custom Op Kernel Context.");
    }
  }

143
  // handle inplace map
144 145 146 147 148 149 150 151
  void MapPlainOutputs(
      const std::vector<std::string>& inputs,
      const std::vector<std::string>& outputs,
      const std::unordered_map<std::string, std::string>& inplace_map);
  void AssignInplaceOutputs();
  std::vector<Tensor*>* AllMutablePlainOutput();
  std::unordered_map<size_t, size_t> GetInplaceTensorMap();

152 153 154 155 156
 private:
  // TODO(chenweihang): replaced be SmallVector
  std::vector<Tensor> inputs_;
  std::vector<Tensor> outputs_;
  std::vector<paddle::any> attrs_;
157
  // handle inplace map
158 159
  std::vector<Tensor*> plain_outputs_;
  std::unordered_map<size_t, size_t> inplace_tensor_map_;
160 161 162 163 164

  std::vector<std::pair<size_t, size_t>> input_range_;
  std::vector<std::pair<size_t, size_t>> output_range_;
};

165 166 167
////////////////////// Kernel Function (PD_KERNEL) ////////////////////////

// Record Op kernel core function
168 169 170 171 172 173
using KernelFunc = void (*)(CustomOpKernelContext*);

#define PD_SPECIALIZE_ComputeCallHelper(attr_type)                             \
  template <typename... Tail>                                                  \
  struct ComputeCallHelper<attr_type, Tail...> {                               \
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs> \
174
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {  \
175 176 177 178 179 180
      attr_type arg = ctx->AttrAt<attr_type>(attr_idx);                        \
      ComputeCallHelper<                                                       \
          Tail...>::template Compute<in_idx, attr_idx + 1, out_idx>(ctx,       \
                                                                    pargs...,  \
                                                                    arg);      \
    }                                                                          \
181 182
  }

183 184 185 186 187 188 189 190
template <typename T>
struct TypeTag {};

template <typename F, F f>
struct KernelFuncImpl;

template <typename Return, typename... Args, Return (*impl_fn)(Args...)>
struct KernelFuncImpl<Return (*)(Args...), impl_fn> {
191 192
  static void Compute(CustomOpKernelContext* ctx) {
    ComputeCallHelper<Args..., TypeTag<int>>::template Compute<0, 0, 0>(ctx);
193 194 195 196 197 198 199 200
  }

 private:
  template <typename... RemainingArgs>
  struct ComputeCallHelper;

  template <typename... Tail>
  struct ComputeCallHelper<const Tensor&, Tail...> {
201
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
202
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
203
      auto& range = ctx->InputRangeAt(in_idx);
204
      auto& arg = ctx->MutableInputAt(range.first);
205 206 207 208
      ComputeCallHelper<
          Tail...>::template Compute<in_idx + 1, attr_idx, out_idx>(ctx,
                                                                    pargs...,
                                                                    arg);
209 210 211
    }
  };

212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230
  template <typename... Tail>
  struct ComputeCallHelper<const paddle::optional<paddle::Tensor>&, Tail...> {
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
      auto& range = ctx->InputRangeAt(in_idx);
      auto& arg = ctx->InputAt(range.first);
      if (!arg.is_initialized()) {
        ComputeCallHelper<Tail...>::
            template Compute<in_idx + 1, attr_idx, out_idx>(
                ctx, pargs..., paddle::none);
      } else {
        ComputeCallHelper<
            Tail...>::template Compute<in_idx + 1, attr_idx, out_idx>(ctx,
                                                                      pargs...,
                                                                      arg);
      }
    }
  };

231 232
  template <typename... Tail>
  struct ComputeCallHelper<const std::vector<Tensor>&, Tail...> {
233
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
234
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
235 236 237 238 239 240
      auto& range = ctx->InputRangeAt(in_idx);
      auto arg = ctx->InputsBetween(range.first, range.second);
      ComputeCallHelper<
          Tail...>::template Compute<in_idx + 1, attr_idx, out_idx>(ctx,
                                                                    pargs...,
                                                                    arg);
241 242 243
    }
  };

244 245 246 247
  PD_SPECIALIZE_ComputeCallHelper(bool);
  PD_SPECIALIZE_ComputeCallHelper(int);
  PD_SPECIALIZE_ComputeCallHelper(float);
  PD_SPECIALIZE_ComputeCallHelper(int64_t);
248 249 250 251 252 253 254
  PD_SPECIALIZE_ComputeCallHelper(const std::string&);
  PD_SPECIALIZE_ComputeCallHelper(const std::vector<int>&);
  PD_SPECIALIZE_ComputeCallHelper(const std::vector<float>&);
  PD_SPECIALIZE_ComputeCallHelper(const std::vector<int64_t>&);
  PD_SPECIALIZE_ComputeCallHelper(const std::vector<std::string>&);
  // TODO(chenweihang): support other attribute type if needed.
  // Why not support other attribute type here?
R
Ruibiao Chen 已提交
255
  // - paddle::blank, std::vector<bool> and std::vector<double>
256 257 258 259 260
  //   are not used in op
  // - BlockDesc* and std::vector<BlockDesc*> are used in framework

  // NOTE(chenweihang): Used to be compatible with the 2.0.1 released
  // interface, and will be deprecated in the future
261 262 263 264
  PD_SPECIALIZE_ComputeCallHelper(const bool&);
  PD_SPECIALIZE_ComputeCallHelper(const int&);
  PD_SPECIALIZE_ComputeCallHelper(const float&);
  PD_SPECIALIZE_ComputeCallHelper(const int64_t&);
265

C
Chen Weihang 已提交
266 267 268 269 270 271 272 273
  // NOTE(chenweihang): Used to be compatible with the 2.1 released
  // interface, but not recommended
  PD_SPECIALIZE_ComputeCallHelper(std::string);
  PD_SPECIALIZE_ComputeCallHelper(std::vector<int>);
  PD_SPECIALIZE_ComputeCallHelper(std::vector<float>);
  PD_SPECIALIZE_ComputeCallHelper(std::vector<int64_t>);
  PD_SPECIALIZE_ComputeCallHelper(std::vector<std::string>);

274 275
  // Used to be compatible with 2.3 released internal inplace interface, not
  // recommended
276 277 278
  template <typename... Tail>
  struct ComputeCallHelper<Tensor*, Tail...> {
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
279
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
280 281 282 283 284 285 286 287 288
      auto& range = ctx->OutputRangeAt(out_idx);
      auto* arg = ctx->MutableOutputAt(range.first);
      ComputeCallHelper<
          Tail...>::template Compute<in_idx, attr_idx, out_idx + 1>(ctx,
                                                                    pargs...,
                                                                    arg);
    }
  };

289 290
  // Used to be compatible with 2.3 released internal inplace interface, not
  // recommended
291 292 293 294 295
  // TODO(chenweihang): What is the appropriate output form?
  // std::vector<Tensor>*? or std::vector<Tensor*>? or std::vector<Tensor*>*
  template <typename... Tail>
  struct ComputeCallHelper<std::vector<Tensor*>, Tail...> {
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
296
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
297 298 299 300 301 302 303 304 305
      auto& range = ctx->OutputRangeAt(out_idx);
      auto arg = ctx->MutableOutputBetweeen(range.first, range.second);
      ComputeCallHelper<
          Tail...>::template Compute<in_idx, attr_idx, out_idx + 1>(ctx,
                                                                    pargs...,
                                                                    arg);
    }
  };

306 307 308 309 310 311 312 313 314 315 316 317 318 319
  // Handle Tensor& for inplace case
  template <typename... Tail>
  struct ComputeCallHelper<Tensor&, Tail...> {
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
      auto& range = ctx->InputRangeAt(in_idx);
      auto& arg = ctx->MutableInputAt(range.first);
      ComputeCallHelper<
          Tail...>::template Compute<in_idx + 1, attr_idx, out_idx>(ctx,
                                                                    pargs...,
                                                                    arg);
    }
  };

320 321 322 323 324 325
  template <int out_idx, typename T>
  struct ComputeReturnHelper;

  // For compatibility with the original custom op form
  template <int out_idx>
  struct ComputeReturnHelper<out_idx, std::vector<Tensor>> {
326
    static void Compute(CustomOpKernelContext* ctx, Args&... args) {
327 328
      static_assert(out_idx == 0,
                    "If return std::vector<Tensor> in Custom OpKernel, "
H
HongyuJia 已提交
329
                    "you cannot pass output by kernel function argument.");
330
      auto outs = impl_fn(args...);
331
      auto* orig_outs = ctx->AllMutablePlainOutput();
332 333 334 335 336 337 338 339
      PD_CHECK(orig_outs->size() == outs.size(),
               "The number of element in custom operator outputs is wrong, "
               "expected contains ",
               orig_outs->size(),
               " Tensors, but actually contains ",
               outs.size(),
               " Tensors.");
      for (size_t i = 0; i < outs.size(); ++i) {
340
        AssignTensorImpl(outs.at(i), orig_outs->at(i));
341 342 343 344 345 346
      }
    }
  };

  template <int out_idx>
  struct ComputeReturnHelper<out_idx, void> {
347
    static void Compute(CustomOpKernelContext* ctx, Args&... args) {
348 349 350 351
      impl_fn(args...);
    }
  };

352 353 354
  // end: base template
  template <typename T>
  struct ComputeCallHelper<TypeTag<T>> {
355
    template <int in_idx, int attr_idx, int out_idx, typename... PreviousArgs>
356
    static void Compute(CustomOpKernelContext* ctx, PreviousArgs&... pargs) {
357
      ComputeReturnHelper<out_idx, Return>::Compute(ctx, pargs...);
358 359 360 361 362 363 364 365 366 367 368
    }
  };
};

#define PD_KERNEL(...) \
  ::paddle::KernelFuncImpl<decltype(&__VA_ARGS__), &__VA_ARGS__>::Compute

/////////////// InferShape Function (PD_INFER_SHAPE) ///////////////

// Record Op infershape core function
using InferShapeFunc = std::vector<std::vector<int64_t>> (*)(
369
    const std::vector<std::vector<int64_t>>& input_shapes,
370
    const std::vector<std::vector<std::vector<int64_t>>>& vec_input_shapes,
371 372
    const std::vector<paddle::any>& attrs);

373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390
#define PD_SPECIALIZE_InferShapeCallHelper_FOR_SHAPE(input_type)     \
  template <typename... Tail>                                        \
  struct InferShapeCallHelper<input_type, Tail...> {                 \
    template <int in_idx,                                            \
              int vec_in_idx,                                        \
              int attr_idx,                                          \
              typename... PreviousArgs>                              \
    static Return InferShape(                                        \
        const std::vector<std::vector<int64_t>>& input_shapes,       \
        const std::vector<std::vector<std::vector<int64_t>>>&        \
            vec_input_shapes,                                        \
        const std::vector<paddle::any>& attrs,                       \
        const PreviousArgs&... pargs) {                              \
      input_type arg = input_shapes[in_idx];                         \
      return InferShapeCallHelper<Tail...>::                         \
          template InferShape<in_idx + 1, vec_in_idx, attr_idx>(     \
              input_shapes, vec_input_shapes, attrs, pargs..., arg); \
    }                                                                \
391 392
  }

393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410
#define PD_SPECIALIZE_InferShapeCallHelper_FOR_SHAPES(input_type)    \
  template <typename... Tail>                                        \
  struct InferShapeCallHelper<input_type, Tail...> {                 \
    template <int in_idx,                                            \
              int vec_in_idx,                                        \
              int attr_idx,                                          \
              typename... PreviousArgs>                              \
    static Return InferShape(                                        \
        const std::vector<std::vector<int64_t>>& input_shapes,       \
        const std::vector<std::vector<std::vector<int64_t>>>&        \
            vec_input_shapes,                                        \
        const std::vector<paddle::any>& attrs,                       \
        const PreviousArgs&... pargs) {                              \
      input_type arg = vec_input_shapes[vec_in_idx];                 \
      return InferShapeCallHelper<Tail...>::                         \
          template InferShape<in_idx, vec_in_idx + 1, attr_idx>(     \
              input_shapes, vec_input_shapes, attrs, pargs..., arg); \
    }                                                                \
411 412
  }

413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440
#define PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(attr_type)               \
  template <typename... Tail>                                                \
  struct InferShapeCallHelper<attr_type, Tail...> {                          \
    template <int in_idx,                                                    \
              int vec_in_idx,                                                \
              int attr_idx,                                                  \
              typename... PreviousArgs>                                      \
    static Return InferShape(                                                \
        const std::vector<std::vector<int64_t>>& input_shapes,               \
        const std::vector<std::vector<std::vector<int64_t>>>&                \
            vec_input_shapes,                                                \
        const std::vector<paddle::any>& attrs,                               \
        const PreviousArgs&... pargs) {                                      \
      try {                                                                  \
        attr_type arg = paddle::any_cast<attr_type>(attrs[attr_idx]);        \
        return InferShapeCallHelper<Tail...>::                               \
            template InferShape<in_idx, vec_in_idx, attr_idx + 1>(           \
                input_shapes, vec_input_shapes, attrs, pargs..., arg);       \
      } catch (paddle::bad_any_cast&) {                                      \
        PD_THROW(                                                            \
            "Attribute cast error in custom operator InferShapeFn. "         \
            "Expected " #attr_type                                           \
            " value. InferShapeFn's attribute list must be exactly same as " \
            "Forward "                                                       \
            "KernelFn's attribute list except std::vector<int64_t> "         \
            "attribute.");                                                   \
      }                                                                      \
    }                                                                        \
441
  }
442 443 444 445 446 447

template <typename F, F f>
struct InferShapeFuncImpl;

template <typename Return, typename... Args, Return (*impl_fn)(Args...)>
struct InferShapeFuncImpl<Return (*)(Args...), impl_fn> {
448
  static Return InferShape(
449
      const std::vector<std::vector<int64_t>>& input_shapes,
450
      const std::vector<std::vector<std::vector<int64_t>>>& vec_input_shapes,
451
      const std::vector<paddle::any>& attrs) {
452 453
    return InferShapeCallHelper<Args..., TypeTag<int>>::
        template InferShape<0, 0, 0>(input_shapes, vec_input_shapes, attrs);
454 455 456 457 458 459
  }

 private:
  template <typename... RemainingArgs>
  struct InferShapeCallHelper;

460 461 462
  PD_SPECIALIZE_InferShapeCallHelper_FOR_SHAPE(const std::vector<int64_t>&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_SHAPES(
      const std::vector<std::vector<int64_t>>&);
463

464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488
  template <typename... Tail>
  struct InferShapeCallHelper<const paddle::optional<std::vector<int64_t>>&,
                              Tail...> {
    template <int in_idx,
              int vec_in_idx,
              int attr_idx,
              typename... PreviousArgs>
    static Return InferShape(
        const std::vector<std::vector<int64_t>>& input_shapes,
        const std::vector<std::vector<std::vector<int64_t>>>& vec_input_shapes,
        const std::vector<paddle::any>& attrs,
        const PreviousArgs&... pargs) {
      const std::vector<int64_t>& arg = input_shapes[in_idx];
      if (arg.empty()) {
        return InferShapeCallHelper<Tail...>::
            template InferShape<in_idx + 1, vec_in_idx, attr_idx>(
                input_shapes, vec_input_shapes, attrs, pargs..., paddle::none);
      } else {
        return InferShapeCallHelper<Tail...>::
            template InferShape<in_idx + 1, vec_in_idx, attr_idx>(
                input_shapes, vec_input_shapes, attrs, pargs..., arg);
      }
    }
  };

489 490 491 492 493
  // NOTE(chenweihang): Used to be compatible with the 2.0.1 released
  // interface, and will be deprecated in the future
  PD_SPECIALIZE_InferShapeCallHelper_FOR_SHAPE(std::vector<int64_t>);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_SHAPES(
      std::vector<std::vector<int64_t>>);
494

495 496 497 498
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(bool);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(int);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(float);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(int64_t);
499 500 501 502 503 504 505 506
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const std::string&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const std::vector<int>&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const std::vector<float>&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const std::vector<std::string>&);
  // NOTE(chenweihang): InferShape can't support std::vector<int64_t> attr type,
  // because the input type is std::vector<int64_t>, only can use one rule to
  // parse std::vector<int64_t> parameter

507 508 509 510 511 512 513
  // NOTE(chenweihang): Used to be compatible with the 2.0.1 released
  // interface, and will be deprecated in the future
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const bool&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const int&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const float&);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(const int64_t&);

C
Chen Weihang 已提交
514 515 516 517 518 519 520
  // NOTE(chenweihang): Used to be compatible with the 2.1 released
  // interface, but not recommended
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(std::string);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(std::vector<int>);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(std::vector<float>);
  PD_SPECIALIZE_InferShapeCallHelper_FOR_ATTR(std::vector<std::string>);

521 522 523
  // end: base template
  template <typename T>
  struct InferShapeCallHelper<TypeTag<T>> {
524
    template <int in_idx, int vec_in_idx, int attr_idx>
525
    static Return InferShape(
526 527
        const std::vector<std::vector<int64_t>>& input_shapes,
        const std::vector<std::vector<std::vector<int64_t>>>& vec_input_shapes,
528 529
        const std::vector<paddle::any>& attrs,
        const Args&... args) {
530 531 532 533 534 535 536 537 538 539 540
      return impl_fn(args...);
    }
  };
};

#define PD_INFER_SHAPE(...) \
  ::paddle::InferShapeFuncImpl<decltype(&__VA_ARGS__), &__VA_ARGS__>::InferShape

/////////////// InferDataType Function (PD_INFER_DTYPE) ///////////////

// Record Op Infer dtype core function
541
using InferDtypeFunc = std::vector<DataType> (*)(
542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559
    const std::vector<DataType>& input_dtypes,
    const std::vector<std::vector<DataType>>& vec_input_dtypes);

#define PD_SPECIALIZE_InferDtypeCallHelper_TO_DTYPE(input_type)              \
  template <typename... Tail>                                                \
  struct InferDtypeCallHelper<input_type, Tail...> {                         \
    template <int in_idx, int vec_in_idx, typename... PreviousArgs>          \
    static Return InferDtype(                                                \
        const std::vector<DataType>& input_dtypes,                           \
        const std::vector<std::vector<DataType>>& vec_input_dtypes,          \
        const PreviousArgs&... pargs) {                                      \
      input_type arg = input_dtypes[in_idx];                                 \
      return InferDtypeCallHelper<Tail...>::template InferDtype<in_idx + 1,  \
                                                                vec_in_idx>( \
          input_dtypes, vec_input_dtypes, pargs..., arg);                    \
    }                                                                        \
  }

560 561 562 563 564 565 566 567 568 569 570 571 572
#define PD_SPECIALIZE_InferDtypeCallHelper_FOR_DTYPES(input_type)   \
  template <typename... Tail>                                       \
  struct InferDtypeCallHelper<input_type, Tail...> {                \
    template <int in_idx, int vec_in_idx, typename... PreviousArgs> \
    static Return InferDtype(                                       \
        const std::vector<DataType>& input_dtypes,                  \
        const std::vector<std::vector<DataType>>& vec_input_dtypes, \
        const PreviousArgs&... pargs) {                             \
      input_type arg = vec_input_dtypes[vec_in_idx];                \
      return InferDtypeCallHelper<Tail...>::                        \
          template InferDtype<in_idx, vec_in_idx + 1>(              \
              input_dtypes, vec_input_dtypes, pargs..., arg);       \
    }                                                               \
573
  }
574 575 576 577 578 579

template <typename F, F f>
struct InferDtypeFuncImpl;

template <typename Return, typename... Args, Return (*impl_fn)(Args...)>
struct InferDtypeFuncImpl<Return (*)(Args...), impl_fn> {
580
  static Return InferDtype(
581 582
      const std::vector<DataType>& input_dtypes,
      const std::vector<std::vector<DataType>>& vec_input_dtypes) {
583 584 585
    return InferDtypeCallHelper<Args..., TypeTag<int>>::template InferDtype<0,
                                                                            0>(
        input_dtypes, vec_input_dtypes);
586 587 588 589 590 591
  }

 private:
  template <typename... RemainingArgs>
  struct InferDtypeCallHelper;

592 593
  PD_SPECIALIZE_InferDtypeCallHelper_TO_DTYPE(const DataType&);
  PD_SPECIALIZE_InferDtypeCallHelper_FOR_DTYPES(const std::vector<DataType>&);
594

595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615
  template <typename... Tail>
  struct InferDtypeCallHelper<const paddle::optional<paddle::DataType>&,
                              Tail...> {
    template <int in_idx, int vec_in_idx, typename... PreviousArgs>
    static Return InferDtype(
        const std::vector<DataType>& input_dtypes,
        const std::vector<std::vector<DataType>>& vec_input_dtypes,
        const PreviousArgs&... pargs) {
      const DataType& arg = input_dtypes[in_idx];
      if (arg == DataType::UNDEFINED) {
        return InferDtypeCallHelper<Tail...>::template InferDtype<in_idx + 1,
                                                                  vec_in_idx>(
            input_dtypes, vec_input_dtypes, pargs..., paddle::none);
      } else {
        return InferDtypeCallHelper<Tail...>::template InferDtype<in_idx + 1,
                                                                  vec_in_idx>(
            input_dtypes, vec_input_dtypes, pargs..., arg);
      }
    }
  };

616 617 618 619
  // NOTE(chenweihang): Used to be compatible with the 2.0.1 released
  // interface, and will be deprecated in the future
  PD_SPECIALIZE_InferDtypeCallHelper_TO_DTYPE(DataType);
  PD_SPECIALIZE_InferDtypeCallHelper_FOR_DTYPES(std::vector<DataType>);
620 621 622 623

  // end: base template
  template <typename T>
  struct InferDtypeCallHelper<TypeTag<T>> {
624 625
    template <int in_idx, int vec_in_idx>
    static Return InferDtype(
626 627
        const std::vector<DataType>& input_dtypes,
        const std::vector<std::vector<DataType>>& vec_input_dtypes,
628
        const Args&... args) {
629 630 631 632 633 634 635 636 637 638
      return impl_fn(args...);
    }
  };
};

#define PD_INFER_DTYPE(...) \
  ::paddle::InferDtypeFuncImpl<decltype(&__VA_ARGS__), &__VA_ARGS__>::InferDtype

////////////////////// Op Meta Info //////////////////////

639
class PADDLE_API OpMetaInfo {
640 641
 public:
  explicit OpMetaInfo(const std::string& op_name) : name_(op_name) {}
642 643

  // format: {"<name1>", "<name2>", ...}
644
  OpMetaInfo& Inputs(std::vector<std::string>&& inputs);
645 646

  // format: {"<name1>", "<name2>", ...}
647
  OpMetaInfo& Outputs(std::vector<std::string>&& outputs);
648

649
  // format: {"<name1>:<type1>", "<name2>:<type2>", ...}
650 651
  OpMetaInfo& Attrs(std::vector<std::string>&& attrs);

652 653
  // format: {"<input_name1>:<output_name1>",
  // "<input_name2>:<output_name2>",...}
654
  OpMetaInfo& SetInplaceMap(
655 656
      std::unordered_map<std::string, std::string>&& inplace_map);

657
  // format: PD_KERNEL(...)
658
  OpMetaInfo& SetKernelFn(KernelFunc&& func);
659 660

  // format: PD_INFER_SHAPE(...)
661
  OpMetaInfo& SetInferShapeFn(InferShapeFunc&& func);
662 663

  // format: PD_INFER_DTYPE(...)
664 665 666
  OpMetaInfo& SetInferDtypeFn(InferDtypeFunc&& func);

 private:
667
  friend class OpMetaInfoHelper;
668 669 670 671 672 673

  // 1. desc info
  std::string name_;
  std::vector<std::string> inputs_;
  std::vector<std::string> outputs_;
  std::vector<std::string> attrs_;
674
  std::unordered_map<std::string, std::string> inplace_map_;
675
  // 2. func info
676 677 678
  KernelFunc kernel_fn_{nullptr};
  InferShapeFunc infer_shape_fn_{nullptr};
  InferDtypeFunc infer_dtype_fn_{nullptr};
679 680
};

681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713
//////////////// Op Meta Info Helper /////////////////
class OpMetaInfoHelper {
 public:
  static const std::string& GetOpName(const paddle::OpMetaInfo& info) {
    return info.name_;
  }
  static const std::vector<std::string>& GetInputs(
      const paddle::OpMetaInfo& info) {
    return info.inputs_;
  }
  static const std::vector<std::string>& GetOutputs(
      const paddle::OpMetaInfo& info) {
    return info.outputs_;
  }
  static const std::vector<std::string>& GetAttrs(
      const paddle::OpMetaInfo& info) {
    return info.attrs_;
  }
  static const std::unordered_map<std::string, std::string>& GetInplaceMap(
      const paddle::OpMetaInfo& info) {
    return info.inplace_map_;
  }
  static const KernelFunc& GetKernelFn(const paddle::OpMetaInfo& info) {
    return info.kernel_fn_;
  }
  static const InferShapeFunc& GetInferShapeFn(const paddle::OpMetaInfo& info) {
    return info.infer_shape_fn_;
  }
  static const InferDtypeFunc& GetInferDtypeFn(const paddle::OpMetaInfo& info) {
    return info.infer_dtype_fn_;
  }
};

714 715
//////////////// Op Meta Info Map /////////////////

716
class PADDLE_API OpMetaInfoMap {
717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739
 public:
  // this function's impl should keep in header file.
  // if move to cc file, meta info can not be added
  // into map
  static OpMetaInfoMap& Instance() {
    static OpMetaInfoMap g_custom_op_meta_info_map;
    return g_custom_op_meta_info_map;
  }

  std::vector<OpMetaInfo>& operator[](const std::string& name);

  const std::unordered_map<std::string, std::vector<OpMetaInfo>>& GetMap()
      const;

 private:
  OpMetaInfoMap() = default;
  std::unordered_map<std::string, std::vector<OpMetaInfo>> map_;

  PD_DISABLE_COPY_AND_ASSIGN(OpMetaInfoMap);
};

//////////////// Op Meta Info Builder /////////////////

740
class PADDLE_API OpMetaInfoBuilder {
741
 public:
742
  explicit OpMetaInfoBuilder(std::string&& name, size_t index);
743 744
  OpMetaInfoBuilder& Inputs(std::vector<std::string>&& inputs);
  OpMetaInfoBuilder& Outputs(std::vector<std::string>&& outputs);
745
  OpMetaInfoBuilder& Attrs(std::vector<std::string>&& attrs);
746
  OpMetaInfoBuilder& SetInplaceMap(
747
      std::unordered_map<std::string, std::string>&& inplace_map);
748 749 750
  OpMetaInfoBuilder& SetKernelFn(KernelFunc func);
  OpMetaInfoBuilder& SetInferShapeFn(InferShapeFunc func);
  OpMetaInfoBuilder& SetInferDtypeFn(InferDtypeFunc func);
751 752 753 754

 private:
  // Forward Op name
  std::string name_;
755
  // ref current info ptr
756
  OpMetaInfo* info_ptr_;
757 758 759
  // The current op meta info index in vector
  // - 0: op, 1: grad_op, 2: grad_grad_op
  size_t index_;
760 761 762 763
};

/////////////////////// Op register Macro /////////////////////////

764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782
#define PD_BUILD_OP(op_name)                                                   \
  STATIC_ASSERT_GLOBAL_NAMESPACE(                                              \
      __reg_op__##op_name, "PD_BUILD_OP must be called in global namespace."); \
  static ::paddle::OpMetaInfoBuilder __op_meta_info_##op_name##__ =            \
      ::paddle::OpMetaInfoBuilder(#op_name, 0)

#define PD_BUILD_GRAD_OP(op_name)                                        \
  STATIC_ASSERT_GLOBAL_NAMESPACE(                                        \
      __reg_grad_op__##op_name,                                          \
      "PD_BUILD_GRAD_OP must be called in global namespace.");           \
  static ::paddle::OpMetaInfoBuilder __grad_op_meta_info_##op_name##__ = \
      ::paddle::OpMetaInfoBuilder(#op_name, 1)

#define PD_BUILD_DOUBLE_GRAD_OP(op_name)                                      \
  STATIC_ASSERT_GLOBAL_NAMESPACE(                                             \
      __reg_grad_grad_op__##op_name,                                          \
      "PD_BUILD_DOUBLE_GRAD_OP must be called in global namespace.");         \
  static ::paddle::OpMetaInfoBuilder __grad_grad_op_meta_info_##op_name##__ = \
      ::paddle::OpMetaInfoBuilder(#op_name, 2)
783

784 785 786 787 788 789 790 791
}  // namespace paddle

///////////////////// C API ///////////////////

#ifdef __cplusplus
extern "C" {
#endif

792
#if defined(_WIN32)
793
// C-API to get global OpMetaInfoMap.
794 795 796 797
__declspec(dllexport) inline paddle::OpMetaInfoMap& PD_GetOpMetaInfoMap() {
  return paddle::OpMetaInfoMap::Instance();
}
#endif  // _WIN32
798 799 800 801

#ifdef __cplusplus
}
#endif