layer.cc 8.3 KB
Newer Older
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23
// Copyright (c) 2018 PaddlePaddle Authors. All Rights Reserved.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
//     http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

#include "paddle/fluid/imperative/layer.h"
#include <deque>
#include <limits>
#include <map>
#include <random>
#include <utility>

#include "paddle/fluid/framework/lod_tensor.h"
#include "paddle/fluid/framework/op_registry.h"
24
#include "paddle/fluid/framework/operator.h"
25 26 27 28 29
#include "paddle/fluid/string/printf.h"

namespace paddle {
namespace imperative {

X
polish  
Xin Pan 已提交
30 31
const char* PyLayer::kFwdInp = "X";
const char* PyLayer::kFwdOut = "Out";
X
polish  
Xin Pan 已提交
32

X
Xin Pan 已提交
33 34
std::map<int, py::object> py_funcs_;

35 36 37 38 39
using framework::Variable;

void AddTo(Variable* src, Variable* dst) {
  framework::LoDTensor* dst_tensor = dst->GetMutable<framework::LoDTensor>();
  framework::LoDTensor* src_tensor = src->GetMutable<framework::LoDTensor>();
M
minqiyang 已提交
40 41 42 43 44
  // FIXME(minqiyang): loss_grad op will pass a zero grad of label
  // ugly fix for it
  if (src_tensor->numel() == 0) {
    return;
  }
45 46 47
  PADDLE_ENFORCE(dst_tensor->numel() == src_tensor->numel(),
                 "dst_numel %lld vs. src_numel %lld", dst_tensor->numel(),
                 src_tensor->numel());
48 49
  float* dst_data = dst_tensor->mutable_data<float>(platform::CPUPlace());
  const float* src_data = src_tensor->data<float>();
M
minqiyang 已提交
50
  for (int64_t i = 0; i < src_tensor->numel(); ++i) {
51 52 53 54 55 56
    dst_data[i] += src_data[i];
  }
}

class Autograd {
 public:
X
Xin Pan 已提交
57
  Autograd() {}
58 59

  void RunBackward(VarBase* var) {
60 61 62
    if (var->stop_gradient_) {
      return;
    }
X
Xin Pan 已提交
63
    VLOG(3) << "start autograd";
64 65 66 67 68 69 70 71 72

    std::deque<OpBase*> ready;
    ready.push_back(var->pre_op_);

    std::map<OpBase*, int> dep_counts = ComputeDepCounts(var->pre_op_);

    while (!ready.empty()) {
      OpBase* ready_op = ready.front();
      ready.pop_front();
X
Xin Pan 已提交
73 74 75 76 77 78 79
      std::map<std::string, std::vector<VarBase*>> input_grads =
          ready_op->ApplyGrad();

      for (auto it : input_grads) {
        const std::vector<VarBase*>& ingrads = it.second;
        for (size_t i = 0; i < ingrads.size(); ++i) {
          if (!ingrads[i]) continue;
80 81 82
          if (ready_op->input_vars_[it.first][i]->stop_gradient_) {
            continue;
          }
X
Xin Pan 已提交
83
          OpBase* pre_op = ready_op->pre_ops_[it.first][i];
X
Xin Pan 已提交
84 85 86 87 88 89 90 91
          if (!pre_op) continue;

          dep_counts[pre_op] -= 1;
          PADDLE_ENFORCE(dep_counts[pre_op] >= 0);
          bool pre_op_ready = dep_counts[pre_op] == 0;
          if (pre_op_ready) {
            ready.push_back(pre_op);
          }
92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107
        }
      }
    }
  }

 private:
  std::map<OpBase*, int> ComputeDepCounts(OpBase* op) {
    std::map<OpBase*, int> ret;

    std::deque<OpBase*> queue;
    queue.push_back(op);
    std::unordered_set<OpBase*> visited;
    visited.insert(op);
    while (!queue.empty()) {
      OpBase* candidate = queue.front();
      queue.pop_front();
X
Xin Pan 已提交
108
      for (auto it : candidate->pre_ops_) {
X
Xin Pan 已提交
109 110 111 112 113 114 115
        for (OpBase* pre_op : it.second) {
          if (!pre_op) continue;
          if (visited.find(pre_op) == visited.end()) {
            visited.insert(pre_op);
            queue.push_back(pre_op);
          }
          ret[pre_op] += 1;
116 117 118 119 120 121 122
        }
      }
    }
    return ret;
  }
};

M
minqiyang 已提交
123
framework::LoDTensor& VarBase::GradValue() {
124
  VLOG(3) << "get var grad " << var_desc_->Name();
M
minqiyang 已提交
125
  return *(grads_->var_->GetMutable<framework::LoDTensor>());
126 127
}

X
Xin Pan 已提交
128
std::map<std::string, std::vector<VarBase*>> OpBase::ApplyGrad() {
X
Xin Pan 已提交
129
  if (!grad_op_desc_ && backward_id_ <= 0) {
130
    LOG(WARNING) << "op with no grad: " << op_desc_->Type();
X
Xin Pan 已提交
131
    return {};
132 133
  }

X
Xin Pan 已提交
134
  std::map<std::string, std::vector<framework::Variable*>> grad_outputs;
X
Xin Pan 已提交
135 136
  if (backward_id_ > 0) {
    VLOG(3) << "py_layer_grad";
X
Xin Pan 已提交
137 138 139
    grad_outputs[framework::GradVarName(PyLayer::kFwdOut)] = PyLayer::ApplyGrad(
        backward_id_,
        grad_input_vars_[framework::GradVarName(PyLayer::kFwdInp)]);
X
Xin Pan 已提交
140 141
  } else {
    VLOG(3) << "op grad " << grad_op_desc_->Type();
X
polish  
Xin Pan 已提交
142 143 144 145 146 147 148 149
    for (auto it : grad_output_vars_) {
      auto& outputs = grad_outputs[it.first];
      for (size_t i = 0; i < it.second.size(); ++i) {
        // Allocate a new variable
        Variable* tmp_var = new framework::Variable();
        tmp_var->GetMutable<framework::LoDTensor>();
        outputs.push_back(tmp_var);
      }
150 151
    }

X
Xin Pan 已提交
152
    framework::RuntimeContext ctx(grad_input_vars_, grad_outputs);
153

X
Xin Pan 已提交
154 155 156
    // No need to do compile time infer shape here.
    // grad_op_desc_->InferShape(*block_);
    grad_op_desc_->InferVarType(block_);
X
Xin Pan 已提交
157

X
Xin Pan 已提交
158 159 160 161 162
    std::unique_ptr<framework::OperatorBase> opbase =
        framework::OpRegistry::CreateOp(*grad_op_desc_);
    framework::OperatorWithKernel* op_kernel =
        dynamic_cast<framework::OperatorWithKernel*>(opbase.get());
    PADDLE_ENFORCE_NOT_NULL(op_kernel, "only support op with kernel");
X
Xin Pan 已提交
163

X
Xin Pan 已提交
164 165 166 167 168 169
    framework::Scope scope;
    platform::CPUPlace place;
    PreparedOp p = PreparedOp::Prepare(ctx, *op_kernel, place);
    p.op.RuntimeInferShape(scope, place, ctx);
    p.func(framework::ExecutionContext(p.op, scope, *p.dev_ctx, p.ctx));
  }
X
Xin Pan 已提交
170 171 172 173

  for (auto it : grad_output_vars_) {
    auto& outputs = grad_outputs[it.first];
    auto& origin_outputs = it.second;
X
polish  
Xin Pan 已提交
174
    PADDLE_ENFORCE_EQ(outputs.size(), origin_outputs.size());
175

X
Xin Pan 已提交
176
    for (size_t i = 0; i < outputs.size(); ++i) {
X
polish  
Xin Pan 已提交
177
      framework::Variable* grad = outputs[i];
M
minqiyang 已提交
178
      framework::Variable* orig_grad = origin_outputs[i];
X
polish  
Xin Pan 已提交
179 180
      AddTo(grad, orig_grad);
      delete grad;
181 182
    }
  }
X
Xin Pan 已提交
183
  return input_vars_;
184 185
}

X
Xin Pan 已提交
186
void VarBase::RunBackward() {
187
  if (!pre_op_) return;
X
Xin Pan 已提交
188

X
Xin Pan 已提交
189
  VLOG(3) << "start backward";
M
minqiyang 已提交
190
  auto grads_t = grads_->var_->GetMutable<framework::LoDTensor>();
X
Xin Pan 已提交
191 192 193
  float* data = grads_t->mutable_data<float>(platform::CPUPlace());
  std::fill(data, data + grads_t->numel(), 1.0);

X
Xin Pan 已提交
194 195 196
  PADDLE_ENFORCE(
      grads_ ==
      pre_op_->output_vars_[pre_op_out_name_][pre_op_out_idx_]->grads_);
X
Xin Pan 已提交
197
  Autograd().RunBackward(this);
198 199
}

X
Xin Pan 已提交
200 201 202 203
void PyLayer::RegisterFunc(int func_id, const py::object& py_func) {
  py_funcs_[func_id] = py_func;
}

X
polish  
Xin Pan 已提交
204 205
int PyLayer::NumFuncs() { return py_funcs_.size(); }

X
Xin Pan 已提交
206
std::vector<VarBase*> PyLayer::Apply(int func_id,
X
Xin Pan 已提交
207
                                     const std::vector<VarBase*>& inputs) {
X
polish  
Xin Pan 已提交
208
  std::vector<framework::Variable*> invars;
X
Xin Pan 已提交
209
  for (const VarBase* in : inputs) {
X
polish  
Xin Pan 已提交
210
    invars.push_back(in->var_);
X
Xin Pan 已提交
211 212
  }
  PADDLE_ENFORCE(py_funcs_.find(func_id) != py_funcs_.end());
X
polish  
Xin Pan 已提交
213 214 215
  std::vector<Variable*> outvars = CallPythonFunc(py_funcs_[func_id], invars);
  std::vector<VarBase*> ret;
  for (Variable* v : outvars) {
216
    ret.push_back(new VarBase(v, new VarBase(true)));
X
polish  
Xin Pan 已提交
217
  }
X
Xin Pan 已提交
218 219 220
  return ret;
}

X
polish  
Xin Pan 已提交
221 222 223 224 225
std::vector<Variable*> PyLayer::ApplyGrad(
    int func_id, const std::vector<framework::Variable*>& inputs) {
  PADDLE_ENFORCE(py_funcs_.find(func_id) != py_funcs_.end());
  return CallPythonFunc(py_funcs_[func_id], inputs);
}
X
Xin Pan 已提交
226

X
polish  
Xin Pan 已提交
227 228 229 230 231 232 233
std::vector<framework::Variable*> PyLayer::CallPythonFunc(
    const py::object& callable, const std::vector<framework::Variable*>& ins) {
  py::gil_scoped_acquire guard;
  py::tuple in_args(ins.size());
  for (size_t i = 0; i < ins.size(); ++i) {
    const framework::LoDTensor& t = ins[i]->Get<framework::LoDTensor>();
    in_args[i] = t.IsInitialized() ? py::cast(t) : py::cast(nullptr);
X
Xin Pan 已提交
234
  }
X
polish  
Xin Pan 已提交
235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257
  VLOG(3) << "pyfunc in " << py::len(in_args);

  // TODO(panyx0718): Who owns the returned LoDTensor.
  auto ret = callable(in_args);
  auto ret_tuple = py::cast<py::tuple>(ret);
  size_t ret_num = py::len(ret_tuple);
  std::vector<framework::Variable*> outs;
  VLOG(3) << "pyfunc out " << ret_num;
  for (size_t i = 0; i < ret_num; ++i) {
    try {
      auto* py_out_tensor = py::cast<framework::LoDTensor*>(ret_tuple[i]);
      PADDLE_ENFORCE_NOT_NULL(py_out_tensor,
                              "Output tensor %d should not be nullptr", i);
      auto* var = new framework::Variable();
      auto* tensor = var->GetMutable<framework::LoDTensor>();
      tensor->ShareDataWith(*py_out_tensor);
      tensor->set_lod(py_out_tensor->lod());
      outs.push_back(var);
    } catch (py::cast_error&) {
      PADDLE_THROW("The %d-th output must be LoDTensor", i);
    }
  }
  return outs;
X
Xin Pan 已提交
258 259
}

260 261
}  // namespace imperative
}  // namespace paddle