Merge branch 'develop' into huber_loss

8 years ago · 7596191821
parent 73ab2d4678 b9a4b2ee52
commit 7596191821
105 changed files with 3073 additions and 577 deletions
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@ -55,6 +55,7 @@ option(WITH_C_API       "Compile PaddlePaddle with C-API(Prediction)"   OFF)
 option(WITH_GOLANG      "Compile PaddlePaddle with GOLANG"              OFF)
 option(GLIDE_INSTALL    "Download and install go dependencies "         ON)
 option(USE_NNPACK       "Compile PaddlePaddle with NNPACK library"      OFF)
 option(USE_EIGEN_FOR_BLAS   "Use matrix multiplication in Eigen"        OFF)
 # CMAKE_BUILD_TYPE
 if(NOT CMAKE_BUILD_TYPE)
--- a/cmake/configure.cmake
+++ b/cmake/configure.cmake
@ -28,6 +28,10 @@ if(NOT WITH_TIMER)
    add_definitions(-DPADDLE_DISABLE_TIMER)
 endif(NOT WITH_TIMER)
 if(USE_EIGEN_FOR_BLAS)
    add_definitions(-DPADDLE_USE_EIGEN_FOR_BLAS)
 endif(USE_EIGEN_FOR_BLAS)
 if(NOT WITH_PROFILER)
    add_definitions(-DPADDLE_DISABLE_PROFILER)
 endif(NOT WITH_PROFILER)
--- a/cmake/cudnn.cmake
+++ b/cmake/cudnn.cmake
@ -2,7 +2,7 @@ if(NOT WITH_GPU)
    return()
 endif()
-set(CUDNN_ROOT "" CACHE PATH "CUDNN ROOT")
+set(CUDNN_ROOT "/usr" CACHE PATH "CUDNN ROOT")
 find_path(CUDNN_INCLUDE_DIR cudnn.h
    PATHS ${CUDNN_ROOT} ${CUDNN_ROOT}/include
    $ENV{CUDNN_ROOT} $ENV{CUDNN_ROOT}/include ${CUDA_TOOLKIT_INCLUDE}
--- a/doc/api/v2/config/layer.rst
+++ b/doc/api/v2/config/layer.rst
@ -257,6 +257,11 @@ seq_concat
 ..  autoclass:: paddle.v2.layer.seq_concat
    :noindex:
 seq_slice
 ---------
 ..  autoclass:: paddle.v2.layer.seq_slice
    :noindex:
 kmax_sequence_score
 -------------------
 ..  autoclass:: paddle.v2.layer.kmax_sequence_score
@ -362,6 +367,11 @@ trans
 ..  autoclass:: paddle.v2.layer.trans
    :noindex:
 scale_shift
 -----------
 ..  autoclass:: paddle.v2.layer.scale_shift
    :noindex:
 Sampling Layers
 ===============
--- a/go/master/client.go
+++ b/go/master/client.go
@ -63,13 +63,24 @@ func WithAddr(addr string) func(c *Client) error {
 // WithEtcd sets the client to use etcd for master discovery.
 func WithEtcd(endpoints []string, timeout time.Duration) func(*Client) error {
 	return func(c *Client) error {
-		cli, err := clientv3.New(clientv3.Config{
+		var cli *clientv3.Client
 		f := func() error {
 			var err error
 			cli, err = clientv3.New(clientv3.Config{
 				Endpoints:   endpoints,
 				DialTimeout: timeout,
 			})
 		if err != nil {
 			return err
 		}
 		for {
 			err := f()
 			if err != nil {
 				log.Warningln(err)
 			} else {
 				break
 			}
 			time.Sleep(time.Second)
 		}
 		ch := make(chan string, 1)
 		a, err := GetKey(cli, DefaultAddrPath, timeout)
@ -101,9 +112,6 @@ func NewClient(opts ...func(*Client) error) (*Client, error) {
 		}
 	}
 	c.ch = make(chan record, c.bufSize)
 	// FIXME: connection is created asyncrosly in monitorMaster go routine,
 	//        ensure the connection is ready for use before calling c.addClient.
 	time.Sleep(time.Second)
 	return c, nil
 }
--- a/paddle/CMakeLists.txt
+++ b/paddle/CMakeLists.txt
@ -15,6 +15,7 @@ if(Boost_FOUND)
  add_subdirectory(platform)
  add_subdirectory(framework)
  add_subdirectory(operators)
  add_subdirectory(pybind)
 endif()
 if(WITH_C_API)
--- a/paddle/framework/CMakeLists.txt
+++ b/paddle/framework/CMakeLists.txt
@ -18,8 +18,8 @@ cc_test(scope_test SRCS scope_test.cc DEPS scope)
 proto_library(framework_proto SRCS framework.proto)
 cc_library(attribute SRCS attribute.cc DEPS framework_proto)
-
+cc_library(op_info SRCS op_info.cc DEPS attribute framework_proto)
-cc_library(operator SRCS operator.cc DEPS framework_proto device_context tensor scope attribute)
+cc_library(operator SRCS operator.cc DEPS op_info device_context tensor scope)
 cc_test(operator_test SRCS operator_test.cc DEPS operator op_registry)
 cc_library(grad_op_builder SRCS grad_op_builder.cc DEPS operator)
@ -39,21 +39,3 @@ add_custom_command(TARGET framework_py_proto POST_BUILD
 cc_library(backward SRCS backward.cc DEPS net_op)
 cc_test(backward_test SRCS backward_test.cc DEPS backward recurrent_op device_context)
 if(WITH_PYTHON)
 cc_library(paddle_pybind SHARED
    SRCS pybind.cc
    DEPS pybind python backward
    sgd_op
    add_op
    mul_op
    rowwise_add_op
    sigmoid_op
    softmax_op
    mean_op
    cross_entropy_op
    recurrent_op
    uniform_random_op
    gaussian_random_op
    fill_zeros_like_op)
 endif(WITH_PYTHON)
--- a/paddle/framework/backward.cc
+++ b/paddle/framework/backward.cc
@ -110,7 +110,7 @@ static std::unique_ptr<OperatorBase> BackwardRecursive(
                       dup_output_ops[out].emplace_back(local_op_id);
                       return false;
                     });
-      net->AddOp(std::move(bwd));
+      net->AppendOp(std::move(bwd));
    }
    // Get unique ID for this method.
    auto uid = uniq_id++;
@ -163,7 +163,8 @@ static std::unique_ptr<OperatorBase> BackwardRecursive(
        // If part of input gradient of that operator is not calculated, fill
        // zero variables to that input gradient.
-        net->AddOp(OpRegistry::CreateOp("fill_zeros_like", {{"Src", {prefix}}},
+        net->AppendOp(OpRegistry::CreateOp("fill_zeros_like",
                                           {{"Src", {prefix}}},
                                           {{"Dst", {grad_input}}}, {}));
      }
      return false;
@ -195,7 +196,7 @@ static std::unique_ptr<OperatorBase> BackwardRecursive(
    if (net->ops_.empty()) {  // Current no aux op is added to network
      return grad_op;
    }
-    net->AddOp(std::move(grad_op));
+    net->AppendOp(std::move(grad_op));
  }
  net->SetType("@GENERATED_BACKWARD@");
  net->CompleteAddOp();
--- a/paddle/framework/backward_test.cc
+++ b/paddle/framework/backward_test.cc
@ -72,16 +72,16 @@ class NoGradOpMaker : public OpProtoAndCheckerMaker {
 class FcOp : public operators::NetOp {
 public:
-  FcOp(const std::string &type, const VarNameMap &inputs,
+  FcOp(const std::string &type, const VariableNameMap &inputs,
-       const VarNameMap &outputs, const AttributeMap &attrs)
+       const VariableNameMap &outputs, const AttributeMap &attrs)
      : NetOp(type, inputs, outputs, attrs) {
-    AddOp(OpRegistry::CreateOp("mul",
+    AppendOp(OpRegistry::CreateOp("mul",
                                  {{"X", {Input("X")}}, {"Y", {Input("W")}}},
                                  {{"Out", {Output("mul_result")}}}, {}));
    auto input_b = Inputs("b");
    std::string before_act = "mul_result";
    if (input_b.size() != 0) {
-      AddOp(OpRegistry::CreateOp(
+      AppendOp(OpRegistry::CreateOp(
          "rowwise_add", {{"X", {Output("mul_result")}}, {"b", {input_b[0]}}},
          {{"Out", {Output("add_result")}}}, {}));
      before_act = "add_result";
@ -92,7 +92,7 @@ class FcOp : public operators::NetOp {
      }
    }
-    AddOp(OpRegistry::CreateOp("sigmoid", {{"X", {Output(before_act)}}},
+    AppendOp(OpRegistry::CreateOp("sigmoid", {{"X", {Output(before_act)}}},
                                  {{"Out", {Output("Out")}}}, {}));
    CompleteAddOp(false);
  }
@ -234,13 +234,13 @@ TEST(Backward, net_fc_backward_not_have_b) {
 TEST(Backward, net_input_of_network_not_need_grad) {
  ops::NetOp net;
-  net.AddOp(f::OpRegistry::CreateOp(
+  net.AppendOp(f::OpRegistry::CreateOp(
      "fc", {{"X", {"x"}}, {"W", {"W1"}}, {"b", {"b1"}}},
      {{"mul_result", {"mul_tmp_0"}},
       {"add_result", {"add_tmp_0"}},
       {"Out", {"hidden0"}}},
      {}));
-  net.AddOp(f::OpRegistry::CreateOp(
+  net.AppendOp(f::OpRegistry::CreateOp(
      "fc", {{"X", {"hidden0"}}, {"W", {"W2"}}, {"b", {"b2"}}},
      {{"mul_result", {"mul_tmp_1"}},
       {"add_result", {"add_tmp_1"}},
@ -273,9 +273,9 @@ TEST(Backward, net_input_of_network_not_need_grad) {
 TEST(Backward, net_shared_weight) {
  ops::NetOp net;
-  net.AddOp(f::OpRegistry::CreateOp("mul", {{"X", {"x"}}, {"Y", {"w"}}},
+  net.AppendOp(f::OpRegistry::CreateOp("mul", {{"X", {"x"}}, {"Y", {"w"}}},
                                       {{"Out", {"out"}}}, {}));
-  net.AddOp(f::OpRegistry::CreateOp("mul", {{"X", {"out"}}, {"Y", {"w"}}},
+  net.AppendOp(f::OpRegistry::CreateOp("mul", {{"X", {"out"}}, {"Y", {"w"}}},
                                       {{"Out", {"FinalOut"}}}, {}));
  net.CompleteAddOp();
@ -357,19 +357,19 @@ TEST(Backward, op_part_of_input_are_not_need) {
 TEST(Backward, linear_net_intermediate_variable_has_no_grad) {
  ops::NetOp net;
-  net.AddOp(f::OpRegistry::CreateOp(
+  net.AppendOp(f::OpRegistry::CreateOp(
      "fc", {{"X", {"x1"}}, {"W", {"w1"}}, {"b", {"b1"}}},
      {{"mul_result", {"mul_out1"}},
       {"add_result", {"add_out1"}},
       {"Out", {"out1"}}},
      {}));
-  net.AddOp(f::OpRegistry::CreateOp(
+  net.AppendOp(f::OpRegistry::CreateOp(
      "fc", {{"X", {"out1"}}, {"W", {"w2"}}, {"b", {"b2"}}},
      {{"mul_result", {"mul_out2"}},
       {"add_result", {"tmp_out2"}},
       {"Out", {"out2"}}},
      {}));
-  net.AddOp(f::OpRegistry::CreateOp(
+  net.AppendOp(f::OpRegistry::CreateOp(
      "fc", {{"X", {"out2"}}, {"W", {"w3"}}, {"b", {"b3"}}},
      {{"mul_result", {"mul_out3"}},
       {"add_result", {"tmp_out3"}},
--- a/paddle/framework/grad_op_builder.cc
+++ b/paddle/framework/grad_op_builder.cc
@ -20,13 +20,13 @@ namespace framework {
 enum class OpArgType { IN, OUT };
 static void TransOpArg(const OperatorBase* src_op, const OpArgType& src_type,
-                       bool is_grad, OperatorBase::VarNameMap* vars) {
+                       bool is_grad, VariableNameMap* vars) {
  const auto& src_inout =
      src_type == OpArgType::IN ? src_op->Inputs() : src_op->Outputs();
  auto& dst_inout = *vars;
-  const OpProto* proto = OpRegistry::op_info_map().at(src_op->Type()).proto_;
+  auto& proto = OpInfoMap::Instance().Get(src_op->Type()).Proto();
  const auto& src_arg_list =
-      src_type == OpArgType::IN ? proto->inputs() : proto->outputs();
+      src_type == OpArgType::IN ? proto.inputs() : proto.outputs();
  for (const auto& arg : src_arg_list) {
    if (arg.not_in_gradient() && !is_grad) continue;
    const std::string src_name = arg.name();
@ -40,26 +40,18 @@ static void TransOpArg(const OperatorBase* src_op, const OpArgType& src_type,
 }
 OperatorBase* BuildGradOp(const OperatorBase* op) {
-  auto it = OpRegistry::op_info_map().find(op->Type());
+  auto& info = OpInfoMap::Instance().Get(op->Type());
-  PADDLE_ENFORCE(it != OpRegistry::op_info_map().end(),
+  PADDLE_ENFORCE(info.HasGradientOp());
                 "'%s' has not been registered.", op->Type());
  PADDLE_ENFORCE(it->second.proto_ != nullptr, "'%s' has no OpProto.",
                 op->Type());
  std::string grad_op_type = it->second.grad_op_type_;
  PADDLE_ENFORCE(!grad_op_type.empty(), "'%s' has no gradient operator.",
                 op->Type());
-  OperatorBase::VarNameMap inputs;
+  VariableNameMap inputs;
-  OperatorBase::VarNameMap outputs;
+  VariableNameMap outputs;
  TransOpArg(op, OpArgType::IN, false, &inputs);   // I
  TransOpArg(op, OpArgType::OUT, false, &inputs);  // O
  TransOpArg(op, OpArgType::OUT, true, &inputs);   // OG
  TransOpArg(op, OpArgType::IN, true, &outputs);   // IG
-  it = OpRegistry::op_info_map().find(grad_op_type);
+  auto& grad_info = OpInfoMap::Instance().Get(info.grad_op_type_);
-  PADDLE_ENFORCE(it != OpRegistry::op_info_map().end(),
+  return grad_info.Creator()(info.grad_op_type_, inputs, outputs, op->Attrs());
                 "'%s' has not been registered.", grad_op_type);
  return it->second.creator_(grad_op_type, inputs, outputs, op->Attrs());
 }
 }  // namespace framework
--- a/paddle/framework/op_info.cc
+++ b/paddle/framework/op_info.cc
@ -0,0 +1,29 @@
 /* Copyright (c) 2016 PaddlePaddle Authors. All Rights Reserve.
   Licensed under the Apache License, Version 2.0 (the "License");
   you may not use this file except in compliance with the License.
   You may obtain a copy of the License at
   http://www.apache.org/licenses/LICENSE-2.0
   Unless required by applicable law or agreed to in writing, software
   distributed under the License is distributed on an "AS IS" BASIS,
   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
   See the License for the specific language governing permissions and
   limitations under the License. */
 #include "paddle/framework/op_info.h"
 namespace paddle {
 namespace framework {
 static OpInfoMap* g_op_info_map = nullptr;
 OpInfoMap& OpInfoMap::Instance() {
  if (g_op_info_map == nullptr) {
    g_op_info_map = new OpInfoMap();
  }
  return *g_op_info_map;
 }
 }  // namespace framework
 }  // namespace paddle
--- a/paddle/framework/op_info.h
+++ b/paddle/framework/op_info.h
@ -0,0 +1,101 @@
 /* Copyright (c) 2016 PaddlePaddle Authors. All Rights Reserve.
   Licensed under the Apache License, Version 2.0 (the "License");
   you may not use this file except in compliance with the License.
   You may obtain a copy of the License at
   http://www.apache.org/licenses/LICENSE-2.0
   Unless required by applicable law or agreed to in writing, software
   distributed under the License is distributed on an "AS IS" BASIS,
   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
   See the License for the specific language governing permissions and
   limitations under the License. */
 #pragma once
 #include <functional>
 #include <map>
 #include <string>
 #include <unordered_map>
 #include "paddle/framework/attribute.h"
 namespace paddle {
 namespace framework {
 class OperatorBase;
 using VariableNameMap = std::map<std::string, std::vector<std::string>>;
 using OpCreator = std::function<OperatorBase*(
    const std::string& /*type*/, const VariableNameMap& /*inputs*/,
    const VariableNameMap& /*outputs*/, const AttributeMap& /*attrs*/)>;
 struct OpInfo {
  OpCreator creator_;
  std::string grad_op_type_;
  OpProto* proto_;
  OpAttrChecker* checker_;
  bool HasOpProtoAndChecker() const {
    return proto_ != nullptr && checker_ != nullptr;
  }
  const OpProto& Proto() const {
    PADDLE_ENFORCE_NOT_NULL(proto_, "Operator Proto has not been registered");
    PADDLE_ENFORCE(proto_->IsInitialized(),
                   "Operator Proto must be initialized in op info");
    return *proto_;
  }
  const OpAttrChecker& Checker() const {
    PADDLE_ENFORCE_NOT_NULL(checker_,
                            "Operator Checker has not been registered");
    return *checker_;
  }
  const OpCreator& Creator() const {
    PADDLE_ENFORCE_NOT_NULL(creator_,
                            "Operator Creator has not been registered");
    return creator_;
  }
  bool HasGradientOp() const { return !grad_op_type_.empty(); }
 };
 class OpInfoMap {
 public:
  static OpInfoMap& Instance();
  OpInfoMap(const OpInfoMap& o) = delete;
  OpInfoMap(OpInfoMap&& o) = delete;
  OpInfoMap& operator=(const OpInfoMap& o) = delete;
  OpInfoMap& operator=(OpInfoMap&& o) = delete;
  bool Has(const std::string& op_type) const {
    return map_.find(op_type) != map_.end();
  }
  void Insert(const std::string& type, const OpInfo& info) {
    PADDLE_ENFORCE(!Has(type), "Operator %s has been registered", type);
    map_.insert({type, info});
  }
  const OpInfo& Get(const std::string& type) const {
    auto it = map_.find(type);
    PADDLE_ENFORCE(it != map_.end(), "Operator %s are not found", type);
    return it->second;
  }
  template <typename Callback>
  void IterAllInfo(Callback callback) {
    for (auto& it : map_) {
      callback(it.first, it.second);
    }
  }
 private:
  OpInfoMap() = default;
  std::unordered_map<std::string, const OpInfo> map_;
 };
 }  // namespace framework
 }  // namespace paddle
--- a/paddle/framework/op_registry.cc
+++ b/paddle/framework/op_registry.cc
@ -19,32 +19,18 @@ limitations under the License. */
 namespace paddle {
 namespace framework {
-std::unique_ptr<OperatorBase> OpRegistry::CreateOp(const std::string& type,
+std::unique_ptr<OperatorBase> OpRegistry::CreateOp(
-                                                   const VarNameMap& inputs,
+    const std::string& type, const VariableNameMap& inputs,
-                                                   const VarNameMap& outputs,
+    const VariableNameMap& outputs, AttributeMap attrs) {
-                                                   AttributeMap attrs) {
+  auto& info = OpInfoMap::Instance().Get(type);
-  auto it = op_info_map().find(type);
+  info.Checker().Check(attrs);
-  PADDLE_ENFORCE(it != op_info_map().end(),
+  auto op = info.Creator()(type, inputs, outputs, attrs);
                 "Operator '%s' has not been registered.", type);
  it->second.checker_->Check(attrs);
  auto op = it->second.creator_(type, inputs, outputs, attrs);
  return std::unique_ptr<OperatorBase>(op);
 }
-std::unique_ptr<OperatorBase> OpRegistry::CreateOp(const OpDesc& op_desc) {
+static VariableNameMap ConvertOpDescVarsToVarNameMap(
  VarNameMap inputs = ConvertOpDescVarsToVarNameMap(op_desc.inputs());
  VarNameMap outputs = ConvertOpDescVarsToVarNameMap(op_desc.outputs());
  AttributeMap attrs;
  for (auto& attr : op_desc.attrs()) {
    attrs[attr.name()] = GetAttrValue(attr);
  }
  return CreateOp(op_desc.type(), inputs, outputs, attrs);
 }
 OperatorBase::VarNameMap OpRegistry::ConvertOpDescVarsToVarNameMap(
    const google::protobuf::RepeatedPtrField<OpDesc::Var>& op_desc_vars) {
-  VarNameMap ret_val;
+  VariableNameMap ret_val;
  for (auto& var : op_desc_vars) {
    auto& var_names = ret_val[var.parameter()];
    auto& var_names_in_proto = var.arguments();
@ -55,6 +41,17 @@ OperatorBase::VarNameMap OpRegistry::ConvertOpDescVarsToVarNameMap(
  return ret_val;
 }
 std::unique_ptr<OperatorBase> OpRegistry::CreateOp(const OpDesc& op_desc) {
  VariableNameMap inputs = ConvertOpDescVarsToVarNameMap(op_desc.inputs());
  VariableNameMap outputs = ConvertOpDescVarsToVarNameMap(op_desc.outputs());
  AttributeMap attrs;
  for (auto& attr : op_desc.attrs()) {
    attrs[attr.name()] = GetAttrValue(attr);
  }
  return CreateOp(op_desc.type(), inputs, outputs, attrs);
 }
 std::unique_ptr<OperatorBase> OpRegistry::CreateGradOp(const OperatorBase& op) {
  PADDLE_ENFORCE(!op.IsNetOp(), "Use framework::Backward to get backward ops");
  return std::unique_ptr<OperatorBase>(BuildGradOp(&op));
--- a/paddle/framework/op_registry.h
+++ b/paddle/framework/op_registry.h
@ -23,6 +23,7 @@ limitations under the License. */
 #include "paddle/framework/attribute.h"
 #include "paddle/framework/framework.pb.h"
 #include "paddle/framework/grad_op_builder.h"
 #include "paddle/framework/op_info.h"
 #include "paddle/framework/operator.h"
 #include "paddle/framework/scope.h"
@ -30,28 +31,16 @@ namespace paddle {
 namespace framework {
 class OpRegistry {
  using VarNameMap = OperatorBase::VarNameMap;
  using OpCreator = std::function<OperatorBase*(
      const std::string& /*type*/, const VarNameMap& /*inputs*/,
      const VarNameMap& /*outputs*/, const AttributeMap& /*attrs*/)>;
 public:
  struct OpInfo {
    OpCreator creator_;
    std::string grad_op_type_;
    OpProto* proto_;
    OpAttrChecker* checker_;
  };
  template <typename OpType, typename ProtoMakerType, typename GradOpType>
  static void RegisterOp(const std::string& op_type,
                         const std::string& grad_op_type) {
-    PADDLE_ENFORCE(op_info_map().count(op_type) == 0,
+    PADDLE_ENFORCE(!OpInfoMap::Instance().Has(op_type),
                   "'%s' is registered more than once.", op_type);
    OpInfo op_info;
-    op_info.creator_ = [](const std::string& type, const VarNameMap& inputs,
+    op_info.creator_ = [](
-                          const VarNameMap& outputs,
+        const std::string& type, const VariableNameMap& inputs,
-                          const AttributeMap& attrs) {
+        const VariableNameMap& outputs, const AttributeMap& attrs) {
      return new OpType(type, inputs, outputs, attrs);
    };
    op_info.grad_op_type_ = grad_op_type;
@ -70,7 +59,7 @@ class OpRegistry {
      op_info.proto_ = nullptr;
      op_info.checker_ = nullptr;
    }
-    op_info_map().insert(std::make_pair(op_type, op_info));
+    OpInfoMap::Instance().Insert(op_type, op_info);
    // register gradient op
    if (!grad_op_type.empty()) {
      RegisterOp<GradOpType, NOPMaker, NOP>(grad_op_type, "");
@ -78,21 +67,13 @@ class OpRegistry {
  }
  static std::unique_ptr<OperatorBase> CreateOp(const std::string& type,
-                                                const VarNameMap& inputs,
+                                                const VariableNameMap& inputs,
-                                                const VarNameMap& outputs,
+                                                const VariableNameMap& outputs,
                                                AttributeMap attrs);
  static std::unique_ptr<OperatorBase> CreateOp(const OpDesc& op_desc);
  static VarNameMap ConvertOpDescVarsToVarNameMap(
      const google::protobuf::RepeatedPtrField<OpDesc::Var>& op_desc_vars);
  static std::unique_ptr<OperatorBase> CreateGradOp(const OperatorBase& op);
  static std::unordered_map<std::string, const OpInfo>& op_info_map() {
    static std::unordered_map<std::string, const OpInfo> op_info_map_;
    return op_info_map_;
  }
 };
 class Registrar {
--- a/paddle/framework/operator.cc
+++ b/paddle/framework/operator.cc
@ -115,8 +115,8 @@ void OperatorBase::Rename(const std::string& old_name,
 }
 OperatorBase::OperatorBase(const std::string& type,
-                           const OperatorBase::VarNameMap& inputs,
+                           const VariableNameMap& inputs,
-                           const OperatorBase::VarNameMap& outputs,
+                           const VariableNameMap& outputs,
                           const AttributeMap& attrs)
    : type_(type), inputs_(inputs), outputs_(outputs), attrs_(attrs) {
  static std::atomic<size_t> gUniqId(0UL);
@ -141,18 +141,10 @@ std::vector<std::string> OperatorBase::OutputVars(bool has_intermediate) const {
    }
    return ret_val;
  }
-  auto it = OpRegistry::op_info_map().find(type_);
+  auto& info = OpInfoMap::Instance().Get(Type());
  PADDLE_ENFORCE(
      it != OpRegistry::op_info_map().end(),
      "Operator %s not registered, cannot figure out intermediate outputs",
      type_);
  PADDLE_ENFORCE(
      it->second.proto_ != nullptr,
      "Operator %s has no OpProto, cannot figure out intermediate outputs",
      type_);
  // get all OpProto::Var for outputs
-  for (auto& o : it->second.proto_->outputs()) {
+  for (auto& o : info.Proto().outputs()) {
    // ignore all intermediate output
    if (o.intermediate()) continue;
    auto out = outputs_.find(o.name());
--- a/paddle/framework/operator.h
+++ b/paddle/framework/operator.h
@ -19,6 +19,7 @@ limitations under the License. */
 #include <unordered_map>
 #include <vector>
 #include "op_info.h"
 #include "paddle/framework/attribute.h"
 #include "paddle/framework/framework.pb.h"
 #include "paddle/framework/scope.h"
@ -62,10 +63,8 @@ class ExecutionContext;
 */
 class OperatorBase {
 public:
-  using VarNameMap = std::map<std::string, std::vector<std::string>>;
+  OperatorBase(const std::string& type, const VariableNameMap& inputs,
-
+               const VariableNameMap& outputs, const AttributeMap& attrs);
  OperatorBase(const std::string& type, const VarNameMap& inputs,
               const VarNameMap& outputs, const AttributeMap& attrs);
  virtual ~OperatorBase() {}
@ -93,8 +92,8 @@ class OperatorBase {
  /// rename inputs outputs name
  void Rename(const std::string& old_name, const std::string& new_name);
-  const VarNameMap& Inputs() const { return inputs_; }
+  const VariableNameMap& Inputs() const { return inputs_; }
-  const VarNameMap& Outputs() const { return outputs_; }
+  const VariableNameMap& Outputs() const { return outputs_; }
  //! Get a input with argument's name described in `op_proto`
  const std::string& Input(const std::string& name) const;
  //! Get a input which has multiple variables.
@ -122,30 +121,32 @@ class OperatorBase {
  // I (Inputs)opear
  // O (Outputs)
  // OG (Output Gradients)
-  VarNameMap inputs_;
+  VariableNameMap inputs_;
  // NOTE: in case of OpGrad, outputs_ contains
  // IG (Inputs Gradients)
-  VarNameMap outputs_;
+  VariableNameMap outputs_;
  AttributeMap attrs_;
 };
 // Macro for define a clone method.
 // If you are writing an kernel operator, `Clone` will be defined when you
 // register it. i.e. `Clone` method is not needed to define by yourself.
-#define DEFINE_OP_CLONE_METHOD(CLS)                       \
+#define DEFINE_OP_CLONE_METHOD(cls)                       \
  std::unique_ptr<OperatorBase> Clone() const final {     \
-    return std::unique_ptr<OperatorBase>(new CLS(*this)); \
+    return std::unique_ptr<OperatorBase>(new cls(*this)); \
  }
 // Macro for define a default constructor for Operator.
 // You can also use
 //   using PARENT_CLASS::PARENT_CLASS;
 // to use parent's constructor.
-#define DEFINE_OP_CONSTRUCTOR(CLS, PARENT_CLS)                                 \
+#define DEFINE_OP_CONSTRUCTOR(cls, parent_cls)             \
-  CLS(const std::string& type, const VarNameMap& inputs,                       \
+  cls(const std::string& type,                             \
-      const VarNameMap& outputs, const paddle::framework::AttributeMap& attrs) \
+      const ::paddle::framework::VariableNameMap& inputs,  \
-      : PARENT_CLS(type, inputs, outputs, attrs) {}
+      const ::paddle::framework::VariableNameMap& outputs, \
      const paddle::framework::AttributeMap& attrs)        \
      : parent_cls(type, inputs, outputs, attrs) {}
 class NOP : public OperatorBase {
 public:
@ -389,8 +390,8 @@ class OperatorWithKernel : public OperatorBase {
  using OpKernelMap =
      std::unordered_map<OpKernelKey, std::unique_ptr<OpKernel>, OpKernelHash>;
-  OperatorWithKernel(const std::string& type, const VarNameMap& inputs,
+  OperatorWithKernel(const std::string& type, const VariableNameMap& inputs,
-                     const VarNameMap& outputs, const AttributeMap& attrs)
+                     const VariableNameMap& outputs, const AttributeMap& attrs)
      : OperatorBase(type, inputs, outputs, attrs) {}
  void InferShape(const Scope& scope) const override {
--- a/paddle/framework/operator_test.cc
+++ b/paddle/framework/operator_test.cc
@ -23,8 +23,8 @@ static int op_run_num = 0;
 class OpWithoutKernelTest : public OperatorBase {
 public:
-  OpWithoutKernelTest(const std::string& type, const VarNameMap& inputs,
+  OpWithoutKernelTest(const std::string& type, const VariableNameMap& inputs,
-                      const VarNameMap& outputs, const AttributeMap& attrs)
+                      const VariableNameMap& outputs, const AttributeMap& attrs)
      : OperatorBase(type, inputs, outputs, attrs), x(1) {}
  void InferShape(const Scope& scope) const override {}
  void Run(const Scope& scope,
@ -249,8 +249,9 @@ TEST(OpKernel, multi_inputs) {
 class OperatorClone : public paddle::framework::OperatorBase {
 public:
  DEFINE_OP_CLONE_METHOD(OperatorClone);
-  OperatorClone(const std::string& type, const VarNameMap& inputs,
+  OperatorClone(const std::string& type,
-                const VarNameMap& outputs,
+                const paddle::framework::VariableNameMap& inputs,
                const paddle::framework::VariableNameMap& outputs,
                const paddle::framework::AttributeMap& attrs)
      : OperatorBase(type, inputs, outputs, attrs) {}
  void InferShape(const paddle::framework::Scope& scope) const override {}
--- a/paddle/framework/tensor.h
+++ b/paddle/framework/tensor.h
@ -105,7 +105,10 @@ class Tensor {
  template <typename T>
  inline Tensor Slice(const int& begin_idx, const int& end_idx) const;
-  platform::Place place() const { return holder_->place(); }
+  platform::Place place() const {
    PADDLE_ENFORCE_NOT_NULL(holder_, "Tensor get place() must contains holder");
    return holder_->place();
  }
 private:
  template <typename T>
--- a/paddle/function/CMakeLists.txt
+++ b/paddle/function/CMakeLists.txt
@ -4,6 +4,10 @@ file(GLOB cpp_files . *Op.cpp)
 list(APPEND h_files Function.h)
 list(APPEND cpp_files Function.cpp)
 list(APPEND cpp_files BufferArg.cpp)
 list(APPEND cpp_files GemmFunctor.cpp)
 if(USE_EIGEN_FOR_BLAS)
  list(APPEND cpp_files EigenGemm.cpp)
 endif(USE_EIGEN_FOR_BLAS)
 if(WITH_GPU)
    file(GLOB cu_files . *OpGpu.cu)
--- a/paddle/function/DepthwiseConvOp.cpp
+++ b/paddle/function/DepthwiseConvOp.cpp
@ -14,7 +14,6 @@ limitations under the License. */
 #include "DepthwiseConvOp.h"
 #include "ConvOp.h"
 #include "GemmFunctor.h"
 namespace paddle {
--- a/paddle/function/DepthwiseConvOpGpu.cu
+++ b/paddle/function/DepthwiseConvOpGpu.cu
@ -13,7 +13,6 @@ See the License for the specific language governing permissions and
 limitations under the License. */
 #include "DepthwiseConvOp.h"
 #include "GemmFunctor.h"
 #include "paddle/math/BaseMatrix.h"
 namespace paddle {
--- a/paddle/function/EigenGemm.cpp
+++ b/paddle/function/EigenGemm.cpp
@ -0,0 +1,91 @@
 /* Copyright (c) 2016 PaddlePaddle Authors. All Rights Reserve.
 Licensed under the Apache License, Version 2.0 (the "License");
 you may not use this file except in compliance with the License.
 You may obtain a copy of the License at
    http://www.apache.org/licenses/LICENSE-2.0
 Unless required by applicable law or agreed to in writing, software
 distributed under the License is distributed on an "AS IS" BASIS,
 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 See the License for the specific language governing permissions and
 limitations under the License. */
 #include <glog/logging.h>
 #include "unsupported/Eigen/CXX11/Tensor"
 namespace paddle {
 template <class T>
 struct EigenBlasGemm {
  typedef Eigen::TensorMap<Eigen::Tensor<T, 2, Eigen::RowMajor, int>,
                           Eigen::Aligned>
      Matrix;
  static void compute(const bool transA,
                      const bool transB,
                      const int M,
                      const int N,
                      const int K,
                      const T alpha,
                      const T* A,
                      const int lda,
                      const T* B,
                      const int ldb,
                      const T beta,
                      T* C,
                      const int ldc) {
    Eigen::array<int, 2> sizeA;
    if (transA) {
      sizeA[0] = K;
      sizeA[1] = M;
      CHECK_EQ(M, lda);
    } else {
      sizeA[0] = M;
      sizeA[1] = K;
      CHECK_EQ(K, lda);
    }
    Eigen::array<int, 2> sizeB;
    if (transB) {
      sizeB[0] = N;
      sizeB[1] = K;
      CHECK_EQ(K, ldb);
    } else {
      sizeB[0] = K;
      sizeB[1] = N;
      CHECK_EQ(N, ldb);
    }
    Eigen::array<int, 2> sizeC;
    sizeC[0] = M;
    sizeC[1] = N;
    CHECK_EQ(N, ldc);
    const Matrix a(const_cast<T*>(A), sizeA);
    const Matrix b(const_cast<T*>(B), sizeB);
    Matrix c(C, sizeC);
    typedef typename Eigen::Tensor<T, 2>::DimensionPair DimPair;
    Eigen::array<DimPair, 1> dims;
    dims[0] = DimPair(1, 0);
    dims[0].first = transA ? 0 : 1;
    dims[0].second = transB ? 1 : 0;
    Eigen::DefaultDevice device;
    if (alpha == T(1) && beta == T(0)) {
      c.device(device) = a.contract(b, dims);
    } else if (alpha == T(1) && beta == T(1)) {
      c.device(device) += a.contract(b, dims);
    } else {
      c.device(device) = alpha * a.contract(b, dims) + beta * c;
    }
  }
 };
 #ifdef PADDLE_TYPE_DOUBLE
 template class EigenBlasGemm<double>;
 #else
 template class EigenBlasGemm<float>;
 #endif
 }  // namespace paddle
--- a/paddle/function/GemmConvOp.cpp
+++ b/paddle/function/GemmConvOp.cpp
@ -85,7 +85,6 @@ public:
    }
    Im2ColFunctor<kCFO, Device, real> im2col;
    GemmFunctor<Device, real> gemm;
    size_t inputOffset = imShape.getElements();
    size_t outputOffset =
        (outputChannels / groups_) * outputHeight * outputWidth;
@ -108,8 +107,8 @@ public:
        int M = outputChannels / groups_;
        int N = outputHeight * outputWidth;
        int K = inputChannels / groups_ * filterHeight * filterWidth;
-        gemm(CblasNoTrans,
+        BlasGemm<Device, real>::compute(false,
-             CblasNoTrans,
+                                        false,
                                        M,
                                        N,
                                        K,
@ -188,8 +187,6 @@ public:
    }
    Col2ImFunctor<kCFO, Device, real> col2im;
    GemmFunctor<Device, real> gemm;
    size_t inputOffset = imShape.getElements();
    size_t outputOffset =
        (outputChannels / groups_) * outputHeight * outputWidth;
@ -205,8 +202,8 @@ public:
          colData = inputGrad + g * inputOffset;
          scale = 1.0f;
        }
-        gemm(CblasTrans,
+        BlasGemm<Device, real>::compute(true,
-             CblasNoTrans,
+                                        false,
                                        M,
                                        N,
                                        K,
@ -299,7 +296,6 @@ public:
    }
    Im2ColFunctor<kCFO, Device, real> im2col;
    GemmFunctor<Device, real> gemm;
    size_t inputOffset = imShape.getElements();
    size_t outputOffset =
        (outputChannels / groups_) * outputHeight * outputWidth;
@ -321,8 +317,8 @@ public:
        int M = outputChannels / groups_;
        int K = outputHeight * outputWidth;
        int N = inputChannels / groups_ * filterHeight * filterWidth;
-        gemm(CblasNoTrans,
+        BlasGemm<Device, real>::compute(false,
-             CblasTrans,
+                                        true,
                                        M,
                                        N,
                                        K,
--- a/paddle/function/GemmFunctor.cpp
+++ b/paddle/function/GemmFunctor.cpp
@ -0,0 +1,90 @@
 /* Copyright (c) 2016 PaddlePaddle Authors. All Rights Reserve.
 Licensed under the Apache License, Version 2.0 (the "License");
 you may not use this file except in compliance with the License.
 You may obtain a copy of the License at
    http://www.apache.org/licenses/LICENSE-2.0
 Unless required by applicable law or agreed to in writing, software
 distributed under the License is distributed on an "AS IS" BASIS,
 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 See the License for the specific language governing permissions and
 limitations under the License. */
 #include "GemmFunctor.h"
 #include "paddle/math/MathFunctions.h"
 namespace paddle {
 template <class T>
 struct BlasGemm<DEVICE_TYPE_CPU, T> {
  static void compute(const bool transA,
                      const bool transB,
                      const int M,
                      const int N,
                      const int K,
                      const T alpha,
                      const T* A,
                      const int lda,
                      const T* B,
                      const int ldb,
                      const T beta,
                      T* C,
                      const int ldc) {
 #ifdef PADDLE_USE_EIGEN_FOR_BLAS
    EigenBlasGemm<T>::compute(
        transA, transB, M, N, K, alpha, A, lda, B, ldb, beta, C, ldc);
 #else
    gemm<T>(transA == false ? CblasNoTrans : CblasTrans,
            transB == false ? CblasNoTrans : CblasTrans,
            M,
            N,
            K,
            alpha,
            A,
            lda,
            B,
            ldb,
            beta,
            C,
            ldc);
 #endif
  }
 };
 template <class T>
 struct BlasGemm<DEVICE_TYPE_GPU, T> {
  static void compute(const bool transA,
                      const bool transB,
                      const int M,
                      const int N,
                      const int K,
                      const T alpha,
                      const T* A,
                      const int lda,
                      const T* B,
                      const int ldb,
                      const T beta,
                      T* C,
                      const int ldc) {
    hl_matrix_mul((T*)A,
                  transA == false ? HPPL_OP_N : HPPL_OP_T,
                  (T*)B,
                  transB == false ? HPPL_OP_N : HPPL_OP_T,
                  C,
                  M,
                  N,
                  K,
                  alpha,
                  beta,
                  lda,
                  ldb,
                  ldc);
  }
 };
 template struct BlasGemm<DEVICE_TYPE_CPU, real>;
 template struct BlasGemm<DEVICE_TYPE_GPU, real>;
 }  // namespace paddle
--- a/paddle/function/GemmFunctor.h
+++ b/paddle/function/GemmFunctor.h
@ -14,7 +14,7 @@ limitations under the License. */
 #pragma once
-#include "paddle/math/MathFunctions.h"
+#include "TensorType.h"
 namespace paddle {
@ -24,10 +24,9 @@ namespace paddle {
 // of MatMulFunction, we need to consider the reconstruction of hl_matrix_mul
 // interface.
 template <DeviceType Device, class T>
-class GemmFunctor {
+struct BlasGemm {
-public:
+  static void compute(const bool transA,
-  void operator()(const CBLAS_TRANSPOSE transA,
+                      const bool transB,
                  const CBLAS_TRANSPOSE TransB,
                      const int M,
                      const int N,
                      const int K,
@ -41,11 +40,15 @@ public:
                      const int ldc);
 };
 // TODO(hedaoyuan): Since the definition of the real type in the Paddle
 // conflicts with the Eigen library, so compile the Eigen code can not
 // include the Paddle header file. And need an EigenBlasGemm template class
 // that does not contain the DeviceType parameter.
 // I will fix this problem and merge BlasGemm and EigenBlasGemm into one.
 template <class T>
-class GemmFunctor<DEVICE_TYPE_CPU, T> {
+struct EigenBlasGemm {
-public:
+  static void compute(const bool transA,
-  void operator()(const CBLAS_TRANSPOSE transA,
+                      const bool transB,
                  const CBLAS_TRANSPOSE TransB,
                      const int M,
                      const int N,
                      const int K,
@ -56,41 +59,7 @@ public:
                      const int ldb,
                      const T beta,
                      T* C,
-                  const int ldc) {
+                      const int ldc);
    gemm<T>(transA, TransB, M, N, K, alpha, A, lda, B, ldb, beta, C, ldc);
  }
 };
 template <class T>
 class GemmFunctor<DEVICE_TYPE_GPU, T> {
 public:
  void operator()(const CBLAS_TRANSPOSE transA,
                  const CBLAS_TRANSPOSE TransB,
                  const int M,
                  const int N,
                  const int K,
                  const T alpha,
                  const T* A,
                  const int lda,
                  const T* B,
                  const int ldb,
                  const T beta,
                  T* C,
                  const int ldc) {
    hl_matrix_mul((T*)A,
                  transA == CblasNoTrans ? HPPL_OP_N : HPPL_OP_T,
                  (T*)B,
                  TransB == CblasNoTrans ? HPPL_OP_N : HPPL_OP_T,
                  C,
                  M,
                  N,
                  K,
                  alpha,
                  beta,
                  lda,
                  ldb,
                  ldc);
  }
 };
 }  // namespace paddle
--- a/Show More
+++ b/Show More