lizexu123
diff --git a/‎c++/ipu/custom_ops/CMakeLists.txt
+64 b/‎c++/ipu/custom_ops/CMakeLists.txt
+64
diff --git a/‎c++/ipu/custom_ops/README.md
+41 b/‎c++/ipu/custom_ops/README.md
+41
diff --git a/‎c++/ipu/custom_ops/compile.sh
+30 b/‎c++/ipu/custom_ops/compile.sh
+30
diff --git a/‎c++/ipu/custom_ops/custom_op_test.cc
+61 b/‎c++/ipu/custom_ops/custom_op_test.cc
+61
diff --git a/‎c++/ipu/custom_ops/custom_relu_op.cc
+86 b/‎c++/ipu/custom_ops/custom_relu_op.cc
+86
@@ -0,0 +1,64 @@
+cmake_minimum_required(VERSION 3.0)
+project(cpp_inference_demo CXX C)
+option(WITH_IPU        "Compile demo with IPU/CPU, default use CPU."                    OFF)
+option(CUSTOM_OPERATOR_FILES "List of file names for custom operators" "")
+
+set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} "${CMAKE_CURRENT_SOURCE_DIR}/cmake")
+
+if(NOT WITH_STATIC_LIB)
+  add_definitions("-DPADDLE_WITH_SHARED_LIB")
+else()
+  # PD_INFER_DECL is mainly used to set the dllimport/dllexport attribute in dynamic library mode.
+  # Set it to empty in static library mode to avoid compilation issues.
+  add_definitions("/DPD_INFER_DECL=")
+endif()
+
+macro(safe_set_static_flag)
+    foreach(flag_var
+        CMAKE_CXX_FLAGS CMAKE_CXX_FLAGS_DEBUG CMAKE_CXX_FLAGS_RELEASE
+        CMAKE_CXX_FLAGS_MINSIZEREL CMAKE_CXX_FLAGS_RELWITHDEBINFO)
+      if(${flag_var} MATCHES "/MD")
+        string(REGEX REPLACE "/MD" "/MT" ${flag_var} "${${flag_var}}")
+      endif(${flag_var} MATCHES "/MD")
+    endforeach(flag_var)
+endmacro()
+
+if(NOT DEFINED PADDLE_LIB)
+  message(FATAL_ERROR "please set PADDLE_LIB with -DPADDLE_LIB=/path/paddle/lib")
+endif()
+if(NOT DEFINED DEMO_NAME)
+  message(FATAL_ERROR "please set DEMO_NAME with -DDEMO_NAME=demo_name")
+endif()
+
+include_directories("${PADDLE_LIB}/")
+set(PADDLE_LIB_THIRD_PARTY_PATH "${PADDLE_LIB}/third_party/install/")
+include_directories("${PADDLE_LIB_THIRD_PARTY_PATH}protobuf/include")
+include_directories("${PADDLE_LIB_THIRD_PARTY_PATH}glog/include")
+include_directories("${PADDLE_LIB_THIRD_PARTY_PATH}gflags/include")
+include_directories("${PADDLE_LIB_THIRD_PARTY_PATH}xxhash/include")
+include_directories("${PADDLE_LIB_THIRD_PARTY_PATH}onnxruntime/include")
+include_directories("${PADDLE_LIB_THIRD_PARTY_PATH}paddle2onnx/include")
+
+link_directories("${PADDLE_LIB_THIRD_PARTY_PATH}protobuf/lib")
+link_directories("${PADDLE_LIB_THIRD_PARTY_PATH}glog/lib")
+link_directories("${PADDLE_LIB_THIRD_PARTY_PATH}gflags/lib")
+link_directories("${PADDLE_LIB_THIRD_PARTY_PATH}xxhash/lib")
+link_directories("${PADDLE_LIB}/paddle/lib")
+link_directories("${PADDLE_LIB_THIRD_PARTY_PATH}onnxruntime/lib")
+link_directories("${PADDLE_LIB_THIRD_PARTY_PATH}paddle2onnx/lib")
+
+set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14")
+if(WITH_IPU)
+  set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DONNX_NAMESPACE=onnx")
+endif()
+
+set(DEPS ${PADDLE_LIB}/paddle/lib/libpaddle_inference${CMAKE_SHARED_LIBRARY_SUFFIX})
+set(EXTERNAL_LIB "-lrt -ldl -lpthread")
+set(DEPS ${DEPS}
+      glog gflags protobuf xxhash
+      ${EXTERNAL_LIB})
+
+add_library(pd_infer_custom_op SHARED ${CUSTOM_OPERATOR_FILES})
+add_executable(${DEMO_NAME} ${DEMO_NAME}.cc)
+set(DEPS ${DEPS} pd_infer_custom_op)
+target_link_libraries(${DEMO_NAME} ${DEPS})
@@ -0,0 +1,41 @@
+# 自定义算子模型构建运行示例
+
+## 一：获取本样例中的自定义算子模型
+下载地址：https://paddle-inference-dist.bj.bcebos.com/inference_demo/custom_operator/custom_relu_infer_model.tgz
+
+执行 `tar zxvf custom_relu_infer_model.tgz` 将模型文件解压至当前目录。
+
+## 二：**样例编译**
+
+文件 `custom_relu_op.cc`、`custom_relu_op_ipu.cc` 为自定义算子源文件，自定义算子编写方式请参考[飞桨官网文档](https://www.paddlepaddle.org.cn/documentation/docs/zh/guides/index_cn.html)。
+注意：自定义算子目前需要与飞桨预测库 `libpaddle_inference.so` 联合构建，启用 IPU 功能需要使用 IPU 版本的飞浆预测库。
+
+文件`custom_op_test.cc` 为预测的样例程序。
+文件`CMakeLists.txt` 为编译构建文件。
+脚本`compile.sh` 包含了第三方库、预编译库的信息配置。
+
+我们首先需要对脚本`compile.sh` 文件中的配置进行修改。
+
+1）**修改`compile.sh`**
+
+打开`compile.sh`，我们对以下的几处信息进行修改：
+
+```shell
+# 根据预编译库中的version.txt信息判断是否将以下标记打开
+WITH_IPU=ON
+
+# 配置预测库的根目录
+LIB_DIR=${work_path}/../lib/paddle_inference
+```
+
+运行 `bash compile.sh`， 会在目录下产生build目录。
+
+
+2） **运行样例**
+
+```shell
+# 运行样例
+./build/custom_op_test
+```
+
+运行结束后，程序会将模型结果打印到屏幕，说明运行成功。
@@ -0,0 +1,30 @@
+#!/bin/bash
+set +x
+set -e
+
+work_path=$(dirname $(readlink -f $0))
+
+# 1. check paddle_inference exists
+if [ ! -d "${work_path}/../../lib/paddle_inference" ]; then
+  echo "Please download paddle_inference lib and move it in Paddle-Inference-Demo/c++/lib"
+  exit 1
+fi
+
+# 2. compile
+mkdir -p build
+cd build
+
+DEMO_NAME=custom_op_test
+
+WITH_IPU=ON
+
+LIB_DIR=${work_path}/../../lib/paddle_inference
+
+
+cmake .. -DPADDLE_LIB=${LIB_DIR} \
+  -DDEMO_NAME=${DEMO_NAME} \
+  -DWITH_IPU=${WITH_IPU} \
+  -DWITH_STATIC_LIB=OFF \
+  -DCUSTOM_OPERATOR_FILES="custom_relu_op.cc;custom_relu_op_ipu.cc"
+
+make -j
@@ -0,0 +1,61 @@
+/* Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved.
+
+Licensed under the Apache License, Version 2.0 (the "License");
+you may not use this file except in compliance with the License.
+You may obtain a copy of the License at
+
+    http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software
+distributed under the License is distributed on an "AS IS" BASIS,
+WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and
+limitations under the License. */
+
+#include <numeric>
+#include <gflags/gflags.h>
+#include <glog/logging.h>
+
+#include "paddle/include/paddle_inference_api.h"
+
+using paddle_infer::Config;
+using paddle_infer::Predictor;
+using paddle_infer::CreatePredictor;
+
+void run(Predictor *predictor, const std::vector<float> &input,
+         const std::vector<int> &input_shape, std::vector<float> *out_data) {
+  auto input_names = predictor->GetInputNames();
+  auto input_t = predictor->GetInputHandle(input_names[0]);
+  input_t->Reshape(input_shape);
+  input_t->CopyFromCpu(input.data());
+
+  CHECK(predictor->Run());
+
+  auto output_names = predictor->GetOutputNames();
+  auto output_t = predictor->GetOutputHandle(output_names[0]);
+  std::vector<int> output_shape = output_t->shape();
+  int out_num = std::accumulate(output_shape.begin(), output_shape.end(), 1,
+                                std::multiplies<int>());
+
+  out_data->resize(out_num);
+  output_t->CopyToCpu(out_data->data());
+}
+
+int main() {
+  paddle::AnalysisConfig config;
+  config.SetModel("./custom_relu_infer_model/custom_relu.pdmodel",
+                  "./custom_relu_infer_model/custom_relu.pdiparams");
+  config.EnableIpu();
+  std::vector<std::vector<std::string>> custom_ops_info {
+                {"custom_relu", "Relu", "custom.ops", "1"}};
+  config.SetIpuCustomInfo(custom_ops_info);
+  auto predictor{paddle_infer::CreatePredictor(config)};
+  std::vector<int> input_shape = {1, 1, 28, 28};
+  std::vector<float> input_data(1 * 1 * 28 * 28, 1);
+  std::vector<float> out_data;
+  run(predictor.get(), input_data, input_shape, &out_data);
+  for (auto e : out_data) {
+    LOG(INFO) << e << '\n';
+  }
+  return 0;
+}
@@ -0,0 +1,86 @@
+// Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved.
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+#include <iostream>
+#include <vector>
+
+#include "paddle/include/experimental/ext_all.h"
+
+template <typename data_t>
+void relu_cpu_forward_kernel(const data_t* x_data,
+                             data_t* out_data,
+                             int64_t x_numel) {
+  for (int i = 0; i < x_numel; ++i) {
+    out_data[i] = std::max(static_cast<data_t>(0.), x_data[i]);
+  }
+}
+
+template <typename data_t>
+void relu_cpu_backward_kernel(const data_t* grad_out_data,
+                              const data_t* out_data,
+                              data_t* grad_x_data,
+                              int64_t out_numel) {
+  for (int i = 0; i < out_numel; ++i) {
+    grad_x_data[i] =
+        grad_out_data[i] * (out_data[i] > static_cast<data_t>(0) ? 1. : 0.);
+  }
+}
+
+std::vector<paddle::Tensor> relu_cpu_forward(const paddle::Tensor& x) {
+  auto out = paddle::Tensor(paddle::PlaceType::kCPU, x.shape());
+
+  PD_DISPATCH_FLOATING_TYPES(
+      x.type(), "relu_cpu_forward", ([&] {
+        relu_cpu_forward_kernel<data_t>(
+            x.data<data_t>(), out.mutable_data<data_t>(x.place()), x.size());
+      }));
+
+  return {out};
+}
+
+std::vector<paddle::Tensor> relu_cpu_backward(const paddle::Tensor& x,
+                                              const paddle::Tensor& out,
+                                              const paddle::Tensor& grad_out) {
+  auto grad_x = paddle::Tensor(paddle::PlaceType::kCPU, x.shape());
+
+  PD_DISPATCH_FLOATING_TYPES(out.type(), "relu_cpu_backward", ([&] {
+                               relu_cpu_backward_kernel<data_t>(
+                                   grad_out.data<data_t>(),
+                                   out.data<data_t>(),
+                                   grad_x.mutable_data<data_t>(x.place()),
+                                   out.size());
+                             }));
+
+  return {grad_x};
+}
+
+std::vector<paddle::Tensor> ReluForward(const paddle::Tensor& x) {
+  return relu_cpu_forward(x);
+}
+
+std::vector<paddle::Tensor> ReluBackward(const paddle::Tensor& x,
+                                         const paddle::Tensor& out,
+                                         const paddle::Tensor& grad_out) {
+  return relu_cpu_backward(x, out, grad_out);
+}
+
+PD_BUILD_OP(custom_relu)
+    .Inputs({"X"})
+    .Outputs({"Out"})
+    .SetKernelFn(PD_KERNEL(ReluForward));
+
+PD_BUILD_GRAD_OP(custom_relu)
+    .Inputs({"X", "Out", paddle::Grad("Out")})
+    .Outputs({paddle::Grad("X")})
+    .SetKernelFn(PD_KERNEL(ReluBackward));